Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -450,8 +450,9 @@ def _run_ifcopenshell_openusd_fallback(
except Exception:
pass

mapping_items: list[dict[str, str]] = []
mapping_items: list[dict[str, Any]] = []
entity_items: list[dict[str, Any]] = []
ifc_class_xforms: set[str] = set()
shape_count = 0
skipped_shape_count = 0
while True:
Expand All @@ -478,33 +479,63 @@ def _run_ifcopenshell_openusd_fallback(
skipped_shape_count += 1
else:
shape_count += 1
prim_path = f"/World/IfcShape_{shape_count:06d}"
ifc_guid = str(getattr(shape, "guid", "") or "")
ifc_name = str(getattr(shape, "name", "") or "")
ifc_type = str(getattr(shape, "type", "") or "")

# streaming-server-fallback-semantic-mapping:IFC-class grouped
# prim path. Per-class Xform 只 Define 一次;GUID/class 任一含
# 非 USD-legal 字元時走 _safe_usd_prim_name 做 sanitize。
# collision suffix 用 while-loop counter:不同原始 GUID
# (例如 `abc$` / `abc!` / `abc-`)可能 sanitize 成同一 token,
# 必須以「下一個未使用的 __N 後綴」確保 prim path 在 stage 內
# 唯一,避免 UsdGeom.Mesh.Define 對既有 prim reapply 而 silently
# overwrite 前一個 shape 的 mesh attr。
class_token = self._resolve_ifc_class_token(ifc_type)
guid_token = self._resolve_guid_token(ifc_guid, shape_count)
if class_token not in ifc_class_xforms:
UsdGeom.Xform.Define(stage, f"/World/{class_token}")
ifc_class_xforms.add(class_token)
candidate = f"/World/{class_token}/{guid_token}"
suffix = 1
while stage.GetPrimAtPath(candidate).IsValid():
candidate = f"/World/{class_token}/{guid_token}__{suffix}"
suffix += 1
prim_path = candidate

mesh = UsdGeom.Mesh.Define(stage, prim_path)
mesh.CreatePointsAttr(points)
mesh.CreateFaceVertexCountsAttr(face_counts)
mesh.CreateFaceVertexIndicesAttr(face_indices)
mesh.CreateExtentAttr(self._mesh_extent(points, vec3_type=Gf.Vec3f))

ifc_guid = str(getattr(shape, "guid", "") or "")
ifc_name = str(getattr(shape, "name", "") or "")
ifc_type = str(getattr(shape, "type", "") or "")
prim = mesh.GetPrim()
if ifc_guid:
prim.SetCustomDataByKey("ifcGlobalId", ifc_guid)
prim.SetCustomDataByKey("ifc_guid", ifc_guid)
mapping_items.append(
{"ifc_guid": ifc_guid, "usd_prim_path": prim_path}
)
if ifc_name:
prim.SetCustomDataByKey("ifcName", ifc_name)
if ifc_type:
prim.SetCustomDataByKey("ifcType", ifc_type)

entity_id = f"entity_{shape_count:06d}"
if ifc_guid:
mapping_items.append(
{
"ifc_guid": ifc_guid,
"usd_prim_path": prim_path,
"ifc_type": ifc_type or None,
"ifc_name": ifc_name or None,
"entity_id": entity_id,
}
)
entity_items.append(
{
"ifc_guid": ifc_guid or None,
"ifc_type": ifc_type or None,
"name": ifc_name or None,
"usd_prim_path": prim_path,
"entity_id": entity_id,
}
)

Expand Down Expand Up @@ -574,6 +605,12 @@ def _run_ifcopenshell_openusd_fallback(
"skipped_shape_count": skipped_shape_count,
"primary_converter_error": primary_error.message,
}
# streaming-server-fallback-semantic-mapping:declare semantic fidelity
# so coordinator /ui 與 viewer 可不必 re-parse mapping_items 即判定
# Semantic ready。flags 取 mapping_items 為主而非 entity_items,因為
# mapping items 才是 viewer 對外消費的 IFC GUID → USD prim 對照表。
mapping_has_ifc_type = any(item.get("ifc_type") for item in mapping_items)
mapping_has_ifc_name = any(item.get("ifc_name") for item in mapping_items)
quality_metrics = {
"source_ifc_entity_count": source_count,
"mapped_count": mapped_count,
Expand All @@ -583,6 +620,9 @@ def _run_ifcopenshell_openusd_fallback(
"materialization_strategy": "ifcopenshell_openusd_fallback",
"sidecar_carrier_count": shape_count,
"minimum_coverage_baseline_locked": False,
"semantic_mapping_fidelity": "ifc_class_grouped_with_name",
"mapping_has_ifc_type": mapping_has_ifc_type,
"mapping_has_ifc_name": mapping_has_ifc_name,
"hard_quality_gates": {
"usdc_openable": True,
"has_renderable_prims": True,
Expand All @@ -603,6 +643,34 @@ def _run_ifcopenshell_openusd_fallback(
json.dumps(quality_metrics, ensure_ascii=False), encoding="utf-8"
)

@staticmethod
def _safe_usd_prim_name(text: str) -> str | None:
"""Sanitize an arbitrary string into a USD-legal prim name segment.

USD prim names must match `[A-Za-z_][A-Za-z0-9_]*`. Returns ``None``
for empty input (callers fall back to a deterministic placeholder).
"""
if not text:
return None
sanitized = "".join(
ch if ch.isalnum() or ch == "_" else "_" for ch in text
)
Comment on lines +656 to +657

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Badge Enforce ASCII-only USD token sanitization

The new _safe_usd_prim_name implementation uses ch.isalnum(), which treats many non-ASCII Unicode characters as valid and therefore preserves them instead of replacing them with _. That breaks this change's own [A-Za-z0-9_] sanitization contract and can emit non-ASCII prim path segments for IFC class/GUID inputs containing localized characters, which may fail downstream consumers that validate or assume ASCII-safe paths. Please switch to an explicit ASCII whitelist check (or regex) when building sanitized tokens.

Useful? React with 👍 / 👎.

if not sanitized:
return None
if not (sanitized[0].isalpha() or sanitized[0] == "_"):
sanitized = "_" + sanitized
return sanitized
Comment on lines +653 to +662
Comment on lines +646 to +662

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟠 Major | ⚡ Quick win

ch.isalnum() allows non-ASCII alphanumeric characters, producing invalid USD prim names.

Python's str.isalnum() returns True for Unicode alphanumeric characters (e.g., '牆'.isalnum() → True, 'é'.isalnum() → True). IFC files from international projects may contain non-ASCII type/name strings. These characters will pass through unsanitized but are not valid in USD prim names ([A-Za-z0-9_] is ASCII-only), potentially causing USD runtime errors.

🐛 Proposed fix to restrict to ASCII alphanumeric
     `@staticmethod`
     def _safe_usd_prim_name(text: str) -> str | None:
         """Sanitize an arbitrary string into a USD-legal prim name segment.

         USD prim names must match `[A-Za-z_][A-Za-z0-9_]*`. Returns ``None``
         for empty input (callers fall back to a deterministic placeholder).
         """
         if not text:
             return None
         sanitized = "".join(
-            ch if ch.isalnum() or ch == "_" else "_" for ch in text
+            ch if (ch.isascii() and ch.isalnum()) or ch == "_" else "_" for ch in text
         )
         if not sanitized:
             return None
         if not (sanitized[0].isalpha() or sanitized[0] == "_"):
             sanitized = "_" + sanitized
         return sanitized
🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In
`@bim-streaming-server/source/extensions/ezplus.bim_review_stream.messaging/ezplus/bim_review_stream/messaging/ifc2usdc_powershell_adapter.py`
around lines 646 - 662, The _safe_usd_prim_name function currently uses
ch.isalnum() which allows non-ASCII characters; update it to only accept ASCII
letters and digits (A-Z, a-z, 0-9) and underscore so output matches USD regex
`[A-Za-z_][A-Za-z0-9_]*`. Change the character filter in _safe_usd_prim_name to
test membership against ASCII sets (e.g., string.ascii_letters and
string.digits) or an ASCII regex, and ensure the first character check enforces
ASCII letter or underscore; keep the same None fallback behavior when input or
sanitized result is empty.


@classmethod
def _resolve_ifc_class_token(cls, ifc_type: str) -> str:
"""Return USD-legal IFC class token, defaulting to ``Unclassified``."""
return cls._safe_usd_prim_name(ifc_type) or "Unclassified"

@classmethod
def _resolve_guid_token(cls, ifc_guid: str, shape_index: int) -> str:
"""Return USD-legal GUID token, defaulting to ``Shape_NNNNNN``."""
return cls._safe_usd_prim_name(ifc_guid) or f"Shape_{shape_index:06d}"

def _usd_mesh_data_from_ifcopenshell(
self,
*,
Expand Down
Loading