diff --git a/config/laboratories/e47-semantic-slam-shadow.json b/config/laboratories/e47-semantic-slam-shadow.json new file mode 100644 index 0000000..08e9f55 --- /dev/null +++ b/config/laboratories/e47-semantic-slam-shadow.json @@ -0,0 +1,10 @@ +{ + "schema_version": "missioncore.laboratory-evidence-definition/v1", + "work_id": "e47-semantic-slam-shadow", + "evidence": { + "runtime_relative_root": "e47/semantic-slam-results", + "result_id_prefix": "e47-semantic-slam", + "document_name": "manifest.json", + "schema_version": "missioncore.e47-semantic-slam-result/v1" + } +} diff --git a/config/laboratory-execution.json b/config/laboratory-execution.json index e971a1c..3b2b02b 100644 --- a/config/laboratory-execution.json +++ b/config/laboratory-execution.json @@ -67,6 +67,25 @@ "run": "missioncore.laboratory-run/v1", "evidence": "missioncore.e46j-raw-fisheye-realtime-result/v1" } + }, + { + "work_id": "e47-semantic-slam-shadow", + "lifecycle": "experimental", + "isolation": "bounded-adapter", + "adapter_id": "experimental.e47-semantic-slam-shadow/v1", + "input_roles": [ + "repository_root", + "semantic_result_root", + "threat_result_root", + "geometry_result_root" + ], + "contracts": { + "source": "missioncore.e47-semantic-slam-source-set/v1", + "provider": "missioncore.semantic-slam-diagnostic-provider/v1", + "graph": "missioncore.e47-semantic-slam-shadow-graph/v1", + "run": "missioncore.laboratory-run/v1", + "evidence": "missioncore.e47-semantic-slam-result/v1" + } } ], "legacy_work_ids": [ diff --git a/config/perception/e47-semantic-slam-shadow-v1.json b/config/perception/e47-semantic-slam-shadow-v1.json new file mode 100644 index 0000000..a733fed --- /dev/null +++ b/config/perception/e47-semantic-slam-shadow-v1.json @@ -0,0 +1,84 @@ +{ + "schema_version": "missioncore.e47-semantic-slam-profile/v1", + "profile_id": "ravnoves00-eomt-kb4-slam-shadow/v1", + "source": { + "source_id": "RAVNOVES00", + "session_id": "20260720T065719Z_viewer_live", + "frame_count": 4489, + "image_width": 800, + "image_height": 600, + "source_pack_id": "e10-lidar-pack-576c994a6c814e2592dd6240ace3902a5db94843312c759a73ba0c9166157d2b", + "source_pack_sha256": "0685d24219d8236caf8b7f1685e93f6d6b59e7fd015a768d88a92bbe8b154944", + "calibration_content_sha256": "05f3ad9b38b3a4fc95388a8ec83da83c745e217709e51787b3d5aad0969f6fa9" + }, + "semantic_provider": { + "provider_id": "eomt-cityscapes-semantic-control/v1", + "model_id": "tue-mps/cityscapes_semantic_eomt_large_1024", + "model_revision": "8d6b6d1a3f7b50d441afd7d247c2ed10db186e8f", + "model_weights_sha256": "c265da9a74f58f5c3f4826d23ca4ca78beac0b106cca5842beca61580de5b782", + "preprocess_id": "raw-kb4-valid-fov-semantic/v1", + "mask_metadata_schema_version": "missioncore.panoptic-frame/v1", + "mask_payload": { + "media_type": "image/png", + "encoding": "uint8-class-id", + "width": 800, + "height": 600, + "sequence_binding": "sequence-0-to-frame-000001" + }, + "role": "fixed-control-not-selected-production-provider" + }, + "taxonomy": [ + {"class_id": 0, "label": "outside_valid_fov", "disposition": "ambiguous", "color_rgb": [0, 0, 0]}, + {"class_id": 1, "label": "person", "disposition": "labeled", "color_rgb": [220, 20, 60]}, + {"class_id": 2, "label": "bicycle", "disposition": "labeled", "color_rgb": [119, 11, 32]}, + {"class_id": 3, "label": "motorcycle", "disposition": "labeled", "color_rgb": [0, 0, 230]}, + {"class_id": 4, "label": "car", "disposition": "labeled", "color_rgb": [0, 0, 142]}, + {"class_id": 5, "label": "heavy_vehicle", "disposition": "labeled", "color_rgb": [0, 0, 70]}, + {"class_id": 6, "label": "building_structure", "disposition": "labeled", "color_rgb": [70, 70, 70]}, + {"class_id": 7, "label": "paved_road", "disposition": "labeled", "color_rgb": [128, 64, 128]}, + {"class_id": 8, "label": "sidewalk_curb", "disposition": "labeled", "color_rgb": [244, 35, 232]}, + {"class_id": 9, "label": "ground_dirt", "disposition": "labeled", "color_rgb": [81, 0, 81]}, + {"class_id": 10, "label": "grass_low_vegetation", "disposition": "labeled", "color_rgb": [152, 251, 152]}, + {"class_id": 11, "label": "tree_woody_vegetation", "disposition": "labeled", "color_rgb": [107, 142, 35]}, + {"class_id": 12, "label": "sky", "disposition": "labeled", "color_rgb": [70, 130, 180]}, + {"class_id": 13, "label": "static_obstacle", "disposition": "labeled", "color_rgb": [220, 220, 0]}, + {"class_id": 14, "label": "animal", "disposition": "labeled", "color_rgb": [255, 127, 80]}, + {"class_id": 15, "label": "other_background", "disposition": "labeled", "color_rgb": [153, 153, 153]} + ], + "fusion": { + "projection": "factory-kb4-current-increment/v1", + "point_index_space": "frame-local-source-point-id/v1", + "observation_aggregation": "dominant-labeled-majority-diagnostic/v1", + "unprojected_status": "unprojected", + "semantic_absence_means_free": false, + "semantic_can_create_obstacle": false, + "semantic_can_change_identity": false, + "semantic_can_change_metric_geometry": false, + "semantic_can_change_occupancy": false, + "semantic_can_change_motion": false, + "semantic_can_change_threat": false + }, + "temporal_binding": { + "semantic_to_camera": "exact-sequence-and-session-time", + "camera_to_lidar": "accepted-e6-nearest-host-arrival-best-effort", + "clock_basis": "recorded-host-monotonic-arrival", + "maximum_lidar_camera_delta_ms": 100.0, + "maximum_pose_point_delta_ms": 100.0, + "physical_synchronization_proven": false + }, + "acceptance": { + "full_frame_accounting_required": true, + "point_accounting_required": true, + "observation_binding_required": true, + "exact_mask_archive_required": true, + "independent_semantic_truth_required_for_provider_promotion": true + }, + "authority": { + "ground_truth": false, + "physical_live": false, + "commands_enabled": false, + "actuation_allowed": false, + "navigation_or_safety_accepted": false, + "semantic_authority": "diagnostic-only" + } +} diff --git a/scripts/build_m4_worker_shadow_artifact.py b/scripts/build_m4_worker_shadow_artifact.py index 2bd082a..b279b05 100644 --- a/scripts/build_m4_worker_shadow_artifact.py +++ b/scripts/build_m4_worker_shadow_artifact.py @@ -25,7 +25,7 @@ WHEEL_NAME = "nodedc_mission_core-0.1.0-py3-none-any.whl" RUNNER_NAME = RUNNER.name PATCH_ID = re.compile(r"^[A-Za-z0-9._-]{1,96}$") EXPECTED_BASELINE_SHA256 = "ea10359339e6cce31b5780a2710299771cab7cc0c1c2a2b56a1621f786b31fa8" -EXPECTED_WHEEL_SHA256 = "2fc53bf3c2cd82a33e62b158a455d813bca64b844707e758792b4bac263b2543" +EXPECTED_WHEEL_SHA256 = "c396a202d5cddc2d22dcc3e8b936519205399d20b0f060c763c02e51ba16c62a" PAYLOAD_FILES = ( RUNNER_NAME, WHEEL_NAME, diff --git a/src/k1link/laboratory/execution.py b/src/k1link/laboratory/execution.py index e7ecc56..bd58932 100644 --- a/src/k1link/laboratory/execution.py +++ b/src/k1link/laboratory/execution.py @@ -315,6 +315,7 @@ def canonical_laboratory_adapters() -> dict[str, LaboratoryAdapter]: "canonical.e33-worker-shadow/v1": _run_e33, "canonical.e35-degradation-recovery/v1": _run_e35, "canonical.e46j-raw-fisheye-realtime/v1": _run_e46j, + "experimental.e47-semantic-slam-shadow/v1": _run_e47, } @@ -375,6 +376,22 @@ def _run_e46j(request: LaboratoryRunRequest) -> LaboratoryAdapterResult: ) +def _run_e47(request: LaboratoryRunRequest) -> LaboratoryAdapterResult: + from k1link.perception.semantic_slam_replay import build_semantic_slam_replay + + result = build_semantic_slam_replay( + repository_root=request.inputs["repository_root"], + semantic_result_root=request.inputs["semantic_result_root"], + threat_result_root=request.inputs["threat_result_root"], + geometry_result_root=request.inputs["geometry_result_root"], + output_root=request.output_root, + ) + return LaboratoryAdapterResult( + result_root=result.result_root, + result_id=result.result_id, + ) + + def _validate_request( request: LaboratoryRunRequest, definition: LaboratoryExecutionDefinition, diff --git a/src/k1link/perception/geometry.py b/src/k1link/perception/geometry.py index 1f27ad3..dba7b73 100644 --- a/src/k1link/perception/geometry.py +++ b/src/k1link/perception/geometry.py @@ -83,6 +83,22 @@ class GeometryFrame: return int(self.points_map.shape[0]) +@dataclass(frozen=True, slots=True) +class RecordedFrameTemporalBinding: + """Digest-bound recorded timing evidence for one camera-indexed increment. + + The shared session time binds the camera ordinal to the E10 pack entry. The + LiDAR and pose deltas retain their admitted E6 meaning: nearest host-arrival + best effort, not hardware synchronization. + """ + + frame_index: int + source_time_ns: int + source_available: bool + lidar_camera_delta_ms: float | None + pose_point_delta_ms: float | None + + @dataclass(frozen=True, slots=True) class ReplayBodyFrameInputs: """Verified inputs required to derive one replay-only virtual body frame.""" @@ -222,13 +238,29 @@ class RecordedGeometryStore: or pose_reference.frame_index != envelope.sequence ): raise GeometryProviderError("packet geometry references are not source-bound") - frame_index = envelope.sequence + frame = self.frame_for_index(envelope.sequence) + if frame is None: + raise GeometryProviderError("packet claims unavailable source geometry as current") + return frame + + def frame_for_index(self, frame_index: int) -> GeometryFrame | None: + """Expose one verified source increment with its pose and KB4 calibration. + + This read-only seam is intentionally narrower than the source archive. It + exists for deterministic replay diagnostics which must project the exact + frame-local point index space without manufacturing a ``SourcePacket``. + An unavailable recorded increment remains ``None``; surface validity is + retained on the returned frame rather than silently filtering its points. + """ + + if not isinstance(frame_index, int) or isinstance(frame_index, bool): + raise GeometryProviderError("replay geometry frame index is invalid") if not 0 <= frame_index < self.profile.frame_count: - raise GeometryProviderError("packet geometry frame index is outside the profile") + raise GeometryProviderError("replay geometry frame is outside the profile") if int(self._source["frame_indices"][frame_index]) != frame_index: raise GeometryProviderError("source pack frame sequence changed") if not bool(self._source["sample_available"][frame_index]): - raise GeometryProviderError("packet claims unavailable source geometry as current") + return None offsets = self._source["cloud_offsets"] start, end = int(offsets[frame_index]), int(offsets[frame_index + 1]) return GeometryFrame( @@ -247,6 +279,42 @@ class RecordedGeometryStore: surface_valid=bool(self._surface["frame_valid"][frame_index]), ) + def temporal_binding_for_index(self, frame_index: int) -> RecordedFrameTemporalBinding: + """Return the sealed ordinal/session binding and admitted best-effort deltas.""" + + if not isinstance(frame_index, int) or isinstance(frame_index, bool): + raise GeometryProviderError("replay temporal frame index is invalid") + if not 0 <= frame_index < self.profile.frame_count: + raise GeometryProviderError("replay temporal frame is outside the profile") + if ( + int(self._source["frame_indices"][frame_index]) != frame_index + or int(self._source["source_frame_indices"][frame_index]) != frame_index + ): + raise GeometryProviderError("source pack temporal sequence changed") + session_seconds = float(self._source["session_seconds"][frame_index]) + if not math.isfinite(session_seconds) or session_seconds < 0.0: + raise GeometryProviderError("source pack session time is invalid") + source_available = bool(self._source["sample_available"][frame_index]) + lidar_delta = float(self._source["lidar_camera_delta_ms"][frame_index]) + pose_delta = float(self._source["pose_point_delta_ms"][frame_index]) + if source_available: + if not math.isfinite(lidar_delta) or not math.isfinite(pose_delta): + raise GeometryProviderError("available source temporal deltas are invalid") + lidar_value: float | None = lidar_delta + pose_value: float | None = pose_delta + else: + if not math.isnan(lidar_delta) or not math.isnan(pose_delta): + raise GeometryProviderError("unavailable source carries temporal deltas") + lidar_value = None + pose_value = None + return RecordedFrameTemporalBinding( + frame_index=frame_index, + source_time_ns=round(session_seconds * 1_000_000_000), + source_available=source_available, + lidar_camera_delta_ms=lidar_value, + pose_point_delta_ms=pose_value, + ) + def current_points(self, packet: SourcePacket) -> FloatArray | None: """Expose the verified frame-local point index space to temporal occupancy.""" @@ -393,6 +461,8 @@ class RecordedGeometryStore: def _validate(self) -> None: source_required = { "frame_indices", + "source_frame_indices", + "session_seconds", "sample_available", "cloud_offsets", "cloud_points_map", @@ -401,6 +471,8 @@ class RecordedGeometryStore: "intrinsic_fx_fy_cx_cy", "distortion_kb4", "t_camera_from_lidar", + "lidar_camera_delta_ms", + "pose_point_delta_ms", } surface_required = {"frame_valid", "point_class"} if not source_required.issubset(self._source): @@ -411,6 +483,8 @@ class RecordedGeometryStore: points = self.profile.point_count shapes = { "frame_indices": (frames,), + "source_frame_indices": (frames,), + "session_seconds": (frames,), "sample_available": (frames,), "cloud_offsets": (frames + 1,), "cloud_points_map": (points, 3), @@ -419,6 +493,8 @@ class RecordedGeometryStore: "intrinsic_fx_fy_cx_cy": (4,), "distortion_kb4": (4,), "t_camera_from_lidar": (4, 4), + "lidar_camera_delta_ms": (frames,), + "pose_point_delta_ms": (frames,), } if any(self._source[name].shape != shape for name, shape in shapes.items()): raise GeometryProviderError("source pack array shapes changed") @@ -428,6 +504,26 @@ class RecordedGeometryStore: raise GeometryProviderError("local surface point shape changed") if int(self._source["cloud_offsets"][-1]) != points: raise GeometryProviderError("source point offsets do not close") + expected_indices = np.arange(frames, dtype=np.int64) + session_seconds = np.asarray(self._source["session_seconds"], dtype=np.float64) + if ( + not np.array_equal(self._source["frame_indices"], expected_indices) + or not np.array_equal(self._source["source_frame_indices"], expected_indices) + or not np.isfinite(session_seconds).all() + or np.any(session_seconds < 0.0) + or np.any(np.diff(session_seconds) <= 0.0) + ): + raise GeometryProviderError("source temporal index changed") + available = np.asarray(self._source["sample_available"], dtype=np.bool_) + lidar_deltas = np.asarray(self._source["lidar_camera_delta_ms"], dtype=np.float64) + pose_deltas = np.asarray(self._source["pose_point_delta_ms"], dtype=np.float64) + if ( + not np.isfinite(lidar_deltas[available]).all() + or not np.isfinite(pose_deltas[available]).all() + or not np.isnan(lidar_deltas[~available]).all() + or not np.isnan(pose_deltas[~available]).all() + ): + raise GeometryProviderError("source temporal delta availability changed") if ( int(np.count_nonzero(self._source["sample_available"])) != self.profile.valid_frame_count @@ -1030,6 +1126,7 @@ __all__ = [ "GeometryProviderError", "GeometryProviderSnapshot", "Ravnoves00GeometryAssociationProvider", + "RecordedFrameTemporalBinding", "RecordedGeometryStore", "load_geometry_profile", ] diff --git a/src/k1link/perception/semantic_fusion.py b/src/k1link/perception/semantic_fusion.py new file mode 100644 index 0000000..7d01611 --- /dev/null +++ b/src/k1link/perception/semantic_fusion.py @@ -0,0 +1,643 @@ +"""Model-neutral semantic diagnostics for admitted geometry observations. + +The seam projects a provider-owned ``uint8`` semantic mask onto the existing +frame-local point index space and aggregates those labels for already-created +``ObstacleObservation`` values. It deliberately returns separate diagnostic +evidence: semantic output cannot create or replace obstacle identity, metric +occupancy, motion, threat, or safety authority. +""" + +from __future__ import annotations + +import math +import re +from collections import Counter +from dataclasses import dataclass +from enum import IntEnum, StrEnum + +import numpy as np +import numpy.typing as npt + +from .contracts import ObstacleObservation +from .geometry_math import ProjectedPointCloud + +Int16Array = npt.NDArray[np.int16] +UInt8Array = npt.NDArray[np.uint8] + +NO_SEMANTIC_CLASS_ID = -1 +_IDENTIFIER = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:/-]{0,159}$") + + +class SemanticFusionError(ValueError): + """Semantic input or its geometry binding violates the diagnostic contract.""" + + +class SemanticClassDisposition(StrEnum): + """Whether one provider class is usable as a label or explicitly uncertain.""" + + LABELED = "labeled" + AMBIGUOUS = "ambiguous" + + +class SemanticEvidenceStatus(IntEnum): + """Compact source-point and observation semantic state.""" + + ABSENT = 0 + UNPROJECTED = 1 + AMBIGUOUS = 2 + LABELED = 3 + + +class SemanticEvidenceAuthority(StrEnum): + """Semantic output is never promoted into planner or safety authority.""" + + DIAGNOSTIC_ONLY = "diagnostic-only" + + +@dataclass(frozen=True, slots=True) +class SemanticClassDefinition: + """Provider-neutral meaning assigned to one raw ``uint8`` mask value.""" + + class_id: int + label: str + disposition: SemanticClassDisposition = SemanticClassDisposition.LABELED + + def __post_init__(self) -> None: + if ( + not isinstance(self.class_id, int) + or isinstance(self.class_id, bool) + or not 0 <= self.class_id <= 255 + ): + raise SemanticFusionError("semantic class id must fit uint8") + _label(self.label, "semantic class label") + if not isinstance(self.disposition, SemanticClassDisposition): + raise SemanticFusionError("semantic class disposition is invalid") + + +@dataclass(frozen=True, slots=True) +class SemanticMask: + """One source-bound hard semantic mask plus its complete class vocabulary.""" + + source_id: str + frame_id: str + provider_id: str + model_id: str + preprocess_id: str + labels: UInt8Array + classes: tuple[SemanticClassDefinition, ...] + + def __post_init__(self) -> None: + for value, label in ( + (self.source_id, "semantic source id"), + (self.frame_id, "semantic frame id"), + (self.provider_id, "semantic provider id"), + (self.model_id, "semantic model id"), + (self.preprocess_id, "semantic preprocess id"), + ): + _identifier(value, label) + if not isinstance(self.labels, np.ndarray): + raise SemanticFusionError("semantic mask must be a numpy array") + if self.labels.dtype != np.uint8 or self.labels.ndim != 2: + raise SemanticFusionError("semantic mask must have uint8 HxW shape") + if self.labels.shape[0] < 1 or self.labels.shape[1] < 1: + raise SemanticFusionError("semantic mask dimensions must be positive") + if not isinstance(self.classes, tuple) or not self.classes: + raise SemanticFusionError("semantic class vocabulary must be a nonempty tuple") + if any(not isinstance(item, SemanticClassDefinition) for item in self.classes): + raise SemanticFusionError("semantic class vocabulary is invalid") + class_ids = tuple(item.class_id for item in self.classes) + if len(set(class_ids)) != len(class_ids): + raise SemanticFusionError("semantic class ids must be unique") + undeclared = set(int(value) for value in np.unique(self.labels)) - set(class_ids) + if undeclared: + raise SemanticFusionError("semantic mask contains undeclared class ids") + frozen = np.array(self.labels, dtype=np.uint8, order="C", copy=True) + frozen.setflags(write=False) + object.__setattr__(self, "labels", frozen) + + @property + def height(self) -> int: + return int(self.labels.shape[0]) + + @property + def width(self) -> int: + return int(self.labels.shape[1]) + + def class_definition(self, class_id: int) -> SemanticClassDefinition: + for definition in self.classes: + if definition.class_id == class_id: + return definition + raise SemanticFusionError("semantic class id is not declared") + + +@dataclass(frozen=True, slots=True) +class PointSemanticLabels: + """Semantic labels aligned to the complete source-point index space. + + ``class_ids`` uses ``-1`` only when the corresponding status is ``ABSENT`` + or ``UNPROJECTED``. Callers must never interpret that sentinel as a model + class. Ambiguous provider classes retain their raw class id for review. + """ + + class_ids: Int16Array + status_codes: UInt8Array + classes: tuple[SemanticClassDefinition, ...] + + def __post_init__(self) -> None: + if not isinstance(self.class_ids, np.ndarray) or self.class_ids.dtype != np.int16: + raise SemanticFusionError("point semantic class ids must be int16") + if self.class_ids.ndim != 1: + raise SemanticFusionError("point semantic class ids must be one-dimensional") + if not isinstance(self.status_codes, np.ndarray) or self.status_codes.dtype != np.uint8: + raise SemanticFusionError("point semantic status codes must be uint8") + if self.status_codes.shape != self.class_ids.shape: + raise SemanticFusionError("point semantic arrays must have equal shape") + if not isinstance(self.classes, tuple) or any( + not isinstance(item, SemanticClassDefinition) for item in self.classes + ): + raise SemanticFusionError("point semantic vocabulary is invalid") + if len({item.class_id for item in self.classes}) != len(self.classes): + raise SemanticFusionError("point semantic class ids must be unique") + valid_statuses = {int(status) for status in SemanticEvidenceStatus} + if set(int(value) for value in np.unique(self.status_codes)) - valid_statuses: + raise SemanticFusionError("point semantic status code is invalid") + unavailable = np.isin( + self.status_codes, + (SemanticEvidenceStatus.ABSENT, SemanticEvidenceStatus.UNPROJECTED), + ) + if np.any(self.class_ids[unavailable] != NO_SEMANTIC_CLASS_ID): + raise SemanticFusionError("unavailable point semantics cannot carry a class id") + available = ~unavailable + if np.any((self.class_ids[available] < 0) | (self.class_ids[available] > 255)): + raise SemanticFusionError("available point semantic class id is invalid") + definitions = {item.class_id: item for item in self.classes} + for class_id, status_code in zip( + self.class_ids[available].tolist(), + self.status_codes[available].tolist(), + strict=True, + ): + definition = definitions.get(int(class_id)) + if definition is None: + raise SemanticFusionError("point semantic class id is not declared") + expected = ( + SemanticEvidenceStatus.AMBIGUOUS + if definition.disposition is SemanticClassDisposition.AMBIGUOUS + else SemanticEvidenceStatus.LABELED + ) + if int(status_code) != int(expected): + raise SemanticFusionError("point semantic status disagrees with its class") + class_ids = np.array(self.class_ids, dtype=np.int16, order="C", copy=True) + status_codes = np.array(self.status_codes, dtype=np.uint8, order="C", copy=True) + class_ids.setflags(write=False) + status_codes.setflags(write=False) + object.__setattr__(self, "class_ids", class_ids) + object.__setattr__(self, "status_codes", status_codes) + + @property + def source_point_count(self) -> int: + return int(self.class_ids.size) + + def status_for(self, source_point_id: int) -> SemanticEvidenceStatus: + _point_id(source_point_id, self.source_point_count) + return SemanticEvidenceStatus(int(self.status_codes[source_point_id])) + + def class_id_for(self, source_point_id: int) -> int | None: + _point_id(source_point_id, self.source_point_count) + value = int(self.class_ids[source_point_id]) + return None if value == NO_SEMANTIC_CLASS_ID else value + + def label_for(self, source_point_id: int) -> str | None: + class_id = self.class_id_for(source_point_id) + if class_id is None: + return None + for definition in self.classes: + if definition.class_id == class_id: + return definition.label + raise AssertionError("validated point class disappeared from its vocabulary") + + +@dataclass(frozen=True, slots=True) +class SemanticClassEvidence: + """Point support for one semantic class inside an existing observation.""" + + class_id: int + label: str + disposition: SemanticClassDisposition + point_count: int + + def __post_init__(self) -> None: + if ( + not isinstance(self.class_id, int) + or isinstance(self.class_id, bool) + or not 0 <= self.class_id <= 255 + ): + raise SemanticFusionError("semantic evidence class id must fit uint8") + _label(self.label, "semantic evidence label") + if not isinstance(self.disposition, SemanticClassDisposition): + raise SemanticFusionError("semantic evidence disposition is invalid") + if ( + not isinstance(self.point_count, int) + or isinstance(self.point_count, bool) + or self.point_count < 1 + ): + raise SemanticFusionError("semantic evidence point count must be positive") + + +@dataclass(frozen=True, slots=True) +class ObservationSemanticEvidence: + """Aggregated, non-authoritative semantics for one immutable observation.""" + + observation_id: str + occupancy_key: str + status: SemanticEvidenceStatus + source_point_count: int + labeled_point_count: int + ambiguous_point_count: int + unprojected_point_count: int + absent_point_count: int + class_evidence: tuple[SemanticClassEvidence, ...] + dominant_class_id: int | None + dominant_label: str | None + dominant_fraction_of_labeled: float | None + reason_code: str + authority: SemanticEvidenceAuthority = SemanticEvidenceAuthority.DIAGNOSTIC_ONLY + + def __post_init__(self) -> None: + _identifier(self.observation_id, "semantic observation id") + _identifier(self.occupancy_key, "semantic occupancy binding") + _identifier(self.reason_code, "semantic evidence reason") + if not isinstance(self.status, SemanticEvidenceStatus): + raise SemanticFusionError("observation semantic status is invalid") + if self.authority is not SemanticEvidenceAuthority.DIAGNOSTIC_ONLY: + raise SemanticFusionError("semantic evidence cannot acquire product authority") + counts = ( + self.source_point_count, + self.labeled_point_count, + self.ambiguous_point_count, + self.unprojected_point_count, + self.absent_point_count, + ) + if any( + not isinstance(value, int) or isinstance(value, bool) or value < 0 + for value in counts + ): + raise SemanticFusionError("semantic evidence counts must be nonnegative integers") + if sum(counts[1:]) != self.source_point_count: + raise SemanticFusionError("semantic evidence accounting is incomplete") + if not isinstance(self.class_evidence, tuple) or any( + not isinstance(item, SemanticClassEvidence) for item in self.class_evidence + ): + raise SemanticFusionError("semantic class evidence is invalid") + if sum(item.point_count for item in self.class_evidence) != ( + self.labeled_point_count + self.ambiguous_point_count + ): + raise SemanticFusionError("semantic class evidence accounting is incomplete") + if len({item.class_id for item in self.class_evidence}) != len(self.class_evidence): + raise SemanticFusionError("semantic class evidence ids must be unique") + if self.status is SemanticEvidenceStatus.ABSENT: + if self.absent_point_count != self.source_point_count: + raise SemanticFusionError("absent semantic evidence accounting is invalid") + elif self.status is SemanticEvidenceStatus.UNPROJECTED: + if self.source_point_count and self.unprojected_point_count != self.source_point_count: + raise SemanticFusionError("unprojected semantic evidence accounting is invalid") + elif self.status is SemanticEvidenceStatus.AMBIGUOUS: + if not self.labeled_point_count and not self.ambiguous_point_count: + raise SemanticFusionError("ambiguous semantic evidence needs projected labels") + elif not self.labeled_point_count: + raise SemanticFusionError("labeled semantic evidence needs labeled points") + dominant_values = ( + self.dominant_class_id, + self.dominant_label, + self.dominant_fraction_of_labeled, + ) + if self.status is SemanticEvidenceStatus.LABELED: + if any(value is None for value in dominant_values): + raise SemanticFusionError("labeled semantic evidence needs a dominant class") + if ( + not isinstance(self.dominant_fraction_of_labeled, float) + or not math.isfinite(self.dominant_fraction_of_labeled) + or not 0.5 < self.dominant_fraction_of_labeled <= 1.0 + ): + raise SemanticFusionError("dominant semantic fraction must be a majority") + elif any(value is not None for value in dominant_values): + raise SemanticFusionError("non-labeled semantic evidence cannot claim a dominant class") + + @property + def semantic_coverage_fraction(self) -> float: + if not self.source_point_count: + return 0.0 + return (self.labeled_point_count + self.ambiguous_point_count) / self.source_point_count + + +@dataclass(frozen=True, slots=True) +class SemanticFusionResult: + """One frame's detached semantic diagnostics.""" + + mask_available: bool + point_labels: PointSemanticLabels + observation_evidence: tuple[ObservationSemanticEvidence, ...] + authority: SemanticEvidenceAuthority = SemanticEvidenceAuthority.DIAGNOSTIC_ONLY + + def __post_init__(self) -> None: + if not isinstance(self.mask_available, bool): + raise SemanticFusionError("semantic mask availability must be boolean") + if not isinstance(self.point_labels, PointSemanticLabels): + raise SemanticFusionError("semantic point labels are invalid") + if not isinstance(self.observation_evidence, tuple) or any( + not isinstance(item, ObservationSemanticEvidence) + for item in self.observation_evidence + ): + raise SemanticFusionError("observation semantic evidence is invalid") + if len({item.observation_id for item in self.observation_evidence}) != len( + self.observation_evidence + ): + raise SemanticFusionError("observation semantic evidence ids must be unique") + if self.authority is not SemanticEvidenceAuthority.DIAGNOSTIC_ONLY: + raise SemanticFusionError("semantic fusion cannot acquire product authority") + if self.mask_available is not bool(self.point_labels.classes): + raise SemanticFusionError("semantic mask availability and vocabulary disagree") + + +def fuse_semantic_diagnostics( + *, + semantic_mask: SemanticMask | None, + projected: ProjectedPointCloud, + observations: tuple[ObstacleObservation, ...], +) -> SemanticFusionResult: + """Attach mask diagnostics to points and observations without mutating authority.""" + + if semantic_mask is not None and not isinstance(semantic_mask, SemanticMask): + raise SemanticFusionError("semantic mask contract is invalid") + if not isinstance(observations, tuple) or any( + not isinstance(item, ObstacleObservation) for item in observations + ): + raise SemanticFusionError("geometry observations must be a tuple") + _validate_projection(projected) + _validate_observation_bindings( + observations, + semantic_mask=semantic_mask, + source_point_count=projected.source_point_count, + ) + point_labels = _project_point_labels(semantic_mask, projected) + evidence = tuple( + _aggregate_observation(observation, point_labels) for observation in observations + ) + return SemanticFusionResult( + mask_available=semantic_mask is not None, + point_labels=point_labels, + observation_evidence=evidence, + ) + + +def _project_point_labels( + semantic_mask: SemanticMask | None, + projected: ProjectedPointCloud, +) -> PointSemanticLabels: + point_count = projected.source_point_count + class_ids = np.full(point_count, NO_SEMANTIC_CLASS_ID, dtype=np.int16) + if semantic_mask is None: + return PointSemanticLabels( + class_ids=class_ids, + status_codes=np.full( + point_count, + int(SemanticEvidenceStatus.ABSENT), + dtype=np.uint8, + ), + classes=(), + ) + + statuses = np.full( + point_count, + int(SemanticEvidenceStatus.UNPROJECTED), + dtype=np.uint8, + ) + if projected.projected_point_count: + pixel_indices = np.floor(projected.pixels_xy).astype(np.int64) + inside = ( + (pixel_indices[:, 0] >= 0) + & (pixel_indices[:, 0] < semantic_mask.width) + & (pixel_indices[:, 1] >= 0) + & (pixel_indices[:, 1] < semantic_mask.height) + ) + rows = np.flatnonzero(inside) + if rows.size: + source_ids = projected.source_indices[rows] + pixels = pixel_indices[rows] + raw_classes = semantic_mask.labels[pixels[:, 1], pixels[:, 0]] + class_ids[source_ids] = raw_classes.astype(np.int16, copy=False) + ambiguous_ids = np.asarray( + [ + item.class_id + for item in semantic_mask.classes + if item.disposition is SemanticClassDisposition.AMBIGUOUS + ], + dtype=np.uint8, + ) + ambiguous = np.isin(raw_classes, ambiguous_ids) + statuses[source_ids] = np.where( + ambiguous, + int(SemanticEvidenceStatus.AMBIGUOUS), + int(SemanticEvidenceStatus.LABELED), + ).astype(np.uint8, copy=False) + return PointSemanticLabels( + class_ids=class_ids, + status_codes=statuses, + classes=semantic_mask.classes, + ) + + +def _aggregate_observation( + observation: ObstacleObservation, + point_labels: PointSemanticLabels, +) -> ObservationSemanticEvidence: + point_ids = np.asarray(observation.source_point_ids, dtype=np.int64) + statuses = point_labels.status_codes[point_ids] + class_ids = point_labels.class_ids[point_ids] + counts = Counter(int(value) for value in statuses.tolist()) + labeled_count = counts[int(SemanticEvidenceStatus.LABELED)] + ambiguous_count = counts[int(SemanticEvidenceStatus.AMBIGUOUS)] + unprojected_count = counts[int(SemanticEvidenceStatus.UNPROJECTED)] + absent_count = counts[int(SemanticEvidenceStatus.ABSENT)] + definitions = {item.class_id: item for item in point_labels.classes} + semantic_class_ids = class_ids[ + np.isin( + statuses, + (SemanticEvidenceStatus.LABELED, SemanticEvidenceStatus.AMBIGUOUS), + ) + ] + class_counts = Counter(int(value) for value in semantic_class_ids.tolist()) + class_evidence = tuple( + SemanticClassEvidence( + class_id=class_id, + label=definitions[class_id].label, + disposition=definitions[class_id].disposition, + point_count=point_count, + ) + for class_id, point_count in sorted(class_counts.items()) + ) + + status: SemanticEvidenceStatus + dominant_class_id: int | None = None + dominant_label: str | None = None + dominant_fraction: float | None = None + if not point_labels.classes: + status = SemanticEvidenceStatus.ABSENT + reason_code = "semantic-mask-unavailable" + elif not observation.source_point_ids or unprojected_count == len( + observation.source_point_ids + ): + status = SemanticEvidenceStatus.UNPROJECTED + reason_code = ( + "observation-has-no-source-points" + if not observation.source_point_ids + else "observation-points-unprojected" + ) + elif not labeled_count: + status = SemanticEvidenceStatus.AMBIGUOUS + reason_code = "semantic-classes-ambiguous" + else: + labeled_ids = class_ids[statuses == int(SemanticEvidenceStatus.LABELED)] + labeled_counts = Counter(int(value) for value in labeled_ids.tolist()) + maximum = max(labeled_counts.values()) + candidates = [ + class_id for class_id, count in labeled_counts.items() if count == maximum + ] + semantic_point_count = labeled_count + ambiguous_count + if len(candidates) != 1 or maximum * 2 <= semantic_point_count: + status = SemanticEvidenceStatus.AMBIGUOUS + reason_code = "semantic-label-majority-ambiguous" + else: + status = SemanticEvidenceStatus.LABELED + dominant_class_id = candidates[0] + dominant_label = definitions[dominant_class_id].label + dominant_fraction = float(maximum / labeled_count) + reason_code = "semantic-label-majority" + return ObservationSemanticEvidence( + observation_id=observation.observation_id, + occupancy_key=observation.occupancy_key, + status=status, + source_point_count=len(observation.source_point_ids), + labeled_point_count=labeled_count, + ambiguous_point_count=ambiguous_count, + unprojected_point_count=unprojected_count, + absent_point_count=absent_count, + class_evidence=class_evidence, + dominant_class_id=dominant_class_id, + dominant_label=dominant_label, + dominant_fraction_of_labeled=dominant_fraction, + reason_code=reason_code, + ) + + +def _validate_projection(projected: ProjectedPointCloud) -> None: + if not isinstance(projected, ProjectedPointCloud): + raise SemanticFusionError("projected point cloud contract is invalid") + if ( + not isinstance(projected.pixels_xy, np.ndarray) + or projected.pixels_xy.dtype != np.float64 + or projected.pixels_xy.ndim != 2 + or projected.pixels_xy.shape[1:] != (2,) + ): + raise SemanticFusionError("projected pixels must have float64 Nx2 shape") + count = projected.projected_point_count + if ( + not isinstance(projected.depths_m, np.ndarray) + or projected.depths_m.dtype != np.float64 + or projected.depths_m.shape != (count,) + ): + raise SemanticFusionError("projected depths must have float64 N shape") + if ( + not isinstance(projected.source_indices, np.ndarray) + or projected.source_indices.dtype != np.int64 + or projected.source_indices.shape != (count,) + ): + raise SemanticFusionError("projected source indices must have int64 N shape") + for value, label in ( + (projected.source_point_count, "source point count"), + (projected.camera_front_point_count, "camera-front point count"), + ): + if not isinstance(value, int) or isinstance(value, bool) or value < 0: + raise SemanticFusionError(f"{label} must be a nonnegative integer") + if not count <= projected.camera_front_point_count <= projected.source_point_count: + raise SemanticFusionError("projected point accounting is invalid") + if ( + not np.isfinite(projected.pixels_xy).all() + or not np.isfinite(projected.depths_m).all() + or np.any(projected.depths_m <= 0.0) + ): + raise SemanticFusionError("projected point values must be finite and in front") + if np.any(projected.source_indices < 0) or np.any( + projected.source_indices >= projected.source_point_count + ): + raise SemanticFusionError("projected source point id is outside the source frame") + if np.unique(projected.source_indices).size != count: + raise SemanticFusionError("projected source point ids must be unique") + + +def _validate_observation_bindings( + observations: tuple[ObstacleObservation, ...], + *, + semantic_mask: SemanticMask | None, + source_point_count: int, +) -> None: + observation_ids: set[str] = set() + point_owners: dict[int, str] = {} + source_frames = {(item.source_id, item.frame_id) for item in observations} + if len(source_frames) > 1: + raise SemanticFusionError("geometry observations escaped their source frame") + for observation in observations: + if observation.observation_id in observation_ids: + raise SemanticFusionError("geometry observation ids must be unique") + observation_ids.add(observation.observation_id) + if semantic_mask is not None and ( + observation.source_id != semantic_mask.source_id + or observation.frame_id != semantic_mask.frame_id + ): + raise SemanticFusionError("semantic mask escaped its observation source frame") + for point_id in observation.source_point_ids: + if point_id >= source_point_count: + raise SemanticFusionError("observation source point id is outside the source frame") + previous = point_owners.setdefault(point_id, observation.observation_id) + if previous != observation.observation_id: + raise SemanticFusionError("source point has duplicate observation ownership") + + +def _identifier(value: str, label: str) -> None: + if not isinstance(value, str) or _IDENTIFIER.fullmatch(value) is None: + raise SemanticFusionError(f"{label} is invalid") + + +def _label(value: str, label: str) -> None: + if ( + not isinstance(value, str) + or not value + or value != value.strip() + or len(value) > 120 + or any(ord(character) < 32 for character in value) + ): + raise SemanticFusionError(f"{label} is invalid") + + +def _point_id(value: int, source_point_count: int) -> None: + if ( + not isinstance(value, int) + or isinstance(value, bool) + or not 0 <= value < source_point_count + ): + raise SemanticFusionError("source point id is outside the point-label frame") + + +__all__ = [ + "NO_SEMANTIC_CLASS_ID", + "ObservationSemanticEvidence", + "PointSemanticLabels", + "SemanticClassDefinition", + "SemanticClassDisposition", + "SemanticClassEvidence", + "SemanticEvidenceAuthority", + "SemanticEvidenceStatus", + "SemanticFusionError", + "SemanticFusionResult", + "SemanticMask", + "fuse_semantic_diagnostics", +] diff --git a/src/k1link/perception/semantic_slam_replay.py b/src/k1link/perception/semantic_slam_replay.py new file mode 100644 index 0000000..4e90c4c --- /dev/null +++ b/src/k1link/perception/semantic_slam_replay.py @@ -0,0 +1,2017 @@ +"""Immutable E47 semantic diagnostics over the accepted M4 geometry ledger. + +The builder admits one exact sealed E4 EoMT result, projects every hard mask +onto the already-recorded frame-local LiDAR point index space, and aggregates +those labels only for observations which already exist in the accepted geometry +ledger. The publication is deliberately detached from occupancy, temporal, +motion, threat, navigation and actuation decisions. +""" + +from __future__ import annotations + +import hashlib +import io +import json +import math +import os +import re +import shutil +import tarfile +import time +import uuid +import zipfile +from collections import Counter +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import Final + +import numpy as np +from PIL import Image, UnidentifiedImageError + +from k1link.sessions import SessionIntegrityError + +from ..compute.perception_epoch import validate_recorded_perception_epoch_result +from .baseline import ( + BASELINE_CAMERA_SOURCE_ID, + BASELINE_RECORDED_JOB_ID, + BASELINE_SESSION_ID, + BASELINE_SOURCE_ID, + BASELINE_SOURCE_PACK_ID, + BASELINE_SOURCE_PACK_SHA256, +) +from .contracts import ObstacleObservation +from .geometry import ( + GeometryFrame, + RecordedFrameTemporalBinding, + RecordedGeometryStore, +) +from .geometry_math import ProjectedPointCloud, project_map_points_kb4 +from .geometry_replay import GeometryReplayResult, read_geometry_replay_result +from .semantic_fusion import ( + ObservationSemanticEvidence, + SemanticClassDefinition, + SemanticClassDisposition, + SemanticEvidenceStatus, + SemanticMask, + fuse_semantic_diagnostics, +) +from .threat_replay import ThreatReplayResult, read_threat_replay_result + +SEMANTIC_SLAM_SCHEMA: Final = "missioncore.e47-semantic-slam-result/v1" +SEMANTIC_SLAM_REPORT_SCHEMA: Final = "missioncore.e47-semantic-slam-report/v1" +SEMANTIC_SLAM_FRAME_SCHEMA: Final = "missioncore.e47-semantic-slam-frame/v1" +SEMANTIC_SLAM_TAXONOMY_SCHEMA: Final = "missioncore.e47-semantic-taxonomy/v1" +SEMANTIC_SLAM_RESULT_PREFIX: Final = "e47-semantic-slam-" +SEMANTIC_SLAM_MANIFEST_NAME: Final = "manifest.json" +SEMANTIC_SLAM_REPORT_NAME: Final = "report.json" +SEMANTIC_SLAM_POINTS_NAME: Final = "semantic-points.npz" +SEMANTIC_SLAM_OBSERVATIONS_NAME: Final = "semantic-observations.jsonl" +SEMANTIC_SLAM_MASKS_NAME: Final = "semantic-masks.zip" +SEMANTIC_SLAM_TAXONOMY_NAME: Final = "taxonomy.json" +DEFAULT_SEMANTIC_SLAM_PROFILE_PATH: Final = "config/perception/e47-semantic-slam-shadow-v1.json" + +E4_SEMANTIC_RESULT_ID: Final = ( + "result-793785170472c519486ccd666be102fb04d169d92383acda3fcc29eecf045d30" +) +E4_SEMANTIC_FRAMES_SHA256: Final = ( + "4780339f5ef94e1f58922dfe2df67ecdf983c4b4e030761f4a990f65b511d5cf" +) +E4_SEMANTIC_MASKS_SHA256: Final = "ef291034e51c5c719dcc1c0902bae1e3daedf919d7b55a21fcb0a31af298071f" +E4_SEMANTIC_CONFIG_PROFILE_SHA256: Final = ( + "ea583966bc3409f5cf563cbf4fad05e366907e67187082eb692aff53d9f5d875" +) +EXPECTED_FRAME_COUNT: Final = 4489 +PUBLICATION_STATUS: Final = "warning-diagnostic-pending-independent-truth" + +_SHA256 = re.compile(r"^[a-f0-9]{64}$") +_SAFE_RESULT_ID = re.compile(r"^e47-semantic-slam-[a-f0-9]{64}$") +_MAX_JSON_BYTES = 64 * 1024 * 1024 +_MAX_JSONL_LINE_BYTES = 16 * 1024 * 1024 +_MAX_MASK_BYTES = 16 * 1024 * 1024 +_FALSE_AUTHORITY_KEYS: Final = ( + "ground_truth", + "physical_live", + "commands_enabled", + "actuation_allowed", + "navigation_or_safety_accepted", +) + + +class SemanticSlamReplayError(RuntimeError): + """An E47 input or publication is mutable, incomplete or inconsistent.""" + + +@dataclass(frozen=True, slots=True) +class SemanticSlamReplayResult: + result_id: str + result_root: Path + status: str + metrics: dict[str, object] + report: dict[str, object] + manifest: dict[str, object] + + +@dataclass(frozen=True, slots=True) +class _TaxonomyClass: + class_id: int + label: str + disposition: SemanticClassDisposition + color_rgb: tuple[int, int, int] + + def definition(self) -> SemanticClassDefinition: + return SemanticClassDefinition( + class_id=self.class_id, + label=self.label, + disposition=self.disposition, + ) + + def document(self) -> dict[str, object]: + return { + "class_id": self.class_id, + "label": self.label, + "disposition": self.disposition.value, + "color_rgb": list(self.color_rgb), + } + + +@dataclass(frozen=True, slots=True) +class _SemanticSlamProfile: + path: Path + sha256: str + profile_id: str + source_id: str + session_id: str + frame_count: int + image_width: int + image_height: int + source_pack_id: str + source_pack_sha256: str + calibration_sha256: str + provider_id: str + model_id: str + model_revision: str + model_weights_sha256: str + preprocess_id: str + mask_metadata_schema_version: str + mask_payload: dict[str, object] + provider_role: str + classes: tuple[_TaxonomyClass, ...] + fusion: dict[str, object] + temporal_binding: dict[str, object] + acceptance: dict[str, object] + authority: dict[str, object] + + +@dataclass(frozen=True, slots=True) +class _SemanticUpstream: + result_id: str + result_root: Path + result_manifest_sha256: str + frames_path: Path + frames_sha256: str + masks_path: Path + masks_sha256: str + created_at_utc: str + job_id: str + input_sha256: str + source_id: str + session_id: str + calibration_sha256: str + configuration_profile_sha256: str + model_id: str + model_revision: str + model_weights_sha256: str + + +@dataclass(frozen=True, slots=True) +class _AdmittedInputs: + profile: _SemanticSlamProfile + semantic: _SemanticUpstream + geometry: GeometryReplayResult + threat: ThreatReplayResult + store: RecordedGeometryStore + geometry_frames_path: Path + geometry_frames_sha256: str + threat_frames_path: Path + threat_frames_sha256: str + + +@dataclass(frozen=True, slots=True) +class _PointArtifactMetrics: + frames: dict[str, int] + points: dict[str, int] + observations: dict[str, int] + frame_point_accounting: tuple[dict[str, int], ...] + + +def build_semantic_slam_replay( + *, + repository_root: Path, + semantic_result_root: Path, + threat_result_root: Path, + geometry_result_root: Path, + output_root: Path, +) -> SemanticSlamReplayResult: + """Build the immutable E47 shadow result from sealed E4 and accepted M4 inputs.""" + + inputs = _admit_inputs( + repository_root=repository_root, + semantic_result_root=semantic_result_root, + threat_result_root=threat_result_root, + geometry_result_root=geometry_result_root, + ) + return _materialize(inputs=inputs, output_root=output_root) + + +def read_semantic_slam_replay_result(root: Path) -> SemanticSlamReplayResult: + """Read and fully verify one self-contained E47 publication.""" + + candidate = root.expanduser().absolute() + if candidate.is_symlink(): + raise SemanticSlamReplayError("semantic SLAM result root is invalid") + resolved = candidate.resolve(strict=True) + if ( + resolved.is_symlink() + or not resolved.is_dir() + or _SAFE_RESULT_ID.fullmatch(resolved.name) is None + ): + raise SemanticSlamReplayError("semantic SLAM result root is invalid") + manifest = _read_json(resolved / SEMANTIC_SLAM_MANIFEST_NAME) + _exact_keys( + manifest, + { + "schema_version", + "result_id", + "identity_sha256", + "created_at_utc", + "status", + "identity", + "artifacts", + "ground_truth", + }, + "semantic SLAM manifest", + ) + identity = _object(manifest.get("identity"), "semantic SLAM identity") + identity_sha256 = hashlib.sha256(_canonical_json(identity)).hexdigest() + if ( + manifest.get("schema_version") != SEMANTIC_SLAM_SCHEMA + or manifest.get("result_id") != resolved.name + or manifest.get("identity_sha256") != identity_sha256 + or resolved.name != f"{SEMANTIC_SLAM_RESULT_PREFIX}{identity_sha256}" + or manifest.get("status") != PUBLICATION_STATUS + or manifest.get("ground_truth") is not False + ): + raise SemanticSlamReplayError("semantic SLAM result identity changed") + _validate_identity(identity) + + artifacts = manifest.get("artifacts") + if not isinstance(artifacts, list): + raise SemanticSlamReplayError("semantic SLAM artifact inventory is invalid") + by_role: dict[str, object] = {} + for raw in artifacts: + descriptor = _object(raw, "semantic SLAM artifact") + role = descriptor.get("role") + if not isinstance(role, str) or role in by_role: + raise SemanticSlamReplayError("semantic SLAM artifact roles are invalid") + by_role[role] = descriptor + expected = { + "semantic-point-labels": ( + SEMANTIC_SLAM_POINTS_NAME, + "application/x-npz", + None, + ), + "semantic-observation-evidence": ( + SEMANTIC_SLAM_OBSERVATIONS_NAME, + "application/x-ndjson", + SEMANTIC_SLAM_FRAME_SCHEMA, + ), + "semantic-mask-archive": (SEMANTIC_SLAM_MASKS_NAME, "application/zip", None), + "semantic-taxonomy": ( + SEMANTIC_SLAM_TAXONOMY_NAME, + "application/json", + SEMANTIC_SLAM_TAXONOMY_SCHEMA, + ), + "semantic-slam-report": ( + SEMANTIC_SLAM_REPORT_NAME, + "application/json", + SEMANTIC_SLAM_REPORT_SCHEMA, + ), + } + if set(by_role) != set(expected): + raise SemanticSlamReplayError("semantic SLAM artifact roles changed") + paths = { + role: _validated_artifact( + resolved, + by_role[role], + expected_role=role, + expected_name=name, + expected_media_type=media_type, + expected_schema_version=schema_version, + ) + for role, (name, media_type, schema_version) in expected.items() + } + core_artifact_keys = { + "semantic-point-labels": "semantic_points_sha256", + "semantic-observation-evidence": "semantic_observations_sha256", + "semantic-mask-archive": "semantic_masks_sha256", + "semantic-taxonomy": "taxonomy_sha256", + } + for role, identity_key in core_artifact_keys.items(): + if _file_sha256(paths[role]) != identity.get(identity_key): + raise SemanticSlamReplayError("semantic SLAM core artifact identity changed") + + taxonomy = _validate_taxonomy(paths["semantic-taxonomy"]) + metrics = _validate_points_artifact(paths["semantic-point-labels"], identity) + frame_metrics, observation_metrics, mask_rows, ledger_point_accounting = ( + _validate_observation_ledger( + paths["semantic-observation-evidence"], + identity, + taxonomy, + ) + ) + _validate_mask_archive(paths["semantic-mask-archive"], mask_rows) + if ( + metrics.frames != frame_metrics + or metrics.observations != observation_metrics + or metrics.frame_point_accounting != ledger_point_accounting + ): + raise SemanticSlamReplayError("semantic SLAM artifact accounting disagrees") + + report = _read_json(paths["semantic-slam-report"]) + _exact_keys( + report, + { + "schema_version", + "result_id", + "identity_sha256", + "created_at_utc", + "status", + "metrics", + "acceptance", + "limitations", + "authority", + }, + "semantic SLAM report", + ) + report_metrics = _object(report.get("metrics"), "semantic SLAM report metrics") + if ( + report.get("schema_version") != SEMANTIC_SLAM_REPORT_SCHEMA + or report.get("result_id") != resolved.name + or report.get("identity_sha256") != identity_sha256 + or report.get("created_at_utc") != manifest.get("created_at_utc") + or report.get("status") != PUBLICATION_STATUS + or report.get("authority") != identity.get("authority") + or report_metrics.get("frames") != frame_metrics + or report_metrics.get("points") != metrics.points + or report_metrics.get("observations") != observation_metrics + or not _valid_runtime_metrics(report_metrics.get("runtime"), frame_metrics["total"]) + or report.get("acceptance") != _acceptance_document() + or report.get("limitations") != _limitations() + ): + raise SemanticSlamReplayError("semantic SLAM report changed") + return SemanticSlamReplayResult( + result_id=resolved.name, + result_root=resolved, + status=PUBLICATION_STATUS, + metrics=report_metrics, + report=report, + manifest=manifest, + ) + + +def _admit_inputs( + *, + repository_root: Path, + semantic_result_root: Path, + threat_result_root: Path, + geometry_result_root: Path, +) -> _AdmittedInputs: + repository = repository_root.expanduser().resolve(strict=True) + if not repository.is_dir(): + raise SemanticSlamReplayError("Mission Core repository root is invalid") + profile = _load_profile(repository / DEFAULT_SEMANTIC_SLAM_PROFILE_PATH) + semantic = _admit_semantic_result(repository, semantic_result_root, profile) + try: + geometry = read_geometry_replay_result(geometry_result_root) + threat = read_threat_replay_result(threat_result_root) + except (OSError, ValueError, RuntimeError) as exc: + raise SemanticSlamReplayError("M4 replay input could not be admitted") from exc + if not geometry.accepted or not threat.accepted: + raise SemanticSlamReplayError("E47 requires accepted M4 geometry and threat results") + geometry_identity = _object(geometry.manifest.get("identity"), "geometry identity") + threat_identity = _object(threat.manifest.get("identity"), "threat identity") + geometry_frames_sha256 = _digest(geometry_identity, "frames_sha256") + threat_frames_sha256 = _digest(threat_identity, "frames_sha256") + if ( + threat_identity.get("geometry_result_id") != geometry.result_id + or threat_identity.get("geometry_frames_sha256") != geometry_frames_sha256 + or geometry_identity.get("source_pack_id") != profile.source_pack_id + or geometry_identity.get("source_pack_sha256") != profile.source_pack_sha256 + or threat_identity.get("source_pack_id") != profile.source_pack_id + or threat_identity.get("source_pack_sha256") != profile.source_pack_sha256 + or threat_identity.get("calibration_content_sha256") != profile.calibration_sha256 + or _nested_total(geometry.metrics, "frames") != profile.frame_count + or _nested_total(threat.metrics, "frames") != profile.frame_count + ): + raise SemanticSlamReplayError("E4 and M4 replay identities do not describe one source") + _require_false_authority(geometry_identity.get("authority"), "geometry authority") + _require_false_authority(threat_identity.get("authority"), "threat authority") + geometry_frames_path = geometry.result_root / "frames.jsonl" + threat_frames_path = threat.result_root / "frames.jsonl" + if ( + _file_sha256(geometry_frames_path) != geometry_frames_sha256 + or _file_sha256(threat_frames_path) != threat_frames_sha256 + ): + raise SemanticSlamReplayError("admitted M4 ledgers changed after validation") + try: + store = RecordedGeometryStore.from_repository(repository) + except (OSError, ValueError, RuntimeError) as exc: + raise SemanticSlamReplayError("recorded SLAM source pack could not be admitted") from exc + if ( + store.profile.source_id != profile.source_id + or store.profile.session_id != profile.session_id + or store.profile.frame_count != profile.frame_count + or store.profile.source_pack_id != profile.source_pack_id + or store.profile.source_pack_sha256 != profile.source_pack_sha256 + ): + raise SemanticSlamReplayError("recorded geometry store escaped the E47 profile") + return _AdmittedInputs( + profile=profile, + semantic=semantic, + geometry=geometry, + threat=threat, + store=store, + geometry_frames_path=geometry_frames_path, + geometry_frames_sha256=geometry_frames_sha256, + threat_frames_path=threat_frames_path, + threat_frames_sha256=threat_frames_sha256, + ) + + +def _admit_semantic_result( + repository: Path, + result_root: Path, + profile: _SemanticSlamProfile, +) -> _SemanticUpstream: + job_root = repository / ".runtime/compute-jobs" / BASELINE_RECORDED_JOB_ID + try: + result = validate_recorded_perception_epoch_result(job_root, result_root) + except (OSError, SessionIntegrityError, ValueError) as exc: + raise SemanticSlamReplayError("sealed E4 semantic result could not be admitted") from exc + result_document = _read_json(result.result_root / "result.json") + identity = _object(result_document.get("identity"), "E4 semantic identity") + configuration = _object(identity.get("configuration"), "E4 semantic configuration") + models = _object(identity.get("models"), "E4 semantic models") + semantic_model = _object(models.get("semantic"), "E4 semantic model") + files = models.get("files") + if not isinstance(files, list): + raise SemanticSlamReplayError("E4 model file identity is unavailable") + weight_rows = [ + _object(item, "E4 model file") + for item in files + if isinstance(item, dict) and item.get("name") == "model.safetensors" + ] + frames = result.artifact("panoptic-frame-metadata") + masks = result.artifact("panoptic-mask-archive") + if ( + result.result_id != E4_SEMANTIC_RESULT_ID + or result.job.job_id != BASELINE_RECORDED_JOB_ID + or result.job.source_id != BASELINE_CAMERA_SOURCE_ID + or result.job.session_id != BASELINE_SESSION_ID + or result.job.segment_count != EXPECTED_FRAME_COUNT + or profile.frame_count != EXPECTED_FRAME_COUNT + or result.calibration_sha256 != profile.calibration_sha256 + or result.calibration_slot != "camera_1" + or frames.sha256 != E4_SEMANTIC_FRAMES_SHA256 + or masks.sha256 != E4_SEMANTIC_MASKS_SHA256 + or configuration.get("profile_sha256") != E4_SEMANTIC_CONFIG_PROFILE_SHA256 + or configuration.get("pipeline") != "recorded-semantic-eomt-fisheye-mask/v1" + or configuration.get("frame_policy") != "all-frames-no-sampling" + or semantic_model.get("id") != profile.model_id + or semantic_model.get("revision") != profile.model_revision + or len(weight_rows) != 1 + or weight_rows[0].get("sha256") != profile.model_weights_sha256 + ): + raise SemanticSlamReplayError("sealed E4 semantic identity changed") + return _SemanticUpstream( + result_id=result.result_id, + result_root=result.result_root, + result_manifest_sha256=_file_sha256(result.result_root / "result.json"), + frames_path=frames.path, + frames_sha256=frames.sha256, + masks_path=masks.path, + masks_sha256=masks.sha256, + created_at_utc=result.created_at_utc, + job_id=result.job.job_id, + input_sha256=result.job.input_sha256, + source_id=result.job.source_id, + session_id=result.job.session_id, + calibration_sha256=result.calibration_sha256, + configuration_profile_sha256=E4_SEMANTIC_CONFIG_PROFILE_SHA256, + model_id=profile.model_id, + model_revision=profile.model_revision, + model_weights_sha256=profile.model_weights_sha256, + ) + + +def _materialize( + *, + inputs: _AdmittedInputs, + output_root: Path, +) -> SemanticSlamReplayResult: + profile = inputs.profile + root = output_root.expanduser().absolute() + if root.exists() and root.is_symlink(): + raise SemanticSlamReplayError("semantic SLAM output root cannot be a symlink") + root.mkdir(mode=0o700, parents=True, exist_ok=True) + staging = root / f".semantic-slam.{uuid.uuid4().hex}.tmp" + staging.mkdir(mode=0o700, exist_ok=False) + started_ns = time.perf_counter_ns() + point_chunks: list[np.ndarray] = [] + status_chunks: list[np.ndarray] = [] + projected_chunks: list[np.ndarray] = [] + frame_offsets = [0] + frame_source_counts: list[int] = [] + frame_projected_counts: list[int] = [] + frame_labeled_counts: list[int] = [] + frame_ambiguous_counts: list[int] = [] + frame_unprojected_counts: list[int] = [] + frame_absent_counts: list[int] = [] + frame_source_available: list[int] = [] + point_counts: Counter[str] = Counter() + observation_counts: Counter[str] = Counter() + source_available_count = 0 + taxonomy_definitions = tuple(item.definition() for item in profile.classes) + try: + observations_path = staging / SEMANTIC_SLAM_OBSERVATIONS_NAME + masks_zip_path = staging / SEMANTIC_SLAM_MASKS_NAME + with ( + inputs.geometry_frames_path.open("rb") as geometry_stream, + inputs.semantic.frames_path.open("rb") as semantic_stream, + tarfile.open(inputs.semantic.masks_path, mode="r:gz") as mask_archive, + zipfile.ZipFile(masks_zip_path, mode="w", compression=zipfile.ZIP_STORED) as masks_zip, + observations_path.open("wb") as observation_output, + ): + _validate_tar_root(mask_archive.next()) + for frame_index in range(profile.frame_count): + semantic_frame = _read_json_line( + semantic_stream.readline(), + "E4 semantic frame", + frame_index, + ) + geometry_frame_document = _read_json_line( + geometry_stream.readline(), + "M4 geometry frame", + frame_index, + ) + mask_member = mask_archive.next() + expected_mask_name = f"semantic-masks/frame-{frame_index + 1:06d}.png" + if ( + mask_member is None + or not mask_member.isfile() + or mask_member.name != expected_mask_name + or not 0 < mask_member.size <= _MAX_MASK_BYTES + ): + raise SemanticSlamReplayError("E4 semantic mask archive order changed") + extracted = mask_archive.extractfile(mask_member) + if extracted is None: + raise SemanticSlamReplayError("E4 semantic mask is unavailable") + mask_bytes = extracted.read(_MAX_MASK_BYTES + 1) + if len(mask_bytes) != mask_member.size: + raise SemanticSlamReplayError("E4 semantic mask length changed") + labels = _decode_mask(mask_bytes, profile) + semantic_time_ns = _validate_semantic_frame( + semantic_frame, + labels, + profile, + frame_index, + ) + mask_sha256 = hashlib.sha256(mask_bytes).hexdigest() + _write_deterministic_zip_member(masks_zip, expected_mask_name, mask_bytes) + + frame_id, source_available, observations = _geometry_frame( + geometry_frame_document, + frame_index, + profile, + ) + source_frame = inputs.store.frame_for_index(frame_index) + temporal_binding = inputs.store.temporal_binding_for_index(frame_index) + _validate_temporal_binding( + semantic_time_ns=semantic_time_ns, + binding=temporal_binding, + profile=profile, + frame_index=frame_index, + ) + if source_available is not (source_frame is not None): + raise SemanticSlamReplayError( + "geometry ledger and source pack availability differ" + ) + projected = _projected_frame(source_frame) + if projected.source_point_count and not source_available: + raise AssertionError("unavailable source published points") + semantic_mask = SemanticMask( + source_id=profile.source_id, + frame_id=frame_id, + provider_id=profile.provider_id, + model_id=profile.model_id, + preprocess_id=profile.preprocess_id, + labels=labels, + classes=taxonomy_definitions, + ) + fused = fuse_semantic_diagnostics( + semantic_mask=semantic_mask, + projected=projected, + observations=observations, + ) + statuses = np.asarray(fused.point_labels.status_codes, dtype=np.uint8) + output_labels = np.zeros(projected.source_point_count, dtype=np.uint8) + available = np.isin( + statuses, + ( + int(SemanticEvidenceStatus.AMBIGUOUS), + int(SemanticEvidenceStatus.LABELED), + ), + ) + output_labels[available] = fused.point_labels.class_ids[available].astype( + np.uint8, + copy=False, + ) + projected_flags = available.astype(np.uint8, copy=False) + point_chunks.append(output_labels) + status_chunks.append(statuses) + projected_chunks.append(projected_flags) + frame_offsets.append(frame_offsets[-1] + projected.source_point_count) + + status_count = Counter(int(value) for value in statuses.tolist()) + labeled = status_count[int(SemanticEvidenceStatus.LABELED)] + ambiguous = status_count[int(SemanticEvidenceStatus.AMBIGUOUS)] + unprojected = status_count[int(SemanticEvidenceStatus.UNPROJECTED)] + absent = status_count[int(SemanticEvidenceStatus.ABSENT)] + projected_count = labeled + ambiguous + frame_source_counts.append(projected.source_point_count) + frame_projected_counts.append(projected_count) + frame_labeled_counts.append(labeled) + frame_ambiguous_counts.append(ambiguous) + frame_unprojected_counts.append(unprojected) + frame_absent_counts.append(absent) + frame_source_available.append(int(source_available)) + source_available_count += int(source_available) + point_counts.update( + { + "total": projected.source_point_count, + "projected": projected_count, + "labeled": labeled, + "ambiguous": ambiguous, + "unprojected": unprojected, + "absent": absent, + } + ) + semantic_observations = [ + _observation_document(observation, evidence) + for observation, evidence in zip( + observations, + fused.observation_evidence, + strict=True, + ) + ] + for evidence in fused.observation_evidence: + observation_counts["total"] += 1 + observation_counts[_status_name(evidence.status)] += 1 + row = { + "schema_version": SEMANTIC_SLAM_FRAME_SCHEMA, + "sequence": frame_index, + "frame_id": frame_id, + "source_time_ns": semantic_time_ns, + "source_available": source_available, + "temporal_binding": { + "semantic_to_camera": profile.temporal_binding["semantic_to_camera"], + "camera_to_lidar": profile.temporal_binding["camera_to_lidar"], + "lidar_camera_delta_ms": temporal_binding.lidar_camera_delta_ms, + "pose_point_delta_ms": temporal_binding.pose_point_delta_ms, + "physical_synchronization_proven": False, + }, + "mask": { + "path": expected_mask_name, + "sha256": mask_sha256, + "width": profile.image_width, + "height": profile.image_height, + }, + "point_range": { + "start": frame_offsets[-2], + "end": frame_offsets[-1], + }, + "point_accounting": { + "total": projected.source_point_count, + "projected": projected_count, + "labeled": labeled, + "ambiguous": ambiguous, + "unprojected": unprojected, + "absent": absent, + }, + "geometry_observation_count": len(observations), + "observations": semantic_observations, + "policy": profile.fusion, + "authority": profile.authority, + } + observation_output.write(_canonical_json(row) + b"\n") + if ( + semantic_stream.readline() + or geometry_stream.readline() + or mask_archive.next() is not None + ): + raise SemanticSlamReplayError("E47 input frame count changed") + + _require_inputs_unchanged(inputs) + frame_count = profile.frame_count + points_path = staging / SEMANTIC_SLAM_POINTS_NAME + _write_deterministic_npz( + points_path, + { + "frame_offsets": np.asarray(frame_offsets, dtype=np.int64), + "point_labels": _concatenate(point_chunks, np.uint8), + "point_status_codes": _concatenate(status_chunks, np.uint8), + "point_projected": _concatenate(projected_chunks, np.uint8), + "frame_source_point_counts": np.asarray(frame_source_counts, dtype=np.int32), + "frame_projected_point_counts": np.asarray(frame_projected_counts, dtype=np.int32), + "frame_labeled_point_counts": np.asarray(frame_labeled_counts, dtype=np.int32), + "frame_ambiguous_point_counts": np.asarray(frame_ambiguous_counts, dtype=np.int32), + "frame_unprojected_point_counts": np.asarray( + frame_unprojected_counts, + dtype=np.int32, + ), + "frame_absent_point_counts": np.asarray(frame_absent_counts, dtype=np.int32), + "frame_source_available": np.asarray(frame_source_available, dtype=np.uint8), + }, + ) + taxonomy_path = staging / SEMANTIC_SLAM_TAXONOMY_NAME + _write_json( + taxonomy_path, + { + "schema_version": SEMANTIC_SLAM_TAXONOMY_SCHEMA, + "classes": [item.document() for item in profile.classes], + }, + ) + points_metrics = _complete_counts(point_counts) + observations_metrics = _complete_counts(observation_counts, include_projected=False) + frames_metrics = { + "total": frame_count, + "mask_available": frame_count, + "source_available": source_available_count, + } + core_artifacts = { + "semantic_points_sha256": _file_sha256(points_path), + "semantic_observations_sha256": _file_sha256(observations_path), + "semantic_masks_sha256": _file_sha256(masks_zip_path), + "taxonomy_sha256": _file_sha256(taxonomy_path), + } + semantic_provider = { + "provider_id": profile.provider_id, + "model_id": profile.model_id, + "model_revision": profile.model_revision, + "model_weights_sha256": profile.model_weights_sha256, + "preprocess_id": profile.preprocess_id, + "mask_metadata_schema_version": profile.mask_metadata_schema_version, + "mask_payload": profile.mask_payload, + "role": profile.provider_role, + } + identity = { + "schema_version": SEMANTIC_SLAM_SCHEMA, + "source_id": profile.source_id, + "source_session_id": profile.session_id, + "profile_id": profile.profile_id, + "profile_sha256": profile.sha256, + "config_sha256": profile.sha256, + "semantic_result_id": inputs.semantic.result_id, + "semantic_result_manifest_sha256": inputs.semantic.result_manifest_sha256, + "semantic_frames_sha256": inputs.semantic.frames_sha256, + "semantic_mask_archive_sha256": inputs.semantic.masks_sha256, + "base_m4_result_id": inputs.threat.result_id, + "base_m4_frames_sha256": inputs.threat_frames_sha256, + "geometry_result_id": inputs.geometry.result_id, + "geometry_frames_sha256": inputs.geometry_frames_sha256, + "source_pack_id": profile.source_pack_id, + "source_pack_sha256": profile.source_pack_sha256, + "calibration_content_sha256": profile.calibration_sha256, + "semantic_provider": semantic_provider, + "policy": profile.fusion, + "temporal_binding": profile.temporal_binding, + "acceptance_policy": profile.acceptance, + "counts": { + "frames": frames_metrics, + "points": points_metrics, + "observations": observations_metrics, + }, + **core_artifacts, + "authority": profile.authority, + } + identity_sha256 = hashlib.sha256(_canonical_json(identity)).hexdigest() + result_id = f"{SEMANTIC_SLAM_RESULT_PREFIX}{identity_sha256}" + elapsed_ms = (time.perf_counter_ns() - started_ns) / 1_000_000 + runtime = { + "elapsed_ms": round(elapsed_ms, 3), + "frames_per_second": round(frame_count * 1000.0 / elapsed_ms, 6), + } + created = datetime.now(UTC).isoformat(timespec="milliseconds").replace("+00:00", "Z") + metrics = { + "frames": frames_metrics, + "points": points_metrics, + "observations": observations_metrics, + "runtime": runtime, + } + report = { + "schema_version": SEMANTIC_SLAM_REPORT_SCHEMA, + "result_id": result_id, + "identity_sha256": identity_sha256, + "created_at_utc": created, + "status": PUBLICATION_STATUS, + "metrics": metrics, + "acceptance": _acceptance_document(), + "limitations": _limitations(), + "authority": profile.authority, + } + report_path = staging / SEMANTIC_SLAM_REPORT_NAME + _write_json(report_path, report) + manifest = { + "schema_version": SEMANTIC_SLAM_SCHEMA, + "result_id": result_id, + "identity_sha256": identity_sha256, + "created_at_utc": created, + "status": PUBLICATION_STATUS, + "identity": identity, + "artifacts": [ + _artifact(points_path, "semantic-point-labels", "application/x-npz"), + _artifact( + observations_path, + "semantic-observation-evidence", + "application/x-ndjson", + SEMANTIC_SLAM_FRAME_SCHEMA, + ), + _artifact(masks_zip_path, "semantic-mask-archive", "application/zip"), + _artifact( + taxonomy_path, + "semantic-taxonomy", + "application/json", + SEMANTIC_SLAM_TAXONOMY_SCHEMA, + ), + _artifact( + report_path, + "semantic-slam-report", + "application/json", + SEMANTIC_SLAM_REPORT_SCHEMA, + ), + ], + "ground_truth": False, + } + _write_json(staging / SEMANTIC_SLAM_MANIFEST_NAME, manifest) + destination = root / result_id + if destination.is_symlink(): + raise SemanticSlamReplayError("semantic SLAM result destination cannot be a symlink") + if destination.exists(): + shutil.rmtree(staging) + return read_semantic_slam_replay_result(destination) + os.replace(staging, destination) + return read_semantic_slam_replay_result(destination) + except BaseException: + shutil.rmtree(staging, ignore_errors=True) + raise + + +def _load_profile(path: Path) -> _SemanticSlamProfile: + document = _read_json(path) + _exact_keys( + document, + { + "schema_version", + "profile_id", + "source", + "semantic_provider", + "taxonomy", + "fusion", + "temporal_binding", + "acceptance", + "authority", + }, + "semantic SLAM profile", + ) + if document.get("schema_version") != "missioncore.e47-semantic-slam-profile/v1": + raise SemanticSlamReplayError("semantic SLAM profile schema is incompatible") + source = _object(document.get("source"), "semantic SLAM source") + provider = _object(document.get("semantic_provider"), "semantic provider") + fusion = _object(document.get("fusion"), "semantic fusion policy") + temporal_binding = _object( + document.get("temporal_binding"), + "semantic temporal binding policy", + ) + acceptance = _object(document.get("acceptance"), "semantic acceptance policy") + authority = _object(document.get("authority"), "semantic authority") + taxonomy_raw = document.get("taxonomy") + if not isinstance(taxonomy_raw, list): + raise SemanticSlamReplayError("semantic taxonomy must be an array") + classes = tuple(_taxonomy_class(item) for item in taxonomy_raw) + if ( + len(classes) != 16 + or tuple(item.class_id for item in classes) != tuple(range(16)) + or classes[0].disposition is not SemanticClassDisposition.AMBIGUOUS + or any(item.disposition is not SemanticClassDisposition.LABELED for item in classes[1:]) + ): + raise SemanticSlamReplayError("semantic taxonomy changed") + expected_source = { + "source_id": BASELINE_SOURCE_ID, + "session_id": BASELINE_SESSION_ID, + "frame_count": EXPECTED_FRAME_COUNT, + "image_width": 800, + "image_height": 600, + "source_pack_id": BASELINE_SOURCE_PACK_ID, + "source_pack_sha256": BASELINE_SOURCE_PACK_SHA256, + "calibration_content_sha256": ( + "05f3ad9b38b3a4fc95388a8ec83da83c745e217709e51787b3d5aad0969f6fa9" + ), + } + if source != expected_source: + raise SemanticSlamReplayError("semantic SLAM source identity changed") + expected_provider = { + "provider_id": "eomt-cityscapes-semantic-control/v1", + "model_id": "tue-mps/cityscapes_semantic_eomt_large_1024", + "model_revision": "8d6b6d1a3f7b50d441afd7d247c2ed10db186e8f", + "model_weights_sha256": ( + "c265da9a74f58f5c3f4826d23ca4ca78beac0b106cca5842beca61580de5b782" + ), + "preprocess_id": "raw-kb4-valid-fov-semantic/v1", + "mask_metadata_schema_version": "missioncore.panoptic-frame/v1", + "mask_payload": { + "media_type": "image/png", + "encoding": "uint8-class-id", + "width": 800, + "height": 600, + "sequence_binding": "sequence-0-to-frame-000001", + }, + "role": "fixed-control-not-selected-production-provider", + } + if provider != expected_provider: + raise SemanticSlamReplayError("semantic provider identity changed") + _validate_policy(fusion, temporal_binding, acceptance, authority) + profile_id = document.get("profile_id") + if profile_id != "ravnoves00-eomt-kb4-slam-shadow/v1": + raise SemanticSlamReplayError("semantic SLAM profile id changed") + return _SemanticSlamProfile( + path=path.resolve(strict=True), + sha256=_file_sha256(path), + profile_id=profile_id, + source_id=BASELINE_SOURCE_ID, + session_id=BASELINE_SESSION_ID, + frame_count=EXPECTED_FRAME_COUNT, + image_width=800, + image_height=600, + source_pack_id=BASELINE_SOURCE_PACK_ID, + source_pack_sha256=BASELINE_SOURCE_PACK_SHA256, + calibration_sha256=("05f3ad9b38b3a4fc95388a8ec83da83c745e217709e51787b3d5aad0969f6fa9"), + provider_id="eomt-cityscapes-semantic-control/v1", + model_id="tue-mps/cityscapes_semantic_eomt_large_1024", + model_revision="8d6b6d1a3f7b50d441afd7d247c2ed10db186e8f", + model_weights_sha256=("c265da9a74f58f5c3f4826d23ca4ca78beac0b106cca5842beca61580de5b782"), + preprocess_id="raw-kb4-valid-fov-semantic/v1", + mask_metadata_schema_version="missioncore.panoptic-frame/v1", + mask_payload={ + "media_type": "image/png", + "encoding": "uint8-class-id", + "width": 800, + "height": 600, + "sequence_binding": "sequence-0-to-frame-000001", + }, + provider_role="fixed-control-not-selected-production-provider", + classes=classes, + fusion=dict(fusion), + temporal_binding=dict(temporal_binding), + acceptance=dict(acceptance), + authority=dict(authority), + ) + + +def _validate_policy( + fusion: dict[str, object], + temporal_binding: dict[str, object], + acceptance: dict[str, object], + authority: dict[str, object], +) -> None: + if fusion != { + "projection": "factory-kb4-current-increment/v1", + "point_index_space": "frame-local-source-point-id/v1", + "observation_aggregation": "dominant-labeled-majority-diagnostic/v1", + "unprojected_status": "unprojected", + "semantic_absence_means_free": False, + "semantic_can_create_obstacle": False, + "semantic_can_change_identity": False, + "semantic_can_change_metric_geometry": False, + "semantic_can_change_occupancy": False, + "semantic_can_change_motion": False, + "semantic_can_change_threat": False, + }: + raise SemanticSlamReplayError("semantic fusion authority changed") + if temporal_binding != { + "semantic_to_camera": "exact-sequence-and-session-time", + "camera_to_lidar": "accepted-e6-nearest-host-arrival-best-effort", + "clock_basis": "recorded-host-monotonic-arrival", + "maximum_lidar_camera_delta_ms": 100.0, + "maximum_pose_point_delta_ms": 100.0, + "physical_synchronization_proven": False, + }: + raise SemanticSlamReplayError("semantic temporal binding policy changed") + if acceptance != { + "full_frame_accounting_required": True, + "point_accounting_required": True, + "observation_binding_required": True, + "exact_mask_archive_required": True, + "independent_semantic_truth_required_for_provider_promotion": True, + }: + raise SemanticSlamReplayError("semantic acceptance policy changed") + if authority != { + "ground_truth": False, + "physical_live": False, + "commands_enabled": False, + "actuation_allowed": False, + "navigation_or_safety_accepted": False, + "semantic_authority": "diagnostic-only", + }: + raise SemanticSlamReplayError("semantic authority must remain diagnostic-only") + + +def _validate_semantic_frame( + document: dict[str, object], + labels: np.ndarray, + profile: _SemanticSlamProfile, + frame_index: int, +) -> int: + session_seconds = document.get("session_seconds") + semantic_rows = document.get("semantic_classes") + if ( + document.get("schema_version") != "missioncore.panoptic-frame/v1" + or document.get("frame_index") != frame_index + or document.get("sequence") != frame_index + 1 + or document.get("instances") != [] + or not isinstance(session_seconds, (int, float)) + or isinstance(session_seconds, bool) + or not math.isfinite(float(session_seconds)) + or not isinstance(semantic_rows, list) + ): + raise SemanticSlamReplayError("E4 semantic frame metadata changed") + taxonomy = {item.class_id: item.label for item in profile.classes} + unique, counts = np.unique(labels, return_counts=True) + pixel_counts = dict( + zip( + (int(value) for value in unique), + (int(value) for value in counts), + strict=True, + ) + ) + if set(pixel_counts) - set(taxonomy): + raise SemanticSlamReplayError("E4 mask contains an undeclared class") + expected_ids = sorted(class_id for class_id in pixel_counts if class_id != 0) + row_ids: list[int] = [] + for value in semantic_rows: + row = _object(value, "E4 semantic class row") + class_id = row.get("id") + if ( + not isinstance(class_id, int) + or isinstance(class_id, bool) + or class_id == 0 + or row.get("label") != taxonomy.get(class_id) + or row.get("pixels") != pixel_counts.get(class_id) + ): + raise SemanticSlamReplayError("E4 semantic class accounting changed") + row_ids.append(class_id) + if sorted(row_ids) != expected_ids or len(set(row_ids)) != len(row_ids): + raise SemanticSlamReplayError("E4 semantic class inventory changed") + return round(float(session_seconds) * 1_000_000_000) + + +def _validate_temporal_binding( + *, + semantic_time_ns: int, + binding: RecordedFrameTemporalBinding, + profile: _SemanticSlamProfile, + frame_index: int, +) -> None: + if binding.frame_index != frame_index or binding.source_time_ns != semantic_time_ns: + raise SemanticSlamReplayError("E4 semantic frame and E10 source-pack session time disagree") + if binding.source_available: + lidar_delta = binding.lidar_camera_delta_ms + pose_delta = binding.pose_point_delta_ms + lidar_bound = _positive_number( + profile.temporal_binding.get("maximum_lidar_camera_delta_ms"), + "maximum LiDAR-camera delta", + ) + pose_bound = _positive_number( + profile.temporal_binding.get("maximum_pose_point_delta_ms"), + "maximum pose-point delta", + ) + if ( + lidar_delta is None + or pose_delta is None + or abs(lidar_delta) > lidar_bound + or abs(pose_delta) > pose_bound + ): + raise SemanticSlamReplayError("E10 best-effort temporal delta exceeds its bound") + elif binding.lidar_camera_delta_ms is not None or binding.pose_point_delta_ms is not None: + raise SemanticSlamReplayError("unavailable E10 frame carries temporal deltas") + + +def _geometry_frame( + document: dict[str, object], + frame_index: int, + profile: _SemanticSlamProfile, +) -> tuple[str, bool, tuple[ObstacleObservation, ...]]: + frame_id = document.get("frame_id") + source_available = document.get("source_available") + raw_observations = document.get("observations") + if ( + document.get("schema_version") != "missioncore.perception-geometry-replay-frame/v1" + or document.get("sequence") != frame_index + or not isinstance(frame_id, str) + or not isinstance(source_available, bool) + or not isinstance(raw_observations, list) + ): + raise SemanticSlamReplayError("M4 geometry frame changed") + try: + observations = tuple(ObstacleObservation.from_dict(value) for value in raw_observations) + except ValueError as exc: + raise SemanticSlamReplayError("M4 geometry observation changed") from exc + if any( + observation.source_id != profile.source_id or observation.frame_id != frame_id + for observation in observations + ): + raise SemanticSlamReplayError("M4 geometry observation escaped its source frame") + return frame_id, source_available, observations + + +def _projected_frame(frame: GeometryFrame | None) -> ProjectedPointCloud: + if frame is None: + return ProjectedPointCloud( + pixels_xy=np.empty((0, 2), dtype=np.float64), + depths_m=np.empty(0, dtype=np.float64), + source_indices=np.empty(0, dtype=np.int64), + source_point_count=0, + camera_front_point_count=0, + ) + return project_map_points_kb4( + frame.points_map, + position_map_xyz=frame.sensor_position_map, + orientation_map_from_lidar_xyzw=frame.sensor_orientation_xyzw, + profile=frame.projection, + ) + + +def _observation_document( + observation: ObstacleObservation, + evidence: ObservationSemanticEvidence, +) -> dict[str, object]: + observation_document = observation.to_dict() + return { + "observation_id": observation.observation_id, + "occupancy_key": observation.occupancy_key, + "geometry_observation_sha256": hashlib.sha256( + _canonical_json(observation_document) + ).hexdigest(), + "source_point_ids_sha256": hashlib.sha256( + _canonical_json(list(observation.source_point_ids)) + ).hexdigest(), + "status": _status_name(evidence.status), + "source_point_count": evidence.source_point_count, + "labeled_point_count": evidence.labeled_point_count, + "ambiguous_point_count": evidence.ambiguous_point_count, + "unprojected_point_count": evidence.unprojected_point_count, + "absent_point_count": evidence.absent_point_count, + "class_evidence": [ + { + "class_id": item.class_id, + "label": item.label, + "disposition": item.disposition.value, + "point_count": item.point_count, + } + for item in evidence.class_evidence + ], + "dominant_class_id": evidence.dominant_class_id, + "dominant_label": evidence.dominant_label, + "dominant_fraction_of_labeled": evidence.dominant_fraction_of_labeled, + "reason_code": evidence.reason_code, + "authority": evidence.authority.value, + } + + +def _decode_mask(payload: bytes, profile: _SemanticSlamProfile) -> np.ndarray: + try: + with Image.open(io.BytesIO(payload)) as image: + if image.format != "PNG" or image.mode not in {"L", "P"}: + raise SemanticSlamReplayError("E4 semantic mask encoding changed") + labels = np.asarray(image, dtype=np.uint8) + except (OSError, UnidentifiedImageError, ValueError) as exc: + raise SemanticSlamReplayError("E4 semantic mask cannot be decoded") from exc + if labels.shape != (profile.image_height, profile.image_width): + raise SemanticSlamReplayError("E4 semantic mask dimensions changed") + return np.array(labels, dtype=np.uint8, order="C", copy=True) + + +def _validate_tar_root(member: tarfile.TarInfo | None) -> None: + if member is None or not member.isdir() or member.name.rstrip("/") != "semantic-masks": + raise SemanticSlamReplayError("E4 semantic mask archive root changed") + + +def _require_inputs_unchanged(inputs: _AdmittedInputs) -> None: + if ( + _file_sha256(inputs.semantic.frames_path) != inputs.semantic.frames_sha256 + or _file_sha256(inputs.semantic.masks_path) != inputs.semantic.masks_sha256 + or _file_sha256(inputs.geometry_frames_path) != inputs.geometry_frames_sha256 + or _file_sha256(inputs.threat_frames_path) != inputs.threat_frames_sha256 + or _file_sha256(inputs.profile.path) != inputs.profile.sha256 + ): + raise SemanticSlamReplayError("an admitted E47 input changed during the build") + + +def _validate_identity(identity: dict[str, object]) -> None: + required = { + "schema_version", + "source_id", + "source_session_id", + "profile_id", + "profile_sha256", + "config_sha256", + "semantic_result_id", + "semantic_result_manifest_sha256", + "semantic_frames_sha256", + "semantic_mask_archive_sha256", + "base_m4_result_id", + "base_m4_frames_sha256", + "geometry_result_id", + "geometry_frames_sha256", + "source_pack_id", + "source_pack_sha256", + "calibration_content_sha256", + "semantic_provider", + "policy", + "temporal_binding", + "acceptance_policy", + "counts", + "semantic_points_sha256", + "semantic_observations_sha256", + "semantic_masks_sha256", + "taxonomy_sha256", + "authority", + } + _exact_keys(identity, required, "semantic SLAM identity") + if identity.get("schema_version") != SEMANTIC_SLAM_SCHEMA or identity.get( + "profile_sha256" + ) != identity.get("config_sha256"): + raise SemanticSlamReplayError("semantic SLAM identity is incompatible") + for key in ( + "profile_sha256", + "config_sha256", + "semantic_result_manifest_sha256", + "semantic_frames_sha256", + "semantic_mask_archive_sha256", + "base_m4_frames_sha256", + "geometry_frames_sha256", + "source_pack_sha256", + "calibration_content_sha256", + "semantic_points_sha256", + "semantic_observations_sha256", + "semantic_masks_sha256", + "taxonomy_sha256", + ): + _digest(identity, key) + provider = _object(identity.get("semantic_provider"), "semantic provider identity") + _exact_keys( + provider, + { + "provider_id", + "model_id", + "model_revision", + "model_weights_sha256", + "preprocess_id", + "mask_metadata_schema_version", + "mask_payload", + "role", + }, + "semantic provider identity", + ) + _digest(provider, "model_weights_sha256") + mask_payload = _object(provider.get("mask_payload"), "semantic mask payload contract") + _exact_keys( + mask_payload, + {"media_type", "encoding", "width", "height", "sequence_binding"}, + "semantic mask payload contract", + ) + if ( + provider.get("mask_metadata_schema_version") != "missioncore.panoptic-frame/v1" + or mask_payload.get("media_type") != "image/png" + or mask_payload.get("encoding") != "uint8-class-id" + or not isinstance(mask_payload.get("width"), int) + or not isinstance(mask_payload.get("height"), int) + or mask_payload.get("sequence_binding") != "sequence-0-to-frame-000001" + ): + raise SemanticSlamReplayError("semantic mask provider contract changed") + _require_false_authority(identity.get("authority"), "semantic SLAM authority") + authority = _object(identity.get("authority"), "semantic SLAM authority") + if authority.get("semantic_authority") != "diagnostic-only": + raise SemanticSlamReplayError("semantic authority changed") + temporal_binding = _object( + identity.get("temporal_binding"), + "semantic temporal binding identity", + ) + if temporal_binding != { + "semantic_to_camera": "exact-sequence-and-session-time", + "camera_to_lidar": "accepted-e6-nearest-host-arrival-best-effort", + "clock_basis": "recorded-host-monotonic-arrival", + "maximum_lidar_camera_delta_ms": 100.0, + "maximum_pose_point_delta_ms": 100.0, + "physical_synchronization_proven": False, + }: + raise SemanticSlamReplayError("semantic temporal binding identity changed") + + +def _validate_points_artifact( + path: Path, + identity: dict[str, object], +) -> _PointArtifactMetrics: + expected_keys = { + "frame_offsets", + "point_labels", + "point_status_codes", + "point_projected", + "frame_source_point_counts", + "frame_projected_point_counts", + "frame_labeled_point_counts", + "frame_ambiguous_point_counts", + "frame_unprojected_point_counts", + "frame_absent_point_counts", + "frame_source_available", + } + try: + with np.load(path, allow_pickle=False) as archive: + if set(archive.files) != expected_keys: + raise SemanticSlamReplayError("semantic point array inventory changed") + arrays = {name: np.asarray(archive[name]) for name in expected_keys} + except (OSError, ValueError, zipfile.BadZipFile) as exc: + raise SemanticSlamReplayError("semantic point artifact is invalid") from exc + offsets = arrays["frame_offsets"] + labels = arrays["point_labels"] + statuses = arrays["point_status_codes"] + projected = arrays["point_projected"] + frame_count = int(offsets.size - 1) + if ( + offsets.dtype != np.int64 + or offsets.ndim != 1 + or offsets.size < 2 + or int(offsets[0]) != 0 + or np.any(np.diff(offsets) < 0) + or labels.dtype != np.uint8 + or statuses.dtype != np.uint8 + or projected.dtype != np.uint8 + or labels.shape != statuses.shape + or labels.shape != projected.shape + or labels.ndim != 1 + or int(offsets[-1]) != labels.size + or np.any((projected != 0) & (projected != 1)) + or np.any((arrays["frame_source_available"] != 0) & (arrays["frame_source_available"] != 1)) + ): + raise SemanticSlamReplayError("semantic point array contract changed") + frame_arrays = { + name: value + for name, value in arrays.items() + if name.startswith("frame_") and name != "frame_offsets" + } + if any( + value.ndim != 1 + or value.shape != (frame_count,) + or (name == "frame_source_available" and value.dtype != np.uint8) + or (name != "frame_source_available" and value.dtype != np.int32) + for name, value in frame_arrays.items() + ): + raise SemanticSlamReplayError("semantic frame point accounting arrays changed") + valid_statuses = {int(value) for value in SemanticEvidenceStatus} + if set(int(value) for value in np.unique(statuses)) - valid_statuses: + raise SemanticSlamReplayError("semantic point status code changed") + semantic_available = np.isin( + statuses, + (int(SemanticEvidenceStatus.AMBIGUOUS), int(SemanticEvidenceStatus.LABELED)), + ) + if not np.array_equal(projected.astype(bool), semantic_available): + raise SemanticSlamReplayError("semantic projected flags disagree with statuses") + unavailable = ~semantic_available + if np.any(labels[unavailable] != 0): + raise SemanticSlamReplayError("unprojected semantic points must remain uint8 zero") + source_counts = np.diff(offsets) + if not np.array_equal(source_counts, arrays["frame_source_point_counts"]): + raise SemanticSlamReplayError("semantic frame offsets disagree with point counts") + per_status = { + "labeled": int(np.count_nonzero(statuses == int(SemanticEvidenceStatus.LABELED))), + "ambiguous": int(np.count_nonzero(statuses == int(SemanticEvidenceStatus.AMBIGUOUS))), + "unprojected": int(np.count_nonzero(statuses == int(SemanticEvidenceStatus.UNPROJECTED))), + "absent": int(np.count_nonzero(statuses == int(SemanticEvidenceStatus.ABSENT))), + } + points = { + "total": int(labels.size), + "projected": int(np.count_nonzero(projected)), + **per_status, + } + expected_counts = _object(identity.get("counts"), "semantic identity counts") + if points != expected_counts.get("points"): + raise SemanticSlamReplayError("semantic point metrics changed") + frame_source_available_count = int(np.count_nonzero(arrays["frame_source_available"])) + frames = { + "total": frame_count, + "mask_available": frame_count, + "source_available": frame_source_available_count, + } + if frames != expected_counts.get("frames"): + raise SemanticSlamReplayError("semantic frame metrics changed") + frame_status_arrays = { + "labeled": arrays["frame_labeled_point_counts"], + "ambiguous": arrays["frame_ambiguous_point_counts"], + "unprojected": arrays["frame_unprojected_point_counts"], + "absent": arrays["frame_absent_point_counts"], + } + if ( + int(arrays["frame_projected_point_counts"].sum()) != points["projected"] + or any(int(values.sum()) != points[name] for name, values in frame_status_arrays.items()) + or np.any( + arrays["frame_projected_point_counts"] + != arrays["frame_labeled_point_counts"] + arrays["frame_ambiguous_point_counts"] + ) + or np.any( + arrays["frame_source_point_counts"] + != arrays["frame_projected_point_counts"] + + arrays["frame_unprojected_point_counts"] + + arrays["frame_absent_point_counts"] + ) + ): + raise SemanticSlamReplayError("semantic point accounting is incomplete") + observations = _count_document( + expected_counts.get("observations"), + include_projected=False, + label="observation counts", + ) + frame_point_accounting = tuple( + { + "total": int(arrays["frame_source_point_counts"][index]), + "projected": int(arrays["frame_projected_point_counts"][index]), + "labeled": int(arrays["frame_labeled_point_counts"][index]), + "ambiguous": int(arrays["frame_ambiguous_point_counts"][index]), + "unprojected": int(arrays["frame_unprojected_point_counts"][index]), + "absent": int(arrays["frame_absent_point_counts"][index]), + } + for index in range(frame_count) + ) + return _PointArtifactMetrics( + frames=frames, + points=points, + observations=observations, + frame_point_accounting=frame_point_accounting, + ) + + +def _validate_observation_ledger( + path: Path, + identity: dict[str, object], + taxonomy: dict[int, _TaxonomyClass], +) -> tuple[ + dict[str, int], + dict[str, int], + list[tuple[str, str]], + tuple[dict[str, int], ...], +]: + counts = _object(identity.get("counts"), "semantic identity counts") + expected_frames = _object(counts.get("frames"), "semantic frame counts") + frame_total = _nonnegative_integer(expected_frames.get("total"), "frame total") + observation_counts: Counter[str] = Counter() + source_available = 0 + mask_rows: list[tuple[str, str]] = [] + point_rows: list[dict[str, int]] = [] + expected_point_start = 0 + provider = _object(identity.get("semantic_provider"), "semantic provider identity") + mask_payload = _object(provider.get("mask_payload"), "semantic mask payload contract") + expected_width = _nonnegative_integer(mask_payload.get("width"), "semantic mask width") + expected_height = _nonnegative_integer(mask_payload.get("height"), "semantic mask height") + temporal_policy = _object( + identity.get("temporal_binding"), + "semantic temporal binding identity", + ) + previous_source_time_ns: int | None = None + try: + with path.open("rb") as stream: + for frame_index, line in enumerate(stream): + if frame_index >= frame_total: + raise SemanticSlamReplayError("semantic observation ledger has extra frames") + row = _read_json_line(line, "semantic observation frame", frame_index) + row_source_available = row.get("source_available") + row_observations = row.get("observations") + _exact_keys( + row, + { + "schema_version", + "sequence", + "frame_id", + "source_time_ns", + "source_available", + "temporal_binding", + "mask", + "point_range", + "point_accounting", + "geometry_observation_count", + "observations", + "policy", + "authority", + }, + "semantic observation frame", + ) + source_time_ns = row.get("source_time_ns") + if ( + row.get("schema_version") != SEMANTIC_SLAM_FRAME_SCHEMA + or row.get("sequence") != frame_index + or not isinstance(row.get("frame_id"), str) + or not isinstance(row_source_available, bool) + or not isinstance(source_time_ns, int) + or isinstance(source_time_ns, bool) + or source_time_ns < 0 + or ( + previous_source_time_ns is not None + and source_time_ns <= previous_source_time_ns + ) + or not isinstance(row_observations, list) + or row.get("geometry_observation_count") != len(row_observations) + ): + raise SemanticSlamReplayError("semantic observation frame changed") + previous_source_time_ns = source_time_ns + source_available += int(row_source_available) + temporal = _object(row.get("temporal_binding"), "frame temporal binding") + _exact_keys( + temporal, + { + "semantic_to_camera", + "camera_to_lidar", + "lidar_camera_delta_ms", + "pose_point_delta_ms", + "physical_synchronization_proven", + }, + "frame temporal binding", + ) + lidar_delta = temporal.get("lidar_camera_delta_ms") + pose_delta = temporal.get("pose_point_delta_ms") + if ( + temporal.get("semantic_to_camera") != temporal_policy.get("semantic_to_camera") + or temporal.get("camera_to_lidar") != temporal_policy.get("camera_to_lidar") + or temporal.get("physical_synchronization_proven") is not False + or ( + row_source_available + and ( + not _bounded_delta( + lidar_delta, + temporal_policy.get("maximum_lidar_camera_delta_ms"), + ) + or not _bounded_delta( + pose_delta, + temporal_policy.get("maximum_pose_point_delta_ms"), + ) + ) + ) + or ( + not row_source_available + and (lidar_delta is not None or pose_delta is not None) + ) + ): + raise SemanticSlamReplayError("frame temporal binding changed") + mask = _object(row.get("mask"), "semantic mask binding") + mask_path = mask.get("path") + mask_sha256 = mask.get("sha256") + if ( + mask_path != f"semantic-masks/frame-{frame_index + 1:06d}.png" + or not isinstance(mask_sha256, str) + or _SHA256.fullmatch(mask_sha256) is None + or mask.get("width") != expected_width + or mask.get("height") != expected_height + ): + raise SemanticSlamReplayError("semantic mask binding changed") + mask_rows.append((mask_path, mask_sha256)) + point_accounting = _object(row.get("point_accounting"), "point accounting") + point_counts = _count_document( + point_accounting, + include_projected=True, + label="semantic frame point accounting", + ) + if ( + point_counts["projected"] != point_counts["labeled"] + point_counts["ambiguous"] + or point_counts["total"] + != point_counts["projected"] + + point_counts["unprojected"] + + point_counts["absent"] + ): + raise SemanticSlamReplayError("semantic frame point accounting is incomplete") + point_range = _object(row.get("point_range"), "semantic point range") + _exact_keys(point_range, {"start", "end"}, "semantic point range") + point_start = _nonnegative_integer(point_range.get("start"), "point range start") + point_end = _nonnegative_integer(point_range.get("end"), "point range end") + if ( + point_start != expected_point_start + or point_end - point_start != point_counts["total"] + ): + raise SemanticSlamReplayError("semantic point range changed") + expected_point_start = point_end + point_rows.append(point_counts) + for raw in row_observations: + observation = _object(raw, "semantic observation") + status = observation.get("status") + if status not in {"labeled", "ambiguous", "unprojected", "absent"}: + raise SemanticSlamReplayError("semantic observation status changed") + observation_counts["total"] += 1 + observation_counts[str(status)] += 1 + for evidence_raw in _array( + observation.get("class_evidence"), + "semantic class evidence", + ): + evidence = _object(evidence_raw, "semantic class evidence") + class_id = evidence.get("class_id") + definition = taxonomy.get(class_id) if isinstance(class_id, int) else None + if ( + definition is None + or evidence.get("label") != definition.label + or evidence.get("disposition") != definition.disposition.value + ): + raise SemanticSlamReplayError("semantic class evidence changed") + if frame_total != len(mask_rows): + raise SemanticSlamReplayError("semantic observation ledger is incomplete") + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: + raise SemanticSlamReplayError("semantic observation ledger is invalid") from exc + frames = { + "total": frame_total, + "mask_available": frame_total, + "source_available": source_available, + } + observations = _complete_counts(observation_counts, include_projected=False) + if frames != expected_frames or observations != counts.get("observations"): + raise SemanticSlamReplayError("semantic observation ledger accounting changed") + return frames, observations, mask_rows, tuple(point_rows) + + +def _validate_mask_archive(path: Path, rows: list[tuple[str, str]]) -> None: + try: + with zipfile.ZipFile(path, mode="r") as archive: + if archive.namelist() != [name for name, _ in rows]: + raise SemanticSlamReplayError("semantic mask archive inventory changed") + for name, expected_sha256 in rows: + payload = archive.read(name) + if hashlib.sha256(payload).hexdigest() != expected_sha256: + raise SemanticSlamReplayError("semantic mask copy changed") + except (OSError, KeyError, zipfile.BadZipFile, RuntimeError) as exc: + if isinstance(exc, SemanticSlamReplayError): + raise + raise SemanticSlamReplayError("semantic mask archive is invalid") from exc + + +def _validate_taxonomy(path: Path) -> dict[int, _TaxonomyClass]: + document = _read_json(path) + _exact_keys(document, {"schema_version", "classes"}, "semantic taxonomy") + if document.get("schema_version") != SEMANTIC_SLAM_TAXONOMY_SCHEMA: + raise SemanticSlamReplayError("semantic taxonomy schema changed") + values = document.get("classes") + if not isinstance(values, list) or not values: + raise SemanticSlamReplayError("semantic taxonomy classes changed") + classes = tuple(_taxonomy_class(value) for value in values) + if len({item.class_id for item in classes}) != len(classes): + raise SemanticSlamReplayError("semantic taxonomy class ids are duplicated") + return {item.class_id: item for item in classes} + + +def _taxonomy_class(value: object) -> _TaxonomyClass: + document = _object(value, "semantic taxonomy class") + _exact_keys( + document, + {"class_id", "label", "disposition", "color_rgb"}, + "semantic taxonomy class", + ) + class_id = document.get("class_id") + label = document.get("label") + color = document.get("color_rgb") + disposition_raw = document.get("disposition") + if not isinstance(disposition_raw, str): + raise SemanticSlamReplayError("semantic taxonomy disposition is invalid") + try: + disposition = SemanticClassDisposition(disposition_raw) + except ValueError as exc: + raise SemanticSlamReplayError("semantic taxonomy disposition is invalid") from exc + if ( + not isinstance(class_id, int) + or isinstance(class_id, bool) + or not 0 <= class_id <= 255 + or not isinstance(label, str) + or not label + or not isinstance(color, list) + or len(color) != 3 + or any( + not isinstance(channel, int) or isinstance(channel, bool) or not 0 <= channel <= 255 + for channel in color + ) + ): + raise SemanticSlamReplayError("semantic taxonomy class is invalid") + return _TaxonomyClass( + class_id=class_id, + label=label, + disposition=disposition, + color_rgb=(color[0], color[1], color[2]), + ) + + +def _write_deterministic_npz(path: Path, arrays: dict[str, np.ndarray]) -> None: + with zipfile.ZipFile( + path, + mode="w", + compression=zipfile.ZIP_DEFLATED, + compresslevel=6, + ) as archive: + for name in sorted(arrays): + buffer = io.BytesIO() + np.lib.format.write_array(buffer, np.asarray(arrays[name]), allow_pickle=False) + _write_deterministic_zip_member(archive, f"{name}.npy", buffer.getvalue()) + + +def _write_deterministic_zip_member( + archive: zipfile.ZipFile, + name: str, + payload: bytes, +) -> None: + info = zipfile.ZipInfo(name, date_time=(1980, 1, 1, 0, 0, 0)) + info.compress_type = archive.compression + info.create_system = 3 + info.external_attr = 0o100600 << 16 + archive.writestr(info, payload) + + +def _concatenate(values: list[np.ndarray], dtype: type[np.generic]) -> np.ndarray: + if not values: + return np.empty(0, dtype=dtype) + return np.concatenate(values).astype(dtype, copy=False) + + +def _status_name(status: SemanticEvidenceStatus) -> str: + return status.name.lower() + + +def _complete_counts( + counts: Counter[str], + *, + include_projected: bool = True, +) -> dict[str, int]: + result = {"total": int(counts["total"])} + if include_projected: + result["projected"] = int(counts["projected"]) + result.update( + { + "labeled": int(counts["labeled"]), + "ambiguous": int(counts["ambiguous"]), + "unprojected": int(counts["unprojected"]), + "absent": int(counts["absent"]), + } + ) + return result + + +def _count_document( + value: object, + *, + include_projected: bool, + label: str, +) -> dict[str, int]: + document = _object(value, label) + expected = {"total", "labeled", "ambiguous", "unprojected", "absent"} + if include_projected: + expected.add("projected") + _exact_keys(document, expected, label) + return { + key: _nonnegative_integer(document.get(key), f"{label} {key}") for key in sorted(expected) + } + + +def _acceptance_document() -> dict[str, bool]: + return { + "diagnostic_publication_accepted": True, + "full_frame_accounting": True, + "point_accounting_complete": True, + "geometry_observation_binding_complete": True, + "exact_mask_copy_complete": True, + "semantic_camera_sequence_and_time_binding_complete": True, + "camera_lidar_best_effort_delta_bound_checked": True, + "camera_lidar_hardware_sync_proven": False, + "geometry_and_threat_ledgers_unchanged": True, + "independent_semantic_truth_complete": False, + "semantic_provider_promoted": False, + "navigation_or_safety_accepted": False, + } + + +def _limitations() -> list[str]: + return [ + "E4 EoMT masks are a sealed engineering control, not independent semantic truth.", + "Class 0 and unprojected source points remain unknown for diagnostic review.", + ( + "Semantic masks match the camera sequence and session clock exactly, but the " + "E10 camera-to-LiDAR association remains nearest-host-arrival best effort " + "within the admitted 100 ms delta bounds; hardware synchronization is unproven." + ), + "Semantic labels cannot create obstacles or change occupancy, motion, threat or safety.", + "The result is recorded-replay evidence and has no physical-live or actuation authority.", + ] + + +def _valid_runtime_metrics(value: object, frame_count: int) -> bool: + if not isinstance(value, dict) or set(value) != {"elapsed_ms", "frames_per_second"}: + return False + elapsed = value.get("elapsed_ms") + fps = value.get("frames_per_second") + return ( + isinstance(elapsed, (int, float)) + and not isinstance(elapsed, bool) + and math.isfinite(float(elapsed)) + and float(elapsed) > 0.0 + and isinstance(fps, (int, float)) + and not isinstance(fps, bool) + and math.isfinite(float(fps)) + and float(fps) > 0.0 + and frame_count > 0 + ) + + +def _bounded_delta(value: object, bound: object) -> bool: + return ( + isinstance(value, (int, float)) + and not isinstance(value, bool) + and math.isfinite(float(value)) + and isinstance(bound, (int, float)) + and not isinstance(bound, bool) + and math.isfinite(float(bound)) + and float(bound) > 0.0 + and abs(float(value)) <= float(bound) + ) + + +def _artifact( + path: Path, + role: str, + media_type: str, + schema_version: str | None = None, +) -> dict[str, object]: + document: dict[str, object] = { + "role": role, + "path": path.name, + "byte_length": path.stat().st_size, + "sha256": _file_sha256(path), + "media_type": media_type, + } + if schema_version is not None: + document["schema_version"] = schema_version + return document + + +def _validated_artifact( + root: Path, + raw: object, + *, + expected_role: str, + expected_name: str, + expected_media_type: str, + expected_schema_version: str | None, +) -> Path: + document = _object(raw, "semantic SLAM artifact") + expected_keys = {"role", "path", "byte_length", "sha256", "media_type"} + if expected_schema_version is not None: + expected_keys.add("schema_version") + _exact_keys(document, expected_keys, "semantic SLAM artifact") + path = root / expected_name + try: + resolved = path.resolve(strict=True) + metadata = resolved.stat() + except OSError as exc: + raise SemanticSlamReplayError("semantic SLAM artifact is missing") from exc + if ( + document.get("role") != expected_role + or document.get("path") != expected_name + or document.get("media_type") != expected_media_type + or document.get("schema_version") != expected_schema_version + or resolved.parent != root + or path.is_symlink() + or not resolved.is_file() + or document.get("byte_length") != metadata.st_size + or document.get("sha256") != _file_sha256(resolved) + ): + raise SemanticSlamReplayError("semantic SLAM artifact digest changed") + return resolved + + +def _write_json(path: Path, value: object) -> None: + path.write_bytes(_canonical_json(value) + b"\n") + + +def _read_json(path: Path) -> dict[str, object]: + try: + if not path.is_file() or path.is_symlink() or path.stat().st_size > _MAX_JSON_BYTES: + raise SemanticSlamReplayError("JSON artifact is missing or outside bounds") + return _object(json.loads(path.read_text("utf-8")), "JSON artifact") + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc: + raise SemanticSlamReplayError("JSON artifact is invalid") from exc + + +def _read_json_line(raw: bytes, label: str, sequence: int) -> dict[str, object]: + if not raw or len(raw) > _MAX_JSONL_LINE_BYTES: + raise SemanticSlamReplayError(f"{label} {sequence} is outside bounds") + try: + return _object(json.loads(raw), label) + except (UnicodeDecodeError, json.JSONDecodeError) as exc: + raise SemanticSlamReplayError(f"{label} {sequence} is invalid") from exc + + +def _canonical_json(value: object) -> bytes: + return json.dumps( + value, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode("utf-8") + + +def _file_sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _object(value: object, label: str) -> dict[str, object]: + if not isinstance(value, dict) or any(not isinstance(key, str) for key in value): + raise SemanticSlamReplayError(f"{label} must be an object") + return value + + +def _array(value: object, label: str) -> list[object]: + if not isinstance(value, list): + raise SemanticSlamReplayError(f"{label} must be an array") + return value + + +def _exact_keys(document: dict[str, object], expected: set[str], label: str) -> None: + if set(document) != expected: + raise SemanticSlamReplayError(f"{label} fields changed") + + +def _digest(document: dict[str, object], key: str) -> str: + value = document.get(key) + if not isinstance(value, str) or _SHA256.fullmatch(value) is None: + raise SemanticSlamReplayError(f"{key} digest is invalid") + return value + + +def _nonnegative_integer(value: object, label: str) -> int: + if not isinstance(value, int) or isinstance(value, bool) or value < 0: + raise SemanticSlamReplayError(f"{label} must be a nonnegative integer") + return value + + +def _positive_number(value: object, label: str) -> float: + if ( + not isinstance(value, (int, float)) + or isinstance(value, bool) + or not math.isfinite(float(value)) + or float(value) <= 0.0 + ): + raise SemanticSlamReplayError(f"{label} must be a positive finite number") + return float(value) + + +def _nested_total(metrics: dict[str, object], key: str) -> int: + return _nonnegative_integer(_object(metrics.get(key), f"{key} metrics").get("total"), key) + + +def _require_false_authority(value: object, label: str) -> None: + authority = _object(value, label) + if any(authority.get(key) is not False for key in _FALSE_AUTHORITY_KEYS if key in authority): + raise SemanticSlamReplayError(f"{label} acquired forbidden authority") + required = {"ground_truth", "physical_live", "commands_enabled", "actuation_allowed"} + if not required.issubset(authority): + raise SemanticSlamReplayError(f"{label} is incomplete") + + +__all__ = [ + "DEFAULT_SEMANTIC_SLAM_PROFILE_PATH", + "SEMANTIC_SLAM_FRAME_SCHEMA", + "SEMANTIC_SLAM_MASKS_NAME", + "SEMANTIC_SLAM_OBSERVATIONS_NAME", + "SEMANTIC_SLAM_POINTS_NAME", + "SEMANTIC_SLAM_REPORT_NAME", + "SEMANTIC_SLAM_REPORT_SCHEMA", + "SEMANTIC_SLAM_RESULT_PREFIX", + "SEMANTIC_SLAM_SCHEMA", + "SEMANTIC_SLAM_TAXONOMY_NAME", + "SEMANTIC_SLAM_TAXONOMY_SCHEMA", + "SemanticSlamReplayError", + "SemanticSlamReplayResult", + "build_semantic_slam_replay", + "read_semantic_slam_replay_result", +] diff --git a/src/k1link/web/app.py b/src/k1link/web/app.py index 3812397..a14596f 100644 --- a/src/k1link/web/app.py +++ b/src/k1link/web/app.py @@ -74,6 +74,7 @@ from k1link.web.e46i_grounding_dino_full_replay_api import ( from k1link.web.e46j_raw_fisheye_realtime_api import ( build_e46j_raw_fisheye_realtime_router, ) +from k1link.web.e47_semantic_slam_api import build_e47_semantic_slam_router from k1link.web.environment_api import build_environment_router from k1link.web.l3_pointpillars_visual_api import ( build_l3_pointpillars_visual_router, @@ -777,6 +778,17 @@ app.include_router( ), ) ) +app.include_router( + build_e47_semantic_slam_router( + root_provider=lambda: ( + REPOSITORY_ROOT + / ".runtime" + / "compute-experiments" + / "e47" + / "semantic-slam-results" + ), + ) +) app.include_router( build_e46e_ready_stack_router( root_provider=lambda: ( diff --git a/src/k1link/web/e47_semantic_slam_api.py b/src/k1link/web/e47_semantic_slam_api.py new file mode 100644 index 0000000..459f107 --- /dev/null +++ b/src/k1link/web/e47_semantic_slam_api.py @@ -0,0 +1,556 @@ +"""Read-only LAB projection of the immutable E47 semantic/SLAM shadow result.""" + +from __future__ import annotations + +import copy +import hashlib +import json +import re +import zipfile +from collections.abc import Callable +from dataclasses import dataclass +from functools import lru_cache +from pathlib import Path +from typing import Final + +import numpy as np +import numpy.typing as npt +from fastapi import APIRouter, HTTPException, Query, Response + +from k1link.perception.semantic_fusion import SemanticEvidenceStatus +from k1link.perception.semantic_slam_replay import ( + PUBLICATION_STATUS, + SEMANTIC_SLAM_MANIFEST_NAME, + SEMANTIC_SLAM_MASKS_NAME, + SEMANTIC_SLAM_OBSERVATIONS_NAME, + SEMANTIC_SLAM_POINTS_NAME, + SEMANTIC_SLAM_REPORT_NAME, + SEMANTIC_SLAM_RESULT_PREFIX, + SEMANTIC_SLAM_TAXONOMY_NAME, + SEMANTIC_SLAM_TAXONOMY_SCHEMA, + SemanticSlamReplayError, + SemanticSlamReplayResult, + read_semantic_slam_replay_result, +) + +E47_SEMANTIC_SLAM_CATALOG_SCHEMA: Final = "missioncore.e47-semantic-slam-catalog/v1" +E47_SEMANTIC_SLAM_VIEW_SCHEMA: Final = "missioncore.e47-semantic-slam-view/v1" +E47_SEMANTIC_SLAM_CHUNK_SCHEMA: Final = "missioncore.e47-semantic-slam-chunk/v1" +E47_SEMANTIC_SLAM_FRAME_SCHEMA: Final = "missioncore.e47-semantic-slam-frame/v1" +E47_SEMANTIC_SLAM_VIEW_STATUS: Final = "diagnostic-semantic-slam-shadow" +E47_SEMANTIC_SLAM_MAX_CHUNK_FRAMES: Final = 24 + +_RESULT_ID = re.compile(rf"^{SEMANTIC_SLAM_RESULT_PREFIX}[a-f0-9]{{64}}$") +_EXPECTED_ARTIFACTS: Final = ( + SEMANTIC_SLAM_MANIFEST_NAME, + SEMANTIC_SLAM_REPORT_NAME, + SEMANTIC_SLAM_POINTS_NAME, + SEMANTIC_SLAM_OBSERVATIONS_NAME, + SEMANTIC_SLAM_MASKS_NAME, + SEMANTIC_SLAM_TAXONOMY_NAME, +) +_MAX_TAXONOMY_BYTES: Final = 1024 * 1024 +_MAX_MASK_BYTES: Final = 16 * 1024 * 1024 + +RootProvider = Callable[[], Path | None] +Int64Array = npt.NDArray[np.int64] +Int32Array = npt.NDArray[np.int32] +UInt8Array = npt.NDArray[np.uint8] + + +@dataclass(frozen=True, slots=True) +class _SemanticPointLedger: + frame_offsets: Int64Array + point_labels: UInt8Array + point_status_codes: UInt8Array + frame_source_point_counts: Int32Array + frame_labeled_point_counts: Int32Array + frame_ambiguous_point_counts: Int32Array + frame_unprojected_point_counts: Int32Array + frame_absent_point_counts: Int32Array + + @property + def frame_count(self) -> int: + return int(self.frame_offsets.size - 1) + + +def build_e47_semantic_slam_router( + *, + root_provider: RootProvider = lambda: None, +) -> APIRouter: + """Expose immutable E47 evidence without granting it safety authority.""" + + router = APIRouter( + prefix="/api/v1/laboratory/e47-semantic-slam", + tags=["laboratory"], + ) + + def result(result_id: str) -> SemanticSlamReplayResult: + if _RESULT_ID.fullmatch(result_id) is None: + raise HTTPException(status_code=404, detail="E47 result не найден") + root = _configured_root(root_provider) + if root is None: + raise HTTPException(status_code=404, detail="E47 result не найден") + candidate = root / result_id + if candidate.is_symlink(): + raise HTTPException(status_code=404, detail="E47 result не найден") + try: + path = candidate.resolve(strict=True) + except OSError: + raise HTTPException(status_code=404, detail="E47 result не найден") from None + if path.parent != root or not path.is_dir(): + raise HTTPException(status_code=404, detail="E47 result не найден") + try: + return _read_semantic_result_cached(str(path), _result_signature(path)) + except (SemanticSlamReplayError, OSError, ValueError, zipfile.BadZipFile): + raise HTTPException(status_code=404, detail="E47 result не найден") from None + + def point_ledger(result_id: str) -> tuple[SemanticSlamReplayResult, _SemanticPointLedger]: + frozen = result(result_id) + try: + signature = _result_signature(frozen.result_root) + return frozen, _read_point_ledger_cached(str(frozen.result_root), signature) + except (SemanticSlamReplayError, OSError, ValueError, zipfile.BadZipFile): + raise HTTPException( + status_code=503, + detail="E47 semantic timeline не прошёл проверку", + ) from None + + @router.get("/results") + def list_results(limit: int = Query(default=1, ge=1, le=10)) -> dict[str, object]: + candidates = _candidates(root_provider) + items: list[dict[str, object]] = [] + invalid_total = 0 + for candidate in candidates: + if len(items) >= limit: + break + try: + items.append(_project_result(result(candidate.name))) + except (HTTPException, OSError, ValueError, json.JSONDecodeError): + invalid_total += 1 + return { + "schema_version": E47_SEMANTIC_SLAM_CATALOG_SCHEMA, + "configured": _configured_root(root_provider) is not None, + "items": items, + "candidate_total": len(candidates), + "invalid_total": invalid_total, + "access": "read-only-diagnostic-shadow", + } + + @router.get("/results/{result_id}/timeline/chunk") + def get_timeline_chunk( + result_id: str, + start: int = Query(default=0, ge=0), + count: int = Query( + default=12, + ge=1, + le=E47_SEMANTIC_SLAM_MAX_CHUNK_FRAMES, + ), + ) -> dict[str, object]: + frozen, ledger = point_ledger(result_id) + if ( + not isinstance(start, int) + or isinstance(start, bool) + or not isinstance(count, int) + or isinstance(count, bool) + or start < 0 + or not 1 <= count <= E47_SEMANTIC_SLAM_MAX_CHUNK_FRAMES + ): + raise HTTPException(status_code=422, detail="Некорректный E47 timeline chunk") + if start >= ledger.frame_count: + raise HTTPException(status_code=404, detail="E47 timeline chunk не найден") + stop = min(start + count, ledger.frame_count) + try: + frames = [_project_frame(ledger, sequence) for sequence in range(start, stop)] + except ValueError: + raise HTTPException( + status_code=503, + detail="E47 semantic timeline не прошёл проверку", + ) from None + return { + "schema_version": E47_SEMANTIC_SLAM_CHUNK_SCHEMA, + "result_id": frozen.result_id, + "start_sequence": start, + "frame_count": len(frames), + "next_sequence": stop if stop < ledger.frame_count else None, + "frames": frames, + "access": "read-only-diagnostic-shadow", + } + + @router.get("/results/{result_id}/masks/{sequence}") + def get_mask(result_id: str, sequence: int) -> Response: + frozen = result(result_id) + frame_total = _frame_total(frozen) + if ( + not isinstance(sequence, int) + or isinstance(sequence, bool) + or not 0 <= sequence < frame_total + ): + raise HTTPException(status_code=404, detail="E47 semantic mask не найдена") + try: + signature = _result_signature(frozen.result_root) + frozen = _read_semantic_result_cached(str(frozen.result_root), signature) + payload = _read_mask(frozen, sequence) + if _result_signature(frozen.result_root) != signature: + raise ValueError("E47 result changed during mask read") + except ( + SemanticSlamReplayError, + OSError, + KeyError, + ValueError, + RuntimeError, + zipfile.BadZipFile, + ): + raise HTTPException( + status_code=503, + detail="E47 semantic mask не прошла проверку", + ) from None + digest = hashlib.sha256(payload).hexdigest() + return Response( + content=payload, + media_type="image/png", + headers={ + "Cache-Control": "private, max-age=31536000, immutable", + "ETag": f'"{digest}"', + "X-Content-Type-Options": "nosniff", + }, + ) + + return router + + +@lru_cache(maxsize=4) +def _read_semantic_result_cached( + root_value: str, + signature: tuple[int, ...], +) -> SemanticSlamReplayResult: + del signature + return read_semantic_slam_replay_result(Path(root_value)) + + +@lru_cache(maxsize=2) +def _read_point_ledger_cached( + root_value: str, + signature: tuple[int, ...], +) -> _SemanticPointLedger: + frozen = _read_semantic_result_cached(root_value, signature) + path = frozen.result_root / SEMANTIC_SLAM_POINTS_NAME + required = { + "frame_offsets", + "point_labels", + "point_status_codes", + "frame_source_point_counts", + "frame_labeled_point_counts", + "frame_ambiguous_point_counts", + "frame_unprojected_point_counts", + "frame_absent_point_counts", + } + with np.load(path, allow_pickle=False) as archive: + if not required.issubset(archive.files): + raise ValueError("E47 semantic point arrays are incomplete") + ledger = _SemanticPointLedger( + frame_offsets=_frozen_int64(archive["frame_offsets"]), + point_labels=_frozen_uint8(archive["point_labels"]), + point_status_codes=_frozen_uint8(archive["point_status_codes"]), + frame_source_point_counts=_frozen_int32(archive["frame_source_point_counts"]), + frame_labeled_point_counts=_frozen_int32(archive["frame_labeled_point_counts"]), + frame_ambiguous_point_counts=_frozen_int32(archive["frame_ambiguous_point_counts"]), + frame_unprojected_point_counts=_frozen_int32(archive["frame_unprojected_point_counts"]), + frame_absent_point_counts=_frozen_int32(archive["frame_absent_point_counts"]), + ) + _validate_point_ledger(ledger, _frame_total(frozen)) + _validate_class_status_bindings(ledger, _read_taxonomy(frozen)) + return ledger + + +def _project_result(result: SemanticSlamReplayResult) -> dict[str, object]: + if result.status != PUBLICATION_STATUS: + raise ValueError("E47 publication status changed") + identity = _object(result.manifest.get("identity"), "E47 identity") + provider = _object(identity.get("semantic_provider"), "E47 provider") + temporal_binding = _object( + identity.get("temporal_binding"), + "E47 temporal binding", + ) + authority = _object(identity.get("authority"), "E47 authority") + if ( + authority.get("ground_truth") is not False + or authority.get("semantic_authority") != "diagnostic-only" + or authority.get("navigation_or_safety_accepted") is not False + or authority.get("actuation_allowed") is not False + ): + raise ValueError("E47 authority changed") + taxonomy = _read_taxonomy(result) + return { + "schema_version": E47_SEMANTIC_SLAM_VIEW_SCHEMA, + "result_id": result.result_id, + "created_at_utc": result.manifest["created_at_utc"], + "status": E47_SEMANTIC_SLAM_VIEW_STATUS, + "profile_id": identity["profile_id"], + "base_m4_result_id": identity["base_m4_result_id"], + "semantic_result_id": identity["semantic_result_id"], + "geometry_result_id": identity["geometry_result_id"], + "source_pack_id": identity["source_pack_id"], + "calibration_content_sha256": identity["calibration_content_sha256"], + "provider": { + "provider_id": provider["provider_id"], + "model_id": provider["model_id"], + "model_revision": provider["model_revision"], + "model_weights_sha256": provider["model_weights_sha256"], + "preprocess_id": provider["preprocess_id"], + }, + "temporal_binding": copy.deepcopy(temporal_binding), + "taxonomy": taxonomy, + "metrics": copy.deepcopy(result.metrics), + "acceptance": { + "artifact_contract_passed": True, + "frame_accounting_passed": True, + "point_accounting_passed": True, + "observation_binding_passed": True, + "temporal_binding_passed": True, + "independent_semantic_truth_passed": False, + "provider_promoted": False, + }, + "limitations": copy.deepcopy(result.report["limitations"]), + "ground_truth": False, + "semantic_authority": "diagnostic-only", + "navigation_or_safety_accepted": False, + "actuation_allowed": False, + "access": "read-only-diagnostic-shadow", + } + + +def _project_frame(ledger: _SemanticPointLedger, sequence: int) -> dict[str, object]: + offset = int(ledger.frame_offsets[sequence]) + stop = int(ledger.frame_offsets[sequence + 1]) + labels = ledger.point_labels[offset:stop].astype(np.int16) + statuses = ledger.point_status_codes[offset:stop] + unavailable = np.isin( + statuses, + ( + int(SemanticEvidenceStatus.ABSENT), + int(SemanticEvidenceStatus.UNPROJECTED), + ), + ) + labels[unavailable] = -1 + counts = { + "labeled": int(ledger.frame_labeled_point_counts[sequence]), + "ambiguous": int(ledger.frame_ambiguous_point_counts[sequence]), + "unprojected": int(ledger.frame_unprojected_point_counts[sequence]), + "absent": int(ledger.frame_absent_point_counts[sequence]), + } + actual_counts = { + "labeled": int(np.count_nonzero(statuses == int(SemanticEvidenceStatus.LABELED))), + "ambiguous": int(np.count_nonzero(statuses == int(SemanticEvidenceStatus.AMBIGUOUS))), + "unprojected": int(np.count_nonzero(statuses == int(SemanticEvidenceStatus.UNPROJECTED))), + "absent": int(np.count_nonzero(statuses == int(SemanticEvidenceStatus.ABSENT))), + } + if counts != actual_counts or sum(counts.values()) != stop - offset: + raise ValueError("E47 frame point accounting changed") + return { + "schema_version": E47_SEMANTIC_SLAM_FRAME_SCHEMA, + "sequence": sequence, + "source_point_count": int(ledger.frame_source_point_counts[sequence]), + "class_ids": labels.tolist(), + "status_codes": statuses.tolist(), + "counts": counts, + } + + +def _read_taxonomy(result: SemanticSlamReplayResult) -> list[dict[str, object]]: + path = result.result_root / SEMANTIC_SLAM_TAXONOMY_NAME + if not path.is_file() or path.is_symlink() or path.stat().st_size > _MAX_TAXONOMY_BYTES: + raise ValueError("E47 taxonomy is invalid") + payload = path.read_bytes() + identity = _object(result.manifest.get("identity"), "E47 identity") + if hashlib.sha256(payload).hexdigest() != identity.get("taxonomy_sha256"): + raise ValueError("E47 taxonomy identity changed") + document = json.loads(payload) + if not isinstance(document, dict) or set(document) != {"schema_version", "classes"}: + raise ValueError("E47 taxonomy contract changed") + if document.get("schema_version") != SEMANTIC_SLAM_TAXONOMY_SCHEMA: + raise ValueError("E47 taxonomy schema changed") + classes = document.get("classes") + if not isinstance(classes, list) or not classes: + raise ValueError("E47 taxonomy classes are invalid") + for item in classes: + if not isinstance(item, dict) or set(item) != { + "class_id", + "label", + "disposition", + "color_rgb", + }: + raise ValueError("E47 taxonomy class changed") + return copy.deepcopy(classes) + + +def _read_mask(result: SemanticSlamReplayResult, sequence: int) -> bytes: + archive_path = result.result_root / SEMANTIC_SLAM_MASKS_NAME + if not archive_path.is_file() or archive_path.is_symlink(): + raise ValueError("E47 semantic mask archive is invalid") + member_name = f"semantic-masks/frame-{sequence + 1:06d}.png" + with zipfile.ZipFile(archive_path, mode="r") as archive: + info = archive.getinfo(member_name) + if info.is_dir() or not 0 < info.file_size <= _MAX_MASK_BYTES: + raise ValueError("E47 semantic mask member is invalid") + payload = archive.read(info) + if len(payload) != info.file_size or not payload.startswith(b"\x89PNG\r\n\x1a\n"): + raise ValueError("E47 semantic mask payload is invalid") + return payload + + +def _validate_point_ledger(ledger: _SemanticPointLedger, frame_total: int) -> None: + arrays = ( + ledger.frame_source_point_counts, + ledger.frame_labeled_point_counts, + ledger.frame_ambiguous_point_counts, + ledger.frame_unprojected_point_counts, + ledger.frame_absent_point_counts, + ) + if ( + ledger.frame_offsets.ndim != 1 + or ledger.frame_offsets.shape != (frame_total + 1,) + or int(ledger.frame_offsets[0]) != 0 + or np.any(np.diff(ledger.frame_offsets) < 0) + or ledger.point_labels.ndim != 1 + or ledger.point_status_codes.shape != ledger.point_labels.shape + or int(ledger.frame_offsets[-1]) != ledger.point_labels.size + or any(value.ndim != 1 or value.shape != (frame_total,) for value in arrays) + or np.any(np.asarray(arrays) < 0) + or not np.array_equal( + np.diff(ledger.frame_offsets), + ledger.frame_source_point_counts, + ) + ): + raise ValueError("E47 semantic point ledger changed") + valid_statuses = {int(status) for status in SemanticEvidenceStatus} + if set(int(value) for value in np.unique(ledger.point_status_codes)) - valid_statuses: + raise ValueError("E47 semantic status changed") + expected_total = ( + ledger.frame_labeled_point_counts + + ledger.frame_ambiguous_point_counts + + ledger.frame_unprojected_point_counts + + ledger.frame_absent_point_counts + ) + if not np.array_equal(expected_total, ledger.frame_source_point_counts): + raise ValueError("E47 semantic frame accounting changed") + + +def _validate_class_status_bindings( + ledger: _SemanticPointLedger, + taxonomy: list[dict[str, object]], +) -> None: + dispositions: dict[int, str] = {} + for item in taxonomy: + class_id = item.get("class_id") + disposition = item.get("disposition") + if ( + not isinstance(class_id, int) + or isinstance(class_id, bool) + or not 0 <= class_id <= 255 + or disposition not in {"labeled", "ambiguous"} + or class_id in dispositions + ): + raise ValueError("E47 semantic taxonomy binding changed") + dispositions[class_id] = str(disposition) + unavailable = np.isin( + ledger.point_status_codes, + ( + int(SemanticEvidenceStatus.ABSENT), + int(SemanticEvidenceStatus.UNPROJECTED), + ), + ) + if np.any(ledger.point_labels[unavailable] != 0): + raise ValueError("E47 unavailable semantic point carried a class") + for status, disposition in ( + (SemanticEvidenceStatus.AMBIGUOUS, "ambiguous"), + (SemanticEvidenceStatus.LABELED, "labeled"), + ): + class_ids = np.unique(ledger.point_labels[ledger.point_status_codes == int(status)]) + if any(dispositions.get(int(class_id)) != disposition for class_id in class_ids): + raise ValueError("E47 semantic point status disagrees with taxonomy") + + +def _frame_total(result: SemanticSlamReplayResult) -> int: + frames = _object(result.metrics.get("frames"), "E47 frame metrics") + value = frames.get("total") + if not isinstance(value, int) or isinstance(value, bool) or value < 1: + raise ValueError("E47 frame count changed") + return value + + +def _frozen_int64(value: npt.ArrayLike) -> Int64Array: + array = np.array(value, dtype=np.int64, order="C", copy=True) + array.setflags(write=False) + return array + + +def _frozen_int32(value: npt.ArrayLike) -> Int32Array: + array = np.array(value, dtype=np.int32, order="C", copy=True) + array.setflags(write=False) + return array + + +def _frozen_uint8(value: npt.ArrayLike) -> UInt8Array: + array = np.array(value, dtype=np.uint8, order="C", copy=True) + array.setflags(write=False) + return array + + +def _configured_root(provider: RootProvider) -> Path | None: + value = provider() + if value is None: + return None + candidate = value.expanduser().absolute() + if candidate.is_symlink(): + return None + try: + root = candidate.resolve(strict=True) + except OSError: + return None + return root if root.is_dir() else None + + +def _result_signature(root: Path) -> tuple[int, ...]: + signature: list[int] = [] + for name in _EXPECTED_ARTIFACTS: + path = root / name + if not path.is_file() or path.is_symlink(): + raise ValueError("E47 result artifact is invalid") + stat = path.stat() + signature.extend((stat.st_ino, stat.st_size, stat.st_mtime_ns, stat.st_ctime_ns)) + return tuple(signature) + + +def _candidates(provider: RootProvider) -> list[Path]: + root = _configured_root(provider) + if root is None: + return [] + try: + return sorted( + ( + item + for item in root.iterdir() + if item.is_dir() and not item.is_symlink() and _RESULT_ID.fullmatch(item.name) + ), + key=lambda item: item.stat().st_mtime_ns, + reverse=True, + ) + except OSError: + return [] + + +def _object(value: object, label: str) -> dict[str, object]: + if not isinstance(value, dict): + raise ValueError(f"{label} is invalid") + return value + + +__all__ = [ + "E47_SEMANTIC_SLAM_CATALOG_SCHEMA", + "E47_SEMANTIC_SLAM_CHUNK_SCHEMA", + "E47_SEMANTIC_SLAM_FRAME_SCHEMA", + "E47_SEMANTIC_SLAM_MAX_CHUNK_FRAMES", + "E47_SEMANTIC_SLAM_VIEW_SCHEMA", + "build_e47_semantic_slam_router", +] diff --git a/tests/test_e47_semantic_slam_api.py b/tests/test_e47_semantic_slam_api.py new file mode 100644 index 0000000..ba23165 --- /dev/null +++ b/tests/test_e47_semantic_slam_api.py @@ -0,0 +1,272 @@ +from __future__ import annotations + +import hashlib +import json +import zipfile +from pathlib import Path + +import numpy as np +import pytest +from fastapi import HTTPException +from fastapi.routing import APIRoute + +from k1link.perception.semantic_slam_replay import ( + PUBLICATION_STATUS, + SemanticSlamReplayResult, +) +from k1link.web import e47_semantic_slam_api as api + +RESULT_ID = f"e47-semantic-slam-{'a' * 64}" +M4_RESULT_ID = f"m4-threat-replay-{'b' * 64}" +PNG_0 = b"\x89PNG\r\n\x1a\nsealed-mask-zero" +PNG_1 = b"\x89PNG\r\n\x1a\nsealed-mask-one" + + +def _endpoint(path: str, root: Path): + router = api.build_e47_semantic_slam_router(root_provider=lambda: root) + return next( + route.endpoint + for route in router.routes + if isinstance(route, APIRoute) and route.path == path + ) + + +@pytest.fixture(autouse=True) +def _clear_api_caches() -> None: + api._read_semantic_result_cached.cache_clear() + api._read_point_ledger_cached.cache_clear() + yield + api._read_semantic_result_cached.cache_clear() + api._read_point_ledger_cached.cache_clear() + + +@pytest.fixture +def publication(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> tuple[Path, Path]: + root = tmp_path / "semantic-slam-results" + result_root = root / RESULT_ID + result_root.mkdir(parents=True) + + taxonomy = { + "schema_version": "missioncore.e47-semantic-taxonomy/v1", + "classes": [ + { + "class_id": 0, + "label": "ambiguous", + "disposition": "ambiguous", + "color_rgb": [0, 0, 0], + }, + { + "class_id": 1, + "label": "road", + "disposition": "labeled", + "color_rgb": [128, 64, 128], + }, + ], + } + taxonomy_payload = json.dumps(taxonomy, separators=(",", ":")).encode() + (result_root / "taxonomy.json").write_bytes(taxonomy_payload) + np.savez( + result_root / "semantic-points.npz", + frame_offsets=np.asarray([0, 4, 7], dtype=np.int64), + point_labels=np.asarray([1, 0, 0, 0, 0, 1, 1], dtype=np.uint8), + point_status_codes=np.asarray([3, 2, 1, 0, 2, 3, 3], dtype=np.uint8), + frame_source_point_counts=np.asarray([4, 3], dtype=np.int32), + frame_labeled_point_counts=np.asarray([1, 2], dtype=np.int32), + frame_ambiguous_point_counts=np.asarray([1, 1], dtype=np.int32), + frame_unprojected_point_counts=np.asarray([1, 0], dtype=np.int32), + frame_absent_point_counts=np.asarray([1, 0], dtype=np.int32), + ) + with zipfile.ZipFile(result_root / "semantic-masks.zip", mode="w") as archive: + archive.writestr("semantic-masks/frame-000001.png", PNG_0) + archive.writestr("semantic-masks/frame-000002.png", PNG_1) + for name, payload in ( + ("manifest.json", b"fixture-manifest"), + ("report.json", b"fixture-report"), + ("semantic-observations.jsonl", b"fixture-observations\n"), + ): + (result_root / name).write_bytes(payload) + + metrics = { + "frames": {"total": 2, "mask_available": 2, "source_available": 2}, + "points": { + "total": 7, + "projected": 5, + "labeled": 3, + "ambiguous": 2, + "unprojected": 1, + "absent": 1, + }, + "observations": { + "total": 3, + "labeled": 1, + "ambiguous": 1, + "unprojected": 1, + "absent": 0, + }, + "runtime": {"elapsed_ms": 10.0, "frames_per_second": 200.0}, + } + identity = { + "profile_id": "ravnoves00-eomt-kb4-slam-shadow/v1", + "base_m4_result_id": M4_RESULT_ID, + "semantic_result_id": f"result-{'c' * 64}", + "geometry_result_id": f"m4-geometry-replay-{'d' * 64}", + "source_pack_id": "ravnoves00-source-pack/v1", + "calibration_content_sha256": "e" * 64, + "taxonomy_sha256": hashlib.sha256(taxonomy_payload).hexdigest(), + "semantic_provider": { + "provider_id": "eomt-cityscapes-semantic-control/v1", + "model_id": "tue-mps/eomt", + "model_revision": "f" * 40, + "model_weights_sha256": "1" * 64, + "preprocess_id": "raw-kb4-valid-fov-semantic/v1", + "role": "fixed-control-not-selected-production-provider", + }, + "temporal_binding": { + "semantic_to_camera": "exact-sequence-and-session-time", + "camera_to_lidar": "accepted-e6-nearest-host-arrival-best-effort", + "clock_basis": "recorded-host-monotonic-arrival", + "maximum_lidar_camera_delta_ms": 100.0, + "maximum_pose_point_delta_ms": 100.0, + "physical_synchronization_proven": False, + }, + "authority": { + "ground_truth": False, + "semantic_authority": "diagnostic-only", + "navigation_or_safety_accepted": False, + "actuation_allowed": False, + }, + } + frozen = SemanticSlamReplayResult( + result_id=RESULT_ID, + result_root=result_root, + status=PUBLICATION_STATUS, + metrics=metrics, + report={ + "status": PUBLICATION_STATUS, + "metrics": metrics, + "limitations": ["No independent semantic truth."], + }, + manifest={ + "result_id": RESULT_ID, + "created_at_utc": "2026-08-06T06:30:00.000Z", + "identity": identity, + }, + ) + monkeypatch.setattr(api, "_read_semantic_result_cached", lambda *_: frozen) + return root, result_root + + +def test_catalog_projects_exact_diagnostic_only_view( + publication: tuple[Path, Path], +) -> None: + root, _ = publication + list_results = _endpoint("/api/v1/laboratory/e47-semantic-slam/results", root) + + catalog = list_results(limit=1) + + assert catalog["schema_version"] == "missioncore.e47-semantic-slam-catalog/v1" + assert catalog["candidate_total"] == 1 + assert catalog["invalid_total"] == 0 + item = catalog["items"][0] + assert item["schema_version"] == "missioncore.e47-semantic-slam-view/v1" + assert item["result_id"] == RESULT_ID + assert item["status"] == "diagnostic-semantic-slam-shadow" + assert item["base_m4_result_id"] == M4_RESULT_ID + assert item["provider"]["provider_id"] == "eomt-cityscapes-semantic-control/v1" + assert item["temporal_binding"]["semantic_to_camera"] == ( + "exact-sequence-and-session-time" + ) + assert item["temporal_binding"]["physical_synchronization_proven"] is False + assert [entry["label"] for entry in item["taxonomy"]] == ["ambiguous", "road"] + assert item["acceptance"] == { + "artifact_contract_passed": True, + "frame_accounting_passed": True, + "point_accounting_passed": True, + "observation_binding_passed": True, + "temporal_binding_passed": True, + "independent_semantic_truth_passed": False, + "provider_promoted": False, + } + assert item["semantic_authority"] == "diagnostic-only" + assert item["navigation_or_safety_accepted"] is False + assert item["actuation_allowed"] is False + + +def test_timeline_chunk_preserves_point_index_space_and_unavailable_sentinel( + publication: tuple[Path, Path], +) -> None: + root, _ = publication + get_chunk = _endpoint( + "/api/v1/laboratory/e47-semantic-slam/results/{result_id}/timeline/chunk", + root, + ) + + chunk = get_chunk(RESULT_ID, start=0, count=2) + + assert chunk["schema_version"] == "missioncore.e47-semantic-slam-chunk/v1" + assert chunk["frame_count"] == 2 + assert chunk["next_sequence"] is None + first = chunk["frames"][0] + assert first == { + "schema_version": "missioncore.e47-semantic-slam-frame/v1", + "sequence": 0, + "source_point_count": 4, + "class_ids": [1, 0, -1, -1], + "status_codes": [3, 2, 1, 0], + "counts": {"labeled": 1, "ambiguous": 1, "unprojected": 1, "absent": 1}, + } + assert chunk["frames"][1]["class_ids"] == [0, 1, 1] + + +def test_timeline_chunk_rejects_out_of_range_and_oversized_requests( + publication: tuple[Path, Path], +) -> None: + root, _ = publication + get_chunk = _endpoint( + "/api/v1/laboratory/e47-semantic-slam/results/{result_id}/timeline/chunk", + root, + ) + + with pytest.raises(HTTPException) as out_of_range: + get_chunk(RESULT_ID, start=2, count=1) + assert out_of_range.value.status_code == 404 + with pytest.raises(HTTPException) as oversized: + get_chunk(RESULT_ID, start=0, count=25) + assert oversized.value.status_code == 422 + + +def test_mask_endpoint_streams_exact_png_with_immutable_identity( + publication: tuple[Path, Path], +) -> None: + root, _ = publication + get_mask = _endpoint( + "/api/v1/laboratory/e47-semantic-slam/results/{result_id}/masks/{sequence}", + root, + ) + + response = get_mask(RESULT_ID, 1) + + assert response.body == PNG_1 + assert response.media_type == "image/png" + assert response.headers["cache-control"] == "private, max-age=31536000, immutable" + assert response.headers["etag"] == f'"{hashlib.sha256(PNG_1).hexdigest()}"' + assert response.headers["x-content-type-options"] == "nosniff" + + +def test_result_resolution_rejects_invalid_id_and_symlink(tmp_path: Path) -> None: + root = tmp_path / "results" + root.mkdir() + outside = tmp_path / RESULT_ID + outside.mkdir() + (root / RESULT_ID).symlink_to(outside, target_is_directory=True) + get_chunk = _endpoint( + "/api/v1/laboratory/e47-semantic-slam/results/{result_id}/timeline/chunk", + root, + ) + + with pytest.raises(HTTPException) as invalid: + get_chunk("../escape", start=0, count=1) + assert invalid.value.status_code == 404 + with pytest.raises(HTTPException) as linked: + get_chunk(RESULT_ID, start=0, count=1) + assert linked.value.status_code == 404 diff --git a/tests/test_laboratory_evidence_registry.py b/tests/test_laboratory_evidence_registry.py index 1692c3c..0f93a6e 100644 --- a/tests/test_laboratory_evidence_registry.py +++ b/tests/test_laboratory_evidence_registry.py @@ -127,10 +127,11 @@ def test_product_registry_declares_every_advanced_evidence_source() -> None: repository_root / "config" / "laboratories" ) - assert len(registry.definitions) == 32 + assert len(registry.definitions) == 33 assert {item.work_id for item in registry.definitions} >= { "e31-source-binding", "e46j-raw-fisheye-realtime", + "e47-semantic-slam-shadow", "l3-pointpillars-visual-audit", "l31-pointpillars-ravnoves", "l32-pointpillars-camera-review", diff --git a/tests/test_laboratory_execution.py b/tests/test_laboratory_execution.py index 1e40a52..286d322 100644 --- a/tests/test_laboratory_execution.py +++ b/tests/test_laboratory_execution.py @@ -94,8 +94,16 @@ def test_repository_registry_classifies_every_evidence_definition() -> None: "e33-worker-shadow", "e35-degradation-recovery", "e46j-raw-fisheye-realtime", + "e47-semantic-slam-shadow", } - assert all(row.lifecycle == "canonical" for row in execution.definitions) + by_work_id = {row.work_id: row for row in execution.definitions} + assert by_work_id["e47-semantic-slam-shadow"].lifecycle == "experimental" + assert by_work_id["e47-semantic-slam-shadow"].isolation == "bounded-adapter" + assert all( + row.lifecycle == "canonical" + for row in execution.definitions + if row.work_id != "e47-semantic-slam-shadow" + ) assert len(execution.definitions) + len(execution.legacy_work_ids) == len( evidence.definitions ) diff --git a/tests/test_semantic_fusion.py b/tests/test_semantic_fusion.py new file mode 100644 index 0000000..28057d2 --- /dev/null +++ b/tests/test_semantic_fusion.py @@ -0,0 +1,306 @@ +from __future__ import annotations + +from dataclasses import replace + +import numpy as np +import pytest + +from k1link.perception.contracts import ( + EvidenceBasis, + EvidenceCurrentness, + MetricGeometry, + ObstacleObservation, +) +from k1link.perception.geometry_math import ProjectedPointCloud +from k1link.perception.semantic_fusion import ( + NO_SEMANTIC_CLASS_ID, + SemanticClassDefinition, + SemanticClassDisposition, + SemanticEvidenceAuthority, + SemanticEvidenceStatus, + SemanticFusionError, + SemanticMask, + fuse_semantic_diagnostics, +) + + +def _classes() -> tuple[SemanticClassDefinition, ...]: + return ( + SemanticClassDefinition(1, "road"), + SemanticClassDefinition(2, "car"), + SemanticClassDefinition( + 255, + "void / uncertain", + SemanticClassDisposition.AMBIGUOUS, + ), + ) + + +def _mask(*, source_id: str = "RAVNOVES00", frame_id: str = "frame-000014") -> SemanticMask: + return SemanticMask( + source_id=source_id, + frame_id=frame_id, + provider_id="semantic-provider/v1", + model_id="semantic-model/v1", + preprocess_id="raw-kb4-semantic/v1", + labels=np.asarray( + [ + [1, 2, 255, 1], + [1, 1, 1, 1], + [1, 1, 1, 1], + ], + dtype=np.uint8, + ), + classes=_classes(), + ) + + +def _projection() -> ProjectedPointCloud: + return ProjectedPointCloud( + pixels_xy=np.asarray( + [ + [0.1, 0.1], + [1.2, 0.2], + [1.8, 0.8], + [2.1, 0.2], + [9.0, 9.0], + ], + dtype=np.float64, + ), + depths_m=np.asarray([2.0, 2.1, 2.2, 2.3, 2.4], dtype=np.float64), + source_indices=np.asarray([0, 1, 2, 3, 4], dtype=np.int64), + source_point_count=5, + camera_front_point_count=5, + ) + + +def _geometry_observation( + *point_ids: int, + observation_id: str = "geometry-observation-1", +) -> ObstacleObservation: + return ObstacleObservation( + observation_id=observation_id, + occupancy_key=f"occupancy-{observation_id}", + source_id="RAVNOVES00", + frame_id="frame-000014", + evidence_time_ns=14_000_000_000, + basis=EvidenceBasis.LIDAR, + currentness=EvidenceCurrentness.CURRENT, + occupied_support=True, + source_point_ids=point_ids, + metric_geometry=MetricGeometry( + coordinate_frame="map", + centroid_xyz_m=(2.0, 0.0, 0.5), + range_m=2.0, + covariance_diagonal_m2=(0.1, 0.1, 0.1), + ), + proposal_ids=(), + semantic_hint=None, + reason_codes=("qualified-lidar-points",), + ) + + +def _camera_only_observation() -> ObstacleObservation: + return ObstacleObservation( + observation_id="camera-observation-1", + occupancy_key="occupancy-camera-observation-1", + source_id="RAVNOVES00", + frame_id="frame-000014", + evidence_time_ns=14_000_000_000, + basis=EvidenceBasis.CAMERA, + currentness=EvidenceCurrentness.CURRENT, + occupied_support=False, + source_point_ids=(), + metric_geometry=None, + proposal_ids=("proposal-1",), + semantic_hint=None, + reason_codes=("camera-only",), + ) + + +def test_semantic_mask_is_strict_source_bound_uint8_and_immutable() -> None: + labels = np.asarray([[1, 2]], dtype=np.uint8) + semantic = SemanticMask( + source_id="RAVNOVES00", + frame_id="frame-000014", + provider_id="semantic-provider/v1", + model_id="semantic-model/v1", + preprocess_id="raw-kb4-semantic/v1", + labels=labels, + classes=_classes(), + ) + labels[0, 0] = 2 + assert semantic.labels.tolist() == [[1, 2]] + assert semantic.labels.flags.writeable is False + with pytest.raises(ValueError): + semantic.labels[0, 0] = 2 + + with pytest.raises(SemanticFusionError, match="uint8 HxW"): + replace(semantic, labels=np.asarray([[1, 2]], dtype=np.int64)) + with pytest.raises(SemanticFusionError, match="undeclared"): + replace(semantic, labels=np.asarray([[1, 7]], dtype=np.uint8)) + with pytest.raises(SemanticFusionError, match="unique"): + replace( + semantic, + classes=(SemanticClassDefinition(1, "road"), SemanticClassDefinition(1, "other")), + ) + + +def test_mask_projection_keeps_absence_ambiguity_and_unprojected_separate() -> None: + result = fuse_semantic_diagnostics( + semantic_mask=_mask(), + projected=_projection(), + observations=(), + ) + labels = result.point_labels + assert [labels.status_for(index) for index in range(5)] == [ + SemanticEvidenceStatus.LABELED, + SemanticEvidenceStatus.LABELED, + SemanticEvidenceStatus.LABELED, + SemanticEvidenceStatus.AMBIGUOUS, + SemanticEvidenceStatus.UNPROJECTED, + ] + assert [labels.class_id_for(index) for index in range(5)] == [1, 2, 2, 255, None] + assert [labels.label_for(index) for index in range(5)] == [ + "road", + "car", + "car", + "void / uncertain", + None, + ] + assert labels.class_ids.tolist() == [1, 2, 2, 255, NO_SEMANTIC_CLASS_ID] + assert labels.class_ids.flags.writeable is False + assert labels.status_codes.flags.writeable is False + assert result.authority is SemanticEvidenceAuthority.DIAGNOSTIC_ONLY + + +def test_observation_aggregation_is_detached_from_geometry_and_safety_authority() -> None: + observation = _geometry_observation(0, 1, 2, 4) + before = observation.to_dict() + result = fuse_semantic_diagnostics( + semantic_mask=_mask(), + projected=_projection(), + observations=(observation,), + ) + evidence = result.observation_evidence[0] + assert observation.to_dict() == before + assert evidence.observation_id == observation.observation_id + assert evidence.occupancy_key == observation.occupancy_identity + assert evidence.status is SemanticEvidenceStatus.LABELED + assert evidence.dominant_class_id == 2 + assert evidence.dominant_label == "car" + assert evidence.dominant_fraction_of_labeled == pytest.approx(2 / 3) + assert evidence.labeled_point_count == 3 + assert evidence.unprojected_point_count == 1 + assert evidence.semantic_coverage_fraction == pytest.approx(0.75) + assert evidence.authority is SemanticEvidenceAuthority.DIAGNOSTIC_ONLY + assert not hasattr(evidence, "occupied_support") + assert not hasattr(evidence, "motion") + assert not hasattr(evidence, "threat") + assert not hasattr(evidence, "actuation_allowed") + + +def test_tied_or_provider_ambiguous_labels_remain_ambiguous() -> None: + tied = _geometry_observation(0, 1, observation_id="geometry-tied") + provider_ambiguous = _geometry_observation(3, observation_id="geometry-void") + result = fuse_semantic_diagnostics( + semantic_mask=_mask(), + projected=_projection(), + observations=(tied, provider_ambiguous), + ) + tie_evidence, void_evidence = result.observation_evidence + assert tie_evidence.status is SemanticEvidenceStatus.AMBIGUOUS + assert tie_evidence.reason_code == "semantic-label-majority-ambiguous" + assert tie_evidence.dominant_class_id is None + assert {item.label: item.point_count for item in tie_evidence.class_evidence} == { + "road": 1, + "car": 1, + } + assert void_evidence.status is SemanticEvidenceStatus.AMBIGUOUS + assert void_evidence.reason_code == "semantic-classes-ambiguous" + assert void_evidence.ambiguous_point_count == 1 + assert void_evidence.class_evidence[0].disposition is SemanticClassDisposition.AMBIGUOUS + + mixed = fuse_semantic_diagnostics( + semantic_mask=_mask(), + projected=_projection(), + observations=(_geometry_observation(1, 3, observation_id="geometry-mixed"),), + ).observation_evidence[0] + assert mixed.status is SemanticEvidenceStatus.AMBIGUOUS + assert mixed.reason_code == "semantic-label-majority-ambiguous" + assert mixed.dominant_class_id is None + + +def test_missing_mask_and_pointless_geometry_have_distinct_outcomes() -> None: + observation = _geometry_observation(0, 1) + absent = fuse_semantic_diagnostics( + semantic_mask=None, + projected=_projection(), + observations=(observation,), + ) + assert absent.mask_available is False + assert [absent.point_labels.status_for(index) for index in range(5)] == [ + SemanticEvidenceStatus.ABSENT + ] * 5 + assert absent.observation_evidence[0].status is SemanticEvidenceStatus.ABSENT + assert absent.observation_evidence[0].absent_point_count == 2 + + unprojected = fuse_semantic_diagnostics( + semantic_mask=_mask(), + projected=_projection(), + observations=(_camera_only_observation(),), + ) + evidence = unprojected.observation_evidence[0] + assert evidence.status is SemanticEvidenceStatus.UNPROJECTED + assert evidence.reason_code == "observation-has-no-source-points" + assert evidence.source_point_count == 0 + + +def test_fusion_rejects_frame_escape_invalid_point_ids_and_duplicate_ownership() -> None: + observation = _geometry_observation(0) + with pytest.raises(SemanticFusionError, match="source frame"): + fuse_semantic_diagnostics( + semantic_mask=_mask(frame_id="frame-000015"), + projected=_projection(), + observations=(observation,), + ) + with pytest.raises(SemanticFusionError, match="outside the source frame"): + fuse_semantic_diagnostics( + semantic_mask=_mask(), + projected=_projection(), + observations=(_geometry_observation(5),), + ) + with pytest.raises(SemanticFusionError, match="duplicate observation ownership"): + fuse_semantic_diagnostics( + semantic_mask=_mask(), + projected=_projection(), + observations=( + observation, + _geometry_observation(0, observation_id="geometry-observation-2"), + ), + ) + with pytest.raises(SemanticFusionError, match="escaped their source frame"): + fuse_semantic_diagnostics( + semantic_mask=None, + projected=_projection(), + observations=( + observation, + replace( + _geometry_observation(1, observation_id="geometry-observation-2"), + frame_id="frame-000015", + ), + ), + ) + + +def test_projection_validator_rejects_malformed_existing_contract_values() -> None: + malformed = replace( + _projection(), + source_indices=np.asarray([0, 1, 2, 3, 5], dtype=np.int64), + ) + with pytest.raises(SemanticFusionError, match="outside the source frame"): + fuse_semantic_diagnostics( + semantic_mask=_mask(), + projected=malformed, + observations=(), + ) diff --git a/tests/test_semantic_slam_replay.py b/tests/test_semantic_slam_replay.py new file mode 100644 index 0000000..f682266 --- /dev/null +++ b/tests/test_semantic_slam_replay.py @@ -0,0 +1,506 @@ +from __future__ import annotations + +import hashlib +import io +import json +import tarfile +from dataclasses import dataclass, replace +from pathlib import Path + +import numpy as np +import pytest +from PIL import Image + +import k1link.perception.semantic_slam_replay as replay +from k1link.perception.contracts import ( + EvidenceBasis, + EvidenceCurrentness, + MetricGeometry, + ObstacleObservation, +) +from k1link.perception.geometry import GeometryFrame, RecordedFrameTemporalBinding +from k1link.perception.geometry_math import Kb4ProjectionProfile +from k1link.perception.geometry_replay import GeometryReplayResult +from k1link.perception.semantic_fusion import ( + SemanticClassDisposition, + SemanticEvidenceStatus, +) +from k1link.perception.threat_replay import ThreatReplayResult + + +def _canonical_json(value: object) -> bytes: + return json.dumps( + value, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode("utf-8") + + +def _sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _write_jsonl(path: Path, rows: list[dict[str, object]]) -> None: + path.write_bytes(b"".join(_canonical_json(row) + b"\n" for row in rows)) + + +def _png(labels: np.ndarray) -> bytes: + buffer = io.BytesIO() + Image.fromarray(labels, mode="L").save(buffer, format="PNG") + return buffer.getvalue() + + +@dataclass(frozen=True) +class _Store: + frame: GeometryFrame + lidar_delta_ms: float = 4.0 + pose_delta_ms: float = 2.0 + + def frame_for_index(self, frame_index: int) -> GeometryFrame | None: + return self.frame if frame_index == 0 else None + + def temporal_binding_for_index(self, frame_index: int) -> RecordedFrameTemporalBinding: + return RecordedFrameTemporalBinding( + frame_index=frame_index, + source_time_ns=(frame_index + 1) * 1_000_000_000, + source_available=frame_index == 0, + lidar_camera_delta_ms=self.lidar_delta_ms if frame_index == 0 else None, + pose_point_delta_ms=self.pose_delta_ms if frame_index == 0 else None, + ) + + +def _observation() -> ObstacleObservation: + return ObstacleObservation( + observation_id="frame-000000:obstacle-0", + occupancy_key="frame-000000:obstacle-0", + source_id="RAVNOVES00", + frame_id="frame-000000", + evidence_time_ns=1, + basis=EvidenceBasis.FUSED, + currentness=EvidenceCurrentness.CURRENT, + occupied_support=True, + source_point_ids=(0, 1), + metric_geometry=MetricGeometry( + coordinate_frame="map", + centroid_xyz_m=(0.0, 0.0, 1.0), + range_m=1.0, + covariance_diagonal_m2=(0.0, 0.0, 0.0), + ), + proposal_ids=("proposal-0",), + semantic_hint="car", + reason_codes=("current-test-support",), + ) + + +def _fixture(tmp_path: Path) -> replay._AdmittedInputs: + source = tmp_path / "source" + source.mkdir() + profile_path = source / "profile.json" + profile_path.write_text('{"fixture":true}\n', encoding="utf-8") + authority = { + "ground_truth": False, + "physical_live": False, + "commands_enabled": False, + "actuation_allowed": False, + "navigation_or_safety_accepted": False, + "semantic_authority": "diagnostic-only", + } + fusion = { + "projection": "factory-kb4-current-increment/v1", + "point_index_space": "frame-local-source-point-id/v1", + "observation_aggregation": "dominant-labeled-majority-diagnostic/v1", + "unprojected_status": "unprojected", + "semantic_absence_means_free": False, + "semantic_can_create_obstacle": False, + "semantic_can_change_identity": False, + "semantic_can_change_metric_geometry": False, + "semantic_can_change_occupancy": False, + "semantic_can_change_motion": False, + "semantic_can_change_threat": False, + } + profile = replay._SemanticSlamProfile( + path=profile_path, + sha256=_sha256(profile_path), + profile_id="fixture-semantic-slam/v1", + source_id="RAVNOVES00", + session_id="fixture-session", + frame_count=2, + image_width=4, + image_height=4, + source_pack_id="fixture-source-pack", + source_pack_sha256="1" * 64, + calibration_sha256="2" * 64, + provider_id="fixture-semantic-provider/v1", + model_id="fixture-model", + model_revision="fixture-revision", + model_weights_sha256="3" * 64, + preprocess_id="fixture-preprocess/v1", + mask_metadata_schema_version="missioncore.panoptic-frame/v1", + mask_payload={ + "media_type": "image/png", + "encoding": "uint8-class-id", + "width": 4, + "height": 4, + "sequence_binding": "sequence-0-to-frame-000001", + }, + provider_role="fixed-control-not-selected-production-provider", + classes=( + replay._TaxonomyClass( + class_id=0, + label="outside_valid_fov", + disposition=SemanticClassDisposition.AMBIGUOUS, + color_rgb=(0, 0, 0), + ), + replay._TaxonomyClass( + class_id=4, + label="car", + disposition=SemanticClassDisposition.LABELED, + color_rgb=(0, 0, 142), + ), + ), + fusion=fusion, + temporal_binding={ + "semantic_to_camera": "exact-sequence-and-session-time", + "camera_to_lidar": "accepted-e6-nearest-host-arrival-best-effort", + "clock_basis": "recorded-host-monotonic-arrival", + "maximum_lidar_camera_delta_ms": 100.0, + "maximum_pose_point_delta_ms": 100.0, + "physical_synchronization_proven": False, + }, + acceptance={ + "full_frame_accounting_required": True, + "point_accounting_required": True, + "observation_binding_required": True, + "exact_mask_archive_required": True, + "independent_semantic_truth_required_for_provider_promotion": True, + }, + authority=authority, + ) + + semantic_root = source / "semantic" + semantic_root.mkdir() + result_json = semantic_root / "result.json" + result_json.write_text('{"sealed":"fixture"}\n', encoding="utf-8") + masks = [] + first = np.zeros((4, 4), dtype=np.uint8) + first[2, 2] = 4 + masks.append(_png(first)) + masks.append(_png(np.zeros((4, 4), dtype=np.uint8))) + mask_archive = semantic_root / "masks.tar.gz" + with tarfile.open(mask_archive, mode="w:gz") as archive: + directory = tarfile.TarInfo("semantic-masks") + directory.type = tarfile.DIRTYPE + archive.addfile(directory) + for sequence, payload in enumerate(masks, start=1): + member = tarfile.TarInfo(f"semantic-masks/frame-{sequence:06d}.png") + member.size = len(payload) + archive.addfile(member, io.BytesIO(payload)) + semantic_frames = semantic_root / "frames.jsonl" + _write_jsonl( + semantic_frames, + [ + { + "schema_version": "missioncore.panoptic-frame/v1", + "frame_index": 0, + "sequence": 1, + "session_seconds": 1.0, + "instances": [], + "semantic_classes": [ + { + "id": 4, + "label": "car", + "pixels": 1, + "fraction_of_valid_fov": 0.0625, + } + ], + }, + { + "schema_version": "missioncore.panoptic-frame/v1", + "frame_index": 1, + "sequence": 2, + "session_seconds": 2.0, + "instances": [], + "semantic_classes": [], + }, + ], + ) + semantic = replay._SemanticUpstream( + result_id="result-" + "4" * 64, + result_root=semantic_root, + result_manifest_sha256=_sha256(result_json), + frames_path=semantic_frames, + frames_sha256=_sha256(semantic_frames), + masks_path=mask_archive, + masks_sha256=_sha256(mask_archive), + created_at_utc="2026-08-06T00:00:00.000Z", + job_id="fixture-job", + input_sha256="5" * 64, + source_id="sensor.camera.right", + session_id="fixture-session", + calibration_sha256="2" * 64, + configuration_profile_sha256="6" * 64, + model_id="fixture-model", + model_revision="fixture-revision", + model_weights_sha256="3" * 64, + ) + + observation = _observation() + geometry_root = source / "geometry" + geometry_root.mkdir() + geometry_frames = geometry_root / "frames.jsonl" + _write_jsonl( + geometry_frames, + [ + { + "schema_version": "missioncore.perception-geometry-replay-frame/v1", + "sequence": 0, + "frame_id": "frame-000000", + "source_available": True, + "observations": [observation.to_dict()], + }, + { + "schema_version": "missioncore.perception-geometry-replay-frame/v1", + "sequence": 1, + "frame_id": "frame-000001", + "source_available": False, + "observations": [], + }, + ], + ) + geometry_sha256 = _sha256(geometry_frames) + geometry = GeometryReplayResult( + result_id="m4-geometry-replay-" + "7" * 64, + result_root=geometry_root, + accepted=True, + metrics={"frames": {"total": 2}}, + report={}, + manifest={"identity": {"frames_sha256": geometry_sha256}}, + ) + + threat_root = source / "threat" + threat_root.mkdir() + threat_frames = threat_root / "frames.jsonl" + _write_jsonl(threat_frames, [{"decision": "unchanged"}]) + threat_sha256 = _sha256(threat_frames) + threat = ThreatReplayResult( + result_id="m4-threat-replay-" + "8" * 64, + result_root=threat_root, + accepted=True, + metrics={"frames": {"total": 2}}, + report={}, + manifest={"identity": {"frames_sha256": threat_sha256}}, + ) + + transform = np.eye(4, dtype=np.float64) + frame = GeometryFrame( + frame_index=0, + points_map=np.asarray(((0.0, 0.0, 1.0), (100.0, 0.0, 1.0)), dtype=np.float64), + point_class=np.zeros(2, dtype=np.uint8), + sensor_position_map=np.zeros(3, dtype=np.float64), + sensor_orientation_xyzw=np.asarray((0.0, 0.0, 0.0, 1.0), dtype=np.float64), + projection=Kb4ProjectionProfile( + width=4, + height=4, + intrinsic_fx_fy_cx_cy=(2.0, 2.0, 2.0, 2.0), + distortion_kb4=(0.0, 0.0, 0.0, 0.0), + t_camera_from_lidar=transform, + ), + surface_valid=True, + ) + return replay._AdmittedInputs( + profile=profile, + semantic=semantic, + geometry=geometry, + threat=threat, + store=_Store(frame), # type: ignore[arg-type] + geometry_frames_path=geometry_frames, + geometry_frames_sha256=geometry_sha256, + threat_frames_path=threat_frames, + threat_frames_sha256=threat_sha256, + ) + + +def _build( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> tuple[replay.SemanticSlamReplayResult, replay._AdmittedInputs]: + admitted = _fixture(tmp_path) + monkeypatch.setattr(replay, "_admit_inputs", lambda **_kwargs: admitted) + result = replay.build_semantic_slam_replay( + repository_root=tmp_path, + semantic_result_root=tmp_path, + threat_result_root=tmp_path, + geometry_result_root=tmp_path, + output_root=tmp_path / "output", + ) + return result, admitted + + +def test_builder_is_idempotent_and_preserves_geometry_and_threat_ledgers( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + admitted = _fixture(tmp_path) + geometry_before = admitted.geometry_frames_path.read_bytes() + threat_before = admitted.threat_frames_path.read_bytes() + monkeypatch.setattr(replay, "_admit_inputs", lambda **_kwargs: admitted) + arguments = { + "repository_root": tmp_path, + "semantic_result_root": tmp_path, + "threat_result_root": tmp_path, + "geometry_result_root": tmp_path, + "output_root": tmp_path / "output", + } + first = replay.build_semantic_slam_replay(**arguments) + manifest_before = (first.result_root / replay.SEMANTIC_SLAM_MANIFEST_NAME).read_bytes() + second = replay.build_semantic_slam_replay(**arguments) + + assert second.result_id == first.result_id + assert (second.result_root / replay.SEMANTIC_SLAM_MANIFEST_NAME).read_bytes() == manifest_before + assert admitted.geometry_frames_path.read_bytes() == geometry_before + assert admitted.threat_frames_path.read_bytes() == threat_before + identity = first.manifest["identity"] + assert identity["geometry_frames_sha256"] == hashlib.sha256(geometry_before).hexdigest() + assert identity["base_m4_frames_sha256"] == hashlib.sha256(threat_before).hexdigest() + + +def test_unprojected_source_point_is_uint8_zero_with_explicit_status( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + result, _ = _build(tmp_path, monkeypatch) + with np.load( + result.result_root / replay.SEMANTIC_SLAM_POINTS_NAME, + allow_pickle=False, + ) as archive: + assert archive["point_labels"].dtype == np.uint8 + assert archive["point_labels"].tolist() == [4, 0] + assert archive["point_status_codes"].tolist() == [ + int(SemanticEvidenceStatus.LABELED), + int(SemanticEvidenceStatus.UNPROJECTED), + ] + assert archive["point_projected"].tolist() == [1, 0] + + rows = [ + json.loads(line) + for line in (result.result_root / replay.SEMANTIC_SLAM_OBSERVATIONS_NAME) + .read_text("utf-8") + .splitlines() + ] + assert rows[0]["observations"][0]["observation_id"] == _observation().observation_id + assert rows[0]["observations"][0]["status"] == "labeled" + assert rows[0]["observations"][0]["unprojected_point_count"] == 1 + assert "threat" not in rows[0]["observations"][0] + assert rows[0]["source_time_ns"] == 1_000_000_000 + assert rows[0]["temporal_binding"] == { + "semantic_to_camera": "exact-sequence-and-session-time", + "camera_to_lidar": "accepted-e6-nearest-host-arrival-best-effort", + "lidar_camera_delta_ms": 4.0, + "pose_point_delta_ms": 2.0, + "physical_synchronization_proven": False, + } + assert rows[1]["temporal_binding"]["lidar_camera_delta_ms"] is None + + +def test_reader_rejects_tampered_point_artifact( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + result, _ = _build(tmp_path, monkeypatch) + points = result.result_root / replay.SEMANTIC_SLAM_POINTS_NAME + points.write_bytes(points.read_bytes() + b"tamper") + + with pytest.raises(replay.SemanticSlamReplayError, match="digest changed"): + replay.read_semantic_slam_replay_result(result.result_root) + + +def test_builder_rejects_semantic_and_source_pack_session_time_mismatch( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + admitted = _fixture(tmp_path) + rows = [ + json.loads(line) for line in admitted.semantic.frames_path.read_text("utf-8").splitlines() + ] + rows[1]["session_seconds"] = 2.001 + _write_jsonl(admitted.semantic.frames_path, rows) + semantic = replace( + admitted.semantic, + frames_sha256=_sha256(admitted.semantic.frames_path), + ) + monkeypatch.setattr( + replay, + "_admit_inputs", + lambda **_kwargs: replace(admitted, semantic=semantic), + ) + + with pytest.raises(replay.SemanticSlamReplayError, match="session time disagree"): + replay.build_semantic_slam_replay( + repository_root=tmp_path, + semantic_result_root=tmp_path, + threat_result_root=tmp_path, + geometry_result_root=tmp_path, + output_root=tmp_path / "output", + ) + + +def test_builder_rejects_best_effort_delta_outside_admitted_bound( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + admitted = _fixture(tmp_path) + frame = admitted.store.frame_for_index(0) + assert frame is not None + monkeypatch.setattr( + replay, + "_admit_inputs", + lambda **_kwargs: replace( + admitted, + store=_Store(frame, lidar_delta_ms=100.001), # type: ignore[arg-type] + ), + ) + + with pytest.raises(replay.SemanticSlamReplayError, match="delta exceeds"): + replay.build_semantic_slam_replay( + repository_root=tmp_path, + semantic_result_root=tmp_path, + threat_result_root=tmp_path, + geometry_result_root=tmp_path, + output_root=tmp_path / "output", + ) + + +def test_reader_rejects_leaf_result_symlink( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + result, _ = _build(tmp_path, monkeypatch) + alias_parent = tmp_path / "alias" + alias_parent.mkdir() + alias = alias_parent / result.result_id + alias.symlink_to(result.result_root, target_is_directory=True) + + with pytest.raises(replay.SemanticSlamReplayError, match="result root is invalid"): + replay.read_semantic_slam_replay_result(alias) + + +def test_builder_rejects_existing_destination_symlink( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + result, admitted = _build(tmp_path, monkeypatch) + monkeypatch.setattr(replay, "_admit_inputs", lambda **_kwargs: admitted) + second_output = tmp_path / "second-output" + second_output.mkdir() + (second_output / result.result_id).symlink_to(result.result_root, target_is_directory=True) + + with pytest.raises(replay.SemanticSlamReplayError, match="destination cannot be a symlink"): + replay.build_semantic_slam_replay( + repository_root=tmp_path, + semantic_result_root=tmp_path, + threat_result_root=tmp_path, + geometry_result_root=tmp_path, + output_root=second_output, + )