"""Fail-closed admission of one GOOSE 3D validation frame. The large source archive and extracted native frame remain on the worker D drive. The only portable products are a path-free admission manifest and a bounded visualization preview; neither is an alternative copy of the dataset. """ from __future__ import annotations import csv import hashlib import json import math import os import shutil import stat import tempfile import zipfile from collections import Counter from datetime import UTC, datetime from pathlib import Path, PurePosixPath from typing import Any, Final import numpy as np from k1link.datasets.gateway import DatasetFrameError, read_semantic_kitti_frame GOOSE_SOURCE_ID: Final = "goose-3d/v2025-08-22" GOOSE_VALIDATION_URL: Final = "https://goose-dataset.de/storage/goose_3d_val.zip" GOOSE_LICENSE: Final = "CC-BY-SA-4.0" GOOSE_ADMISSION_SCHEMA: Final = "missioncore.dataset-admission/v1" GOOSE_PREVIEW_SCHEMA: Final = "missioncore.dataset-native-scan-preview/v1" GOOSE_ARCHIVE_FILENAME: Final = "goose_3d_val.zip" GOOSE_ARCHIVE_OBSERVED_BYTES: Final = 3_498_402_435 MAX_ARCHIVE_ENTRIES: Final = 100_000 MAX_UNCOMPRESSED_BYTES: Final = 128 * 1024**3 MAX_METADATA_BYTES: Final = 4 * 1024**2 MAX_PREVIEW_POINTS: Final = 50_000 GOOSE_CHALLENGE_GROUPS: Final[dict[str, frozenset[str]]] = { "other": frozenset({"undefined", "ego_vehicle", "outlier"}), "artificial_structures": frozenset({"building", "wall", "bridge", "tunnel"}), "artificial_ground": frozenset( { "cobble", "bikeway", "pedestrian_crossing", "road_marking", "sidewalk", "curb", "asphalt", } ), "natural_ground": frozenset( {"snow", "leaves", "gravel", "soil", "low_grass", "water"} ), "obstacle": frozenset( { "traffic_cone", "obstacle", "street_light", "road_block", "traffic_light", "boom_barrier", "rail_track", "debris", "animal", "rock", "fence", "guard_rail", "pole", "traffic_sign", "misc_sign", "barrier_tape", "wire", "container", "barrel", "pipe", } ), "vehicle": frozenset( { "car", "bicycle", "bus", "motorcycle", "truck", "on_rails", "caravan", "trailer", "kick_scooter", "heavy_machinery", "military_vehicle", } ), "vegetation": frozenset( { "forest", "bush", "moss", "tree_crown", "tree_trunk", "crops", "high_grass", "scenery_vegetation", "hedge", "tree_root", } ), "human": frozenset({"person", "rider"}), "sky": frozenset({"sky"}), } GOOSE_CHALLENGE_IDS: Final = { name: identifier for identifier, name in enumerate( ( "other", "artificial_structures", "artificial_ground", "natural_ground", "obstacle", "vehicle", "vegetation", "human", "sky", ) ) } class GooseAdmissionError(RuntimeError): """The source archive cannot satisfy the pinned GOOSE admission contract.""" def admit_goose_validation( dataset_root: Path, *, archive_path: Path | None = None, preview_points: int = MAX_PREVIEW_POINTS, ) -> dict[str, Any]: """Verify a GOOSE validation archive and admit one deterministic frame.""" root = dataset_root.expanduser().absolute() if not _is_canonical_worker_root(root): raise GooseAdmissionError("GOOSE admission requires the canonical worker D dataset root") archive = ( archive_path.expanduser().absolute() if archive_path is not None else root / "goose-3d/v2025-08-22/archives" / GOOSE_ARCHIVE_FILENAME ) if not archive.is_file(): raise GooseAdmissionError("GOOSE validation archive is unavailable") size_bytes = archive.stat().st_size if size_bytes != GOOSE_ARCHIVE_OBSERVED_BYTES: raise GooseAdmissionError( "GOOSE validation archive size differs from the pinned observed release" ) if not 1 <= preview_points <= MAX_PREVIEW_POINTS: raise GooseAdmissionError("preview point limit is outside the admitted range") archive_sha256 = _sha256_file(archive) install_root = root / "goose-3d/v2025-08-22/installs" / archive_sha256 state_root = root / "state" state_root.mkdir(parents=True, exist_ok=True) try: with zipfile.ZipFile(archive) as source: members = _admitted_members(source) point_member, label_member = _first_frame_pair(members) mapping_member = _unique_metadata_member(members, "goose_label_mapping.csv") license_member = _unique_metadata_member(members, "LICENSE") mapping_bytes = _bounded_member_bytes(source, mapping_member) license_bytes = _bounded_member_bytes(source, license_member) labels = parse_goose_label_mapping(mapping_bytes) license_text = license_bytes.decode("utf-8", errors="replace") if "Attribution-ShareAlike 4.0 International" not in license_text: raise GooseAdmissionError("archive LICENSE does not declare CC BY-SA 4.0") frame_id = _frame_id(point_member.filename) frame_root = install_root / "frames" / frame_id frame_root.mkdir(parents=True, exist_ok=True) point_path = frame_root / "points.bin" label_path = frame_root / "labels.label" mapping_path = install_root / "goose_label_mapping.csv" license_path = install_root / "LICENSE" _extract_once(source, point_member, point_path) _extract_once(source, label_member, label_path) _write_once(mapping_path, mapping_bytes) _write_once(license_path, license_bytes) except (OSError, zipfile.BadZipFile) as exc: raise GooseAdmissionError("GOOSE archive could not be verified") from exc try: frame = read_semantic_kitti_frame(point_path, label_path) except DatasetFrameError as exc: raise GooseAdmissionError("first GOOSE frame violates XYZI/label alignment") from exc unknown_labels = sorted( int(value) for value in np.unique(frame.semantic_labels) if int(value) not in labels ) if unknown_labels: raise GooseAdmissionError("frame contains semantic ids absent from the label mapping") preview = _native_scan_preview( frame_id, frame.points_xyz_m, frame.remission, frame.semantic_labels, labels, maximum_points=preview_points, ) preview_path = install_root / "previews" / f"{frame_id}.json" _atomic_json(preview_path, preview) preview_sha256 = _sha256_file(preview_path) present_classes = Counter(int(value) for value in frame.semantic_labels) ground_points = sum( count for label_id, count in present_classes.items() if labels[label_id]["challenge_category_id"] in (2, 3) ) manifest = { "schema_version": GOOSE_ADMISSION_SCHEMA, "source_id": GOOSE_SOURCE_ID, "observed_at_utc": datetime.now(UTC).isoformat().replace("+00:00", "Z"), "status": "frame-ready", "storage": { "policy": "worker-d-only", "admitted": True, "canonical_root": True, "path_exposed": False, }, "archive": { "filename": GOOSE_ARCHIVE_FILENAME, "source_url": GOOSE_VALIDATION_URL, "bytes_transferred": size_bytes, "total_bytes": size_bytes, "size_bytes": size_bytes, "sha256": archive_sha256, "integrity": "zip-structure-and-content-digest", "vendor_checksum_available": False, }, "license": { "spdx": GOOSE_LICENSE, "artifact_present": True, }, "frame": { "frame_id": frame_id, "representation": "native-scan", "point_count": frame.point_count, "semantic_class_count": len(present_classes), "ground_truth_ground_points": ground_points, "ground_truth_ground_fraction": ground_points / frame.point_count, "preview_point_count": int(preview["point_count"]), "preview_sha256": preview_sha256, "preview_available": True, }, "next_action": "review-first-native-scan", } _atomic_json(state_root / "goose-3d-v2025-08-22.json", manifest) return manifest def _is_canonical_worker_root(root: Path) -> bool: normalized = str(root).replace("\\", "/").rstrip("/").lower() return normalized == "/mnt/d/ndc_missioncore/datasets" def _sha256_file(path: Path) -> str: digest = hashlib.sha256() try: with path.open("rb") as source: for chunk in iter(lambda: source.read(8 * 1024**2), b""): digest.update(chunk) except OSError as exc: raise GooseAdmissionError("dataset artifact cannot be hashed") from exc return digest.hexdigest() def _admitted_members(source: zipfile.ZipFile) -> tuple[zipfile.ZipInfo, ...]: members = tuple(source.infolist()) if not members or len(members) > MAX_ARCHIVE_ENTRIES: raise GooseAdmissionError("archive entry count is outside the admitted range") total_bytes = 0 for member in members: path = PurePosixPath(member.filename.replace("\\", "/")) if path.is_absolute() or ".." in path.parts or not path.parts: raise GooseAdmissionError("archive contains an unsafe member path") mode = member.external_attr >> 16 if stat.S_ISLNK(mode): raise GooseAdmissionError("archive contains a symbolic link") total_bytes += member.file_size if total_bytes > MAX_UNCOMPRESSED_BYTES: raise GooseAdmissionError("archive expands beyond the admitted size") return members def _first_frame_pair( members: tuple[zipfile.ZipInfo, ...], ) -> tuple[zipfile.ZipInfo, zipfile.ZipInfo]: points = { _frame_id(member.filename): member for member in members if not member.is_dir() and any( prefix in "/" + member.filename.replace("\\", "/") for prefix in ("/lidar/val/", "/velodyne/val/") ) and member.filename.endswith("_vls128.bin") } labels = { _frame_id(member.filename): member for member in members if not member.is_dir() and "/labels/val/" in "/" + member.filename.replace("\\", "/") and member.filename.endswith("_goose.label") } shared = sorted(points.keys() & labels.keys()) if not shared: raise GooseAdmissionError("archive has no aligned GOOSE validation frame") frame_id = shared[0] return points[frame_id], labels[frame_id] def _frame_id(filename: str) -> str: name = PurePosixPath(filename.replace("\\", "/")).name for suffix in ("_vls128.bin", "_goose.label"): if name.endswith(suffix): identifier = name[: -len(suffix)] if identifier: return identifier raise GooseAdmissionError("GOOSE frame filename is incompatible") def _unique_metadata_member( members: tuple[zipfile.ZipInfo, ...], filename: str, ) -> zipfile.ZipInfo: matches = [member for member in members if PurePosixPath(member.filename).name == filename] if len(matches) != 1 or matches[0].is_dir(): raise GooseAdmissionError(f"archive must contain exactly one {filename}") return matches[0] def _bounded_member_bytes(source: zipfile.ZipFile, member: zipfile.ZipInfo) -> bytes: if member.file_size <= 0 or member.file_size > MAX_METADATA_BYTES: raise GooseAdmissionError("archive metadata size is outside the admitted range") value = source.read(member) if len(value) != member.file_size: raise GooseAdmissionError("archive metadata was truncated") return value def _extract_once(source: zipfile.ZipFile, member: zipfile.ZipInfo, target: Path) -> None: if target.exists(): if target.is_file() and target.stat().st_size == member.file_size: return raise GooseAdmissionError("immutable extracted artifact already exists with another size") target.parent.mkdir(parents=True, exist_ok=True) with tempfile.NamedTemporaryFile(dir=target.parent, delete=False) as temporary: temporary_path = Path(temporary.name) try: with source.open(member) as member_source: shutil.copyfileobj(member_source, temporary, length=1024**2) temporary.flush() os.fsync(temporary.fileno()) except Exception: temporary_path.unlink(missing_ok=True) raise if temporary_path.stat().st_size != member.file_size: temporary_path.unlink(missing_ok=True) raise GooseAdmissionError("extracted artifact size does not match the archive") os.replace(temporary_path, target) def _write_once(target: Path, value: bytes) -> None: if target.exists(): if target.is_file() and target.read_bytes() == value: return raise GooseAdmissionError("immutable metadata artifact already exists with other content") target.parent.mkdir(parents=True, exist_ok=True) with tempfile.NamedTemporaryFile(dir=target.parent, delete=False) as temporary: temporary_path = Path(temporary.name) temporary.write(value) temporary.flush() os.fsync(temporary.fileno()) os.replace(temporary_path, target) def read_goose_label_mapping(path: Path) -> dict[int, dict[str, Any]]: try: value = path.read_bytes() except OSError as exc: raise GooseAdmissionError("GOOSE label mapping is unavailable") from exc if not 0 < len(value) <= MAX_METADATA_BYTES: raise GooseAdmissionError("GOOSE label mapping size is outside the admitted range") return parse_goose_label_mapping(value) def parse_goose_label_mapping(value: bytes) -> dict[int, dict[str, Any]]: try: text = value.decode("utf-8-sig") rows = tuple(csv.DictReader(text.splitlines())) except (UnicodeDecodeError, csv.Error) as exc: raise GooseAdmissionError("GOOSE label mapping cannot be decoded") from exc required = { "class_name", "label_key", "hex", } if not rows or not required.issubset(rows[0]): raise GooseAdmissionError("GOOSE label mapping columns are incompatible") labels: dict[int, dict[str, Any]] = {} try: for row in rows: label_id = int(row["label_key"]) color = row["hex"].strip().lower() if label_id in labels or not 0 <= label_id <= 65_535: raise GooseAdmissionError("GOOSE label ids are invalid") if len(color) != 7 or color[0] != "#": raise GooseAdmissionError("GOOSE label color is invalid") int(color[1:], 16) labels[label_id] = { "class_name": row["class_name"].strip(), "hex": color, **_challenge_category(row["class_name"].strip()), } except (KeyError, TypeError, ValueError) as exc: raise GooseAdmissionError("GOOSE label mapping rows are invalid") from exc return labels def _challenge_category(class_name: str) -> dict[str, Any]: matches = [ name for name, members in GOOSE_CHALLENGE_GROUPS.items() if class_name in members ] if len(matches) != 1: raise GooseAdmissionError("GOOSE class is absent from the pinned challenge mapping") name = matches[0] return { "challenge_category_id": GOOSE_CHALLENGE_IDS[name], "challenge_category_name": name, } def _native_scan_preview( frame_id: str, points_xyz_m: np.ndarray[Any, Any], remission: np.ndarray[Any, Any], semantic_labels: np.ndarray[Any, Any], labels: dict[int, dict[str, Any]], *, maximum_points: int, ) -> dict[str, Any]: point_count = int(points_xyz_m.shape[0]) sample_count = min(point_count, maximum_points) indices = np.linspace(0, point_count - 1, sample_count, dtype=np.int64) sampled_points = points_xyz_m[indices] sampled_remission = remission[indices] sampled_labels = semantic_labels[indices] remission_min = float(np.min(sampled_remission)) remission_max = float(np.max(sampled_remission)) remission_span = remission_max - remission_min if not math.isfinite(remission_span): raise GooseAdmissionError("GOOSE remission range is non-finite") if remission_span > 0: scaled_remission = np.rint( (sampled_remission - remission_min) * (255.0 / remission_span) ).astype(np.uint8) else: scaled_remission = np.zeros(sample_count, dtype=np.uint8) semantic_rgb: list[int] = [] ground_mask: list[int] = [] for value in sampled_labels: label = labels[int(value)] color = label["hex"] semantic_rgb.extend( (int(color[1:3], 16), int(color[3:5], 16), int(color[5:7], 16)) ) ground_mask.append(1 if label["challenge_category_id"] in (2, 3) else 0) present_ids = sorted({int(value) for value in sampled_labels}) return { "schema_version": GOOSE_PREVIEW_SCHEMA, "source_id": GOOSE_SOURCE_ID, "frame_id": frame_id, "representation": "native-scan", "sampling": "deterministic-even-index", "source_point_count": point_count, "point_count": sample_count, "points_xyz_m": sampled_points.tolist(), "remission_0_to_255": scaled_remission.tolist(), "semantic_label_ids": sampled_labels.astype(np.uint16).tolist(), "semantic_rgb_0_to_255": semantic_rgb, "ground_truth_ground": ground_mask, "classes": [ { "label_id": label_id, **labels[label_id], } for label_id in present_ids ], "safety": { "visualization_only": True, "navigation_or_safety_accepted": False, }, } def _atomic_json(path: Path, value: dict[str, Any]) -> None: path.parent.mkdir(parents=True, exist_ok=True) encoded = ( json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode( "utf-8" ) + b"\n" ) with tempfile.NamedTemporaryFile(dir=path.parent, delete=False) as temporary: temporary_path = Path(temporary.name) temporary.write(encoded) temporary.flush() os.fsync(temporary.fileno()) os.replace(temporary_path, path)