NODEDC_MISSION_CORE/tests/test_perception_qualificati...

381 lines
13 KiB
Python

from __future__ import annotations
import hashlib
import json
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
import numpy as np
import pytest
from PIL import Image
from k1link.compute import (
EvaluationFrameRequest,
RecordedEvaluationPackError,
RecordedQualificationSliceError,
prepare_camera_compute_job,
prepare_recorded_evaluation_pack,
prepare_recorded_qualification_slice,
validate_recorded_evaluation_pack,
validate_recorded_qualification_slice,
)
from k1link.device_plugins.xgrids_k1.analyze import (
K1ValidFovMaskError,
prepare_k1_valid_fov_mask,
validate_k1_valid_fov_mask,
)
from k1link.device_plugins.xgrids_k1.calibration_schema import (
parse_k1_factory_calibration,
)
from k1link.device_plugins.xgrids_k1.calibration_snapshot import (
CALIBRATION_SNAPSHOT_MANIFEST_VERSION,
)
from k1link.web.camera_archive import CameraArchiveWriter
CALIBRATION_FIXTURES = Path(__file__).parent / "fixtures" / "k1" / "calibration"
def _box(box_type: bytes, payload: bytes = b"") -> bytes:
return (8 + len(payload)).to_bytes(4, "big") + box_type + payload
def _full_box(box_type: bytes, payload: bytes = b"", *, flags: int = 0) -> bytes:
return _box(box_type, bytes([0]) + flags.to_bytes(3, "big") + payload)
def _recorded_h264_fixture(base_decode_time: int = 0) -> tuple[bytes, bytes]:
track_id = 1
sample_duration = 500
tkhd = _full_box(
b"tkhd",
b"\x00" * 8 + track_id.to_bytes(4, "big") + b"\x00" * 4,
)
mdhd = _full_box(
b"mdhd",
b"\x00" * 8 + (1_000).to_bytes(4, "big") + b"\x00" * 4,
)
hdlr = _full_box(b"hdlr", b"\x00" * 4 + b"vide")
trak = _box(b"trak", tkhd + _box(b"mdia", mdhd + hdlr))
trex = _full_box(
b"trex",
track_id.to_bytes(4, "big")
+ (1).to_bytes(4, "big")
+ sample_duration.to_bytes(4, "big")
+ b"\x00" * 8,
)
init = _box(b"ftyp", b"isom") + _box(
b"moov",
trak + _box(b"mvex", trex) + _box(b"avcC", b"\x01\x64\x00\x28"),
)
tfhd = _full_box(b"tfhd", track_id.to_bytes(4, "big"), flags=0x020000)
tfdt = _full_box(b"tfdt", base_decode_time.to_bytes(4, "big"))
trun = _full_box(b"trun", (1).to_bytes(4, "big"))
fragment = _box(b"moof", _box(b"traf", tfhd + tfdt + trun)) + _box(
b"mdat",
b"frame",
)
return init, fragment
def _camera_job(tmp_path: Path, frame_count: int = 9) -> Path:
session = tmp_path / "session-qualification"
session.mkdir()
init, _fragment = _recorded_h264_fixture()
writer = CameraArchiveWriter(session, "sensor.camera.right", 1)
writer.append(
"init",
init,
host_epoch_ns=1_000_000_000,
host_monotonic_ns=2_000_000_000,
)
for index in range(frame_count):
_init, fragment = _recorded_h264_fixture(index * 500)
writer.append(
"media",
fragment,
host_epoch_ns=1_500_000_000 + index * 500_000_000,
host_monotonic_ns=2_500_000_000 + index * 500_000_000,
)
writer.close()
return prepare_camera_compute_job(
session_root=session,
source_id="sensor.camera.right",
codec_epoch=1,
origin_epoch_ns=1_000_000_000,
origin_monotonic_ns=2_000_000_000,
output_root=tmp_path / "jobs",
).job_root
def _calibration_snapshot(tmp_path: Path) -> Path:
root = tmp_path / "calibration"
root.mkdir()
camera = (CALIBRATION_FIXTURES / "camera.yaml").read_bytes()
extrinsic = (CALIBRATION_FIXTURES / "extrinsic_camera_lidar.yaml").read_bytes()
(root / "camera.yaml").write_bytes(camera)
(root / "extrinsic_camera_lidar.yaml").write_bytes(extrinsic)
vendor_device_id = "fixture-device"
device_serial = "fixture-serial"
artifact_values = (
(
"/mnt/system/factory-data/config/camera.yaml",
"camera.yaml",
camera,
),
(
"/mnt/system/factory-data/config/extrinsic_camera_lidar.yaml",
"extrinsic_camera_lidar.yaml",
extrinsic,
),
)
artifacts = [
{
"source_path": source_path,
"artifact_name": name,
"sha256": hashlib.sha256(payload).hexdigest(),
"bytes": len(payload),
"encoding": "utf-8",
}
for source_path, name, payload in artifact_values
]
content_identity = hashlib.sha256()
content_identity.update(vendor_device_id.encode("ascii"))
content_identity.update(b"\x00")
content_identity.update(device_serial.encode("ascii"))
for artifact in sorted(artifacts, key=lambda item: str(item["source_path"])):
content_identity.update(b"\x00")
content_identity.update(str(artifact["source_path"]).encode("utf-8"))
content_identity.update(bytes.fromhex(str(artifact["sha256"])))
calibration = parse_k1_factory_calibration(camera, extrinsic)
manifest = {
"schema_version": CALIBRATION_SNAPSHOT_MANIFEST_VERSION,
"captured_at_utc": datetime.now(UTC).isoformat(),
"content_identity_sha256": content_identity.hexdigest(),
"device": {
"vendor_device_id": vendor_device_id,
"device_serial": device_serial,
},
"artifacts": artifacts,
"normalized_calibration": calibration.normalized_profile(),
}
(root / "manifest.json").write_text(json.dumps(manifest), encoding="utf-8")
return root
def test_valid_fov_mask_is_calibration_bound_binary_and_reusable(tmp_path: Path) -> None:
snapshot = _calibration_snapshot(tmp_path)
first = prepare_k1_valid_fov_mask(
calibration_snapshot_root=snapshot,
source_id="sensor.camera.right",
output_root=tmp_path / "masks",
)
repeated = prepare_k1_valid_fov_mask(
calibration_snapshot_root=snapshot,
source_id="sensor.camera.right",
output_root=tmp_path / "masks",
)
assert repeated == first
assert first.calibration_slot == "camera_1"
assert (first.width, first.height) == (800, 600)
calibration = parse_k1_factory_calibration(
(CALIBRATION_FIXTURES / "camera.yaml").read_bytes(),
(CALIBRATION_FIXTURES / "extrinsic_camera_lidar.yaml").read_bytes(),
)
camera = calibration.camera("camera_1")
expected_center = (camera.intrinsic[2] * 0.2, camera.intrinsic[3] * 0.2)
expected_radius = min(
expected_center[0],
799 - expected_center[0],
expected_center[1],
599 - expected_center[1],
) - 4.0
assert first.center_xy == pytest.approx(expected_center)
assert first.radius_pixels == pytest.approx(expected_radius)
assert 0.55 < first.valid_fraction < 0.57
mask = np.asarray(Image.open(first.mask_path), dtype=np.uint8)
assert set(np.unique(mask)) == {0, 255}
assert mask[0, 0] == 0
assert mask[round(first.center_xy[1]), round(first.center_xy[0])] == 255
assert validate_k1_valid_fov_mask(first.root) == first
def test_valid_fov_mask_rejects_changed_png(tmp_path: Path) -> None:
result = prepare_k1_valid_fov_mask(
calibration_snapshot_root=_calibration_snapshot(tmp_path),
source_id="sensor.camera.right",
output_root=tmp_path / "masks",
)
result.mask_path.write_bytes(b"changed")
with pytest.raises(K1ValidFovMaskError, match="artifact changed"):
validate_k1_valid_fov_mask(result.root)
def test_qualification_slice_is_uniform_job_bound_and_reusable(tmp_path: Path) -> None:
job_root = _camera_job(tmp_path)
first = prepare_recorded_qualification_slice(
job_root=job_root,
output_root=tmp_path / "slices",
sample_count=4,
)
repeated = prepare_recorded_qualification_slice(
job_root=job_root,
output_root=tmp_path / "slices",
sample_count=4,
)
assert repeated == first
assert first.source_frame_count == 9
assert tuple(frame.frame_index for frame in first.frames) == (0, 3, 5, 8)
assert tuple(frame.sequence for frame in first.frames) == (1, 4, 6, 9)
assert validate_recorded_qualification_slice(first.root, job_root=job_root) == first
def test_qualification_slice_rejects_changed_selection(tmp_path: Path) -> None:
job_root = _camera_job(tmp_path)
result = prepare_recorded_qualification_slice(
job_root=job_root,
output_root=tmp_path / "slices",
sample_count=4,
)
manifest = json.loads(result.manifest_path.read_text(encoding="utf-8"))
manifest["frames"][1]["frame_index"] = 2
result.manifest_path.write_text(json.dumps(manifest), encoding="utf-8")
with pytest.raises(RecordedQualificationSliceError, match="binding changed"):
validate_recorded_qualification_slice(result.root, job_root=job_root)
def _evaluation_fixture(tmp_path: Path) -> tuple[Path, Path, Path, Path, Path, tuple[Any, ...]]:
job_root = _camera_job(tmp_path, frame_count=64)
qualification = prepare_recorded_qualification_slice(
job_root=job_root,
output_root=tmp_path / "slices",
sample_count=16,
)
valid_fov = prepare_k1_valid_fov_mask(
calibration_snapshot_root=_calibration_snapshot(tmp_path),
source_id="sensor.camera.right",
output_root=tmp_path / "masks",
)
anchors = {frame.frame_index for frame in qualification.frames}
temporal = {
1: "clip-a",
2: "clip-a",
3: "clip-a",
60: "clip-b",
61: "clip-b",
62: "clip-b",
}
requests = tuple(
EvaluationFrameRequest(
frame_index=index,
role="anchor" if index in anchors else "temporal",
group_id=f"anchor-{index:06d}" if index in anchors else temporal[index],
)
for index in sorted(anchors | set(temporal))
)
frames_root = tmp_path / "decoded"
frames_root.mkdir()
for request in requests:
image = Image.new(
"RGB",
(800, 600),
(request.frame_index, request.frame_index // 2, 255 - request.frame_index),
)
image.save(frames_root / f"frame-{request.frame_index:06d}.png")
timeline_path = tmp_path / "timeline.jsonl"
timeline_path.write_text(
"".join(
json.dumps({"frame_index": index, "session_seconds": 0.5 + index * 0.5}) + "\n"
for index in range(64)
),
encoding="utf-8",
)
return (
job_root,
qualification.root,
valid_fov.root,
frames_root,
timeline_path,
requests,
)
def test_evaluation_pack_is_bound_reusable_and_separates_annotation_state(
tmp_path: Path,
) -> None:
job_root, qualification_root, valid_fov_root, frames_root, timeline_path, requests = (
_evaluation_fixture(tmp_path)
)
first = prepare_recorded_evaluation_pack(
job_root=job_root,
qualification_root=qualification_root,
valid_fov_root=valid_fov_root,
decoded_frames_root=frames_root,
timeline_path=timeline_path,
output_root=tmp_path / "evaluation-packs",
selection=requests,
decoder_version="fixture-decoder/v1",
selection_document_sha256="1" * 64,
producer_files=(("fixture.py", "2" * 64),),
)
repeated = prepare_recorded_evaluation_pack(
job_root=job_root,
qualification_root=qualification_root,
valid_fov_root=valid_fov_root,
decoded_frames_root=frames_root,
timeline_path=timeline_path,
output_root=tmp_path / "evaluation-packs",
selection=requests,
decoder_version="fixture-decoder/v1",
selection_document_sha256="1" * 64,
producer_files=(("fixture.py", "2" * 64),),
)
assert repeated == first
assert len(first.frames) == 22
assert sum(frame.role == "anchor" for frame in first.frames) == 16
assert sum(frame.role == "temporal" for frame in first.frames) == 6
template = json.loads(first.annotation_template_path.read_text(encoding="utf-8"))
assert template["state"] == "unannotated"
assert {row["annotation_status"] for row in template["reviews"]} == {"unannotated"}
assert (
validate_recorded_evaluation_pack(
first.root,
job_root=job_root,
qualification_root=qualification_root,
valid_fov_root=valid_fov_root,
)
== first
)
def test_evaluation_pack_rejects_changed_image(tmp_path: Path) -> None:
job_root, qualification_root, valid_fov_root, frames_root, timeline_path, requests = (
_evaluation_fixture(tmp_path)
)
result = prepare_recorded_evaluation_pack(
job_root=job_root,
qualification_root=qualification_root,
valid_fov_root=valid_fov_root,
decoded_frames_root=frames_root,
timeline_path=timeline_path,
output_root=tmp_path / "evaluation-packs",
selection=requests,
decoder_version="fixture-decoder/v1",
selection_document_sha256="1" * 64,
producer_files=(("fixture.py", "2" * 64),),
)
result.frames[0].valid_fov_fill_path.write_bytes(b"changed")
with pytest.raises(RecordedEvaluationPackError, match="artifact changed"):
validate_recorded_evaluation_pack(
result.root,
job_root=job_root,
qualification_root=qualification_root,
valid_fov_root=valid_fov_root,
)