feat(perception): define semantic object understanding
This commit is contained in:
@@ -0,0 +1,37 @@
|
||||
{
|
||||
"schema_version": "missioncore.m48s-mask-grounding-dino-profile/v0",
|
||||
"profile_id": "m48s-mask-grounding-dino-research-shadow/v0",
|
||||
"model": {
|
||||
"catalog": "NVIDIA NGC",
|
||||
"catalog_url": "https://catalog.ngc.nvidia.com/orgs/nvidia/tao/models/pretrained_mask_grounding_dino_v2/mask_grounding_dino_swin_tiny_research_deployable_v2.0/version-history",
|
||||
"artifact": "mask_grounding_dino_epoch_021.onnx",
|
||||
"artifact_sha256": "4e8b066cf7782ae6b4269255b861c205f4b07b33ff752e5eaf297c42ecfb2f5e",
|
||||
"license_scope": "research-only",
|
||||
"commercial_checkpoint_acquired": false
|
||||
},
|
||||
"runtime": {
|
||||
"worker_node": "DESKTOP-OPJ8J04",
|
||||
"gpu": "NVIDIA GeForce RTX 4090",
|
||||
"container_reference": "nvcr.io/nvidia/tao/tao-toolkit:7.0.1-deploy",
|
||||
"container_image_id": "sha256:2a3095330dd83e4314aada21fc7d184fcdb442a9131d591e5593ca5497394e7a",
|
||||
"tensorrt_precision": "FP16",
|
||||
"engine_sha256": "1d127107d15f18dcddafd7345a0a766acfee204d63c5ab697eaabada2c7afcd2",
|
||||
"engine_size_bytes": 407588756
|
||||
},
|
||||
"binding": {
|
||||
"minimum_mask_iou": 0.9,
|
||||
"minimum_projected_points": 4,
|
||||
"minimum_support_fraction": 0.5,
|
||||
"maximum_secondary_support_fraction": 0.25,
|
||||
"minimum_label_confidence": 0.3,
|
||||
"minimum_label_margin": 0.1
|
||||
},
|
||||
"authority": {
|
||||
"ground_truth": false,
|
||||
"independent_truth": false,
|
||||
"candidate_accepted": false,
|
||||
"commands_enabled": false,
|
||||
"actuation_allowed": false,
|
||||
"navigation_or_safety_accepted": false
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
model_name: missioncore-m48s-mask-grounding-dino-shadow
|
||||
results_dir: /workspace/probe/results
|
||||
wandb:
|
||||
enable: false
|
||||
model:
|
||||
backbone: swin_tiny_224_1k
|
||||
num_queries: 900
|
||||
num_feature_levels: 4
|
||||
dec_layers: 6
|
||||
enc_layers: 6
|
||||
dim_feedforward: 2048
|
||||
dropout_ratio: 0.0
|
||||
log_scale: auto
|
||||
class_embed_bias: true
|
||||
num_region_queries: 100
|
||||
loss_types:
|
||||
- labels
|
||||
- boxes
|
||||
- masks
|
||||
- rela
|
||||
dataset:
|
||||
infer_data_sources:
|
||||
image_dir: /workspace/probe/input
|
||||
data_type: OD
|
||||
captions:
|
||||
- adult person
|
||||
- child
|
||||
- dog
|
||||
- bicycle
|
||||
- kick scooter
|
||||
- stroller
|
||||
- shopping cart
|
||||
- car
|
||||
- bus
|
||||
- truck
|
||||
- motorcycle
|
||||
- trash bin
|
||||
- bollard
|
||||
- post
|
||||
- concrete hemisphere
|
||||
- traffic cone
|
||||
- road sign
|
||||
- temporary barrier
|
||||
batch_size: 1
|
||||
workers: 1
|
||||
pin_memory: true
|
||||
max_labels: 80
|
||||
has_mask: true
|
||||
inference:
|
||||
trt_engine: /workspace/model/mask_grounding_dino_swin_tiny_research_fp16.engine
|
||||
results_dir: /workspace/probe/results
|
||||
batch_size: -1
|
||||
input_width: 960
|
||||
input_height: 544
|
||||
conf_threshold: 0.25
|
||||
ioi_threshold: 0.5
|
||||
nms_threshold: 0.2
|
||||
text_threshold: 0.3
|
||||
outline_width: 3
|
||||
gen_trt_engine:
|
||||
onnx_file: /workspace/model/mask_grounding_dino_epoch_021.onnx
|
||||
trt_engine: /workspace/model/mask_grounding_dino_swin_tiny_research_fp16.engine
|
||||
results_dir: /workspace/model/engine-build
|
||||
batch_size: -1
|
||||
verbose: false
|
||||
tensorrt:
|
||||
workspace_size: 8192
|
||||
min_batch_size: 1
|
||||
opt_batch_size: 1
|
||||
max_batch_size: 1
|
||||
data_type: FP16
|
||||
@@ -0,0 +1,26 @@
|
||||
{
|
||||
"schema_version": "missioncore.semantic-object-quality-profile/v0",
|
||||
"profile_id": "m48s-urban-semantic-object-quality/v0",
|
||||
"vocabulary_id": "missioncore.urban-object-semantics/v0",
|
||||
"top_k": 5,
|
||||
"thresholds": {
|
||||
"minimum_prediction_coverage": 1.0,
|
||||
"minimum_exact_top1_accuracy": 0.8,
|
||||
"minimum_coarse_group_accuracy": 0.95,
|
||||
"minimum_exact_top_k_recall": 0.95,
|
||||
"maximum_unresolved_fraction": 0.1,
|
||||
"maximum_conflict_fraction": 0.05
|
||||
},
|
||||
"scope": {
|
||||
"object_presence_scored": false,
|
||||
"semantic_class_scored": true,
|
||||
"risk_policy_scored": false,
|
||||
"requires_separate_adjudicated_semantic_truth": true
|
||||
},
|
||||
"authority": {
|
||||
"physical_live": false,
|
||||
"commands_enabled": false,
|
||||
"actuation_allowed": false,
|
||||
"navigation_or_safety_accepted": false
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,233 @@
|
||||
{
|
||||
"schema_version": "missioncore.object-semantic-vocabulary/v0",
|
||||
"vocabulary_id": "missioncore.urban-object-semantics/v0",
|
||||
"status": "experimental",
|
||||
"scope": "urban-ground-robot",
|
||||
"classes": [
|
||||
{
|
||||
"class_id": "object.unknown",
|
||||
"parent_id": null,
|
||||
"aliases": ["unknown", "unknown_object"],
|
||||
"agency_prior": "unknown",
|
||||
"risk_traits": ["unknown-clearance", "unknown-motion-potential"]
|
||||
},
|
||||
{
|
||||
"class_id": "human.unknown",
|
||||
"parent_id": "object.unknown",
|
||||
"aliases": ["person", "human", "pedestrian"],
|
||||
"agency_prior": "animate",
|
||||
"risk_traits": ["vulnerable-road-user", "unpredictable-motion"]
|
||||
},
|
||||
{
|
||||
"class_id": "human.adult",
|
||||
"parent_id": "human.unknown",
|
||||
"aliases": ["adult", "adult_person"],
|
||||
"agency_prior": "animate",
|
||||
"risk_traits": ["vulnerable-road-user", "unpredictable-motion"]
|
||||
},
|
||||
{
|
||||
"class_id": "human.child",
|
||||
"parent_id": "human.unknown",
|
||||
"aliases": ["child", "kid", "toddler"],
|
||||
"agency_prior": "animate",
|
||||
"risk_traits": ["vulnerable-road-user", "high-unpredictability"]
|
||||
},
|
||||
{
|
||||
"class_id": "animal.unknown",
|
||||
"parent_id": "object.unknown",
|
||||
"aliases": ["animal", "unknown_animal"],
|
||||
"agency_prior": "animate",
|
||||
"risk_traits": ["unpredictable-motion"]
|
||||
},
|
||||
{
|
||||
"class_id": "animal.dog",
|
||||
"parent_id": "animal.unknown",
|
||||
"aliases": ["dog", "canine"],
|
||||
"agency_prior": "animate",
|
||||
"risk_traits": ["unpredictable-motion", "close-interaction-risk"]
|
||||
},
|
||||
{
|
||||
"class_id": "vehicle.unknown",
|
||||
"parent_id": "object.unknown",
|
||||
"aliases": ["vehicle", "unknown_vehicle"],
|
||||
"agency_prior": "self-propelled",
|
||||
"risk_traits": ["may-start-moving", "large-footprint"]
|
||||
},
|
||||
{
|
||||
"class_id": "vehicle.car",
|
||||
"parent_id": "vehicle.unknown",
|
||||
"aliases": ["car", "automobile"],
|
||||
"agency_prior": "self-propelled",
|
||||
"risk_traits": ["may-start-moving", "reversing-risk"]
|
||||
},
|
||||
{
|
||||
"class_id": "vehicle.heavy",
|
||||
"parent_id": "vehicle.unknown",
|
||||
"aliases": ["heavy_vehicle", "truck", "bus"],
|
||||
"agency_prior": "self-propelled",
|
||||
"risk_traits": ["may-start-moving", "large-footprint", "blind-zone"]
|
||||
},
|
||||
{
|
||||
"class_id": "vehicle.bicycle",
|
||||
"parent_id": "vehicle.unknown",
|
||||
"aliases": ["bicycle", "bike"],
|
||||
"agency_prior": "self-propelled",
|
||||
"risk_traits": ["may-start-moving", "vulnerable-road-user"]
|
||||
},
|
||||
{
|
||||
"class_id": "vehicle.motorcycle",
|
||||
"parent_id": "vehicle.unknown",
|
||||
"aliases": ["motorcycle", "motorbike"],
|
||||
"agency_prior": "self-propelled",
|
||||
"risk_traits": ["may-start-moving", "rapid-motion"]
|
||||
},
|
||||
{
|
||||
"class_id": "vehicle.kick-scooter",
|
||||
"parent_id": "vehicle.unknown",
|
||||
"aliases": ["kick_scooter", "scooter"],
|
||||
"agency_prior": "self-propelled",
|
||||
"risk_traits": ["may-start-moving", "vulnerable-road-user"]
|
||||
},
|
||||
{
|
||||
"class_id": "mobility.unknown",
|
||||
"parent_id": "object.unknown",
|
||||
"aliases": ["passive_mobility", "wheeled_object"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["may-be-pushed", "rolling-potential"]
|
||||
},
|
||||
{
|
||||
"class_id": "mobility.stroller",
|
||||
"parent_id": "mobility.unknown",
|
||||
"aliases": ["stroller", "pram"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["may-be-pushed", "vulnerable-occupant"]
|
||||
},
|
||||
{
|
||||
"class_id": "mobility.cart",
|
||||
"parent_id": "mobility.unknown",
|
||||
"aliases": ["cart", "shopping_cart", "trolley"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["may-be-pushed", "rolling-potential"]
|
||||
},
|
||||
{
|
||||
"class_id": "static.unknown",
|
||||
"parent_id": "object.unknown",
|
||||
"aliases": ["static_obstacle", "generic_obstacle", "obstacle"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["static-clearance"]
|
||||
},
|
||||
{
|
||||
"class_id": "static.trash-bin",
|
||||
"parent_id": "static.unknown",
|
||||
"aliases": ["trash_bin", "garbage_bin", "waste_bin"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["static-clearance", "movable-by-external-force"]
|
||||
},
|
||||
{
|
||||
"class_id": "static.bollard",
|
||||
"parent_id": "static.unknown",
|
||||
"aliases": ["bollard", "road_bollard"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["small-static", "rigid-clearance"]
|
||||
},
|
||||
{
|
||||
"class_id": "static.post",
|
||||
"parent_id": "static.unknown",
|
||||
"aliases": ["post", "pole", "pipe_post"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["small-static", "rigid-clearance"]
|
||||
},
|
||||
{
|
||||
"class_id": "static.concrete-hemisphere",
|
||||
"parent_id": "static.unknown",
|
||||
"aliases": ["concrete_hemisphere", "hemisphere", "dome_barrier"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["low-profile", "rigid-clearance"]
|
||||
},
|
||||
{
|
||||
"class_id": "static.traffic-cone",
|
||||
"parent_id": "static.unknown",
|
||||
"aliases": ["traffic_cone", "cone"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["small-static", "temporary-road-control"]
|
||||
},
|
||||
{
|
||||
"class_id": "static.road-sign",
|
||||
"parent_id": "static.unknown",
|
||||
"aliases": ["road_sign", "traffic_sign"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["static-clearance", "road-control-device"]
|
||||
},
|
||||
{
|
||||
"class_id": "static.temporary-barrier",
|
||||
"parent_id": "static.unknown",
|
||||
"aliases": ["temporary_barrier", "road_barrier", "barrier"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["static-clearance", "temporary-road-control"]
|
||||
},
|
||||
{
|
||||
"class_id": "terrain.unknown",
|
||||
"parent_id": "object.unknown",
|
||||
"aliases": ["terrain_hazard", "surface_hazard"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["surface-discontinuity"]
|
||||
},
|
||||
{
|
||||
"class_id": "terrain.curb",
|
||||
"parent_id": "terrain.unknown",
|
||||
"aliases": ["curb", "sidewalk_curb"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["surface-discontinuity", "vehicle-dependent-traversability"]
|
||||
},
|
||||
{
|
||||
"class_id": "terrain.step",
|
||||
"parent_id": "terrain.unknown",
|
||||
"aliases": ["step", "stair"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["surface-discontinuity", "vehicle-dependent-traversability"]
|
||||
},
|
||||
{
|
||||
"class_id": "terrain.pothole",
|
||||
"parent_id": "terrain.unknown",
|
||||
"aliases": ["pothole", "road_hole"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["negative-obstacle", "vehicle-dependent-traversability"]
|
||||
},
|
||||
{
|
||||
"class_id": "terrain.drainage-grate",
|
||||
"parent_id": "terrain.unknown",
|
||||
"aliases": ["drainage_grate", "drain", "road_grate"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["surface-discontinuity", "wheel-entrapment"]
|
||||
},
|
||||
{
|
||||
"class_id": "natural.unknown",
|
||||
"parent_id": "object.unknown",
|
||||
"aliases": ["natural_obstacle", "debris"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["irregular-clearance"]
|
||||
},
|
||||
{
|
||||
"class_id": "natural.rock",
|
||||
"parent_id": "natural.unknown",
|
||||
"aliases": ["rock", "stone"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["irregular-clearance", "rigid-clearance"]
|
||||
},
|
||||
{
|
||||
"class_id": "natural.branch",
|
||||
"parent_id": "natural.unknown",
|
||||
"aliases": ["branch", "fallen_branch"],
|
||||
"agency_prior": "inert",
|
||||
"risk_traits": ["irregular-clearance", "deformable-obstacle"]
|
||||
}
|
||||
],
|
||||
"policies": {
|
||||
"occupancy_independent_of_semantics": true,
|
||||
"unknown_preserves_obstacle": true,
|
||||
"class_prior_is_not_observed_state": true,
|
||||
"risk_is_advisory_only": true,
|
||||
"planner_command_authority": false,
|
||||
"max_hypotheses": 5
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
{
|
||||
"schema_version": "missioncore.open-vocabulary-semantic-shadow-profile/v0",
|
||||
"profile_id": "m48s-raw-kb4-urban-semantics-shadow/v0",
|
||||
"status": "experimental-shadow",
|
||||
"source": {
|
||||
"session_id": "20260720T065719Z_viewer_live",
|
||||
"source_id": "RAVNOVES00",
|
||||
"camera_source_id": "sensor.camera.right",
|
||||
"coordinate_space": "raw-kb4",
|
||||
"width": 800,
|
||||
"height": 600,
|
||||
"decoded_frame_basis": "zero-based-decoded-video-frame",
|
||||
"decoded_image_encoding": "jpeg-q2-yuvj420p",
|
||||
"valid_fov_result_id": "valid-fov-mask-b4dd8ddf2b87c1d520ee8a0868c4fea062d7c14d1bae73ccabd3abe1f3acbac2",
|
||||
"valid_fov_mask_sha256": "a40cee06b7c6f69b6a09a11563dcfd237f3de833b1ccd31459e66692e528ba63",
|
||||
"valid_fov_fill_value": 114
|
||||
},
|
||||
"provider": {
|
||||
"provider_id": "nvidia-tao-grounding-dino-trt/v1",
|
||||
"name": "NVIDIA TAO Grounding DINO Swin-Tiny Commercial",
|
||||
"version": "1.0",
|
||||
"model_id": "grounding-dino-swin-tiny-commercial/v1",
|
||||
"model_file": "grounding_dino_swin_tiny_commercial_deployable.onnx",
|
||||
"model_sha256": "6895acdc6b588e923f753e37b3bd18869e064256e5ecc1b2b9853e8c51125f94",
|
||||
"model_revision": "nvidia-tao-7.0.1-commercial-v1.0",
|
||||
"engine_file": "grounding_dino_swin_tiny_commercial_fp16.engine",
|
||||
"engine_sha256": "63213f60730ce8577047ff508dfe06113269347501e2dd832aeb56c89bc11489",
|
||||
"engine_precision": "FP16",
|
||||
"container_reference": "nvcr.io/nvidia/tao/tao-toolkit:7.0.1-deploy",
|
||||
"container_image_id": "sha256:2a3095330dd83e4314aada21fc7d184fcdb442a9131d591e5593ca5497394e7a",
|
||||
"preprocess_id": "raw-kb4-valid-fov-fill114-800x600-to-960x544/v1",
|
||||
"engine_input_width": 960,
|
||||
"engine_input_height": 544
|
||||
},
|
||||
"postprocessing": {
|
||||
"minimum_input_confidence": 0.25,
|
||||
"minimum_box_area_fraction": 0.00002,
|
||||
"maximum_box_area_fraction": 0.35,
|
||||
"minimum_valid_fov_fraction": 0.5,
|
||||
"require_center_inside_valid_fov": true,
|
||||
"fusion_iou_threshold": 0.7,
|
||||
"selected_minimum_confidence": 0.5,
|
||||
"selected_minimum_margin": 0.1,
|
||||
"max_hypotheses": 5
|
||||
},
|
||||
"prompt_groups": [
|
||||
{
|
||||
"prompt_set_id": "urban-static/v0",
|
||||
"captions": [
|
||||
"trash bin",
|
||||
"shopping cart",
|
||||
"bollard",
|
||||
"post",
|
||||
"concrete hemisphere",
|
||||
"traffic cone",
|
||||
"road sign",
|
||||
"temporary barrier"
|
||||
]
|
||||
},
|
||||
{
|
||||
"prompt_set_id": "urban-agents/v0",
|
||||
"captions": [
|
||||
"adult person",
|
||||
"child",
|
||||
"dog",
|
||||
"stroller",
|
||||
"kick scooter",
|
||||
"bicycle"
|
||||
]
|
||||
},
|
||||
{
|
||||
"prompt_set_id": "urban-vehicles/v0",
|
||||
"captions": ["car", "truck", "bus", "motorcycle"]
|
||||
}
|
||||
],
|
||||
"semantic_vocabulary_id": "missioncore.urban-object-semantics/v0",
|
||||
"risk_policy_id": "urban-object-risk-shadow/v0",
|
||||
"authority": {
|
||||
"ground_truth": false,
|
||||
"independent_truth": false,
|
||||
"candidate_accepted": false,
|
||||
"commands_enabled": false,
|
||||
"actuation_allowed": false,
|
||||
"navigation_or_safety_accepted": false
|
||||
}
|
||||
}
|
||||
@@ -75,6 +75,48 @@ device protocol, transport, identity, or command state.
|
||||
12. A plugin UI contribution is not ownership of Mission Core navigation,
|
||||
scene layout, persistence, or global application state.
|
||||
|
||||
## Object understanding projection
|
||||
|
||||
Mission Core also has one bounded, executable perception vocabulary profile:
|
||||
`missioncore.urban-object-semantics/v0`. It is referenced by the separate
|
||||
`missioncore.object-understanding/v1` document and remains experimental. This
|
||||
is a versioned projection over an existing `ObstacleObservation`; it is not a
|
||||
second occupancy model, a long-term identity store, or a runtime ontology
|
||||
service.
|
||||
|
||||
| Term | Meaning |
|
||||
|---|---|
|
||||
| `ObjectUnderstanding` | A source/frame-bound composition of immutable geometry, ranked semantic hypotheses, explicit semantic resolution, object state, advisory risk and evidence provenance. |
|
||||
| `SemanticHypothesis` | One provider label mapped to a canonical class candidate with confidence, rank and evidence IDs. It is not class truth. |
|
||||
| `SemanticDecision` | An explicit `unresolved`, `selected`, `ambiguous` or `conflict` result over the ranked hypotheses. |
|
||||
| `ObjectStateEstimate` | Observed motion plus a separately qualified agency claim. A class prior is carried as a prior and never rewritten as observed motion. |
|
||||
| `AdvisoryRiskAssessment` | Evidence-qualified risk level and suggested response for a future policy consumer. It never grants planner, navigation, safety or actuation authority. |
|
||||
| `EvidenceProvenance` | The exact provider/model/revision/preprocess/prompt-set identity and source-frame binding supporting a claim. |
|
||||
|
||||
The v0 class profile is intentionally shallow. It covers the current named
|
||||
consumers: human/adult/child, animal/dog, vehicle subtypes, passive wheeled
|
||||
objects, common small static street obstacles, terrain discontinuities and
|
||||
natural debris. Unknown and group-level classes remain valid outcomes; the
|
||||
profile does not require a fragile leaf-class decision.
|
||||
|
||||
Mandatory object-perception non-conflation rules:
|
||||
|
||||
1. Occupancy identity is owned by `ObstacleObservation`, not by a semantic
|
||||
class, detector label or provider tracklet.
|
||||
2. A missing or unresolved semantic class never means free space and never
|
||||
removes an occupied obstacle.
|
||||
3. A raw provider label is not a canonical class; a ranked class hypothesis is
|
||||
not a selected class; a selected class is not independent truth.
|
||||
4. Observed motion is not a class motion prior. In particular, a currently
|
||||
stationary vehicle may retain a self-propelled agency prior.
|
||||
5. Semantic risk is advisory. `reduce-speed`, `yield`, `stop` and
|
||||
`route-around` values describe policy hints, not issued commands.
|
||||
6. Vehicle-dependent traversability is not encoded as an intrinsic property
|
||||
of a curb, step, hemisphere or other obstacle class.
|
||||
7. The class-free M4.8 obstacle-quality truth remains class-free. Semantic
|
||||
evaluation requires a separate sealed label generation and cannot rewrite
|
||||
an existing M4.8 result.
|
||||
|
||||
## Deferred compatibility work
|
||||
|
||||
The following meanings are intentionally unresolved in v0alpha2. They must be
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
# M48S Mask Grounding DINO shadow report
|
||||
|
||||
Date: 2026-08-25
|
||||
Status: completed shadow; rejected for semantic and navigation authority
|
||||
|
||||
## Question
|
||||
|
||||
Can a ready open-vocabulary instance-segmentation model name the urban obstacles
|
||||
already discovered by Mission Core geometry, without a manual annotation program?
|
||||
|
||||
## Frozen setup
|
||||
|
||||
- Source: the same 11 raw 800 × 600 KB4 frames and 73 occupied-geometry
|
||||
observations admitted by the M4.8 semantic slice.
|
||||
- Model: NVIDIA TAO Mask Grounding DINO Swin-Tiny research deployable v2.0,
|
||||
ONNX SHA-256
|
||||
`4e8b066cf7782ae6b4269255b861c205f4b07b33ff752e5eaf297c42ecfb2f5e`.
|
||||
- Runtime: TAO 7.0.1 deploy container on the Worker RTX 4090; generated FP16
|
||||
TensorRT engine SHA-256
|
||||
`1d127107d15f18dcddafd7345a0a766acfee204d63c5ab697eaabada2c7afcd2`.
|
||||
- Prompts: separate static, agent, and vehicle groups. Adult, child, and dog were
|
||||
also rerun as three independent single-class probes on frames 253 and 1228.
|
||||
- Binding: masks were clustered at IoU ≥ 0.90 and bound only through exact
|
||||
projected LiDAR point identities. A mask had to cover ≥ 50% of one obstacle's
|
||||
projected points and ≤ 25% of every secondary obstacle.
|
||||
- Authority: false for ground truth, acceptance, commands, actuation, navigation,
|
||||
and safety throughout.
|
||||
|
||||
The model choice and TensorRT path follow NVIDIA's official
|
||||
[Mask Grounding DINO documentation](https://docs.nvidia.com/tao/tao-toolkit/latest/text/cv_finetuning/pytorch/instance_segmentation/mask_grounding_dino.html)
|
||||
and
|
||||
[TAO Deploy documentation](https://docs.nvidia.com/tao/tao-toolkit/latest/text/tao_deploy/mask_grounding_dino.html).
|
||||
The checkpoint is the official
|
||||
[NGC research deployable v2.0](https://catalog.ngc.nvidia.com/orgs/nvidia/tao/models/pretrained_mask_grounding_dino_v2/mask_grounding_dino_swin_tiny_research_deployable_v2.0/version-history).
|
||||
No commercial-model license was accepted on the user's behalf.
|
||||
|
||||
## Result
|
||||
|
||||
| Measure | Result |
|
||||
|---|---:|
|
||||
| Frames completed | 11 / 11 |
|
||||
| Geometry obstacles retained | 73 |
|
||||
| Raw mask detections | 25 |
|
||||
| Pixel-instance clusters after caption deduplication | 7 |
|
||||
| Instances exclusively bound to one LiDAR obstacle | 4 |
|
||||
| Instances covering multiple LiDAR obstacles | 2 |
|
||||
| Instances without occupied-geometry support | 1 |
|
||||
| Final named instances | 0 |
|
||||
| Adult single-class detections | 0 |
|
||||
| Child single-class detections | 0 |
|
||||
| Dog single-class detections | 0 |
|
||||
|
||||
The mask head was useful: four instances had strong exclusive LiDAR support.
|
||||
The language classification was not. The same near-identical mask was repeatedly
|
||||
named `trash bin`, `shopping cart`, `road sign`, and `truck`. After the required
|
||||
confidence margin, six of seven instances were ambiguous and one unresolved.
|
||||
|
||||
The full evidence result is
|
||||
`m48s-mask-grounding-dino-shadow-b0a37f265223b4138754f76d5d7b8d17e7c4f5395481990f169fccc1645baba8`.
|
||||
|
||||
## Decision
|
||||
|
||||
This research checkpoint is rejected as an object-name provider and is not a
|
||||
candidate for navigation or safety authority. The zero-result single-class probes
|
||||
show that prompt competition was not the reason adult, child, and dog were missed.
|
||||
|
||||
The next candidate separates the two jobs:
|
||||
|
||||
1. LiDAR geometry supplies point/box prompts to a promptable mask segmenter.
|
||||
2. An independent zero-shot image encoder ranks canonical Mission Core classes on
|
||||
the geometry-owned crop and masked crop.
|
||||
3. The name stays unresolved unless the two views agree and the class margin passes.
|
||||
4. Unknown objects remain occupied and route-around; no physical traversability
|
||||
model is introduced.
|
||||
|
||||
SAM 2 officially supports point and box prompts through its
|
||||
[image predictor](https://github.com/facebookresearch/sam2/blob/main/sam2/sam2_image_predictor.py).
|
||||
OpenCLIP is the current open implementation candidate for independent zero-shot
|
||||
crop classification; its official repository is
|
||||
[mlfoundations/open_clip](https://github.com/mlfoundations/open_clip).
|
||||
|
||||
No manual annotation program is authorized by this result. A small two-reviewer
|
||||
truth slice remains necessary only to score candidates; it is evidence QA, not a
|
||||
training-dataset commitment.
|
||||
@@ -0,0 +1,208 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Materialize geometry-owned raw-KB4 crops for the M48S Worker shadow."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import shutil
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
from k1link.perception.baseline import BASELINE_RECORDED_JOB_ID
|
||||
from k1link.perception.contracts import SourceEnvelope
|
||||
from k1link.perception.detector_replay_result import (
|
||||
read_detector_replay_result,
|
||||
require_m4_detector_replay_acceptance,
|
||||
)
|
||||
from k1link.perception.geometry import (
|
||||
DEFAULT_GEOMETRY_PROFILE_PATH,
|
||||
Ravnoves00GeometryAssociationProvider,
|
||||
RecordedGeometryStore,
|
||||
load_geometry_profile,
|
||||
)
|
||||
from k1link.perception.geometry_semantic_roi import (
|
||||
GeometrySemanticRoiProfile,
|
||||
build_geometry_semantic_rois,
|
||||
materialize_geometry_semantic_crop,
|
||||
)
|
||||
from k1link.perception.graph_validation import validate_observations
|
||||
from k1link.perception.providers import SourcePacket
|
||||
from k1link.perception.recorded_source import RECORDED_SOURCE_PACK_ID, RecordedFrameReference
|
||||
|
||||
FRAME_INDICES = (121, 131, 253, 275, 443, 463, 1094, 1228, 1454, 1856, 2386)
|
||||
DETECTOR_RESULT_ID = (
|
||||
"m4-detector-replay-11f83f2e0b81758ac2a5a5fc54e9d293b501678df5f6ef97b5c6069ba08605c5"
|
||||
)
|
||||
SCHEMA_VERSION = "missioncore.m48s-geometry-semantic-roi-package/v0"
|
||||
|
||||
|
||||
def main() -> int:
|
||||
repository = Path(__file__).resolve().parents[2]
|
||||
runtime = repository / ".runtime/compute-experiments/m48s-semantic-shadow"
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--repository-root", type=Path, default=repository)
|
||||
parser.add_argument(
|
||||
"--input-root",
|
||||
type=Path,
|
||||
default=runtime / "raw-11-valid-fov-fill-v1",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output-root",
|
||||
type=Path,
|
||||
default=runtime / "geometry-rois-v1",
|
||||
)
|
||||
arguments = parser.parse_args()
|
||||
root = arguments.repository_root.resolve(strict=True)
|
||||
inputs = arguments.input_root.resolve(strict=True)
|
||||
output = arguments.output_root.expanduser().absolute()
|
||||
if output.exists():
|
||||
raise RuntimeError("geometry semantic ROI package already exists")
|
||||
|
||||
detector = read_detector_replay_result(root / ".runtime/worker-results" / DETECTOR_RESULT_ID)
|
||||
require_m4_detector_replay_acceptance(detector)
|
||||
by_sequence = {item.sequence: item for item in detector.frames}
|
||||
if any(index not in by_sequence for index in FRAME_INDICES):
|
||||
raise RuntimeError("geometry semantic frame escaped the accepted detector timeline")
|
||||
geometry_profile_path = root / DEFAULT_GEOMETRY_PROFILE_PATH
|
||||
geometry_profile = load_geometry_profile(geometry_profile_path)
|
||||
store = RecordedGeometryStore.from_repository(root, profile=geometry_profile)
|
||||
provider = Ravnoves00GeometryAssociationProvider(store=store)
|
||||
roi_profile = GeometrySemanticRoiProfile()
|
||||
|
||||
staging = output.parent / f".{output.name}.{uuid.uuid4().hex}.tmp"
|
||||
staging.mkdir(mode=0o700, parents=True, exist_ok=False)
|
||||
try:
|
||||
input_destination = staging / "input"
|
||||
input_destination.mkdir(mode=0o700)
|
||||
frame_documents: list[dict[str, object]] = []
|
||||
artifacts: list[dict[str, object]] = []
|
||||
total_observations = 0
|
||||
total_rois = 0
|
||||
for frame_index in FRAME_INDICES:
|
||||
detector_frame = by_sequence[frame_index]
|
||||
if detector_frame.outcome != "completed":
|
||||
raise RuntimeError("accepted detector frame is incomplete")
|
||||
packet = _packet(detector_frame.envelope)
|
||||
observations = provider.associate(packet, ())
|
||||
validate_observations(packet, (), observations)
|
||||
frame = store.frame(packet)
|
||||
if frame is None:
|
||||
raise RuntimeError("selected geometry frame is unavailable")
|
||||
roi_frame = build_geometry_semantic_rois(
|
||||
frame=frame,
|
||||
observations=observations,
|
||||
profile=roi_profile,
|
||||
)
|
||||
source = inputs / f"frame-{frame_index:06d}.png"
|
||||
frame_rois: list[dict[str, object]] = []
|
||||
for roi in roi_frame.rois:
|
||||
destination = input_destination / roi.crop_name
|
||||
materialize_geometry_semantic_crop(
|
||||
image_path=source,
|
||||
roi=roi,
|
||||
destination=destination,
|
||||
)
|
||||
document = roi.to_dict()
|
||||
document["crop_sha256"] = _sha256(destination)
|
||||
document["crop_width"] = int(roi.crop_region.x_max - roi.crop_region.x_min)
|
||||
document["crop_height"] = int(roi.crop_region.y_max - roi.crop_region.y_min)
|
||||
frame_rois.append(document)
|
||||
artifacts.append(
|
||||
{
|
||||
"path": f"input/{roi.crop_name}",
|
||||
"sha256": document["crop_sha256"],
|
||||
}
|
||||
)
|
||||
frame_documents.append(
|
||||
{
|
||||
"frame_index": frame_index,
|
||||
"frame_id": detector_frame.envelope.frame_id,
|
||||
"source_image": source.name,
|
||||
"source_image_sha256": _sha256(source),
|
||||
"geometry_observation_count": len(observations),
|
||||
"roi_count": len(roi_frame.rois),
|
||||
"not_projected_observations": [
|
||||
item.to_dict() for item in roi_frame.not_projected_observations
|
||||
],
|
||||
"rois": frame_rois,
|
||||
}
|
||||
)
|
||||
total_observations += len(observations)
|
||||
total_rois += len(roi_frame.rois)
|
||||
identity = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"detector_result_id": detector.result_id,
|
||||
"geometry_profile_id": geometry_profile.profile_id,
|
||||
"geometry_profile_sha256": _sha256(geometry_profile_path),
|
||||
"roi_producer_sha256": _sha256(
|
||||
root / "src/k1link/perception/geometry_semantic_roi.py"
|
||||
),
|
||||
"frame_indices": list(FRAME_INDICES),
|
||||
"roi_profile": {
|
||||
"minimum_projected_points": roi_profile.minimum_projected_points,
|
||||
"minimum_crop_width": roi_profile.minimum_crop_width,
|
||||
"minimum_crop_height": roi_profile.minimum_crop_height,
|
||||
"padding_fraction": roi_profile.padding_fraction,
|
||||
"minimum_padding_pixels": roi_profile.minimum_padding_pixels,
|
||||
},
|
||||
"geometry_observation_count": total_observations,
|
||||
"roi_count": total_rois,
|
||||
"not_projected_observation_count": total_observations - total_rois,
|
||||
"frames": frame_documents,
|
||||
"artifacts": artifacts,
|
||||
"authority": _false_authority(),
|
||||
}
|
||||
digest = hashlib.sha256(_canonical_json(identity)).hexdigest()
|
||||
manifest = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"package_id": f"m48s-geometry-semantic-rois-{digest}",
|
||||
"identity_sha256": digest,
|
||||
"identity": identity,
|
||||
}
|
||||
(staging / "manifest.json").write_bytes(_canonical_json(manifest) + b"\n")
|
||||
staging.rename(output)
|
||||
except Exception:
|
||||
shutil.rmtree(staging, ignore_errors=True)
|
||||
raise
|
||||
print(json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=True))
|
||||
return 0
|
||||
|
||||
|
||||
def _packet(envelope: SourceEnvelope) -> SourcePacket:
|
||||
image = RecordedFrameReference(BASELINE_RECORDED_JOB_ID, envelope.sequence)
|
||||
geometry = RecordedFrameReference(RECORDED_SOURCE_PACK_ID, envelope.sequence)
|
||||
return SourcePacket(
|
||||
envelope=envelope,
|
||||
image_payload=image,
|
||||
registered_point_increment_payload=geometry,
|
||||
pose_payload=geometry,
|
||||
)
|
||||
|
||||
|
||||
def _canonical_json(value: object) -> bytes:
|
||||
return json.dumps(value, sort_keys=True, separators=(",", ":"), allow_nan=False).encode()
|
||||
|
||||
|
||||
def _sha256(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.resolve(strict=True).open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def _false_authority() -> dict[str, bool]:
|
||||
return {
|
||||
"ground_truth": False,
|
||||
"independent_truth": False,
|
||||
"candidate_accepted": False,
|
||||
"commands_enabled": False,
|
||||
"actuation_allowed": False,
|
||||
"navigation_or_safety_accepted": False,
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,91 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate bounded prompt-group specs for Mask Grounding DINO Worker replay."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import copy
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
|
||||
PROMPT_SUFFIXES = {
|
||||
"urban-static/v0": "static",
|
||||
"urban-agents/v0": "agents",
|
||||
"urban-vehicles/v0": "vehicles",
|
||||
}
|
||||
AGENT_PROBES = {
|
||||
"adult": "adult person",
|
||||
"child": "child",
|
||||
"dog": "dog",
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
repository = Path(__file__).resolve().parents[2]
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--repository-root", type=Path, default=repository)
|
||||
parser.add_argument(
|
||||
"--output-root",
|
||||
type=Path,
|
||||
default=(
|
||||
repository
|
||||
/ ".runtime/compute-experiments/m48s-semantic-shadow"
|
||||
/ "mask-grounding-dino-worker-specs-v0"
|
||||
),
|
||||
)
|
||||
arguments = parser.parse_args()
|
||||
root = arguments.repository_root.resolve(strict=True)
|
||||
output = arguments.output_root.expanduser().absolute()
|
||||
if output.exists():
|
||||
raise RuntimeError("Mask Grounding DINO Worker specs already exist")
|
||||
base = yaml.safe_load(
|
||||
(root / "config/perception/m48s-mask-grounding-dino-shadow-v0.yaml").read_text(
|
||||
"utf-8"
|
||||
)
|
||||
)
|
||||
profile = json.loads(
|
||||
(root / "config/perception/open-vocabulary-semantic-shadow-v0.json").read_text(
|
||||
"utf-8"
|
||||
)
|
||||
)
|
||||
if not isinstance(base, dict) or not isinstance(profile, dict):
|
||||
raise RuntimeError("M48S mask semantic source configs are incompatible")
|
||||
groups = profile.get("prompt_groups")
|
||||
if not isinstance(groups, list):
|
||||
raise RuntimeError("M48S semantic prompt groups are unavailable")
|
||||
output.mkdir(mode=0o700, parents=True)
|
||||
for raw_group in groups:
|
||||
if not isinstance(raw_group, dict):
|
||||
raise RuntimeError("M48S semantic prompt group is incompatible")
|
||||
prompt_id = raw_group.get("prompt_set_id")
|
||||
captions = raw_group.get("captions")
|
||||
if prompt_id not in PROMPT_SUFFIXES or not isinstance(captions, list):
|
||||
raise RuntimeError("M48S semantic prompt group escaped the bounded profile")
|
||||
suffix = PROMPT_SUFFIXES[prompt_id]
|
||||
spec = copy.deepcopy(base)
|
||||
spec["model_name"] = f"missioncore-m48s-mask-grounding-dino-{suffix}"
|
||||
spec["results_dir"] = f"/workspace/probe/results-{suffix}"
|
||||
spec["dataset"]["infer_data_sources"]["captions"] = captions
|
||||
spec["inference"]["results_dir"] = f"/workspace/probe/results-{suffix}"
|
||||
(output / f"mask-{suffix}.yaml").write_text(
|
||||
yaml.safe_dump(spec, sort_keys=False),
|
||||
"utf-8",
|
||||
)
|
||||
for suffix, caption in AGENT_PROBES.items():
|
||||
spec = copy.deepcopy(base)
|
||||
spec["model_name"] = f"missioncore-m48s-mask-grounding-dino-probe-{suffix}"
|
||||
spec["results_dir"] = f"/workspace/probe/results-{suffix}"
|
||||
spec["dataset"]["infer_data_sources"]["captions"] = [caption]
|
||||
spec["inference"]["results_dir"] = f"/workspace/probe/results-{suffix}"
|
||||
(output / f"mask-{suffix}.yaml").write_text(
|
||||
yaml.safe_dump(spec, sort_keys=False),
|
||||
"utf-8",
|
||||
)
|
||||
print(output)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,72 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Build the bounded geometry-first M48S semantic shadow result."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from k1link.perception.geometry_semantic_shadow_replay import (
|
||||
build_geometry_semantic_shadow_replay,
|
||||
)
|
||||
|
||||
FRAME_INDICES = (121, 131, 253, 275, 443, 463, 1094, 1228, 1454, 1856, 2386)
|
||||
DETECTOR_RESULT_ID = (
|
||||
"m4-detector-replay-11f83f2e0b81758ac2a5a5fc54e9d293b501678df5f6ef97b5c6069ba08605c5"
|
||||
)
|
||||
VALID_FOV_RESULT_ID = (
|
||||
"valid-fov-mask-b4dd8ddf2b87c1d520ee8a0868c4fea062d7c14d1bae73ccabd3abe1f3acbac2"
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
root = Path(__file__).resolve().parents[2]
|
||||
runtime = root / ".runtime/compute-experiments/m48s-semantic-shadow"
|
||||
worker = runtime / "worker-results/geometry-roi58-20260825T0804Z"
|
||||
result = build_geometry_semantic_shadow_replay(
|
||||
repository_root=root,
|
||||
profile_path=root / "config/perception/open-vocabulary-semantic-shadow-v0.json",
|
||||
vocabulary_path=root / "config/perception/object-semantic-vocabulary-v0.json",
|
||||
detector_result_root=root / ".runtime/worker-results" / DETECTOR_RESULT_ID,
|
||||
roi_package_root=runtime / "geometry-rois-v1",
|
||||
valid_fov_mask_path=(
|
||||
root / ".runtime/compute-experiments/e1/valid-fov" / VALID_FOV_RESULT_ID / "mask.png"
|
||||
),
|
||||
worker_result_roots={
|
||||
"urban-static/v0": worker / "results-static/trt_inference",
|
||||
"urban-agents/v0": worker / "results-agents/trt_inference",
|
||||
"urban-vehicles/v0": worker / "results-vehicles/trt_inference",
|
||||
},
|
||||
frame_indices=FRAME_INDICES,
|
||||
worker_execution={
|
||||
"worker_node": "DESKTOP-OPJ8J04",
|
||||
"gpu_name": "NVIDIA GeForce RTX 4090",
|
||||
"container_reference": "nvcr.io/nvidia/tao/tao-toolkit:7.0.1-deploy",
|
||||
"container_image_id": (
|
||||
"sha256:2a3095330dd83e4314aada21fc7d184fcdb442a9131d591e5593ca5497394e7a"
|
||||
),
|
||||
"remote_root": (
|
||||
"D:/NDC_MISSIONCORE/runtime/experiments/"
|
||||
"m48s-geometry-roi58-20260825T0804Z"
|
||||
),
|
||||
"network_observation": (
|
||||
"TAO performed Hugging Face metadata requests despite a populated cache"
|
||||
),
|
||||
"authority": {
|
||||
"ground_truth": False,
|
||||
"independent_truth": False,
|
||||
"candidate_accepted": False,
|
||||
"commands_enabled": False,
|
||||
"actuation_allowed": False,
|
||||
"navigation_or_safety_accepted": False,
|
||||
},
|
||||
},
|
||||
output_root=runtime / "geometry-first-results",
|
||||
)
|
||||
print(result.result_id)
|
||||
print(json.dumps(result.metrics, ensure_ascii=False, indent=2, sort_keys=True))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,323 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Build immutable Mask Grounding DINO × LiDAR shadow evidence for M48S."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
from k1link.perception.detector_replay_result import (
|
||||
read_detector_replay_result,
|
||||
require_m4_detector_replay_acceptance,
|
||||
)
|
||||
from k1link.perception.geometry import Ravnoves00GeometryAssociationProvider, RecordedGeometryStore
|
||||
from k1link.perception.geometry_math import project_map_points_kb4
|
||||
from k1link.perception.mask_grounding_semantics import (
|
||||
MaskBindingResolution,
|
||||
MaskGeometryBindingProfile,
|
||||
MaskGroundingDetection,
|
||||
MaskLabelResolution,
|
||||
bind_mask_instances_to_geometry,
|
||||
cluster_mask_instances,
|
||||
load_mask_grounding_evidence,
|
||||
resolve_mask_instance_label,
|
||||
)
|
||||
from k1link.perception.semantic_shadow_replay import semantic_replay_packet
|
||||
|
||||
SCHEMA: Final = "missioncore.m48s-mask-grounding-dino-shadow-analysis/v0"
|
||||
FRAME_INDICES: Final = (121, 131, 253, 275, 443, 463, 1094, 1228, 1454, 1856, 2386)
|
||||
DETECTOR_RESULT_ID: Final = (
|
||||
"m4-detector-replay-11f83f2e0b81758ac2a5a5fc54e9d293b501678df5f6ef97b5c6069ba08605c5"
|
||||
)
|
||||
GROUPS: Final = {
|
||||
"static": "urban-static/v0",
|
||||
"agents": "urban-agents/v0",
|
||||
"vehicles": "urban-vehicles/v0",
|
||||
}
|
||||
AUTHORITY: Final = {
|
||||
"ground_truth": False,
|
||||
"independent_truth": False,
|
||||
"candidate_accepted": False,
|
||||
"commands_enabled": False,
|
||||
"actuation_allowed": False,
|
||||
"navigation_or_safety_accepted": False,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
root = Path(__file__).resolve().parents[2]
|
||||
runtime = root / ".runtime/compute-experiments/m48s-semantic-shadow"
|
||||
split_root = (
|
||||
runtime
|
||||
/ "worker-results/mask-grounding-dino-split11-20260825T0833Z"
|
||||
).resolve(strict=True)
|
||||
agent_root = (
|
||||
runtime
|
||||
/ "worker-results/mask-grounding-dino-agent2-20260825T0836Z"
|
||||
).resolve(strict=True)
|
||||
profile_path = root / "config/perception/m48s-mask-grounding-dino-evidence-v0.json"
|
||||
profile_document = json.loads(profile_path.read_text("utf-8"))
|
||||
binding_profile = MaskGeometryBindingProfile(**profile_document["binding"])
|
||||
detector = read_detector_replay_result(
|
||||
root / ".runtime/worker-results" / DETECTOR_RESULT_ID
|
||||
)
|
||||
require_m4_detector_replay_acceptance(detector)
|
||||
detector_by_sequence = {item.sequence: item for item in detector.frames}
|
||||
store = RecordedGeometryStore.from_repository(root)
|
||||
geometry = Ravnoves00GeometryAssociationProvider(store=store)
|
||||
totals: Counter[str] = Counter()
|
||||
geometry_resolutions: Counter[str] = Counter()
|
||||
label_resolutions: Counter[str] = Counter()
|
||||
combined_resolutions: Counter[str] = Counter()
|
||||
frame_documents = []
|
||||
input_artifacts: dict[str, str] = {}
|
||||
evidence_writer_sha256 = _matching_evidence_writer_sha256(split_root, agent_root)
|
||||
for frame_index in FRAME_INDICES:
|
||||
detector_frame = detector_by_sequence[frame_index]
|
||||
packet = semantic_replay_packet(detector_frame.envelope)
|
||||
observations = geometry.associate(packet, ())
|
||||
geometry_frame = store.frame(packet)
|
||||
if geometry_frame is None:
|
||||
raise RuntimeError("selected M48S geometry frame is unavailable")
|
||||
projected = project_map_points_kb4(
|
||||
geometry_frame.points_map,
|
||||
position_map_xyz=geometry_frame.sensor_position_map,
|
||||
orientation_map_from_lidar_xyzw=geometry_frame.sensor_orientation_xyzw,
|
||||
profile=geometry_frame.projection,
|
||||
)
|
||||
detections: list[MaskGroundingDetection] = []
|
||||
source_file_sha256: str | None = None
|
||||
group_counts = {}
|
||||
for suffix, prompt_set_id in GROUPS.items():
|
||||
evidence_path = (
|
||||
split_root / f"results-{suffix}/masks/frame-{frame_index:06d}.npz"
|
||||
)
|
||||
evidence = load_mask_grounding_evidence(
|
||||
evidence_path,
|
||||
prompt_set_id=prompt_set_id,
|
||||
)
|
||||
if source_file_sha256 not in (None, evidence.source_file_sha256):
|
||||
raise RuntimeError("Mask Grounding DINO groups used different source rasters")
|
||||
source_file_sha256 = evidence.source_file_sha256
|
||||
detections.extend(evidence.detections)
|
||||
group_counts[prompt_set_id] = len(evidence.detections)
|
||||
input_artifacts[str(evidence_path.relative_to(root))] = _sha256(evidence_path)
|
||||
for name in ("experiment.yaml", "status.json"):
|
||||
artifact = split_root / f"results-{suffix}" / name
|
||||
input_artifacts[str(artifact.relative_to(root))] = _sha256(artifact)
|
||||
instances = cluster_mask_instances(tuple(detections), profile=binding_profile)
|
||||
bindings = bind_mask_instances_to_geometry(
|
||||
instances,
|
||||
observations=observations,
|
||||
projected=projected,
|
||||
profile=binding_profile,
|
||||
)
|
||||
decisions = tuple(
|
||||
resolve_mask_instance_label(item, profile=binding_profile) for item in instances
|
||||
)
|
||||
instance_documents = []
|
||||
for instance, binding, decision in zip(
|
||||
instances, bindings, decisions, strict=True
|
||||
):
|
||||
geometry_resolutions[binding.resolution.value] += 1
|
||||
label_resolutions[decision.resolution.value] += 1
|
||||
combined = _combined_resolution(binding.resolution, decision.resolution)
|
||||
combined_resolutions[combined] += 1
|
||||
instance_documents.append(
|
||||
{
|
||||
"instance_id": instance.instance_id,
|
||||
"member_detection_ids": [
|
||||
item.detection_id for item in instance.detections
|
||||
],
|
||||
"member_mask_sha256": sorted(
|
||||
{item.mask_sha256 for item in instance.detections}
|
||||
),
|
||||
"ranked_labels": [
|
||||
{"label": label, "confidence": confidence}
|
||||
for label, confidence in decision.ranked_labels
|
||||
],
|
||||
"geometry_binding": {
|
||||
"resolution": binding.resolution.value,
|
||||
"selected_observation_id": binding.selected_observation_id,
|
||||
"reason_code": binding.reason_code,
|
||||
"supports": [
|
||||
{
|
||||
"observation_id": item.observation_id,
|
||||
"projected_point_count": item.projected_point_count,
|
||||
"support_fraction": item.support_fraction,
|
||||
}
|
||||
for item in binding.supports
|
||||
],
|
||||
},
|
||||
"label_decision": {
|
||||
"resolution": decision.resolution.value,
|
||||
"selected_label": decision.selected_label,
|
||||
"reason_code": decision.reason_code,
|
||||
},
|
||||
"combined_resolution": combined,
|
||||
"authority": AUTHORITY,
|
||||
}
|
||||
)
|
||||
totals["frame_count"] += 1
|
||||
totals["geometry_observation_count"] += len(observations)
|
||||
totals["raw_detection_count"] += len(detections)
|
||||
totals["mask_instance_count"] += len(instances)
|
||||
frame_documents.append(
|
||||
{
|
||||
"frame_index": frame_index,
|
||||
"frame_id": detector_frame.envelope.frame_id,
|
||||
"source_file_sha256": source_file_sha256,
|
||||
"geometry_observation_count": len(observations),
|
||||
"group_detection_counts": group_counts,
|
||||
"instances": instance_documents,
|
||||
"authority": AUTHORITY,
|
||||
}
|
||||
)
|
||||
agent_probe = _agent_probe(agent_root, root=root, input_artifacts=input_artifacts)
|
||||
metrics = {
|
||||
"frames": {
|
||||
"requested": len(FRAME_INDICES),
|
||||
"completed": totals["frame_count"],
|
||||
},
|
||||
"geometry_observation_count": totals["geometry_observation_count"],
|
||||
"raw_detection_count": totals["raw_detection_count"],
|
||||
"mask_instance_count": totals["mask_instance_count"],
|
||||
"geometry_binding_resolution_counts": dict(sorted(geometry_resolutions.items())),
|
||||
"label_resolution_counts": dict(sorted(label_resolutions.items())),
|
||||
"combined_resolution_counts": dict(sorted(combined_resolutions.items())),
|
||||
"single_class_agent_probe": agent_probe,
|
||||
"authority": AUTHORITY,
|
||||
}
|
||||
frames_bytes = b"".join(
|
||||
_canonical_json(item) + b"\n" for item in frame_documents
|
||||
)
|
||||
identity = {
|
||||
"schema_version": SCHEMA,
|
||||
"profile_sha256": _sha256(profile_path),
|
||||
"detector_result_id": detector.result_id,
|
||||
"frame_indices": list(FRAME_INDICES),
|
||||
"worker_input_artifacts": dict(sorted(input_artifacts.items())),
|
||||
"evidence_writer_sha256": evidence_writer_sha256,
|
||||
"producer_sha256": {
|
||||
"mask_grounding_semantics.py": _sha256(
|
||||
root / "src/k1link/perception/mask_grounding_semantics.py"
|
||||
),
|
||||
"run_m48s_mask_grounding_dino_analysis.py": _sha256(Path(__file__))
|
||||
},
|
||||
"frames_sha256": hashlib.sha256(frames_bytes).hexdigest(),
|
||||
"metrics": metrics,
|
||||
"completed": totals["frame_count"] == len(FRAME_INDICES),
|
||||
"accepted": False,
|
||||
"authority": AUTHORITY,
|
||||
}
|
||||
result_id = "m48s-mask-grounding-dino-shadow-" + hashlib.sha256(
|
||||
_canonical_json(identity)
|
||||
).hexdigest()
|
||||
output = runtime / "mask-grounding-dino-results" / result_id
|
||||
if output.exists():
|
||||
raise RuntimeError("immutable Mask Grounding DINO result already exists")
|
||||
output.mkdir(mode=0o700, parents=True)
|
||||
(output / "frames.jsonl").write_bytes(frames_bytes)
|
||||
manifest = {"result_id": result_id, **identity}
|
||||
(output / "manifest.json").write_bytes(_canonical_json(manifest) + b"\n")
|
||||
report = {
|
||||
"schema_version": SCHEMA,
|
||||
"result_id": result_id,
|
||||
"created_at_utc": datetime.now(UTC).isoformat(timespec="milliseconds").replace(
|
||||
"+00:00", "Z"
|
||||
),
|
||||
"completed": identity["completed"],
|
||||
"accepted": False,
|
||||
"metrics": metrics,
|
||||
"decision": {
|
||||
"mask_instance_geometry_binding_completed": True,
|
||||
"semantic_quality_accepted": False,
|
||||
"agent_semantics_accepted": False,
|
||||
"reason_codes": [
|
||||
"single-class-adult-child-dog-probes-returned-zero-masks",
|
||||
"geometry-supported-instances-retained-conflicting-class-labels",
|
||||
"two-reviewer-independent-truth-unavailable",
|
||||
"research-checkpoint-not-for-commercial-deployment",
|
||||
],
|
||||
"next_candidate": (
|
||||
"geometry-prompted-mask-segmentation-plus-independent-zero-shot-"
|
||||
"crop-classifier"
|
||||
),
|
||||
},
|
||||
"authority": AUTHORITY,
|
||||
}
|
||||
(output / "report.json").write_bytes(_canonical_json(report) + b"\n")
|
||||
print(result_id)
|
||||
print(json.dumps(metrics, ensure_ascii=False, indent=2, sort_keys=True))
|
||||
return 0
|
||||
|
||||
|
||||
def _combined_resolution(
|
||||
geometry: MaskBindingResolution,
|
||||
label: MaskLabelResolution,
|
||||
) -> str:
|
||||
if geometry is MaskBindingResolution.UNRESOLVED:
|
||||
return "unresolved"
|
||||
if geometry is MaskBindingResolution.AMBIGUOUS:
|
||||
return "ambiguous"
|
||||
return label.value
|
||||
|
||||
|
||||
def _agent_probe(
|
||||
agent_root: Path,
|
||||
*,
|
||||
root: Path,
|
||||
input_artifacts: dict[str, str],
|
||||
) -> dict[str, object]:
|
||||
counts: Counter[str] = Counter()
|
||||
for label in ("adult", "child", "dog"):
|
||||
for frame_index in (253, 1228):
|
||||
path = agent_root / f"results-{label}/masks/frame-{frame_index:06d}.npz"
|
||||
evidence = load_mask_grounding_evidence(
|
||||
path,
|
||||
prompt_set_id=f"agent-probe-{label}/v0",
|
||||
)
|
||||
counts[label] += len(evidence.detections)
|
||||
input_artifacts[str(path.relative_to(root))] = _sha256(path)
|
||||
for name in ("experiment.yaml", "status.json"):
|
||||
artifact = agent_root / f"results-{label}" / name
|
||||
input_artifacts[str(artifact.relative_to(root))] = _sha256(artifact)
|
||||
return {
|
||||
"frame_indices": [253, 1228],
|
||||
"detection_counts": dict(sorted(counts.items())),
|
||||
"passed": False,
|
||||
}
|
||||
|
||||
|
||||
def _matching_evidence_writer_sha256(split_root: Path, agent_root: Path) -> str:
|
||||
split = _sha256(split_root / "sitecustomize.py")
|
||||
agent = _sha256(agent_root / "sitecustomize.py")
|
||||
if split != agent:
|
||||
raise RuntimeError("Mask Grounding DINO evidence writers changed between runs")
|
||||
return split
|
||||
|
||||
|
||||
def _sha256(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as stream:
|
||||
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def _canonical_json(value: object) -> bytes:
|
||||
return json.dumps(
|
||||
value,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode("utf-8")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,114 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Persist raw Mask Grounding DINO tensors before TAO visualization.
|
||||
|
||||
This Worker-only wrapper leaves NVIDIA's inference and postprocessing intact. It
|
||||
intercepts the documented OD drawing seam solely to persist the already-filtered
|
||||
class, score, box and binary-mask evidence in a lossless NPZ ledger.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
from nvidia_tao_deploy.cv.mask_grounding_dino.entrypoint.mask_grounding_dino import main
|
||||
from nvidia_tao_deploy.cv.mask_grounding_dino.inferencer import MaskGDINOInferencer
|
||||
from PIL import Image
|
||||
|
||||
INPUT_ROOT = Path(os.environ.get("M48S_INPUT_ROOT", "/workspace/probe/input"))
|
||||
MASK_ROOT = Path(os.environ.get("M48S_MASK_ROOT", "/workspace/probe/results/masks"))
|
||||
|
||||
|
||||
def _pixel_sha256(image: Image.Image) -> str:
|
||||
digest = hashlib.sha256()
|
||||
normalized = image.convert("RGB")
|
||||
digest.update(normalized.width.to_bytes(4, "big"))
|
||||
digest.update(normalized.height.to_bytes(4, "big"))
|
||||
digest.update(normalized.tobytes())
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def _source_index() -> dict[str, tuple[str, str]]:
|
||||
result: dict[str, tuple[str, str]] = {}
|
||||
for path in sorted(INPUT_ROOT.glob("*.png")):
|
||||
with Image.open(path) as image:
|
||||
pixel_sha256 = _pixel_sha256(image)
|
||||
file_sha256 = hashlib.sha256(path.read_bytes()).hexdigest()
|
||||
if pixel_sha256 in result:
|
||||
raise RuntimeError("M48S mask input rasters are duplicated")
|
||||
result[pixel_sha256] = (path.stem, file_sha256)
|
||||
if not result:
|
||||
raise RuntimeError("M48S mask input root is empty")
|
||||
return result
|
||||
|
||||
|
||||
SOURCE_INDEX = _source_index()
|
||||
ORIGINAL_DRAW_BBOX = MaskGDINOInferencer.draw_bbox
|
||||
|
||||
|
||||
def _evidence_draw_bbox(
|
||||
self: MaskGDINOInferencer,
|
||||
image: Image.Image,
|
||||
prediction: np.ndarray[Any, Any],
|
||||
masks: np.ndarray[Any, Any],
|
||||
class_mapping: dict[int, str],
|
||||
threshold: float = 0.3,
|
||||
color_map: dict[str, object] | None = None,
|
||||
) -> tuple[Image.Image, list[str]]:
|
||||
source = SOURCE_INDEX.get(_pixel_sha256(image))
|
||||
if source is None:
|
||||
raise RuntimeError("M48S mask inference image escaped the admitted input set")
|
||||
stem, source_file_sha256 = source
|
||||
if masks.ndim != 3 or masks.shape[-1] != prediction.shape[0]:
|
||||
raise RuntimeError("M48S mask output shape is incompatible")
|
||||
selected = tuple(
|
||||
index
|
||||
for index, item in enumerate(prediction)
|
||||
if int(item[0]) in class_mapping and float(item[1]) >= threshold
|
||||
)
|
||||
class_ids = np.asarray([int(prediction[index, 0]) for index in selected], dtype=np.int16)
|
||||
class_names = np.asarray([class_mapping[int(item)] for item in class_ids], dtype="U128")
|
||||
scores = np.asarray([prediction[index, 1] for index in selected], dtype=np.float32)
|
||||
boxes = np.asarray(
|
||||
[prediction[index, 2:6] for index in selected], dtype=np.float32
|
||||
).reshape((-1, 4))
|
||||
if selected:
|
||||
binary_masks = np.transpose(masks[..., selected] > 0.5, (2, 0, 1)).astype(np.uint8)
|
||||
else:
|
||||
binary_masks = np.empty((0, masks.shape[0], masks.shape[1]), dtype=np.uint8)
|
||||
MASK_ROOT.mkdir(mode=0o700, parents=True, exist_ok=True)
|
||||
destination = MASK_ROOT / f"{stem}.npz"
|
||||
if destination.exists():
|
||||
raise RuntimeError("M48S mask evidence destination already exists")
|
||||
np.savez_compressed(
|
||||
destination,
|
||||
schema_version=np.asarray("missioncore.m48s-mask-grounding-dino-evidence/v0"),
|
||||
source_file_sha256=np.asarray(source_file_sha256),
|
||||
source_pixel_sha256=np.asarray(_pixel_sha256(image)),
|
||||
class_ids=class_ids,
|
||||
class_names=class_names,
|
||||
scores=scores,
|
||||
boxes_xyxy=boxes,
|
||||
masks=binary_masks,
|
||||
)
|
||||
return ORIGINAL_DRAW_BBOX(
|
||||
self,
|
||||
image,
|
||||
prediction,
|
||||
masks,
|
||||
class_mapping,
|
||||
threshold,
|
||||
color_map,
|
||||
)
|
||||
|
||||
|
||||
MaskGDINOInferencer.draw_bbox = _evidence_draw_bbox
|
||||
print("M48S raw-mask evidence hook installed", file=sys.stderr)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,78 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Build the bounded raw-KB4 M48S semantic shadow result."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from k1link.perception.semantic_shadow_replay import build_semantic_shadow_replay
|
||||
|
||||
FRAME_INDICES = (121, 131, 253, 275, 443, 463, 1094, 1228, 1454, 1856, 2386)
|
||||
DETECTOR_RESULT_ID = (
|
||||
"m4-detector-replay-11f83f2e0b81758ac2a5a5fc54e9d293b501678df5f6ef97b5c6069ba08605c5"
|
||||
)
|
||||
VALID_FOV_RESULT_ID = (
|
||||
"valid-fov-mask-b4dd8ddf2b87c1d520ee8a0868c4fea062d7c14d1bae73ccabd3abe1f3acbac2"
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
repository = Path(__file__).resolve().parents[2]
|
||||
default_runtime = (
|
||||
repository
|
||||
/ ".runtime/compute-experiments/m48s-semantic-shadow/worker-results"
|
||||
/ "fovfill11-20260825T0755Z"
|
||||
)
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--repository-root", type=Path, default=repository)
|
||||
parser.add_argument("--worker-results", type=Path, default=default_runtime)
|
||||
parser.add_argument(
|
||||
"--source-frames",
|
||||
type=Path,
|
||||
default=(repository / ".runtime/compute-experiments/m48s-semantic-shadow/raw-11-frames-v1"),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--inference-frames",
|
||||
type=Path,
|
||||
default=(
|
||||
repository
|
||||
/ ".runtime/compute-experiments/m48s-semantic-shadow"
|
||||
/ "raw-11-valid-fov-fill-v1"
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output-root",
|
||||
type=Path,
|
||||
default=(repository / ".runtime/compute-experiments/m48s-semantic-shadow/results"),
|
||||
)
|
||||
arguments = parser.parse_args()
|
||||
root = arguments.repository_root.resolve(strict=True)
|
||||
worker = arguments.worker_results.resolve(strict=True)
|
||||
result = build_semantic_shadow_replay(
|
||||
repository_root=root,
|
||||
profile_path=root / "config/perception/open-vocabulary-semantic-shadow-v0.json",
|
||||
vocabulary_path=root / "config/perception/object-semantic-vocabulary-v0.json",
|
||||
detector_result_root=root / ".runtime/worker-results" / DETECTOR_RESULT_ID,
|
||||
source_frames_root=arguments.source_frames,
|
||||
inference_frames_root=arguments.inference_frames,
|
||||
valid_fov_mask_path=(
|
||||
root / ".runtime/compute-experiments/e1/valid-fov" / VALID_FOV_RESULT_ID / "mask.png"
|
||||
),
|
||||
worker_result_roots={
|
||||
"urban-static/v0": worker / "results-static/trt_inference",
|
||||
"urban-agents/v0": worker / "results-agents/trt_inference",
|
||||
"urban-vehicles/v0": worker / "results-vehicles/trt_inference",
|
||||
},
|
||||
worker_identity_path=worker / "worker-identity.json",
|
||||
frame_indices=FRAME_INDICES,
|
||||
output_root=arguments.output_root,
|
||||
)
|
||||
print(result.result_id)
|
||||
print(json.dumps(result.metrics, ensure_ascii=False, indent=2, sort_keys=True))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,454 @@
|
||||
"""Separate semantic-class evaluation over geometry-bound object projections.
|
||||
|
||||
This companion contour does not mutate the class-free M4.8 truth or score
|
||||
object presence. It evaluates canonical semantic resolution only after a
|
||||
separate two-reviewer, adjudicated semantic label set exists.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
from k1link.perception.object_understanding import (
|
||||
ObjectSemanticVocabulary,
|
||||
ObjectUnderstanding,
|
||||
SemanticResolution,
|
||||
validate_object_understanding,
|
||||
)
|
||||
|
||||
SEMANTIC_TRUTH_LABEL_SCHEMA: Final = "missioncore.semantic-object-truth-label/v0"
|
||||
SEMANTIC_QUALITY_PROFILE_SCHEMA: Final = "missioncore.semantic-object-quality-profile/v0"
|
||||
SEMANTIC_QUALITY_REPORT_SCHEMA: Final = "missioncore.semantic-object-quality-report/v0"
|
||||
|
||||
_IDENTIFIER = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:/-]{0,159}$")
|
||||
_FALSE_AUTHORITY: Final = {
|
||||
"physical_live": False,
|
||||
"commands_enabled": False,
|
||||
"actuation_allowed": False,
|
||||
"navigation_or_safety_accepted": False,
|
||||
}
|
||||
|
||||
|
||||
class SemanticObjectQualityError(ValueError):
|
||||
"""A semantic truth, profile or evaluation input is incompatible."""
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SemanticTruthLabel:
|
||||
"""One independently reviewed canonical class bound to existing geometry."""
|
||||
|
||||
label_id: str
|
||||
observation_id: str
|
||||
source_id: str
|
||||
frame_id: str
|
||||
class_id: str
|
||||
reviewer_count: int
|
||||
adjudicated: bool
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
for value, label in (
|
||||
(self.label_id, "semantic truth label id"),
|
||||
(self.observation_id, "semantic truth observation id"),
|
||||
(self.source_id, "semantic truth source id"),
|
||||
(self.frame_id, "semantic truth frame id"),
|
||||
(self.class_id, "semantic truth class id"),
|
||||
):
|
||||
_identifier(value, label)
|
||||
if (
|
||||
not isinstance(self.reviewer_count, int)
|
||||
or isinstance(self.reviewer_count, bool)
|
||||
or self.reviewer_count < 2
|
||||
):
|
||||
raise SemanticObjectQualityError("semantic truth requires two independent reviewers")
|
||||
if self.adjudicated is not True:
|
||||
raise SemanticObjectQualityError("semantic truth must be adjudicated")
|
||||
|
||||
def to_dict(self) -> dict[str, object]:
|
||||
return {
|
||||
"schema_version": SEMANTIC_TRUTH_LABEL_SCHEMA,
|
||||
"label_id": self.label_id,
|
||||
"observation_id": self.observation_id,
|
||||
"source_id": self.source_id,
|
||||
"frame_id": self.frame_id,
|
||||
"class_id": self.class_id,
|
||||
"reviewer_count": self.reviewer_count,
|
||||
"adjudicated": self.adjudicated,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: object) -> SemanticTruthLabel:
|
||||
document = _object(value, "semantic truth label")
|
||||
_exact_keys(
|
||||
document,
|
||||
{
|
||||
"schema_version",
|
||||
"label_id",
|
||||
"observation_id",
|
||||
"source_id",
|
||||
"frame_id",
|
||||
"class_id",
|
||||
"reviewer_count",
|
||||
"adjudicated",
|
||||
},
|
||||
"semantic truth label",
|
||||
)
|
||||
if document.get("schema_version") != SEMANTIC_TRUTH_LABEL_SCHEMA:
|
||||
raise SemanticObjectQualityError("semantic truth label schema is incompatible")
|
||||
return cls(
|
||||
label_id=_string(document, "label_id"),
|
||||
observation_id=_string(document, "observation_id"),
|
||||
source_id=_string(document, "source_id"),
|
||||
frame_id=_string(document, "frame_id"),
|
||||
class_id=_string(document, "class_id"),
|
||||
reviewer_count=_integer(document, "reviewer_count"),
|
||||
adjudicated=_boolean(document, "adjudicated"),
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SemanticObjectQualityProfile:
|
||||
"""Bounded evaluation policy for a frozen semantic provider candidate."""
|
||||
|
||||
profile_id: str
|
||||
vocabulary_id: str
|
||||
top_k: int
|
||||
minimum_prediction_coverage: float
|
||||
minimum_exact_top1_accuracy: float
|
||||
minimum_coarse_group_accuracy: float
|
||||
minimum_exact_top_k_recall: float
|
||||
maximum_unresolved_fraction: float
|
||||
maximum_conflict_fraction: float
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_identifier(self.profile_id, "semantic quality profile id")
|
||||
_identifier(self.vocabulary_id, "semantic quality vocabulary id")
|
||||
if (
|
||||
not isinstance(self.top_k, int)
|
||||
or isinstance(self.top_k, bool)
|
||||
or not 1 <= self.top_k <= 5
|
||||
):
|
||||
raise SemanticObjectQualityError("semantic quality top-k is invalid")
|
||||
for value, label in (
|
||||
(self.minimum_prediction_coverage, "minimum prediction coverage"),
|
||||
(self.minimum_exact_top1_accuracy, "minimum exact top-1 accuracy"),
|
||||
(self.minimum_coarse_group_accuracy, "minimum coarse group accuracy"),
|
||||
(self.minimum_exact_top_k_recall, "minimum exact top-k recall"),
|
||||
(self.maximum_unresolved_fraction, "maximum unresolved fraction"),
|
||||
(self.maximum_conflict_fraction, "maximum conflict fraction"),
|
||||
):
|
||||
_fraction(value, label)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SemanticQualityCase:
|
||||
"""One deterministic truth/prediction comparison row."""
|
||||
|
||||
label_id: str
|
||||
observation_id: str
|
||||
truth_class_id: str
|
||||
resolution: str
|
||||
selected_class_id: str | None
|
||||
exact_top1_correct: bool
|
||||
coarse_group_correct: bool
|
||||
exact_top_k_hit: bool
|
||||
|
||||
def to_dict(self) -> dict[str, object]:
|
||||
return {
|
||||
"label_id": self.label_id,
|
||||
"observation_id": self.observation_id,
|
||||
"truth_class_id": self.truth_class_id,
|
||||
"resolution": self.resolution,
|
||||
"selected_class_id": self.selected_class_id,
|
||||
"exact_top1_correct": self.exact_top1_correct,
|
||||
"coarse_group_correct": self.coarse_group_correct,
|
||||
"exact_top_k_hit": self.exact_top_k_hit,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SemanticObjectQualityResult:
|
||||
"""In-memory deterministic semantic report and comparison ledger."""
|
||||
|
||||
report: dict[str, object]
|
||||
cases: tuple[SemanticQualityCase, ...]
|
||||
|
||||
|
||||
def load_semantic_object_quality_profile(path: Path) -> SemanticObjectQualityProfile:
|
||||
"""Read a strict semantic evaluation profile."""
|
||||
|
||||
try:
|
||||
document = json.loads(path.expanduser().resolve(strict=True).read_text("utf-8"))
|
||||
except (OSError, json.JSONDecodeError) as exc:
|
||||
raise SemanticObjectQualityError("semantic quality profile cannot be read") from exc
|
||||
root = _object(document, "semantic quality profile")
|
||||
_exact_keys(
|
||||
root,
|
||||
{
|
||||
"schema_version",
|
||||
"profile_id",
|
||||
"vocabulary_id",
|
||||
"top_k",
|
||||
"thresholds",
|
||||
"scope",
|
||||
"authority",
|
||||
},
|
||||
"semantic quality profile",
|
||||
)
|
||||
if root.get("schema_version") != SEMANTIC_QUALITY_PROFILE_SCHEMA:
|
||||
raise SemanticObjectQualityError("semantic quality profile schema is incompatible")
|
||||
scope = _object(root.get("scope"), "semantic quality scope")
|
||||
_exact_keys(
|
||||
scope,
|
||||
{
|
||||
"object_presence_scored",
|
||||
"semantic_class_scored",
|
||||
"risk_policy_scored",
|
||||
"requires_separate_adjudicated_semantic_truth",
|
||||
},
|
||||
"semantic quality scope",
|
||||
)
|
||||
if scope != {
|
||||
"object_presence_scored": False,
|
||||
"semantic_class_scored": True,
|
||||
"risk_policy_scored": False,
|
||||
"requires_separate_adjudicated_semantic_truth": True,
|
||||
}:
|
||||
raise SemanticObjectQualityError("semantic quality scope changed")
|
||||
if _object(root.get("authority"), "semantic quality authority") != _FALSE_AUTHORITY:
|
||||
raise SemanticObjectQualityError("semantic quality authority changed")
|
||||
thresholds = _object(root.get("thresholds"), "semantic quality thresholds")
|
||||
_exact_keys(
|
||||
thresholds,
|
||||
{
|
||||
"minimum_prediction_coverage",
|
||||
"minimum_exact_top1_accuracy",
|
||||
"minimum_coarse_group_accuracy",
|
||||
"minimum_exact_top_k_recall",
|
||||
"maximum_unresolved_fraction",
|
||||
"maximum_conflict_fraction",
|
||||
},
|
||||
"semantic quality thresholds",
|
||||
)
|
||||
return SemanticObjectQualityProfile(
|
||||
profile_id=_string(root, "profile_id"),
|
||||
vocabulary_id=_string(root, "vocabulary_id"),
|
||||
top_k=_integer(root, "top_k"),
|
||||
minimum_prediction_coverage=_number(thresholds, "minimum_prediction_coverage"),
|
||||
minimum_exact_top1_accuracy=_number(thresholds, "minimum_exact_top1_accuracy"),
|
||||
minimum_coarse_group_accuracy=_number(thresholds, "minimum_coarse_group_accuracy"),
|
||||
minimum_exact_top_k_recall=_number(thresholds, "minimum_exact_top_k_recall"),
|
||||
maximum_unresolved_fraction=_number(thresholds, "maximum_unresolved_fraction"),
|
||||
maximum_conflict_fraction=_number(thresholds, "maximum_conflict_fraction"),
|
||||
)
|
||||
|
||||
|
||||
def score_semantic_object_quality(
|
||||
*,
|
||||
predictions: tuple[ObjectUnderstanding, ...],
|
||||
truth: tuple[SemanticTruthLabel, ...],
|
||||
vocabulary: ObjectSemanticVocabulary,
|
||||
profile: SemanticObjectQualityProfile,
|
||||
) -> SemanticObjectQualityResult:
|
||||
"""Score semantics on geometry-bound truth without scoring object presence."""
|
||||
|
||||
if not truth or any(not isinstance(item, SemanticTruthLabel) for item in truth):
|
||||
raise SemanticObjectQualityError("semantic truth set is invalid")
|
||||
if any(not isinstance(item, ObjectUnderstanding) for item in predictions):
|
||||
raise SemanticObjectQualityError("semantic prediction set is invalid")
|
||||
if profile.vocabulary_id != vocabulary.vocabulary_id:
|
||||
raise SemanticObjectQualityError("semantic quality vocabulary changed")
|
||||
truth_ids = tuple(item.observation_id for item in truth)
|
||||
if len(set(truth_ids)) != len(truth_ids):
|
||||
raise SemanticObjectQualityError("semantic truth observations must be unique")
|
||||
prediction_ids = tuple(item.observation.observation_id for item in predictions)
|
||||
if len(set(prediction_ids)) != len(prediction_ids):
|
||||
raise SemanticObjectQualityError("semantic prediction observations must be unique")
|
||||
declared_classes = {item.class_id for item in vocabulary.classes}
|
||||
if any(item.class_id not in declared_classes for item in truth):
|
||||
raise SemanticObjectQualityError("semantic truth uses an undeclared class")
|
||||
for candidate in predictions:
|
||||
validate_object_understanding(candidate, vocabulary)
|
||||
|
||||
by_observation = {item.observation.observation_id: item for item in predictions}
|
||||
cases: list[SemanticQualityCase] = []
|
||||
predicted_count = 0
|
||||
selected_count = 0
|
||||
conflict_count = 0
|
||||
for label in truth:
|
||||
prediction = by_observation.get(label.observation_id)
|
||||
if prediction is None:
|
||||
cases.append(
|
||||
SemanticQualityCase(
|
||||
label_id=label.label_id,
|
||||
observation_id=label.observation_id,
|
||||
truth_class_id=label.class_id,
|
||||
resolution="unavailable",
|
||||
selected_class_id=None,
|
||||
exact_top1_correct=False,
|
||||
coarse_group_correct=False,
|
||||
exact_top_k_hit=False,
|
||||
)
|
||||
)
|
||||
continue
|
||||
predicted_count += 1
|
||||
observation = prediction.observation
|
||||
if observation.source_id != label.source_id or observation.frame_id != label.frame_id:
|
||||
raise SemanticObjectQualityError("semantic prediction escaped its truth source frame")
|
||||
selected = prediction.semantic.selected_class_id
|
||||
if prediction.semantic.resolution is SemanticResolution.SELECTED:
|
||||
selected_count += 1
|
||||
if prediction.semantic.resolution is SemanticResolution.CONFLICT:
|
||||
conflict_count += 1
|
||||
exact = selected == label.class_id
|
||||
coarse = selected is not None and _coarse_group(selected, vocabulary) == _coarse_group(
|
||||
label.class_id, vocabulary
|
||||
)
|
||||
top_k_ids = {item.class_id for item in prediction.hypotheses[: profile.top_k]}
|
||||
cases.append(
|
||||
SemanticQualityCase(
|
||||
label_id=label.label_id,
|
||||
observation_id=label.observation_id,
|
||||
truth_class_id=label.class_id,
|
||||
resolution=prediction.semantic.resolution.value,
|
||||
selected_class_id=selected,
|
||||
exact_top1_correct=exact,
|
||||
coarse_group_correct=coarse,
|
||||
exact_top_k_hit=label.class_id in top_k_ids,
|
||||
)
|
||||
)
|
||||
|
||||
truth_count = len(truth)
|
||||
exact_accuracy = sum(item.exact_top1_correct for item in cases) / truth_count
|
||||
coarse_accuracy = sum(item.coarse_group_correct for item in cases) / truth_count
|
||||
top_k_recall = sum(item.exact_top_k_hit for item in cases) / truth_count
|
||||
prediction_coverage = predicted_count / truth_count
|
||||
unresolved_fraction = (truth_count - selected_count) / truth_count
|
||||
conflict_fraction = conflict_count / truth_count
|
||||
gates = {
|
||||
"prediction_coverage": (prediction_coverage >= profile.minimum_prediction_coverage),
|
||||
"exact_top1_accuracy": (exact_accuracy >= profile.minimum_exact_top1_accuracy),
|
||||
"coarse_group_accuracy": (coarse_accuracy >= profile.minimum_coarse_group_accuracy),
|
||||
"exact_top_k_recall": (top_k_recall >= profile.minimum_exact_top_k_recall),
|
||||
"unresolved_fraction": (unresolved_fraction <= profile.maximum_unresolved_fraction),
|
||||
"conflict_fraction": (conflict_fraction <= profile.maximum_conflict_fraction),
|
||||
}
|
||||
report: dict[str, object] = {
|
||||
"schema_version": SEMANTIC_QUALITY_REPORT_SCHEMA,
|
||||
"profile_id": profile.profile_id,
|
||||
"vocabulary_id": vocabulary.vocabulary_id,
|
||||
"scope": {
|
||||
"object_presence_scored": False,
|
||||
"semantic_class_scored": True,
|
||||
"risk_policy_scored": False,
|
||||
},
|
||||
"metrics": {
|
||||
"truth_count": truth_count,
|
||||
"prediction_count": predicted_count,
|
||||
"selected_count": selected_count,
|
||||
"conflict_count": conflict_count,
|
||||
"prediction_coverage": prediction_coverage,
|
||||
"exact_top1_accuracy": exact_accuracy,
|
||||
"coarse_group_accuracy": coarse_accuracy,
|
||||
"exact_top_k_recall": top_k_recall,
|
||||
"unresolved_fraction": unresolved_fraction,
|
||||
"conflict_fraction": conflict_fraction,
|
||||
},
|
||||
"gates": gates,
|
||||
"candidate_semantic_gate_passed": all(gates.values()),
|
||||
"limitations": [
|
||||
"Object presence and geometry quality remain owned by the class-free contour.",
|
||||
"Semantic qualification does not qualify risk policy or physical motion.",
|
||||
"No navigation, safety, command or actuation authority is granted.",
|
||||
],
|
||||
"authority": dict(_FALSE_AUTHORITY),
|
||||
}
|
||||
return SemanticObjectQualityResult(report=report, cases=tuple(cases))
|
||||
|
||||
|
||||
def _coarse_group(
|
||||
class_id: str,
|
||||
vocabulary: ObjectSemanticVocabulary,
|
||||
) -> str:
|
||||
ancestors = vocabulary.ancestors(class_id)
|
||||
if not ancestors:
|
||||
return class_id
|
||||
non_root = tuple(item for item in ancestors if item != "object.unknown")
|
||||
return non_root[-1] if non_root else class_id
|
||||
|
||||
|
||||
def _object(value: object, label: str) -> dict[str, object]:
|
||||
if not isinstance(value, dict) or any(not isinstance(key, str) for key in value):
|
||||
raise SemanticObjectQualityError(f"{label} must be an object")
|
||||
return value
|
||||
|
||||
|
||||
def _exact_keys(document: dict[str, object], expected: set[str], label: str) -> None:
|
||||
if set(document) != expected:
|
||||
raise SemanticObjectQualityError(f"{label} fields are incompatible")
|
||||
|
||||
|
||||
def _string(document: dict[str, object], key: str) -> str:
|
||||
value = document.get(key)
|
||||
if not isinstance(value, str) or not value:
|
||||
raise SemanticObjectQualityError(f"{key} must be a nonempty string")
|
||||
return value
|
||||
|
||||
|
||||
def _integer(document: dict[str, object], key: str) -> int:
|
||||
value = document.get(key)
|
||||
if not isinstance(value, int) or isinstance(value, bool):
|
||||
raise SemanticObjectQualityError(f"{key} must be an integer")
|
||||
return value
|
||||
|
||||
|
||||
def _boolean(document: dict[str, object], key: str) -> bool:
|
||||
value = document.get(key)
|
||||
if not isinstance(value, bool):
|
||||
raise SemanticObjectQualityError(f"{key} must be boolean")
|
||||
return value
|
||||
|
||||
|
||||
def _number(document: dict[str, object], key: str) -> float:
|
||||
value = document.get(key)
|
||||
if (
|
||||
not isinstance(value, (int, float))
|
||||
or isinstance(value, bool)
|
||||
or not math.isfinite(float(value))
|
||||
):
|
||||
raise SemanticObjectQualityError(f"{key} must be finite")
|
||||
return float(value)
|
||||
|
||||
|
||||
def _fraction(value: object, label: str) -> float:
|
||||
if (
|
||||
not isinstance(value, (int, float))
|
||||
or isinstance(value, bool)
|
||||
or not math.isfinite(float(value))
|
||||
or not 0.0 <= float(value) <= 1.0
|
||||
):
|
||||
raise SemanticObjectQualityError(f"{label} must be within [0, 1]")
|
||||
return float(value)
|
||||
|
||||
|
||||
def _identifier(value: str, label: str) -> None:
|
||||
if not isinstance(value, str) or _IDENTIFIER.fullmatch(value) is None:
|
||||
raise SemanticObjectQualityError(f"{label} is not a safe identifier")
|
||||
|
||||
|
||||
__all__ = [
|
||||
"SEMANTIC_QUALITY_PROFILE_SCHEMA",
|
||||
"SEMANTIC_QUALITY_REPORT_SCHEMA",
|
||||
"SEMANTIC_TRUTH_LABEL_SCHEMA",
|
||||
"SemanticObjectQualityError",
|
||||
"SemanticObjectQualityProfile",
|
||||
"SemanticObjectQualityResult",
|
||||
"SemanticQualityCase",
|
||||
"SemanticTruthLabel",
|
||||
"load_semantic_object_quality_profile",
|
||||
"score_semantic_object_quality",
|
||||
]
|
||||
@@ -0,0 +1,272 @@
|
||||
"""Geometry-first raw-KB4 regions for semantic classification shadows."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
from .contracts import BoundingRegion2D, ObstacleObservation
|
||||
from .geometry import GeometryFrame
|
||||
from .geometry_math import project_map_points_kb4
|
||||
from .open_vocabulary_semantics import OpenVocabularyDetection, SemanticFusionResult
|
||||
|
||||
|
||||
class GeometrySemanticRoiError(ValueError):
|
||||
"""A geometry-owned semantic ROI or crop is incompatible."""
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class GeometrySemanticRoiProfile:
|
||||
minimum_projected_points: int = 4
|
||||
minimum_crop_width: int = 96
|
||||
minimum_crop_height: int = 96
|
||||
padding_fraction: float = 0.25
|
||||
minimum_padding_pixels: int = 16
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if (
|
||||
self.minimum_projected_points < 2
|
||||
or self.minimum_crop_width < 32
|
||||
or self.minimum_crop_height < 32
|
||||
or not math.isfinite(self.padding_fraction)
|
||||
or not 0.0 <= self.padding_fraction <= 1.0
|
||||
or self.minimum_padding_pixels < 0
|
||||
):
|
||||
raise GeometrySemanticRoiError("geometry semantic ROI profile is invalid")
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class GeometrySemanticRoi:
|
||||
roi_id: str
|
||||
frame_index: int
|
||||
observation: ObstacleObservation
|
||||
core_region: BoundingRegion2D
|
||||
crop_region: BoundingRegion2D
|
||||
projected_point_count: int
|
||||
crop_name: str
|
||||
|
||||
def to_dict(self) -> dict[str, object]:
|
||||
return {
|
||||
"roi_id": self.roi_id,
|
||||
"frame_index": self.frame_index,
|
||||
"observation": self.observation.to_dict(),
|
||||
"core_region": self.core_region.to_dict(),
|
||||
"crop_region": self.crop_region.to_dict(),
|
||||
"projected_point_count": self.projected_point_count,
|
||||
"crop_name": self.crop_name,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class GeometrySemanticRoiFrame:
|
||||
frame_index: int
|
||||
rois: tuple[GeometrySemanticRoi, ...]
|
||||
not_projected_observations: tuple[ObstacleObservation, ...]
|
||||
|
||||
|
||||
def build_geometry_semantic_rois(
|
||||
*,
|
||||
frame: GeometryFrame,
|
||||
observations: tuple[ObstacleObservation, ...],
|
||||
profile: GeometrySemanticRoiProfile | None = None,
|
||||
) -> GeometrySemanticRoiFrame:
|
||||
"""Project geometry-owned point identities into bounded semantic crops."""
|
||||
|
||||
selected_profile = profile or GeometrySemanticRoiProfile()
|
||||
if frame.frame_index < 0 or not frame.surface_valid:
|
||||
raise GeometrySemanticRoiError("geometry semantic ROI frame is unavailable")
|
||||
if any(
|
||||
item.frame_id != f"frame-{frame.frame_index:06d}"
|
||||
or item.metric_geometry is None
|
||||
or not item.occupied_support
|
||||
for item in observations
|
||||
):
|
||||
raise GeometrySemanticRoiError("geometry semantic ROI observations are incompatible")
|
||||
projected = project_map_points_kb4(
|
||||
frame.points_map,
|
||||
position_map_xyz=frame.sensor_position_map,
|
||||
orientation_map_from_lidar_xyzw=frame.sensor_orientation_xyzw,
|
||||
profile=frame.projection,
|
||||
)
|
||||
rois: list[GeometrySemanticRoi] = []
|
||||
missing: list[ObstacleObservation] = []
|
||||
for observation in observations:
|
||||
source_ids = np.asarray(observation.source_point_ids, dtype=np.int64)
|
||||
selected = np.isin(projected.source_indices, source_ids)
|
||||
pixels = projected.pixels_xy[selected]
|
||||
if pixels.shape[0] < selected_profile.minimum_projected_points:
|
||||
missing.append(observation)
|
||||
continue
|
||||
x_min, y_min = pixels.min(axis=0)
|
||||
x_max, y_max = pixels.max(axis=0)
|
||||
core = _nonempty_region(
|
||||
float(x_min),
|
||||
float(y_min),
|
||||
float(x_max),
|
||||
float(y_max),
|
||||
width=frame.projection.width,
|
||||
height=frame.projection.height,
|
||||
)
|
||||
crop = _crop_region(
|
||||
core,
|
||||
width=frame.projection.width,
|
||||
height=frame.projection.height,
|
||||
profile=selected_profile,
|
||||
)
|
||||
ordinal = len(rois)
|
||||
roi_id = f"{observation.frame_id}:geometry-roi-{ordinal:03d}"
|
||||
rois.append(
|
||||
GeometrySemanticRoi(
|
||||
roi_id=roi_id,
|
||||
frame_index=frame.frame_index,
|
||||
observation=observation,
|
||||
core_region=core,
|
||||
crop_region=crop,
|
||||
projected_point_count=int(pixels.shape[0]),
|
||||
crop_name=(
|
||||
f"frame-{frame.frame_index:06d}-geometry-roi-{ordinal:03d}.png"
|
||||
),
|
||||
)
|
||||
)
|
||||
return GeometrySemanticRoiFrame(
|
||||
frame_index=frame.frame_index,
|
||||
rois=tuple(rois),
|
||||
not_projected_observations=tuple(missing),
|
||||
)
|
||||
|
||||
|
||||
def materialize_geometry_semantic_crop(
|
||||
*,
|
||||
image_path: Path,
|
||||
roi: GeometrySemanticRoi,
|
||||
destination: Path,
|
||||
) -> None:
|
||||
"""Write one lossless crop while retaining raw-image coordinate lineage."""
|
||||
|
||||
try:
|
||||
with Image.open(image_path.resolve(strict=True)) as opened:
|
||||
if opened.size != (800, 600):
|
||||
raise GeometrySemanticRoiError("semantic crop source raster changed")
|
||||
image = opened.convert("RGB")
|
||||
crop = image.crop(_integer_box(roi.crop_region))
|
||||
except OSError as exc:
|
||||
raise GeometrySemanticRoiError("semantic crop source cannot be read") from exc
|
||||
target = destination.resolve()
|
||||
if target.exists() or target.name != roi.crop_name:
|
||||
raise GeometrySemanticRoiError("semantic crop destination is invalid")
|
||||
target.parent.mkdir(mode=0o700, parents=True, exist_ok=True)
|
||||
crop.save(target, format="PNG", optimize=False)
|
||||
|
||||
|
||||
def select_geometry_roi_detections(
|
||||
roi: GeometrySemanticRoi,
|
||||
fusion: SemanticFusionResult,
|
||||
) -> tuple[OpenVocabularyDetection, ...]:
|
||||
"""Select the semantic cluster covering the geometry-owned core, if any."""
|
||||
|
||||
center_x = (roi.core_region.x_min + roi.core_region.x_max) / 2.0
|
||||
center_y = (roi.core_region.y_min + roi.core_region.y_max) / 2.0
|
||||
candidates = []
|
||||
for binding in fusion.bindings:
|
||||
region = binding.proposal.region
|
||||
center_inside = (
|
||||
region.x_min <= center_x <= region.x_max
|
||||
and region.y_min <= center_y <= region.y_max
|
||||
)
|
||||
core_coverage = _intersection_area(region, roi.core_region) / _area(roi.core_region)
|
||||
if center_inside or core_coverage >= 0.25:
|
||||
candidates.append((binding.proposal.objectness, core_coverage, binding))
|
||||
if not candidates:
|
||||
return ()
|
||||
selected = max(candidates, key=lambda item: (item[0], item[1], item[2].proposal.proposal_id))
|
||||
return selected[2].detections
|
||||
|
||||
|
||||
def _nonempty_region(
|
||||
x_min: float,
|
||||
y_min: float,
|
||||
x_max: float,
|
||||
y_max: float,
|
||||
*,
|
||||
width: int,
|
||||
height: int,
|
||||
) -> BoundingRegion2D:
|
||||
if x_max - x_min < 1.0:
|
||||
center = (x_min + x_max) / 2.0
|
||||
x_min, x_max = center - 0.5, center + 0.5
|
||||
if y_max - y_min < 1.0:
|
||||
center = (y_min + y_max) / 2.0
|
||||
y_min, y_max = center - 0.5, center + 0.5
|
||||
x_min = max(0.0, min(x_min, width - 1.0))
|
||||
y_min = max(0.0, min(y_min, height - 1.0))
|
||||
x_max = min(float(width), max(x_max, x_min + 1.0))
|
||||
y_max = min(float(height), max(y_max, y_min + 1.0))
|
||||
return BoundingRegion2D(x_min, y_min, x_max, y_max)
|
||||
|
||||
|
||||
def _crop_region(
|
||||
core: BoundingRegion2D,
|
||||
*,
|
||||
width: int,
|
||||
height: int,
|
||||
profile: GeometrySemanticRoiProfile,
|
||||
) -> BoundingRegion2D:
|
||||
core_width = core.x_max - core.x_min
|
||||
core_height = core.y_max - core.y_min
|
||||
padding_x = max(profile.minimum_padding_pixels, core_width * profile.padding_fraction)
|
||||
padding_y = max(profile.minimum_padding_pixels, core_height * profile.padding_fraction)
|
||||
target_width = min(
|
||||
width,
|
||||
max(profile.minimum_crop_width, math.ceil(core_width + 2.0 * padding_x)),
|
||||
)
|
||||
target_height = min(
|
||||
height,
|
||||
max(profile.minimum_crop_height, math.ceil(core_height + 2.0 * padding_y)),
|
||||
)
|
||||
center_x = (core.x_min + core.x_max) / 2.0
|
||||
center_y = (core.y_min + core.y_max) / 2.0
|
||||
x_min = int(round(center_x - target_width / 2.0))
|
||||
y_min = int(round(center_y - target_height / 2.0))
|
||||
x_min = min(max(0, x_min), width - target_width)
|
||||
y_min = min(max(0, y_min), height - target_height)
|
||||
return BoundingRegion2D(
|
||||
float(x_min),
|
||||
float(y_min),
|
||||
float(x_min + target_width),
|
||||
float(y_min + target_height),
|
||||
)
|
||||
|
||||
|
||||
def _integer_box(region: BoundingRegion2D) -> tuple[int, int, int, int]:
|
||||
return (
|
||||
int(region.x_min),
|
||||
int(region.y_min),
|
||||
int(region.x_max),
|
||||
int(region.y_max),
|
||||
)
|
||||
|
||||
|
||||
def _area(region: BoundingRegion2D) -> float:
|
||||
return (region.x_max - region.x_min) * (region.y_max - region.y_min)
|
||||
|
||||
|
||||
def _intersection_area(left: BoundingRegion2D, right: BoundingRegion2D) -> float:
|
||||
return max(0.0, min(left.x_max, right.x_max) - max(left.x_min, right.x_min)) * max(
|
||||
0.0,
|
||||
min(left.y_max, right.y_max) - max(left.y_min, right.y_min),
|
||||
)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"GeometrySemanticRoi",
|
||||
"GeometrySemanticRoiError",
|
||||
"GeometrySemanticRoiFrame",
|
||||
"GeometrySemanticRoiProfile",
|
||||
"build_geometry_semantic_rois",
|
||||
"materialize_geometry_semantic_crop",
|
||||
"select_geometry_roi_detections",
|
||||
]
|
||||
@@ -0,0 +1,580 @@
|
||||
"""Immutable geometry-first M48S semantic shadow replay."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import shutil
|
||||
import uuid
|
||||
from collections import Counter
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
from .detector_replay_result import (
|
||||
read_detector_replay_result,
|
||||
require_m4_detector_replay_acceptance,
|
||||
)
|
||||
from .geometry import Ravnoves00GeometryAssociationProvider, RecordedGeometryStore
|
||||
from .geometry_semantic_roi import (
|
||||
GeometrySemanticRoi,
|
||||
build_geometry_semantic_rois,
|
||||
select_geometry_roi_detections,
|
||||
)
|
||||
from .graph_validation import validate_observations
|
||||
from .object_understanding import (
|
||||
ObjectUnderstanding,
|
||||
SemanticResolution,
|
||||
load_object_semantic_vocabulary,
|
||||
)
|
||||
from .open_vocabulary_semantics import (
|
||||
OpenVocabularyDetection,
|
||||
fuse_open_vocabulary_detections,
|
||||
load_open_vocabulary_semantic_profile,
|
||||
parse_tao_grounding_dino_labels,
|
||||
understand_geometry_observation,
|
||||
)
|
||||
from .semantic_shadow_replay import semantic_replay_packet
|
||||
from .yolox_object_detector import load_valid_fov_mask
|
||||
|
||||
GEOMETRY_SEMANTIC_SHADOW_REPLAY_SCHEMA: Final = (
|
||||
"missioncore.m48s-geometry-semantic-shadow-replay/v0"
|
||||
)
|
||||
GEOMETRY_SEMANTIC_SHADOW_FRAME_SCHEMA: Final = (
|
||||
"missioncore.m48s-geometry-semantic-shadow-frame/v0"
|
||||
)
|
||||
GEOMETRY_SEMANTIC_ROI_PACKAGE_SCHEMA: Final = (
|
||||
"missioncore.m48s-geometry-semantic-roi-package/v0"
|
||||
)
|
||||
RESULT_PREFIX: Final = "m48s-geometry-semantic-shadow-"
|
||||
|
||||
|
||||
class GeometrySemanticShadowReplayError(RuntimeError):
|
||||
"""Geometry-first semantic evidence is incomplete or incompatible."""
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class GeometrySemanticShadowReplayResult:
|
||||
result_id: str
|
||||
result_root: Path
|
||||
completed: bool
|
||||
accepted: bool
|
||||
metrics: dict[str, object]
|
||||
report: dict[str, object]
|
||||
manifest: dict[str, object]
|
||||
|
||||
|
||||
def build_geometry_semantic_shadow_replay(
|
||||
*,
|
||||
repository_root: Path,
|
||||
profile_path: Path,
|
||||
vocabulary_path: Path,
|
||||
detector_result_root: Path,
|
||||
roi_package_root: Path,
|
||||
valid_fov_mask_path: Path,
|
||||
worker_result_roots: Mapping[str, Path],
|
||||
frame_indices: tuple[int, ...],
|
||||
worker_execution: Mapping[str, object],
|
||||
output_root: Path,
|
||||
) -> GeometrySemanticShadowReplayResult:
|
||||
"""Name geometry-owned obstacles while leaving occupancy and authority untouched."""
|
||||
|
||||
repository = repository_root.resolve(strict=True)
|
||||
profile = load_open_vocabulary_semantic_profile(profile_path)
|
||||
vocabulary = load_object_semantic_vocabulary(vocabulary_path)
|
||||
if vocabulary.vocabulary_id != profile.vocabulary_id:
|
||||
raise GeometrySemanticShadowReplayError("semantic profile and vocabulary disagree")
|
||||
frames = _validate_frame_indices(frame_indices)
|
||||
roi_root = roi_package_root.resolve(strict=True)
|
||||
roi_manifest = _read_object(roi_root / "manifest.json", "ROI package manifest")
|
||||
roi_identity = _validate_roi_manifest(roi_manifest, frames=frames)
|
||||
crop_names = tuple(
|
||||
_string(_object(roi, "ROI"), "crop_name")
|
||||
for frame in _array(roi_identity, "frames")
|
||||
for roi in _array(_object(frame, "ROI frame"), "rois")
|
||||
)
|
||||
_validate_worker_results(worker_result_roots, profile=profile, crop_names=crop_names)
|
||||
valid_fov_mask = load_valid_fov_mask(
|
||||
valid_fov_mask_path,
|
||||
expected_sha256=profile.valid_fov_mask_sha256,
|
||||
)
|
||||
detector = read_detector_replay_result(detector_result_root)
|
||||
require_m4_detector_replay_acceptance(detector)
|
||||
detector_by_sequence = {item.sequence: item for item in detector.frames}
|
||||
if any(index not in detector_by_sequence for index in frames):
|
||||
raise GeometrySemanticShadowReplayError("semantic frame escaped detector timeline")
|
||||
store = RecordedGeometryStore.from_repository(repository)
|
||||
geometry = Ravnoves00GeometryAssociationProvider(store=store)
|
||||
|
||||
output = output_root.expanduser().absolute()
|
||||
output.mkdir(mode=0o700, parents=True, exist_ok=True)
|
||||
staging = output / f".geometry-semantic-shadow.{uuid.uuid4().hex}.tmp"
|
||||
staging.mkdir(mode=0o700, exist_ok=False)
|
||||
totals: Counter[str] = Counter()
|
||||
resolution_counts: Counter[str] = Counter()
|
||||
class_counts: Counter[str] = Counter()
|
||||
try:
|
||||
frames_path = staging / "frames.jsonl"
|
||||
with frames_path.open("wb") as ledger:
|
||||
for frame_index in frames:
|
||||
detector_frame = detector_by_sequence[frame_index]
|
||||
if detector_frame.outcome != "completed":
|
||||
raise GeometrySemanticShadowReplayError("accepted detector frame failed")
|
||||
packet = semantic_replay_packet(detector_frame.envelope)
|
||||
observations = geometry.associate(packet, ())
|
||||
validate_observations(packet, (), observations)
|
||||
geometry_frame = store.frame(packet)
|
||||
if geometry_frame is None:
|
||||
raise GeometrySemanticShadowReplayError("selected geometry frame unavailable")
|
||||
roi_frame = build_geometry_semantic_rois(
|
||||
frame=geometry_frame,
|
||||
observations=observations,
|
||||
)
|
||||
stored_frame = _roi_manifest_frame(roi_identity, frame_index)
|
||||
if [item.to_dict() for item in roi_frame.rois] != [
|
||||
_roi_without_materialization(_object(item, "stored ROI"))
|
||||
for item in _array(stored_frame, "rois")
|
||||
]:
|
||||
raise GeometrySemanticShadowReplayError("stored ROI geometry changed")
|
||||
understandings_by_observation: dict[str, ObjectUnderstanding] = {}
|
||||
roi_documents: list[dict[str, object]] = []
|
||||
for roi in roi_frame.rois:
|
||||
detections = _roi_detections(
|
||||
roi,
|
||||
worker_result_roots=worker_result_roots,
|
||||
profile=profile,
|
||||
)
|
||||
fusion = fuse_open_vocabulary_detections(
|
||||
detections,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
valid_fov_mask=valid_fov_mask,
|
||||
)
|
||||
selected = select_geometry_roi_detections(roi, fusion)
|
||||
understanding = understand_geometry_observation(
|
||||
roi.observation,
|
||||
detections=selected,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
generated_monotonic_ns=detector_frame.envelope.timestamps.monotonic_ns,
|
||||
)
|
||||
understandings_by_observation[roi.observation.observation_id] = understanding
|
||||
totals["roi_count"] += 1
|
||||
totals["raw_detection_count"] += len(detections)
|
||||
totals["below_confidence_count"] += fusion.below_confidence_count
|
||||
totals["invalid_area_count"] += fusion.invalid_area_count
|
||||
totals["outside_valid_fov_count"] += fusion.outside_valid_fov_count
|
||||
totals["retained_detection_count"] += fusion.retained_detection_count
|
||||
totals["fused_proposal_count"] += len(fusion.proposals)
|
||||
totals["selected_cluster_detection_count"] += len(selected)
|
||||
totals["roi_with_selected_cluster_count"] += bool(selected)
|
||||
roi_documents.append(
|
||||
{
|
||||
"roi": roi.to_dict(),
|
||||
"detections": [_detection_document(item) for item in detections],
|
||||
"fused_proposals": [item.to_dict() for item in fusion.proposals],
|
||||
"selected_detection_ids": [item.detection_id for item in selected],
|
||||
"understanding": understanding.to_dict(),
|
||||
}
|
||||
)
|
||||
for observation in roi_frame.not_projected_observations:
|
||||
understandings_by_observation[observation.observation_id] = (
|
||||
understand_geometry_observation(
|
||||
observation,
|
||||
detections=(),
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
generated_monotonic_ns=(
|
||||
detector_frame.envelope.timestamps.monotonic_ns
|
||||
),
|
||||
)
|
||||
)
|
||||
totals["not_projected_observation_count"] += 1
|
||||
ordered = tuple(
|
||||
understandings_by_observation[item.observation_id] for item in observations
|
||||
)
|
||||
if len(ordered) != len(observations):
|
||||
raise GeometrySemanticShadowReplayError("geometry obstacle accounting changed")
|
||||
for understanding in ordered:
|
||||
resolution_counts[understanding.semantic.resolution.value] += 1
|
||||
if (
|
||||
understanding.semantic.resolution is SemanticResolution.SELECTED
|
||||
and understanding.semantic.selected_class_id is not None
|
||||
):
|
||||
class_counts[understanding.semantic.selected_class_id] += 1
|
||||
totals["geometry_observation_count"] += len(observations)
|
||||
frame_document = {
|
||||
"schema_version": GEOMETRY_SEMANTIC_SHADOW_FRAME_SCHEMA,
|
||||
"frame_index": frame_index,
|
||||
"frame_id": detector_frame.envelope.frame_id,
|
||||
"geometry_observation_count": len(observations),
|
||||
"roi_count": len(roi_frame.rois),
|
||||
"not_projected_observation_count": len(
|
||||
roi_frame.not_projected_observations
|
||||
),
|
||||
"rois": roi_documents,
|
||||
"understandings": [item.to_dict() for item in ordered],
|
||||
"authority": _false_authority(),
|
||||
}
|
||||
ledger.write(_canonical_json(frame_document) + b"\n")
|
||||
metrics: dict[str, object] = {
|
||||
"frames": {"requested": len(frames), "completed": len(frames)},
|
||||
"geometry": {
|
||||
"observation_count": totals["geometry_observation_count"],
|
||||
"roi_count": totals["roi_count"],
|
||||
"not_projected_observation_count": totals[
|
||||
"not_projected_observation_count"
|
||||
],
|
||||
},
|
||||
"semantic_inference": {
|
||||
"raw_detection_count": totals["raw_detection_count"],
|
||||
"below_confidence_count": totals["below_confidence_count"],
|
||||
"invalid_area_count": totals["invalid_area_count"],
|
||||
"outside_valid_fov_count": totals["outside_valid_fov_count"],
|
||||
"retained_detection_count": totals["retained_detection_count"],
|
||||
"fused_proposal_count": totals["fused_proposal_count"],
|
||||
"roi_with_selected_cluster_count": totals[
|
||||
"roi_with_selected_cluster_count"
|
||||
],
|
||||
"selected_cluster_detection_count": totals[
|
||||
"selected_cluster_detection_count"
|
||||
],
|
||||
},
|
||||
"semantics": {
|
||||
"resolution_counts": dict(sorted(resolution_counts.items())),
|
||||
"selected_class_counts": dict(sorted(class_counts.items())),
|
||||
},
|
||||
"authority": _false_authority(),
|
||||
}
|
||||
completed = (
|
||||
totals["geometry_observation_count"]
|
||||
== totals["roi_count"] + totals["not_projected_observation_count"]
|
||||
and sum(resolution_counts.values()) == totals["geometry_observation_count"]
|
||||
)
|
||||
identity = {
|
||||
"schema_version": GEOMETRY_SEMANTIC_SHADOW_REPLAY_SCHEMA,
|
||||
"profile_id": profile.profile_id,
|
||||
"profile_sha256": _sha256(profile_path),
|
||||
"vocabulary_id": vocabulary.vocabulary_id,
|
||||
"vocabulary_sha256": _sha256(vocabulary_path),
|
||||
"detector_result_id": detector.result_id,
|
||||
"roi_package_id": _string(roi_manifest, "package_id"),
|
||||
"roi_package_sha256": _sha256(roi_root / "manifest.json"),
|
||||
"frame_indices": list(frames),
|
||||
"worker_execution": dict(worker_execution),
|
||||
"worker_artifacts": _worker_artifacts(worker_result_roots, profile, crop_names),
|
||||
"producer_sha256": {
|
||||
name: _sha256(repository / "src/k1link/perception" / name)
|
||||
for name in (
|
||||
"geometry_semantic_roi.py",
|
||||
"geometry_semantic_shadow_replay.py",
|
||||
"object_understanding.py",
|
||||
"open_vocabulary_semantics.py",
|
||||
)
|
||||
},
|
||||
"frames_sha256": _sha256(frames_path),
|
||||
"metrics": metrics,
|
||||
"completed": completed,
|
||||
"accepted": False,
|
||||
"authority": _false_authority(),
|
||||
}
|
||||
identity_sha256 = hashlib.sha256(_canonical_json(identity)).hexdigest()
|
||||
result_id = f"{RESULT_PREFIX}{identity_sha256}"
|
||||
report = {
|
||||
"schema_version": GEOMETRY_SEMANTIC_SHADOW_REPLAY_SCHEMA,
|
||||
"result_id": result_id,
|
||||
"created_at_utc": datetime.now(UTC).isoformat(timespec="milliseconds").replace(
|
||||
"+00:00", "Z"
|
||||
),
|
||||
"completed": completed,
|
||||
"accepted": False,
|
||||
"metrics": metrics,
|
||||
"decision": {
|
||||
"geometry_first_binding_completed": completed,
|
||||
"semantic_quality_accepted": False,
|
||||
"navigation_or_safety_accepted": False,
|
||||
"reason_codes": [
|
||||
"two-reviewer-independent-truth-unavailable",
|
||||
"open-vocabulary-geometry-roi-remains-experimental-shadow",
|
||||
"unknown-and-unresolved-objects-remain-route-around-obstacles",
|
||||
],
|
||||
},
|
||||
"authority": _false_authority(),
|
||||
}
|
||||
manifest: dict[str, object] = {
|
||||
"schema_version": GEOMETRY_SEMANTIC_SHADOW_REPLAY_SCHEMA,
|
||||
"result_id": result_id,
|
||||
"identity_sha256": identity_sha256,
|
||||
"identity": identity,
|
||||
"artifacts": {
|
||||
"frames.jsonl": _sha256(frames_path),
|
||||
"report.json": hashlib.sha256(_canonical_json(report) + b"\n").hexdigest(),
|
||||
},
|
||||
}
|
||||
(staging / "report.json").write_bytes(_canonical_json(report) + b"\n")
|
||||
(staging / "manifest.json").write_bytes(_canonical_json(manifest) + b"\n")
|
||||
destination = output / result_id
|
||||
if destination.exists():
|
||||
shutil.rmtree(staging)
|
||||
else:
|
||||
staging.rename(destination)
|
||||
except Exception:
|
||||
shutil.rmtree(staging, ignore_errors=True)
|
||||
raise
|
||||
return GeometrySemanticShadowReplayResult(
|
||||
result_id=result_id,
|
||||
result_root=destination,
|
||||
completed=completed,
|
||||
accepted=False,
|
||||
metrics=metrics,
|
||||
report=report,
|
||||
manifest=manifest,
|
||||
)
|
||||
|
||||
|
||||
def read_geometry_semantic_shadow_replay(
|
||||
result_root: Path,
|
||||
) -> GeometrySemanticShadowReplayResult:
|
||||
"""Read and revalidate one immutable geometry-first semantic result."""
|
||||
|
||||
root = result_root.resolve(strict=True)
|
||||
manifest = _read_object(root / "manifest.json", "geometry semantic manifest")
|
||||
report = _read_object(root / "report.json", "geometry semantic report")
|
||||
identity = _object(manifest.get("identity"), "geometry semantic identity")
|
||||
identity_sha256 = hashlib.sha256(_canonical_json(identity)).hexdigest()
|
||||
result_id = f"{RESULT_PREFIX}{identity_sha256}"
|
||||
artifacts = _object(manifest.get("artifacts"), "geometry semantic artifacts")
|
||||
if (
|
||||
manifest.get("schema_version") != GEOMETRY_SEMANTIC_SHADOW_REPLAY_SCHEMA
|
||||
or manifest.get("identity_sha256") != identity_sha256
|
||||
or manifest.get("result_id") != result_id
|
||||
or root.name != result_id
|
||||
or report.get("schema_version") != GEOMETRY_SEMANTIC_SHADOW_REPLAY_SCHEMA
|
||||
or report.get("result_id") != result_id
|
||||
or artifacts
|
||||
!= {
|
||||
"frames.jsonl": _sha256(root / "frames.jsonl"),
|
||||
"report.json": _sha256(root / "report.json"),
|
||||
}
|
||||
):
|
||||
raise GeometrySemanticShadowReplayError("geometry semantic result identity changed")
|
||||
metrics = _object(report.get("metrics"), "geometry semantic metrics")
|
||||
completed = report.get("completed")
|
||||
accepted = report.get("accepted")
|
||||
if not isinstance(completed, bool) or not isinstance(accepted, bool):
|
||||
raise GeometrySemanticShadowReplayError("geometry semantic result state changed")
|
||||
if identity.get("metrics") != metrics or identity.get("completed") != completed:
|
||||
raise GeometrySemanticShadowReplayError("geometry semantic result accounting changed")
|
||||
return GeometrySemanticShadowReplayResult(
|
||||
result_id=result_id,
|
||||
result_root=root,
|
||||
completed=completed,
|
||||
accepted=accepted,
|
||||
metrics=metrics,
|
||||
report=report,
|
||||
manifest=manifest,
|
||||
)
|
||||
|
||||
|
||||
def _roi_detections(
|
||||
roi: GeometrySemanticRoi,
|
||||
*,
|
||||
worker_result_roots: Mapping[str, Path],
|
||||
profile: object,
|
||||
) -> tuple[OpenVocabularyDetection, ...]:
|
||||
from .open_vocabulary_semantics import OpenVocabularySemanticProfile
|
||||
|
||||
if not isinstance(profile, OpenVocabularySemanticProfile):
|
||||
raise GeometrySemanticShadowReplayError("semantic profile is incompatible")
|
||||
width = int(roi.crop_region.x_max - roi.crop_region.x_min)
|
||||
height = int(roi.crop_region.y_max - roi.crop_region.y_min)
|
||||
detections: list[OpenVocabularyDetection] = []
|
||||
for group in profile.prompt_groups:
|
||||
labels = (
|
||||
worker_result_roots[group.prompt_set_id].resolve(strict=True)
|
||||
/ "labels"
|
||||
/ roi.crop_name.replace(".png", ".txt")
|
||||
)
|
||||
detections.extend(
|
||||
parse_tao_grounding_dino_labels(
|
||||
labels,
|
||||
source_id=profile.source_id,
|
||||
frame_id=roi.observation.frame_id,
|
||||
prompt_set_id=group.prompt_set_id,
|
||||
profile=profile,
|
||||
image_width=width,
|
||||
image_height=height,
|
||||
offset_x=roi.crop_region.x_min,
|
||||
offset_y=roi.crop_region.y_min,
|
||||
detection_scope_id=roi.roi_id,
|
||||
)
|
||||
)
|
||||
return tuple(detections)
|
||||
|
||||
|
||||
def _validate_worker_results(
|
||||
roots: Mapping[str, Path],
|
||||
*,
|
||||
profile: object,
|
||||
crop_names: tuple[str, ...],
|
||||
) -> None:
|
||||
from .open_vocabulary_semantics import OpenVocabularySemanticProfile
|
||||
|
||||
if not isinstance(profile, OpenVocabularySemanticProfile):
|
||||
raise GeometrySemanticShadowReplayError("semantic profile is incompatible")
|
||||
if set(roots) != {item.prompt_set_id for item in profile.prompt_groups}:
|
||||
raise GeometrySemanticShadowReplayError("Worker prompt roots changed")
|
||||
expected = {item.replace(".png", ".txt") for item in crop_names}
|
||||
for root in roots.values():
|
||||
resolved = root.resolve(strict=True)
|
||||
actual = {item.name for item in (resolved / "labels").glob("*.txt")}
|
||||
if actual != expected:
|
||||
raise GeometrySemanticShadowReplayError("Worker label coverage changed")
|
||||
status = (resolved / "status.json").read_text("utf-8")
|
||||
if '"status": "SUCCESS"' not in status:
|
||||
raise GeometrySemanticShadowReplayError("Worker semantic run did not succeed")
|
||||
|
||||
|
||||
def _worker_artifacts(
|
||||
roots: Mapping[str, Path],
|
||||
profile: object,
|
||||
crop_names: tuple[str, ...],
|
||||
) -> dict[str, object]:
|
||||
from .open_vocabulary_semantics import OpenVocabularySemanticProfile
|
||||
|
||||
if not isinstance(profile, OpenVocabularySemanticProfile):
|
||||
raise GeometrySemanticShadowReplayError("semantic profile is incompatible")
|
||||
artifacts: dict[str, object] = {}
|
||||
for group in profile.prompt_groups:
|
||||
root = roots[group.prompt_set_id].resolve(strict=True)
|
||||
labels = [
|
||||
{
|
||||
"name": crop_name.replace(".png", ".txt"),
|
||||
"sha256": _sha256(root / "labels" / crop_name.replace(".png", ".txt")),
|
||||
}
|
||||
for crop_name in crop_names
|
||||
]
|
||||
artifacts[group.prompt_set_id] = {
|
||||
"status_sha256": _sha256(root / "status.json"),
|
||||
"experiment_sha256": _sha256(root / "experiment.yaml"),
|
||||
"labels": labels,
|
||||
}
|
||||
return artifacts
|
||||
|
||||
|
||||
def _validate_roi_manifest(
|
||||
manifest: dict[str, object],
|
||||
*,
|
||||
frames: tuple[int, ...],
|
||||
) -> dict[str, object]:
|
||||
if manifest.get("schema_version") != GEOMETRY_SEMANTIC_ROI_PACKAGE_SCHEMA:
|
||||
raise GeometrySemanticShadowReplayError("ROI package schema changed")
|
||||
identity = _object(manifest.get("identity"), "ROI identity")
|
||||
if identity.get("frame_indices") != list(frames):
|
||||
raise GeometrySemanticShadowReplayError("ROI frame selection changed")
|
||||
if identity.get("roi_count") != 58 or identity.get("geometry_observation_count") != 73:
|
||||
raise GeometrySemanticShadowReplayError("bounded ROI package accounting changed")
|
||||
digest = hashlib.sha256(_canonical_json(identity)).hexdigest()
|
||||
if manifest.get("identity_sha256") != digest or manifest.get("package_id") != (
|
||||
f"m48s-geometry-semantic-rois-{digest}"
|
||||
):
|
||||
raise GeometrySemanticShadowReplayError("ROI package identity changed")
|
||||
return identity
|
||||
|
||||
|
||||
def _roi_manifest_frame(identity: dict[str, object], frame_index: int) -> dict[str, object]:
|
||||
matches = [
|
||||
_object(item, "ROI frame")
|
||||
for item in _array(identity, "frames")
|
||||
if _object(item, "ROI frame").get("frame_index") == frame_index
|
||||
]
|
||||
if len(matches) != 1:
|
||||
raise GeometrySemanticShadowReplayError("ROI frame manifest is ambiguous")
|
||||
return matches[0]
|
||||
|
||||
|
||||
def _roi_without_materialization(value: dict[str, object]) -> dict[str, object]:
|
||||
return {
|
||||
key: item
|
||||
for key, item in value.items()
|
||||
if key not in {"crop_sha256", "crop_width", "crop_height"}
|
||||
}
|
||||
|
||||
|
||||
def _validate_frame_indices(value: tuple[int, ...]) -> tuple[int, ...]:
|
||||
if not value or tuple(sorted(set(value))) != value or any(item < 0 for item in value):
|
||||
raise GeometrySemanticShadowReplayError("semantic frame selection is invalid")
|
||||
return value
|
||||
|
||||
|
||||
def _detection_document(value: OpenVocabularyDetection) -> dict[str, object]:
|
||||
return {
|
||||
"detection_id": value.detection_id,
|
||||
"source_id": value.source_id,
|
||||
"frame_id": value.frame_id,
|
||||
"prompt_set_id": value.prompt_set_id,
|
||||
"raw_label": value.raw_label,
|
||||
"confidence": value.confidence,
|
||||
"region": value.region.to_dict(),
|
||||
}
|
||||
|
||||
|
||||
def _read_object(path: Path, label: str) -> dict[str, object]:
|
||||
try:
|
||||
value = json.loads(path.resolve(strict=True).read_text("utf-8"))
|
||||
except (OSError, json.JSONDecodeError) as exc:
|
||||
raise GeometrySemanticShadowReplayError(f"{label} cannot be read") from exc
|
||||
return _object(value, label)
|
||||
|
||||
|
||||
def _object(value: object, label: str) -> dict[str, object]:
|
||||
if not isinstance(value, dict) or any(not isinstance(key, str) for key in value):
|
||||
raise GeometrySemanticShadowReplayError(f"{label} must be an object")
|
||||
return value
|
||||
|
||||
|
||||
def _array(value: dict[str, object], key: str) -> list[object]:
|
||||
item = value.get(key)
|
||||
if not isinstance(item, list):
|
||||
raise GeometrySemanticShadowReplayError(f"{key} must be an array")
|
||||
return item
|
||||
|
||||
|
||||
def _string(value: dict[str, object], key: str) -> str:
|
||||
item = value.get(key)
|
||||
if not isinstance(item, str) or not item:
|
||||
raise GeometrySemanticShadowReplayError(f"{key} must be a nonempty string")
|
||||
return item
|
||||
|
||||
|
||||
def _canonical_json(value: object) -> bytes:
|
||||
return json.dumps(value, sort_keys=True, separators=(",", ":"), allow_nan=False).encode()
|
||||
|
||||
|
||||
def _sha256(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.resolve(strict=True).open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def _false_authority() -> dict[str, bool]:
|
||||
return {
|
||||
"ground_truth": False,
|
||||
"independent_truth": False,
|
||||
"candidate_accepted": False,
|
||||
"commands_enabled": False,
|
||||
"actuation_allowed": False,
|
||||
"navigation_or_safety_accepted": False,
|
||||
}
|
||||
|
||||
|
||||
__all__ = [
|
||||
"GeometrySemanticShadowReplayError",
|
||||
"GeometrySemanticShadowReplayResult",
|
||||
"build_geometry_semantic_shadow_replay",
|
||||
"read_geometry_semantic_shadow_replay",
|
||||
]
|
||||
@@ -0,0 +1,436 @@
|
||||
"""Fail-closed Mask Grounding DINO evidence and geometry binding.
|
||||
|
||||
The mask model is allowed to suggest instance masks and class hypotheses. It
|
||||
cannot create occupied geometry, and contradictory captions for one pixel mask
|
||||
remain explicitly ambiguous.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import math
|
||||
from collections import defaultdict
|
||||
from dataclasses import dataclass
|
||||
from enum import StrEnum
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import numpy as np
|
||||
import numpy.typing as npt
|
||||
|
||||
from .contracts import ObstacleObservation
|
||||
from .geometry_math import ProjectedPointCloud
|
||||
|
||||
MASK_GROUNDING_EVIDENCE_SCHEMA: Final = (
|
||||
"missioncore.m48s-mask-grounding-dino-evidence/v0"
|
||||
)
|
||||
|
||||
BoolArray = npt.NDArray[np.bool_]
|
||||
|
||||
|
||||
class MaskGroundingSemanticError(ValueError):
|
||||
"""Mask semantic evidence is malformed or cannot be bound safely."""
|
||||
|
||||
|
||||
class MaskBindingResolution(StrEnum):
|
||||
SELECTED = "selected"
|
||||
AMBIGUOUS = "ambiguous"
|
||||
UNRESOLVED = "unresolved"
|
||||
|
||||
|
||||
class MaskLabelResolution(StrEnum):
|
||||
SELECTED = "selected"
|
||||
AMBIGUOUS = "ambiguous"
|
||||
UNRESOLVED = "unresolved"
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class MaskGroundingDetection:
|
||||
detection_id: str
|
||||
prompt_set_id: str
|
||||
class_id: int
|
||||
class_name: str
|
||||
confidence: float
|
||||
box_xyxy: tuple[float, float, float, float]
|
||||
mask_sha256: str
|
||||
mask: BoolArray
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if (
|
||||
not self.detection_id
|
||||
or not self.prompt_set_id
|
||||
or not self.class_name
|
||||
or self.class_id < 0
|
||||
or not math.isfinite(self.confidence)
|
||||
or not 0.0 <= self.confidence <= 1.0
|
||||
or len(self.box_xyxy) != 4
|
||||
or not np.isfinite(self.box_xyxy).all()
|
||||
or self.box_xyxy[2] <= self.box_xyxy[0]
|
||||
or self.box_xyxy[3] <= self.box_xyxy[1]
|
||||
or self.mask.shape != (600, 800)
|
||||
or not self.mask.any()
|
||||
or len(self.mask_sha256) != 64
|
||||
):
|
||||
raise MaskGroundingSemanticError("mask grounding detection is invalid")
|
||||
frozen = np.asarray(self.mask, dtype=np.bool_).copy()
|
||||
frozen.setflags(write=False)
|
||||
object.__setattr__(self, "mask", frozen)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class MaskGroundingEvidence:
|
||||
source_file_sha256: str
|
||||
source_pixel_sha256: str
|
||||
detections: tuple[MaskGroundingDetection, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class MaskInstance:
|
||||
instance_id: str
|
||||
detections: tuple[MaskGroundingDetection, ...]
|
||||
|
||||
@property
|
||||
def ranked_labels(self) -> tuple[tuple[str, float], ...]:
|
||||
confidence_by_label: dict[str, float] = {}
|
||||
for item in self.detections:
|
||||
confidence_by_label[item.class_name] = max(
|
||||
item.confidence,
|
||||
confidence_by_label.get(item.class_name, 0.0),
|
||||
)
|
||||
return tuple(
|
||||
sorted(
|
||||
confidence_by_label.items(),
|
||||
key=lambda item: (-item[1], item[0]),
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class MaskObservationSupport:
|
||||
observation_id: str
|
||||
projected_point_count: int
|
||||
support_fraction: float
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class MaskGeometryBinding:
|
||||
instance_id: str
|
||||
resolution: MaskBindingResolution
|
||||
selected_observation_id: str | None
|
||||
supports: tuple[MaskObservationSupport, ...]
|
||||
reason_code: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class MaskLabelDecision:
|
||||
instance_id: str
|
||||
resolution: MaskLabelResolution
|
||||
selected_label: str | None
|
||||
ranked_labels: tuple[tuple[str, float], ...]
|
||||
reason_code: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class MaskGeometryBindingProfile:
|
||||
minimum_mask_iou: float = 0.9
|
||||
minimum_projected_points: int = 4
|
||||
minimum_support_fraction: float = 0.5
|
||||
maximum_secondary_support_fraction: float = 0.25
|
||||
minimum_label_confidence: float = 0.3
|
||||
minimum_label_margin: float = 0.1
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
fractions = (
|
||||
self.minimum_mask_iou,
|
||||
self.minimum_support_fraction,
|
||||
self.maximum_secondary_support_fraction,
|
||||
self.minimum_label_confidence,
|
||||
self.minimum_label_margin,
|
||||
)
|
||||
if (
|
||||
self.minimum_projected_points < 1
|
||||
or not all(math.isfinite(item) and 0.0 <= item <= 1.0 for item in fractions)
|
||||
or self.maximum_secondary_support_fraction
|
||||
>= self.minimum_support_fraction
|
||||
):
|
||||
raise MaskGroundingSemanticError("mask geometry binding profile is invalid")
|
||||
|
||||
|
||||
def load_mask_grounding_evidence(
|
||||
path: Path,
|
||||
*,
|
||||
prompt_set_id: str,
|
||||
) -> MaskGroundingEvidence:
|
||||
"""Load one lossless Worker ledger without allowing object arrays."""
|
||||
|
||||
if not prompt_set_id:
|
||||
raise MaskGroundingSemanticError("mask prompt set id is required")
|
||||
try:
|
||||
source = path.resolve(strict=True)
|
||||
with np.load(source, allow_pickle=False) as archive:
|
||||
expected = {
|
||||
"schema_version",
|
||||
"source_file_sha256",
|
||||
"source_pixel_sha256",
|
||||
"class_ids",
|
||||
"class_names",
|
||||
"scores",
|
||||
"boxes_xyxy",
|
||||
"masks",
|
||||
}
|
||||
if set(archive.files) != expected:
|
||||
raise MaskGroundingSemanticError("mask evidence fields changed")
|
||||
schema = str(archive["schema_version"].item())
|
||||
source_file_sha256 = str(archive["source_file_sha256"].item())
|
||||
source_pixel_sha256 = str(archive["source_pixel_sha256"].item())
|
||||
class_ids = np.asarray(archive["class_ids"], dtype=np.int64)
|
||||
class_names = np.asarray(archive["class_names"])
|
||||
scores = np.asarray(archive["scores"], dtype=np.float64)
|
||||
boxes = np.asarray(archive["boxes_xyxy"], dtype=np.float64)
|
||||
masks = np.asarray(archive["masks"], dtype=np.uint8)
|
||||
except (OSError, ValueError) as exc:
|
||||
raise MaskGroundingSemanticError("mask evidence cannot be read") from exc
|
||||
if schema != MASK_GROUNDING_EVIDENCE_SCHEMA:
|
||||
raise MaskGroundingSemanticError("mask evidence schema changed")
|
||||
count = int(class_ids.size)
|
||||
if count == 0 and boxes.shape == (0,):
|
||||
# Early v0 shadow ledgers encoded an empty list before the writer was
|
||||
# tightened to (0, 4). No values are inferred by this normalization.
|
||||
boxes = boxes.reshape((0, 4))
|
||||
if (
|
||||
len(source_file_sha256) != 64
|
||||
or len(source_pixel_sha256) != 64
|
||||
or class_ids.shape != (count,)
|
||||
or class_names.shape != (count,)
|
||||
or scores.shape != (count,)
|
||||
or boxes.shape != (count, 4)
|
||||
or masks.shape != (count, 600, 800)
|
||||
or np.any((masks != 0) & (masks != 1))
|
||||
):
|
||||
raise MaskGroundingSemanticError("mask evidence tensors are incompatible")
|
||||
detections = []
|
||||
for index in range(count):
|
||||
mask = masks[index].astype(np.bool_)
|
||||
mask_sha256 = hashlib.sha256(masks[index].tobytes()).hexdigest()
|
||||
detections.append(
|
||||
MaskGroundingDetection(
|
||||
detection_id=f"{path.stem}:{prompt_set_id}:{index:04d}",
|
||||
prompt_set_id=prompt_set_id,
|
||||
class_id=int(class_ids[index]),
|
||||
class_name=str(class_names[index]),
|
||||
confidence=float(scores[index]),
|
||||
box_xyxy=tuple(float(item) for item in boxes[index]), # type: ignore[arg-type]
|
||||
mask_sha256=mask_sha256,
|
||||
mask=mask,
|
||||
)
|
||||
)
|
||||
return MaskGroundingEvidence(
|
||||
source_file_sha256=source_file_sha256,
|
||||
source_pixel_sha256=source_pixel_sha256,
|
||||
detections=tuple(detections),
|
||||
)
|
||||
|
||||
|
||||
def cluster_mask_instances(
|
||||
detections: tuple[MaskGroundingDetection, ...],
|
||||
*,
|
||||
profile: MaskGeometryBindingProfile | None = None,
|
||||
) -> tuple[MaskInstance, ...]:
|
||||
"""Collapse caption duplicates into deterministic pixel-owned instances."""
|
||||
|
||||
selected_profile = profile or MaskGeometryBindingProfile()
|
||||
ordered = tuple(sorted(detections, key=lambda item: item.detection_id))
|
||||
parent = list(range(len(ordered)))
|
||||
|
||||
def find(index: int) -> int:
|
||||
while parent[index] != index:
|
||||
parent[index] = parent[parent[index]]
|
||||
index = parent[index]
|
||||
return index
|
||||
|
||||
def union(left: int, right: int) -> None:
|
||||
left_root, right_root = find(left), find(right)
|
||||
if left_root != right_root:
|
||||
parent[max(left_root, right_root)] = min(left_root, right_root)
|
||||
|
||||
for left in range(len(ordered)):
|
||||
for right in range(left):
|
||||
if _mask_iou(ordered[left].mask, ordered[right].mask) >= (
|
||||
selected_profile.minimum_mask_iou
|
||||
):
|
||||
union(left, right)
|
||||
grouped: dict[int, list[MaskGroundingDetection]] = defaultdict(list)
|
||||
for index, detection in enumerate(ordered):
|
||||
grouped[find(index)].append(detection)
|
||||
instances = []
|
||||
for ordinal, members in enumerate(grouped.values()):
|
||||
member_ids = "\n".join(item.detection_id for item in members).encode()
|
||||
digest = hashlib.sha256(member_ids).hexdigest()[:16]
|
||||
instances.append(
|
||||
MaskInstance(
|
||||
instance_id=f"mask-instance-{ordinal:03d}-{digest}",
|
||||
detections=tuple(members),
|
||||
)
|
||||
)
|
||||
return tuple(instances)
|
||||
|
||||
|
||||
def bind_mask_instances_to_geometry(
|
||||
instances: tuple[MaskInstance, ...],
|
||||
*,
|
||||
observations: tuple[ObstacleObservation, ...],
|
||||
projected: ProjectedPointCloud,
|
||||
profile: MaskGeometryBindingProfile | None = None,
|
||||
) -> tuple[MaskGeometryBinding, ...]:
|
||||
"""Bind masks only to exclusively supported occupied-geometry observations."""
|
||||
|
||||
selected_profile = profile or MaskGeometryBindingProfile()
|
||||
preliminary = tuple(
|
||||
_bind_instance(
|
||||
instance,
|
||||
observations=observations,
|
||||
projected=projected,
|
||||
profile=selected_profile,
|
||||
)
|
||||
for instance in instances
|
||||
)
|
||||
claims: dict[str, list[int]] = defaultdict(list)
|
||||
for index, binding in enumerate(preliminary):
|
||||
if binding.selected_observation_id is not None:
|
||||
claims[binding.selected_observation_id].append(index)
|
||||
collisions = {index for indices in claims.values() if len(indices) > 1 for index in indices}
|
||||
return tuple(
|
||||
MaskGeometryBinding(
|
||||
instance_id=item.instance_id,
|
||||
resolution=MaskBindingResolution.AMBIGUOUS,
|
||||
selected_observation_id=None,
|
||||
supports=item.supports,
|
||||
reason_code="observation-claimed-by-multiple-mask-instances",
|
||||
)
|
||||
if index in collisions
|
||||
else item
|
||||
for index, item in enumerate(preliminary)
|
||||
)
|
||||
|
||||
|
||||
def resolve_mask_instance_label(
|
||||
instance: MaskInstance,
|
||||
*,
|
||||
profile: MaskGeometryBindingProfile | None = None,
|
||||
) -> MaskLabelDecision:
|
||||
"""Resolve a name only when confidence and inter-label margin both pass."""
|
||||
|
||||
selected_profile = profile or MaskGeometryBindingProfile()
|
||||
ranked = instance.ranked_labels
|
||||
if not ranked or ranked[0][1] < selected_profile.minimum_label_confidence:
|
||||
return MaskLabelDecision(
|
||||
instance.instance_id,
|
||||
MaskLabelResolution.UNRESOLVED,
|
||||
None,
|
||||
ranked,
|
||||
"top-label-below-confidence",
|
||||
)
|
||||
runner_up = ranked[1][1] if len(ranked) > 1 else 0.0
|
||||
if ranked[0][1] - runner_up < selected_profile.minimum_label_margin:
|
||||
return MaskLabelDecision(
|
||||
instance.instance_id,
|
||||
MaskLabelResolution.AMBIGUOUS,
|
||||
None,
|
||||
ranked,
|
||||
"top-label-margin-insufficient",
|
||||
)
|
||||
return MaskLabelDecision(
|
||||
instance.instance_id,
|
||||
MaskLabelResolution.SELECTED,
|
||||
ranked[0][0],
|
||||
ranked,
|
||||
"confidence-and-margin-passed",
|
||||
)
|
||||
|
||||
|
||||
def _bind_instance(
|
||||
instance: MaskInstance,
|
||||
*,
|
||||
observations: tuple[ObstacleObservation, ...],
|
||||
projected: ProjectedPointCloud,
|
||||
profile: MaskGeometryBindingProfile,
|
||||
) -> MaskGeometryBinding:
|
||||
supports = []
|
||||
for observation in observations:
|
||||
selected = np.isin(
|
||||
projected.source_indices,
|
||||
np.asarray(observation.source_point_ids, dtype=np.int64),
|
||||
)
|
||||
pixels = projected.pixels_xy[selected]
|
||||
if pixels.shape[0] < profile.minimum_projected_points:
|
||||
continue
|
||||
integer_pixels = np.floor(pixels).astype(np.int64)
|
||||
x = integer_pixels[:, 0]
|
||||
y = integer_pixels[:, 1]
|
||||
support = max(
|
||||
float(detection.mask[y, x].mean()) for detection in instance.detections
|
||||
)
|
||||
supports.append(
|
||||
MaskObservationSupport(
|
||||
observation_id=observation.observation_id,
|
||||
projected_point_count=int(pixels.shape[0]),
|
||||
support_fraction=support,
|
||||
)
|
||||
)
|
||||
ordered = tuple(
|
||||
sorted(
|
||||
supports,
|
||||
key=lambda item: (-item.support_fraction, item.observation_id),
|
||||
)
|
||||
)
|
||||
if not ordered or ordered[0].support_fraction < profile.minimum_support_fraction:
|
||||
return MaskGeometryBinding(
|
||||
instance.instance_id,
|
||||
MaskBindingResolution.UNRESOLVED,
|
||||
None,
|
||||
ordered,
|
||||
"occupied-geometry-support-insufficient",
|
||||
)
|
||||
if (
|
||||
len(ordered) > 1
|
||||
and ordered[1].support_fraction > profile.maximum_secondary_support_fraction
|
||||
):
|
||||
return MaskGeometryBinding(
|
||||
instance.instance_id,
|
||||
MaskBindingResolution.AMBIGUOUS,
|
||||
None,
|
||||
ordered,
|
||||
"mask-covers-multiple-geometry-observations",
|
||||
)
|
||||
return MaskGeometryBinding(
|
||||
instance.instance_id,
|
||||
MaskBindingResolution.SELECTED,
|
||||
ordered[0].observation_id,
|
||||
ordered,
|
||||
"exclusive-occupied-geometry-support-passed",
|
||||
)
|
||||
|
||||
|
||||
def _mask_iou(left: BoolArray, right: BoolArray) -> float:
|
||||
intersection = int(np.logical_and(left, right).sum())
|
||||
union = int(np.logical_or(left, right).sum())
|
||||
return intersection / union if union else 0.0
|
||||
|
||||
|
||||
__all__ = [
|
||||
"MASK_GROUNDING_EVIDENCE_SCHEMA",
|
||||
"MaskBindingResolution",
|
||||
"MaskGeometryBinding",
|
||||
"MaskGeometryBindingProfile",
|
||||
"MaskGroundingDetection",
|
||||
"MaskGroundingEvidence",
|
||||
"MaskGroundingSemanticError",
|
||||
"MaskInstance",
|
||||
"MaskLabelDecision",
|
||||
"MaskLabelResolution",
|
||||
"MaskObservationSupport",
|
||||
"bind_mask_instances_to_geometry",
|
||||
"cluster_mask_instances",
|
||||
"load_mask_grounding_evidence",
|
||||
"resolve_mask_instance_label",
|
||||
]
|
||||
@@ -0,0 +1,929 @@
|
||||
"""Versioned semantic, state and advisory-risk projection for an obstacle.
|
||||
|
||||
The projection composes an immutable :class:`ObstacleObservation` instead of
|
||||
changing the strict v1 geometry contract. Semantic identity, observed state,
|
||||
class priors and advisory risk remain separate claims with explicit evidence.
|
||||
None of them can create occupancy or acquire navigation, safety or actuation
|
||||
authority.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from enum import StrEnum
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
from .contracts import FalseAuthority, MotionState, ObstacleObservation
|
||||
|
||||
OBJECT_UNDERSTANDING_SCHEMA: Final = "missioncore.object-understanding/v1"
|
||||
OBJECT_SEMANTIC_VOCABULARY_SCHEMA: Final = "missioncore.object-semantic-vocabulary/v0"
|
||||
MAX_SEMANTIC_HYPOTHESES: Final = 5
|
||||
|
||||
_IDENTIFIER = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:/-]{0,159}$")
|
||||
_NORMALIZED_LABEL = re.compile(r"^[a-z0-9][a-z0-9_]{0,79}$")
|
||||
|
||||
|
||||
class ObjectUnderstandingError(ValueError):
|
||||
"""An object-understanding document or vocabulary is incompatible."""
|
||||
|
||||
|
||||
class EvidenceKind(StrEnum):
|
||||
DETECTOR = "detector"
|
||||
SEMANTIC_MASK = "semantic-mask"
|
||||
HUMAN_REVIEW = "human-review"
|
||||
GEOMETRY = "geometry"
|
||||
TEMPORAL = "temporal"
|
||||
POLICY = "policy"
|
||||
|
||||
|
||||
class SemanticResolution(StrEnum):
|
||||
UNRESOLVED = "unresolved"
|
||||
SELECTED = "selected"
|
||||
AMBIGUOUS = "ambiguous"
|
||||
CONFLICT = "conflict"
|
||||
|
||||
|
||||
class AgencyState(StrEnum):
|
||||
UNKNOWN = "unknown"
|
||||
INERT = "inert"
|
||||
ANIMATE = "animate"
|
||||
SELF_PROPELLED = "self-propelled"
|
||||
|
||||
|
||||
class StateBasis(StrEnum):
|
||||
UNKNOWN = "unknown"
|
||||
OBSERVED = "observed"
|
||||
CLASS_PRIOR = "class-prior"
|
||||
FUSED = "fused"
|
||||
|
||||
|
||||
class RiskLevel(StrEnum):
|
||||
UNKNOWN = "unknown"
|
||||
LOW = "low"
|
||||
ELEVATED = "elevated"
|
||||
HIGH = "high"
|
||||
CRITICAL = "critical"
|
||||
|
||||
|
||||
class RiskBasis(StrEnum):
|
||||
UNKNOWN = "unknown"
|
||||
SEMANTIC_PRIOR = "semantic-prior"
|
||||
OBSERVED_STATE = "observed-state"
|
||||
GEOMETRY = "geometry"
|
||||
FUSED = "fused"
|
||||
|
||||
|
||||
class AdvisoryResponse(StrEnum):
|
||||
MONITOR = "monitor"
|
||||
REDUCE_SPEED = "reduce-speed"
|
||||
YIELD = "yield"
|
||||
STOP = "stop"
|
||||
ROUTE_AROUND = "route-around"
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class EvidenceProvenance:
|
||||
"""One source-bound evidence item used by semantic, state or risk claims."""
|
||||
|
||||
evidence_id: str
|
||||
kind: EvidenceKind
|
||||
source_id: str
|
||||
frame_id: str
|
||||
provider_id: str
|
||||
model_id: str | None
|
||||
model_revision: str | None
|
||||
preprocess_id: str | None
|
||||
prompt_set_id: str | None = None
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not isinstance(self.kind, EvidenceKind):
|
||||
raise ObjectUnderstandingError("evidence kind is invalid")
|
||||
for value, label in (
|
||||
(self.evidence_id, "evidence id"),
|
||||
(self.source_id, "evidence source id"),
|
||||
(self.frame_id, "evidence frame id"),
|
||||
(self.provider_id, "evidence provider id"),
|
||||
):
|
||||
_identifier(value, label)
|
||||
for optional_value, label in (
|
||||
(self.model_id, "evidence model id"),
|
||||
(self.model_revision, "evidence model revision"),
|
||||
(self.preprocess_id, "evidence preprocess id"),
|
||||
(self.prompt_set_id, "evidence prompt-set id"),
|
||||
):
|
||||
_optional_identifier(optional_value, label)
|
||||
if self.kind in {EvidenceKind.DETECTOR, EvidenceKind.SEMANTIC_MASK} and (
|
||||
self.model_id is None or self.model_revision is None or self.preprocess_id is None
|
||||
):
|
||||
raise ObjectUnderstandingError(
|
||||
"model evidence requires model, revision and preprocess identity"
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict[str, object]:
|
||||
return {
|
||||
"evidence_id": self.evidence_id,
|
||||
"kind": self.kind.value,
|
||||
"source_id": self.source_id,
|
||||
"frame_id": self.frame_id,
|
||||
"provider_id": self.provider_id,
|
||||
"model_id": self.model_id,
|
||||
"model_revision": self.model_revision,
|
||||
"preprocess_id": self.preprocess_id,
|
||||
"prompt_set_id": self.prompt_set_id,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: object) -> EvidenceProvenance:
|
||||
document = _object(value, "evidence provenance")
|
||||
_exact_keys(
|
||||
document,
|
||||
{
|
||||
"evidence_id",
|
||||
"kind",
|
||||
"source_id",
|
||||
"frame_id",
|
||||
"provider_id",
|
||||
"model_id",
|
||||
"model_revision",
|
||||
"preprocess_id",
|
||||
"prompt_set_id",
|
||||
},
|
||||
"evidence provenance",
|
||||
)
|
||||
return cls(
|
||||
evidence_id=_string(document, "evidence_id"),
|
||||
kind=_enum(EvidenceKind, document.get("kind"), "evidence kind"),
|
||||
source_id=_string(document, "source_id"),
|
||||
frame_id=_string(document, "frame_id"),
|
||||
provider_id=_string(document, "provider_id"),
|
||||
model_id=_optional_string(document.get("model_id"), "model id"),
|
||||
model_revision=_optional_string(document.get("model_revision"), "model revision"),
|
||||
preprocess_id=_optional_string(document.get("preprocess_id"), "preprocess id"),
|
||||
prompt_set_id=_optional_string(document.get("prompt_set_id"), "prompt-set id"),
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SemanticHypothesis:
|
||||
"""One ranked canonical class hypothesis, never an occupancy identity."""
|
||||
|
||||
rank: int
|
||||
class_id: str
|
||||
raw_label: str
|
||||
confidence: float
|
||||
evidence_ids: tuple[str, ...]
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_positive_integer(self.rank, "semantic rank")
|
||||
_identifier(self.class_id, "semantic class id")
|
||||
_label(self.raw_label, "raw semantic label")
|
||||
_confidence(self.confidence, "semantic confidence")
|
||||
_unique_identifiers(self.evidence_ids, "semantic evidence ids")
|
||||
|
||||
def to_dict(self) -> dict[str, object]:
|
||||
return {
|
||||
"rank": self.rank,
|
||||
"class_id": self.class_id,
|
||||
"raw_label": self.raw_label,
|
||||
"confidence": self.confidence,
|
||||
"evidence_ids": list(self.evidence_ids),
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: object) -> SemanticHypothesis:
|
||||
document = _object(value, "semantic hypothesis")
|
||||
_exact_keys(
|
||||
document,
|
||||
{"rank", "class_id", "raw_label", "confidence", "evidence_ids"},
|
||||
"semantic hypothesis",
|
||||
)
|
||||
return cls(
|
||||
rank=_integer(document, "rank"),
|
||||
class_id=_string(document, "class_id"),
|
||||
raw_label=_string(document, "raw_label"),
|
||||
confidence=_number(document, "confidence"),
|
||||
evidence_ids=_string_tuple(document.get("evidence_ids"), "evidence ids"),
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SemanticDecision:
|
||||
"""Resolution over ranked hypotheses; ambiguity remains first-class."""
|
||||
|
||||
resolution: SemanticResolution
|
||||
selected_class_id: str | None
|
||||
selected_confidence: float | None
|
||||
reason_codes: tuple[str, ...]
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not isinstance(self.resolution, SemanticResolution):
|
||||
raise ObjectUnderstandingError("semantic resolution is invalid")
|
||||
_optional_identifier(self.selected_class_id, "selected semantic class id")
|
||||
if self.selected_confidence is not None:
|
||||
_confidence(self.selected_confidence, "selected semantic confidence")
|
||||
_unique_identifiers(self.reason_codes, "semantic decision reasons")
|
||||
has_selection = self.selected_class_id is not None and self.selected_confidence is not None
|
||||
if self.resolution is SemanticResolution.SELECTED:
|
||||
if not has_selection:
|
||||
raise ObjectUnderstandingError("selected semantics require class and confidence")
|
||||
elif self.selected_class_id is not None or self.selected_confidence is not None:
|
||||
raise ObjectUnderstandingError("non-selected semantics cannot publish a selected class")
|
||||
|
||||
def to_dict(self) -> dict[str, object]:
|
||||
return {
|
||||
"resolution": self.resolution.value,
|
||||
"selected_class_id": self.selected_class_id,
|
||||
"selected_confidence": self.selected_confidence,
|
||||
"reason_codes": list(self.reason_codes),
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: object) -> SemanticDecision:
|
||||
document = _object(value, "semantic decision")
|
||||
_exact_keys(
|
||||
document,
|
||||
{
|
||||
"resolution",
|
||||
"selected_class_id",
|
||||
"selected_confidence",
|
||||
"reason_codes",
|
||||
},
|
||||
"semantic decision",
|
||||
)
|
||||
return cls(
|
||||
resolution=_enum(
|
||||
SemanticResolution,
|
||||
document.get("resolution"),
|
||||
"semantic resolution",
|
||||
),
|
||||
selected_class_id=_optional_string(
|
||||
document.get("selected_class_id"), "selected class id"
|
||||
),
|
||||
selected_confidence=_optional_number(
|
||||
document.get("selected_confidence"), "selected confidence"
|
||||
),
|
||||
reason_codes=_string_tuple(document.get("reason_codes"), "reason codes"),
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ObjectStateEstimate:
|
||||
"""Observed motion and agency prior, with their bases kept explicit."""
|
||||
|
||||
motion: MotionState
|
||||
motion_confidence: float
|
||||
agency: AgencyState
|
||||
agency_basis: StateBasis
|
||||
evidence_ids: tuple[str, ...]
|
||||
reason_codes: tuple[str, ...]
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not isinstance(self.motion, MotionState):
|
||||
raise ObjectUnderstandingError("motion state is invalid")
|
||||
if not isinstance(self.agency, AgencyState):
|
||||
raise ObjectUnderstandingError("agency state is invalid")
|
||||
if not isinstance(self.agency_basis, StateBasis):
|
||||
raise ObjectUnderstandingError("agency basis is invalid")
|
||||
_confidence(self.motion_confidence, "motion confidence")
|
||||
_unique_identifiers(self.evidence_ids, "state evidence ids", allow_empty=True)
|
||||
_unique_identifiers(self.reason_codes, "state reason codes")
|
||||
if self.agency is AgencyState.UNKNOWN:
|
||||
if self.agency_basis is not StateBasis.UNKNOWN:
|
||||
raise ObjectUnderstandingError("unknown agency must retain unknown evidence basis")
|
||||
elif self.agency_basis is StateBasis.UNKNOWN:
|
||||
raise ObjectUnderstandingError("agency claim requires an explicit basis")
|
||||
|
||||
def to_dict(self) -> dict[str, object]:
|
||||
return {
|
||||
"motion": self.motion.value,
|
||||
"motion_confidence": self.motion_confidence,
|
||||
"agency": self.agency.value,
|
||||
"agency_basis": self.agency_basis.value,
|
||||
"evidence_ids": list(self.evidence_ids),
|
||||
"reason_codes": list(self.reason_codes),
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: object) -> ObjectStateEstimate:
|
||||
document = _object(value, "object state")
|
||||
_exact_keys(
|
||||
document,
|
||||
{
|
||||
"motion",
|
||||
"motion_confidence",
|
||||
"agency",
|
||||
"agency_basis",
|
||||
"evidence_ids",
|
||||
"reason_codes",
|
||||
},
|
||||
"object state",
|
||||
)
|
||||
return cls(
|
||||
motion=_enum(MotionState, document.get("motion"), "motion state"),
|
||||
motion_confidence=_number(document, "motion_confidence"),
|
||||
agency=_enum(AgencyState, document.get("agency"), "agency state"),
|
||||
agency_basis=_enum(StateBasis, document.get("agency_basis"), "agency basis"),
|
||||
evidence_ids=_string_tuple(document.get("evidence_ids"), "evidence ids"),
|
||||
reason_codes=_string_tuple(document.get("reason_codes"), "reason codes"),
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class AdvisoryRiskAssessment:
|
||||
"""Evidence-qualified risk hint that is never a planner command."""
|
||||
|
||||
policy_id: str
|
||||
level: RiskLevel
|
||||
confidence: float
|
||||
basis: RiskBasis
|
||||
responses: tuple[AdvisoryResponse, ...]
|
||||
evidence_ids: tuple[str, ...]
|
||||
reason_codes: tuple[str, ...]
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not isinstance(self.level, RiskLevel):
|
||||
raise ObjectUnderstandingError("risk level is invalid")
|
||||
if not isinstance(self.basis, RiskBasis):
|
||||
raise ObjectUnderstandingError("risk basis is invalid")
|
||||
_identifier(self.policy_id, "risk policy id")
|
||||
_confidence(self.confidence, "risk confidence")
|
||||
_unique_enum_values(self.responses, "advisory responses", allow_empty=True)
|
||||
_unique_identifiers(self.evidence_ids, "risk evidence ids", allow_empty=True)
|
||||
_unique_identifiers(self.reason_codes, "risk reason codes")
|
||||
if self.level is RiskLevel.UNKNOWN:
|
||||
if self.basis is not RiskBasis.UNKNOWN or self.confidence != 0.0:
|
||||
raise ObjectUnderstandingError(
|
||||
"unknown risk must retain unknown basis and zero confidence"
|
||||
)
|
||||
elif self.basis is RiskBasis.UNKNOWN:
|
||||
raise ObjectUnderstandingError("risk claim requires an explicit basis")
|
||||
|
||||
def to_dict(self) -> dict[str, object]:
|
||||
return {
|
||||
"policy_id": self.policy_id,
|
||||
"level": self.level.value,
|
||||
"confidence": self.confidence,
|
||||
"basis": self.basis.value,
|
||||
"responses": [item.value for item in self.responses],
|
||||
"evidence_ids": list(self.evidence_ids),
|
||||
"reason_codes": list(self.reason_codes),
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: object) -> AdvisoryRiskAssessment:
|
||||
document = _object(value, "advisory risk")
|
||||
_exact_keys(
|
||||
document,
|
||||
{
|
||||
"policy_id",
|
||||
"level",
|
||||
"confidence",
|
||||
"basis",
|
||||
"responses",
|
||||
"evidence_ids",
|
||||
"reason_codes",
|
||||
},
|
||||
"advisory risk",
|
||||
)
|
||||
return cls(
|
||||
policy_id=_string(document, "policy_id"),
|
||||
level=_enum(RiskLevel, document.get("level"), "risk level"),
|
||||
confidence=_number(document, "confidence"),
|
||||
basis=_enum(RiskBasis, document.get("basis"), "risk basis"),
|
||||
responses=tuple(
|
||||
_enum(AdvisoryResponse, item, "advisory response")
|
||||
for item in _array(document, "responses")
|
||||
),
|
||||
evidence_ids=_string_tuple(document.get("evidence_ids"), "evidence ids"),
|
||||
reason_codes=_string_tuple(document.get("reason_codes"), "reason codes"),
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ObjectUnderstanding:
|
||||
"""Complete machine projection: geometry, semantics, state, risk and lineage."""
|
||||
|
||||
understanding_id: str
|
||||
vocabulary_id: str
|
||||
generated_monotonic_ns: int
|
||||
observation: ObstacleObservation
|
||||
hypotheses: tuple[SemanticHypothesis, ...]
|
||||
semantic: SemanticDecision
|
||||
state: ObjectStateEstimate
|
||||
risk: AdvisoryRiskAssessment
|
||||
provenance: tuple[EvidenceProvenance, ...]
|
||||
authority: FalseAuthority = FalseAuthority()
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_identifier(self.understanding_id, "understanding id")
|
||||
_identifier(self.vocabulary_id, "semantic vocabulary id")
|
||||
_nonnegative_integer(self.generated_monotonic_ns, "generation time")
|
||||
if not isinstance(self.observation, ObstacleObservation):
|
||||
raise ObjectUnderstandingError("object geometry observation is invalid")
|
||||
if (
|
||||
not isinstance(self.hypotheses, tuple)
|
||||
or any(not isinstance(item, SemanticHypothesis) for item in self.hypotheses)
|
||||
or len(self.hypotheses) > MAX_SEMANTIC_HYPOTHESES
|
||||
):
|
||||
raise ObjectUnderstandingError("semantic hypothesis set is invalid")
|
||||
if tuple(item.rank for item in self.hypotheses) != tuple(
|
||||
range(1, len(self.hypotheses) + 1)
|
||||
):
|
||||
raise ObjectUnderstandingError("semantic hypotheses must have contiguous ranks")
|
||||
class_ids = tuple(item.class_id for item in self.hypotheses)
|
||||
if len(set(class_ids)) != len(class_ids):
|
||||
raise ObjectUnderstandingError("semantic hypothesis classes must be unique")
|
||||
if any(
|
||||
self.hypotheses[index].confidence < self.hypotheses[index + 1].confidence
|
||||
for index in range(len(self.hypotheses) - 1)
|
||||
):
|
||||
raise ObjectUnderstandingError("semantic hypotheses must be ordered by confidence")
|
||||
if not isinstance(self.semantic, SemanticDecision):
|
||||
raise ObjectUnderstandingError("semantic decision is invalid")
|
||||
if not isinstance(self.state, ObjectStateEstimate):
|
||||
raise ObjectUnderstandingError("object state is invalid")
|
||||
if not isinstance(self.risk, AdvisoryRiskAssessment):
|
||||
raise ObjectUnderstandingError("advisory risk is invalid")
|
||||
if not isinstance(self.authority, FalseAuthority):
|
||||
raise ObjectUnderstandingError("object understanding authority is invalid")
|
||||
if not isinstance(self.provenance, tuple) or any(
|
||||
not isinstance(item, EvidenceProvenance) for item in self.provenance
|
||||
):
|
||||
raise ObjectUnderstandingError("evidence provenance is invalid")
|
||||
evidence_ids = tuple(item.evidence_id for item in self.provenance)
|
||||
if len(set(evidence_ids)) != len(evidence_ids):
|
||||
raise ObjectUnderstandingError("evidence provenance ids must be unique")
|
||||
if any(
|
||||
item.source_id != self.observation.source_id
|
||||
or item.frame_id != self.observation.frame_id
|
||||
for item in self.provenance
|
||||
):
|
||||
raise ObjectUnderstandingError("object evidence escaped its geometry source frame")
|
||||
known_evidence = set(evidence_ids)
|
||||
claimed_evidence = (
|
||||
{evidence_id for item in self.hypotheses for evidence_id in item.evidence_ids}
|
||||
| set(self.state.evidence_ids)
|
||||
| set(self.risk.evidence_ids)
|
||||
)
|
||||
if claimed_evidence - known_evidence:
|
||||
raise ObjectUnderstandingError("object claim references unknown evidence")
|
||||
if (
|
||||
self.semantic.resolution
|
||||
in {
|
||||
SemanticResolution.AMBIGUOUS,
|
||||
SemanticResolution.CONFLICT,
|
||||
}
|
||||
and not self.hypotheses
|
||||
):
|
||||
raise ObjectUnderstandingError("ambiguous or conflicting semantics require hypotheses")
|
||||
if self.semantic.resolution is SemanticResolution.CONFLICT and len(self.hypotheses) < 2:
|
||||
raise ObjectUnderstandingError("semantic conflict requires two hypotheses")
|
||||
if self.semantic.resolution is SemanticResolution.SELECTED:
|
||||
selected = next(
|
||||
(
|
||||
item
|
||||
for item in self.hypotheses
|
||||
if item.class_id == self.semantic.selected_class_id
|
||||
),
|
||||
None,
|
||||
)
|
||||
if selected is None or selected.confidence != self.semantic.selected_confidence:
|
||||
raise ObjectUnderstandingError(
|
||||
"selected semantics must match one ranked hypothesis"
|
||||
)
|
||||
|
||||
@property
|
||||
def occupancy_identity(self) -> str:
|
||||
"""Semantic or risk changes never replace the geometry-owned identity."""
|
||||
|
||||
return self.observation.occupancy_identity
|
||||
|
||||
def to_dict(self) -> dict[str, object]:
|
||||
return {
|
||||
"schema_version": OBJECT_UNDERSTANDING_SCHEMA,
|
||||
"understanding_id": self.understanding_id,
|
||||
"vocabulary_id": self.vocabulary_id,
|
||||
"generated_monotonic_ns": self.generated_monotonic_ns,
|
||||
"observation": self.observation.to_dict(),
|
||||
"hypotheses": [item.to_dict() for item in self.hypotheses],
|
||||
"semantic": self.semantic.to_dict(),
|
||||
"state": self.state.to_dict(),
|
||||
"risk": self.risk.to_dict(),
|
||||
"provenance": [item.to_dict() for item in self.provenance],
|
||||
"authority": self.authority.to_dict(),
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: object) -> ObjectUnderstanding:
|
||||
document = _contract(
|
||||
value,
|
||||
OBJECT_UNDERSTANDING_SCHEMA,
|
||||
{
|
||||
"understanding_id",
|
||||
"vocabulary_id",
|
||||
"generated_monotonic_ns",
|
||||
"observation",
|
||||
"hypotheses",
|
||||
"semantic",
|
||||
"state",
|
||||
"risk",
|
||||
"provenance",
|
||||
"authority",
|
||||
},
|
||||
"object understanding",
|
||||
)
|
||||
return cls(
|
||||
understanding_id=_string(document, "understanding_id"),
|
||||
vocabulary_id=_string(document, "vocabulary_id"),
|
||||
generated_monotonic_ns=_integer(document, "generated_monotonic_ns"),
|
||||
observation=ObstacleObservation.from_dict(document.get("observation")),
|
||||
hypotheses=tuple(
|
||||
SemanticHypothesis.from_dict(item) for item in _array(document, "hypotheses")
|
||||
),
|
||||
semantic=SemanticDecision.from_dict(document.get("semantic")),
|
||||
state=ObjectStateEstimate.from_dict(document.get("state")),
|
||||
risk=AdvisoryRiskAssessment.from_dict(document.get("risk")),
|
||||
provenance=tuple(
|
||||
EvidenceProvenance.from_dict(item) for item in _array(document, "provenance")
|
||||
),
|
||||
authority=FalseAuthority.from_dict(document.get("authority")),
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class CanonicalObjectClass:
|
||||
"""One class in the bounded experimental object vocabulary."""
|
||||
|
||||
class_id: str
|
||||
parent_id: str | None
|
||||
aliases: tuple[str, ...]
|
||||
agency_prior: AgencyState
|
||||
risk_traits: tuple[str, ...]
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not isinstance(self.agency_prior, AgencyState):
|
||||
raise ObjectUnderstandingError("canonical agency prior is invalid")
|
||||
_identifier(self.class_id, "canonical class id")
|
||||
_optional_identifier(self.parent_id, "canonical parent id")
|
||||
if not self.aliases:
|
||||
raise ObjectUnderstandingError("canonical class aliases must be nonempty")
|
||||
normalized = tuple(normalize_raw_label(item) for item in self.aliases)
|
||||
if len(set(normalized)) != len(normalized):
|
||||
raise ObjectUnderstandingError("canonical class aliases must be unique")
|
||||
_unique_identifiers(self.risk_traits, "class risk traits", allow_empty=True)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ObjectSemanticVocabulary:
|
||||
"""Loaded executable vocabulary profile; not a runtime ontology service."""
|
||||
|
||||
vocabulary_id: str
|
||||
status: str
|
||||
scope: str
|
||||
classes: tuple[CanonicalObjectClass, ...]
|
||||
max_hypotheses: int
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_identifier(self.vocabulary_id, "vocabulary id")
|
||||
if self.status != "experimental":
|
||||
raise ObjectUnderstandingError("object vocabulary must remain experimental")
|
||||
_identifier(self.scope, "vocabulary scope")
|
||||
if not 1 <= self.max_hypotheses <= MAX_SEMANTIC_HYPOTHESES:
|
||||
raise ObjectUnderstandingError("vocabulary top-k bound is invalid")
|
||||
if not self.classes:
|
||||
raise ObjectUnderstandingError("object vocabulary must declare classes")
|
||||
by_id = {item.class_id: item for item in self.classes}
|
||||
if len(by_id) != len(self.classes):
|
||||
raise ObjectUnderstandingError("canonical class ids must be unique")
|
||||
for item in self.classes:
|
||||
if item.parent_id is not None and item.parent_id not in by_id:
|
||||
raise ObjectUnderstandingError("canonical class parent is undeclared")
|
||||
seen = {item.class_id}
|
||||
parent_id = item.parent_id
|
||||
while parent_id is not None:
|
||||
if parent_id in seen:
|
||||
raise ObjectUnderstandingError("canonical class hierarchy is cyclic")
|
||||
seen.add(parent_id)
|
||||
parent_id = by_id[parent_id].parent_id
|
||||
aliases = [normalize_raw_label(alias) for item in self.classes for alias in item.aliases]
|
||||
if len(set(aliases)) != len(aliases):
|
||||
raise ObjectUnderstandingError("canonical aliases must be globally unique")
|
||||
|
||||
def class_definition(self, class_id: str) -> CanonicalObjectClass:
|
||||
for item in self.classes:
|
||||
if item.class_id == class_id:
|
||||
return item
|
||||
raise ObjectUnderstandingError("canonical class is undeclared")
|
||||
|
||||
def resolve_label(self, raw_label: str) -> str | None:
|
||||
normalized = normalize_raw_label(raw_label)
|
||||
for item in self.classes:
|
||||
if normalized in {normalize_raw_label(alias) for alias in item.aliases}:
|
||||
return item.class_id
|
||||
return None
|
||||
|
||||
def ancestors(self, class_id: str) -> tuple[str, ...]:
|
||||
by_id = {item.class_id: item for item in self.classes}
|
||||
current = self.class_definition(class_id)
|
||||
result: list[str] = []
|
||||
while current.parent_id is not None:
|
||||
result.append(current.parent_id)
|
||||
current = by_id[current.parent_id]
|
||||
return tuple(result)
|
||||
|
||||
|
||||
def load_object_semantic_vocabulary(path: Path) -> ObjectSemanticVocabulary:
|
||||
"""Load and fail-close an executable vocabulary profile."""
|
||||
|
||||
try:
|
||||
document = json.loads(path.expanduser().resolve(strict=True).read_text("utf-8"))
|
||||
except (OSError, json.JSONDecodeError) as exc:
|
||||
raise ObjectUnderstandingError("object vocabulary cannot be read") from exc
|
||||
root = _object(document, "object semantic vocabulary")
|
||||
_exact_keys(
|
||||
root,
|
||||
{
|
||||
"schema_version",
|
||||
"vocabulary_id",
|
||||
"status",
|
||||
"scope",
|
||||
"classes",
|
||||
"policies",
|
||||
},
|
||||
"object semantic vocabulary",
|
||||
)
|
||||
if root.get("schema_version") != OBJECT_SEMANTIC_VOCABULARY_SCHEMA:
|
||||
raise ObjectUnderstandingError("object vocabulary schema is incompatible")
|
||||
policies = _object(root.get("policies"), "object vocabulary policies")
|
||||
_exact_keys(
|
||||
policies,
|
||||
{
|
||||
"occupancy_independent_of_semantics",
|
||||
"unknown_preserves_obstacle",
|
||||
"class_prior_is_not_observed_state",
|
||||
"risk_is_advisory_only",
|
||||
"planner_command_authority",
|
||||
"max_hypotheses",
|
||||
},
|
||||
"object vocabulary policies",
|
||||
)
|
||||
required_true = (
|
||||
"occupancy_independent_of_semantics",
|
||||
"unknown_preserves_obstacle",
|
||||
"class_prior_is_not_observed_state",
|
||||
"risk_is_advisory_only",
|
||||
)
|
||||
if (
|
||||
any(policies.get(key) is not True for key in required_true)
|
||||
or policies.get("planner_command_authority") is not False
|
||||
):
|
||||
raise ObjectUnderstandingError("object vocabulary authority policy changed")
|
||||
classes: list[CanonicalObjectClass] = []
|
||||
for raw in _array(root, "classes"):
|
||||
row = _object(raw, "canonical object class")
|
||||
_exact_keys(
|
||||
row,
|
||||
{"class_id", "parent_id", "aliases", "agency_prior", "risk_traits"},
|
||||
"canonical object class",
|
||||
)
|
||||
classes.append(
|
||||
CanonicalObjectClass(
|
||||
class_id=_string(row, "class_id"),
|
||||
parent_id=_optional_string(row.get("parent_id"), "parent id"),
|
||||
aliases=_string_tuple(row.get("aliases"), "aliases"),
|
||||
agency_prior=_enum(AgencyState, row.get("agency_prior"), "agency prior"),
|
||||
risk_traits=_string_tuple(row.get("risk_traits"), "risk traits"),
|
||||
)
|
||||
)
|
||||
return ObjectSemanticVocabulary(
|
||||
vocabulary_id=_string(root, "vocabulary_id"),
|
||||
status=_string(root, "status"),
|
||||
scope=_string(root, "scope"),
|
||||
classes=tuple(classes),
|
||||
max_hypotheses=_integer(policies, "max_hypotheses"),
|
||||
)
|
||||
|
||||
|
||||
def validate_object_understanding(
|
||||
value: ObjectUnderstanding,
|
||||
vocabulary: ObjectSemanticVocabulary,
|
||||
) -> None:
|
||||
"""Validate canonical class references without changing the document."""
|
||||
|
||||
if not isinstance(value, ObjectUnderstanding):
|
||||
raise ObjectUnderstandingError("object understanding is invalid")
|
||||
if value.vocabulary_id != vocabulary.vocabulary_id:
|
||||
raise ObjectUnderstandingError("object understanding vocabulary changed")
|
||||
declared = {item.class_id for item in vocabulary.classes}
|
||||
referenced = {item.class_id for item in value.hypotheses}
|
||||
if value.semantic.selected_class_id is not None:
|
||||
referenced.add(value.semantic.selected_class_id)
|
||||
if referenced - declared:
|
||||
raise ObjectUnderstandingError("object understanding uses undeclared classes")
|
||||
if len(value.hypotheses) > vocabulary.max_hypotheses:
|
||||
raise ObjectUnderstandingError("object understanding exceeds vocabulary top-k")
|
||||
|
||||
|
||||
def normalize_raw_label(value: str) -> str:
|
||||
"""Normalize a provider label only for alias lookup, never as class truth."""
|
||||
|
||||
_label(value, "raw semantic label")
|
||||
normalized = re.sub(r"[_\s-]+", "_", value.strip().lower())
|
||||
if _NORMALIZED_LABEL.fullmatch(normalized) is None:
|
||||
raise ObjectUnderstandingError("raw semantic label cannot be normalized")
|
||||
return normalized
|
||||
|
||||
|
||||
def _contract(
|
||||
value: object,
|
||||
schema: str,
|
||||
fields: set[str],
|
||||
label: str,
|
||||
) -> dict[str, object]:
|
||||
document = _object(value, label)
|
||||
_exact_keys(document, {"schema_version", *fields}, label)
|
||||
if document.get("schema_version") != schema:
|
||||
raise ObjectUnderstandingError(f"{label} schema is incompatible")
|
||||
return document
|
||||
|
||||
|
||||
def _object(value: object, label: str) -> dict[str, object]:
|
||||
if not isinstance(value, dict) or any(not isinstance(key, str) for key in value):
|
||||
raise ObjectUnderstandingError(f"{label} must be an object")
|
||||
return value
|
||||
|
||||
|
||||
def _exact_keys(document: dict[str, object], expected: set[str], label: str) -> None:
|
||||
if set(document) != expected:
|
||||
raise ObjectUnderstandingError(f"{label} fields are incompatible")
|
||||
|
||||
|
||||
def _array(document: dict[str, object], key: str) -> list[object]:
|
||||
value = document.get(key)
|
||||
if not isinstance(value, list):
|
||||
raise ObjectUnderstandingError(f"{key} must be an array")
|
||||
return value
|
||||
|
||||
|
||||
def _string(document: dict[str, object], key: str) -> str:
|
||||
return _string_value(document.get(key), key)
|
||||
|
||||
|
||||
def _string_value(value: object, label: str) -> str:
|
||||
if not isinstance(value, str) or not value:
|
||||
raise ObjectUnderstandingError(f"{label} must be a nonempty string")
|
||||
return value
|
||||
|
||||
|
||||
def _optional_string(value: object, label: str) -> str | None:
|
||||
if value is None:
|
||||
return None
|
||||
return _string_value(value, label)
|
||||
|
||||
|
||||
def _integer(document: dict[str, object], key: str) -> int:
|
||||
value = document.get(key)
|
||||
if not isinstance(value, int) or isinstance(value, bool):
|
||||
raise ObjectUnderstandingError(f"{key} must be an integer")
|
||||
return value
|
||||
|
||||
|
||||
def _number(document: dict[str, object], key: str) -> float:
|
||||
value = document.get(key)
|
||||
if (
|
||||
not isinstance(value, (int, float))
|
||||
or isinstance(value, bool)
|
||||
or not math.isfinite(float(value))
|
||||
):
|
||||
raise ObjectUnderstandingError(f"{key} must be finite")
|
||||
return float(value)
|
||||
|
||||
|
||||
def _optional_number(value: object, label: str) -> float | None:
|
||||
if value is None:
|
||||
return None
|
||||
if (
|
||||
not isinstance(value, (int, float))
|
||||
or isinstance(value, bool)
|
||||
or not math.isfinite(float(value))
|
||||
):
|
||||
raise ObjectUnderstandingError(f"{label} must be finite")
|
||||
return float(value)
|
||||
|
||||
|
||||
def _identifier(value: str, label: str) -> None:
|
||||
if not isinstance(value, str) or _IDENTIFIER.fullmatch(value) is None:
|
||||
raise ObjectUnderstandingError(f"{label} is not a safe identifier")
|
||||
|
||||
|
||||
def _optional_identifier(value: str | None, label: str) -> None:
|
||||
if value is not None:
|
||||
_identifier(value, label)
|
||||
|
||||
|
||||
def _label(value: str, label: str) -> None:
|
||||
if (
|
||||
not isinstance(value, str)
|
||||
or not value
|
||||
or value != value.strip()
|
||||
or len(value) > 120
|
||||
or any(ord(character) < 32 for character in value)
|
||||
):
|
||||
raise ObjectUnderstandingError(f"{label} is invalid")
|
||||
|
||||
|
||||
def _nonnegative_integer(value: object, label: str) -> int:
|
||||
if not isinstance(value, int) or isinstance(value, bool) or value < 0:
|
||||
raise ObjectUnderstandingError(f"{label} must be a nonnegative integer")
|
||||
return value
|
||||
|
||||
|
||||
def _positive_integer(value: object, label: str) -> int:
|
||||
result = _nonnegative_integer(value, label)
|
||||
if result == 0:
|
||||
raise ObjectUnderstandingError(f"{label} must be positive")
|
||||
return result
|
||||
|
||||
|
||||
def _confidence(value: object, label: str) -> float:
|
||||
if (
|
||||
not isinstance(value, (int, float))
|
||||
or isinstance(value, bool)
|
||||
or not math.isfinite(float(value))
|
||||
or not 0.0 <= float(value) <= 1.0
|
||||
):
|
||||
raise ObjectUnderstandingError(f"{label} must be within [0, 1]")
|
||||
return float(value)
|
||||
|
||||
|
||||
def _unique_identifiers(
|
||||
values: tuple[str, ...],
|
||||
label: str,
|
||||
*,
|
||||
allow_empty: bool = False,
|
||||
) -> None:
|
||||
if (not values and not allow_empty) or len(set(values)) != len(values):
|
||||
raise ObjectUnderstandingError(f"{label} must be unique")
|
||||
for value in values:
|
||||
_identifier(value, label)
|
||||
|
||||
|
||||
def _unique_enum_values(
|
||||
values: tuple[AdvisoryResponse, ...],
|
||||
label: str,
|
||||
*,
|
||||
allow_empty: bool,
|
||||
) -> None:
|
||||
if (not values and not allow_empty) or len(set(values)) != len(values):
|
||||
raise ObjectUnderstandingError(f"{label} must be unique")
|
||||
if any(not isinstance(value, AdvisoryResponse) for value in values):
|
||||
raise ObjectUnderstandingError(f"{label} are invalid")
|
||||
|
||||
|
||||
def _string_tuple(value: object, label: str) -> tuple[str, ...]:
|
||||
if not isinstance(value, list):
|
||||
raise ObjectUnderstandingError(f"{label} must be an array")
|
||||
return tuple(_string_value(item, label) for item in value)
|
||||
|
||||
|
||||
def _enum[ENUM: StrEnum](
|
||||
enum_type: type[ENUM],
|
||||
value: object,
|
||||
label: str,
|
||||
) -> ENUM:
|
||||
if not isinstance(value, str):
|
||||
raise ObjectUnderstandingError(f"{label} must be a string")
|
||||
try:
|
||||
return enum_type(value)
|
||||
except ValueError as exc:
|
||||
raise ObjectUnderstandingError(f"{label} is incompatible") from exc
|
||||
|
||||
|
||||
__all__ = [
|
||||
"MAX_SEMANTIC_HYPOTHESES",
|
||||
"OBJECT_SEMANTIC_VOCABULARY_SCHEMA",
|
||||
"OBJECT_UNDERSTANDING_SCHEMA",
|
||||
"AdvisoryResponse",
|
||||
"AdvisoryRiskAssessment",
|
||||
"AgencyState",
|
||||
"CanonicalObjectClass",
|
||||
"EvidenceKind",
|
||||
"EvidenceProvenance",
|
||||
"ObjectSemanticVocabulary",
|
||||
"ObjectStateEstimate",
|
||||
"ObjectUnderstanding",
|
||||
"ObjectUnderstandingError",
|
||||
"RiskBasis",
|
||||
"RiskLevel",
|
||||
"SemanticDecision",
|
||||
"SemanticHypothesis",
|
||||
"SemanticResolution",
|
||||
"StateBasis",
|
||||
"load_object_semantic_vocabulary",
|
||||
"normalize_raw_label",
|
||||
"validate_object_understanding",
|
||||
]
|
||||
@@ -0,0 +1,784 @@
|
||||
"""Bounded open-vocabulary detections projected onto immutable obstacle geometry.
|
||||
|
||||
The adapter deliberately consumes detections in the raw KB4 image coordinate
|
||||
space used by the admitted geometry provider. It fuses prompt collisions into
|
||||
one spatial proposal, retains ranked canonical hypotheses and never turns a
|
||||
semantic label into occupancy, risk authority or a planner command.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import numpy as np
|
||||
from numpy.typing import NDArray
|
||||
|
||||
from .contracts import BoundingRegion2D, MotionState, ObjectProposal2D, ObstacleObservation
|
||||
from .object_understanding import (
|
||||
AdvisoryResponse,
|
||||
AdvisoryRiskAssessment,
|
||||
AgencyState,
|
||||
EvidenceKind,
|
||||
EvidenceProvenance,
|
||||
ObjectSemanticVocabulary,
|
||||
ObjectStateEstimate,
|
||||
ObjectUnderstanding,
|
||||
RiskBasis,
|
||||
RiskLevel,
|
||||
SemanticDecision,
|
||||
SemanticHypothesis,
|
||||
SemanticResolution,
|
||||
StateBasis,
|
||||
normalize_raw_label,
|
||||
validate_object_understanding,
|
||||
)
|
||||
|
||||
OPEN_VOCABULARY_SEMANTIC_SHADOW_PROFILE_SCHEMA: Final = (
|
||||
"missioncore.open-vocabulary-semantic-shadow-profile/v0"
|
||||
)
|
||||
TAO_GROUNDING_DINO_TRAILING_FIELDS: Final = 15
|
||||
|
||||
|
||||
class OpenVocabularySemanticError(ValueError):
|
||||
"""An open-vocabulary profile, label ledger or binding is incompatible."""
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class PromptGroup:
|
||||
prompt_set_id: str
|
||||
captions: tuple[str, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class OpenVocabularySemanticProfile:
|
||||
profile_id: str
|
||||
session_id: str
|
||||
source_id: str
|
||||
camera_source_id: str
|
||||
coordinate_space: str
|
||||
width: int
|
||||
height: int
|
||||
valid_fov_result_id: str
|
||||
valid_fov_mask_sha256: str
|
||||
valid_fov_fill_value: int
|
||||
provider_id: str
|
||||
provider_name: str
|
||||
model_id: str
|
||||
model_revision: str
|
||||
model_sha256: str
|
||||
engine_sha256: str
|
||||
container_reference: str
|
||||
container_image_id: str
|
||||
preprocess_id: str
|
||||
engine_input_width: int
|
||||
engine_input_height: int
|
||||
minimum_input_confidence: float
|
||||
minimum_box_area_fraction: float
|
||||
maximum_box_area_fraction: float
|
||||
minimum_valid_fov_fraction: float
|
||||
require_center_inside_valid_fov: bool
|
||||
fusion_iou_threshold: float
|
||||
selected_minimum_confidence: float
|
||||
selected_minimum_margin: float
|
||||
max_hypotheses: int
|
||||
prompt_groups: tuple[PromptGroup, ...]
|
||||
vocabulary_id: str
|
||||
risk_policy_id: str
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if self.coordinate_space != "raw-kb4" or (self.width, self.height) != (800, 600):
|
||||
raise OpenVocabularySemanticError("semantic shadow must remain in raw KB4 800x600")
|
||||
if self.width < 1 or self.height < 1:
|
||||
raise OpenVocabularySemanticError("semantic shadow raster is invalid")
|
||||
if self.valid_fov_fill_value != 114 or len(self.valid_fov_mask_sha256) != 64:
|
||||
raise OpenVocabularySemanticError("semantic valid-FOV preprocessing changed")
|
||||
if (self.engine_input_width, self.engine_input_height) != (960, 544):
|
||||
raise OpenVocabularySemanticError("semantic engine input raster changed")
|
||||
for value, label in (
|
||||
(self.minimum_input_confidence, "minimum input confidence"),
|
||||
(self.minimum_box_area_fraction, "minimum box area fraction"),
|
||||
(self.maximum_box_area_fraction, "maximum box area fraction"),
|
||||
(self.minimum_valid_fov_fraction, "minimum valid-FOV fraction"),
|
||||
(self.fusion_iou_threshold, "fusion IoU threshold"),
|
||||
(self.selected_minimum_confidence, "selection confidence"),
|
||||
(self.selected_minimum_margin, "selection margin"),
|
||||
):
|
||||
if not math.isfinite(value) or not 0.0 <= value <= 1.0:
|
||||
raise OpenVocabularySemanticError(f"{label} must be within [0, 1]")
|
||||
if (
|
||||
self.minimum_box_area_fraction <= 0.0
|
||||
or self.maximum_box_area_fraction <= self.minimum_box_area_fraction
|
||||
or not 1 <= self.max_hypotheses <= 5
|
||||
or not self.prompt_groups
|
||||
or self.require_center_inside_valid_fov is not True
|
||||
):
|
||||
raise OpenVocabularySemanticError("semantic shadow postprocessing is invalid")
|
||||
prompt_ids = tuple(item.prompt_set_id for item in self.prompt_groups)
|
||||
if len(set(prompt_ids)) != len(prompt_ids):
|
||||
raise OpenVocabularySemanticError("semantic prompt-set ids are duplicated")
|
||||
captions = tuple(
|
||||
normalize_raw_label(caption)
|
||||
for group in self.prompt_groups
|
||||
for caption in group.captions
|
||||
)
|
||||
if len(set(captions)) != len(captions):
|
||||
raise OpenVocabularySemanticError("semantic captions are duplicated")
|
||||
|
||||
def prompt_group(self, prompt_set_id: str) -> PromptGroup:
|
||||
for item in self.prompt_groups:
|
||||
if item.prompt_set_id == prompt_set_id:
|
||||
return item
|
||||
raise OpenVocabularySemanticError("semantic prompt set is undeclared")
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class OpenVocabularyDetection:
|
||||
detection_id: str
|
||||
source_id: str
|
||||
frame_id: str
|
||||
prompt_set_id: str
|
||||
raw_label: str
|
||||
confidence: float
|
||||
region: BoundingRegion2D
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not self.detection_id or not self.source_id or not self.frame_id:
|
||||
raise OpenVocabularySemanticError("semantic detection identity is invalid")
|
||||
normalize_raw_label(self.raw_label)
|
||||
if not math.isfinite(self.confidence) or not 0.0 <= self.confidence <= 1.0:
|
||||
raise OpenVocabularySemanticError("semantic detection confidence is invalid")
|
||||
|
||||
@property
|
||||
def evidence_id(self) -> str:
|
||||
return f"evidence:{self.detection_id}"
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SemanticProposalBinding:
|
||||
proposal: ObjectProposal2D
|
||||
detections: tuple[OpenVocabularyDetection, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SemanticFusionResult:
|
||||
bindings: tuple[SemanticProposalBinding, ...]
|
||||
input_detection_count: int
|
||||
below_confidence_count: int
|
||||
invalid_area_count: int
|
||||
outside_valid_fov_count: int
|
||||
|
||||
@property
|
||||
def proposals(self) -> tuple[ObjectProposal2D, ...]:
|
||||
return tuple(item.proposal for item in self.bindings)
|
||||
|
||||
@property
|
||||
def retained_detection_count(self) -> int:
|
||||
return sum(len(item.detections) for item in self.bindings)
|
||||
|
||||
|
||||
def load_open_vocabulary_semantic_profile(path: Path) -> OpenVocabularySemanticProfile:
|
||||
"""Load the experimental shadow profile and fail closed on authority drift."""
|
||||
|
||||
try:
|
||||
root = json.loads(path.expanduser().resolve(strict=True).read_text("utf-8"))
|
||||
except (OSError, json.JSONDecodeError) as exc:
|
||||
raise OpenVocabularySemanticError("semantic shadow profile cannot be read") from exc
|
||||
if not isinstance(root, dict) or root.get("schema_version") != (
|
||||
OPEN_VOCABULARY_SEMANTIC_SHADOW_PROFILE_SCHEMA
|
||||
):
|
||||
raise OpenVocabularySemanticError("semantic shadow profile schema changed")
|
||||
expected_root = {
|
||||
"schema_version",
|
||||
"profile_id",
|
||||
"status",
|
||||
"source",
|
||||
"provider",
|
||||
"postprocessing",
|
||||
"prompt_groups",
|
||||
"semantic_vocabulary_id",
|
||||
"risk_policy_id",
|
||||
"authority",
|
||||
}
|
||||
if set(root) != expected_root or root.get("status") != "experimental-shadow":
|
||||
raise OpenVocabularySemanticError("semantic shadow profile fields changed")
|
||||
source = _object(root.get("source"), "semantic source")
|
||||
provider = _object(root.get("provider"), "semantic provider")
|
||||
postprocessing = _object(root.get("postprocessing"), "semantic postprocessing")
|
||||
authority = _object(root.get("authority"), "semantic authority")
|
||||
if authority != _false_authority():
|
||||
raise OpenVocabularySemanticError("semantic shadow acquired authority")
|
||||
raw_groups = root.get("prompt_groups")
|
||||
if not isinstance(raw_groups, list):
|
||||
raise OpenVocabularySemanticError("semantic prompt groups must be an array")
|
||||
groups: list[PromptGroup] = []
|
||||
for raw_group in raw_groups:
|
||||
group = _object(raw_group, "semantic prompt group")
|
||||
if set(group) != {"prompt_set_id", "captions"}:
|
||||
raise OpenVocabularySemanticError("semantic prompt group fields changed")
|
||||
raw_captions = group.get("captions")
|
||||
if (
|
||||
not isinstance(raw_captions, list)
|
||||
or not raw_captions
|
||||
or any(not isinstance(item, str) or not item for item in raw_captions)
|
||||
):
|
||||
raise OpenVocabularySemanticError("semantic prompt captions are invalid")
|
||||
groups.append(
|
||||
PromptGroup(
|
||||
prompt_set_id=_string(group, "prompt_set_id"),
|
||||
captions=tuple(raw_captions),
|
||||
)
|
||||
)
|
||||
return OpenVocabularySemanticProfile(
|
||||
profile_id=_string(root, "profile_id"),
|
||||
session_id=_string(source, "session_id"),
|
||||
source_id=_string(source, "source_id"),
|
||||
camera_source_id=_string(source, "camera_source_id"),
|
||||
coordinate_space=_string(source, "coordinate_space"),
|
||||
width=_integer(source, "width"),
|
||||
height=_integer(source, "height"),
|
||||
valid_fov_result_id=_string(source, "valid_fov_result_id"),
|
||||
valid_fov_mask_sha256=_string(source, "valid_fov_mask_sha256"),
|
||||
valid_fov_fill_value=_integer(source, "valid_fov_fill_value"),
|
||||
provider_id=_string(provider, "provider_id"),
|
||||
provider_name=_string(provider, "name"),
|
||||
model_id=_string(provider, "model_id"),
|
||||
model_revision=_string(provider, "model_revision"),
|
||||
model_sha256=_string(provider, "model_sha256"),
|
||||
engine_sha256=_string(provider, "engine_sha256"),
|
||||
container_reference=_string(provider, "container_reference"),
|
||||
container_image_id=_string(provider, "container_image_id"),
|
||||
preprocess_id=_string(provider, "preprocess_id"),
|
||||
engine_input_width=_integer(provider, "engine_input_width"),
|
||||
engine_input_height=_integer(provider, "engine_input_height"),
|
||||
minimum_input_confidence=_number(postprocessing, "minimum_input_confidence"),
|
||||
minimum_box_area_fraction=_number(postprocessing, "minimum_box_area_fraction"),
|
||||
maximum_box_area_fraction=_number(postprocessing, "maximum_box_area_fraction"),
|
||||
minimum_valid_fov_fraction=_number(postprocessing, "minimum_valid_fov_fraction"),
|
||||
require_center_inside_valid_fov=_boolean(postprocessing, "require_center_inside_valid_fov"),
|
||||
fusion_iou_threshold=_number(postprocessing, "fusion_iou_threshold"),
|
||||
selected_minimum_confidence=_number(postprocessing, "selected_minimum_confidence"),
|
||||
selected_minimum_margin=_number(postprocessing, "selected_minimum_margin"),
|
||||
max_hypotheses=_integer(postprocessing, "max_hypotheses"),
|
||||
prompt_groups=tuple(groups),
|
||||
vocabulary_id=_string(root, "semantic_vocabulary_id"),
|
||||
risk_policy_id=_string(root, "risk_policy_id"),
|
||||
)
|
||||
|
||||
|
||||
def parse_tao_grounding_dino_labels(
|
||||
path: Path,
|
||||
*,
|
||||
source_id: str,
|
||||
frame_id: str,
|
||||
prompt_set_id: str,
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
image_width: int | None = None,
|
||||
image_height: int | None = None,
|
||||
offset_x: float = 0.0,
|
||||
offset_y: float = 0.0,
|
||||
detection_scope_id: str | None = None,
|
||||
) -> tuple[OpenVocabularyDetection, ...]:
|
||||
"""Parse one TAO label file while preserving original source-image boxes."""
|
||||
|
||||
group = profile.prompt_group(prompt_set_id)
|
||||
allowed = {normalize_raw_label(item) for item in group.captions}
|
||||
local_width = profile.width if image_width is None else image_width
|
||||
local_height = profile.height if image_height is None else image_height
|
||||
if (
|
||||
local_width < 1
|
||||
or local_height < 1
|
||||
or offset_x < 0.0
|
||||
or offset_y < 0.0
|
||||
or offset_x + local_width > profile.width
|
||||
or offset_y + local_height > profile.height
|
||||
):
|
||||
raise OpenVocabularySemanticError("semantic label image window is invalid")
|
||||
scope_id = frame_id if detection_scope_id is None else detection_scope_id
|
||||
if not scope_id:
|
||||
raise OpenVocabularySemanticError("semantic detection scope is invalid")
|
||||
resolved = path.resolve(strict=True)
|
||||
if resolved.is_symlink() or not resolved.is_file():
|
||||
raise OpenVocabularySemanticError("semantic label ledger must be a regular file")
|
||||
detections: list[OpenVocabularyDetection] = []
|
||||
for line_number, raw_line in enumerate(resolved.read_text("utf-8").splitlines(), start=1):
|
||||
if not raw_line.strip():
|
||||
continue
|
||||
parts = raw_line.split()
|
||||
if len(parts) <= TAO_GROUNDING_DINO_TRAILING_FIELDS:
|
||||
raise OpenVocabularySemanticError("TAO semantic label row is incomplete")
|
||||
label_parts = parts[:-TAO_GROUNDING_DINO_TRAILING_FIELDS]
|
||||
numeric_parts = parts[-TAO_GROUNDING_DINO_TRAILING_FIELDS:]
|
||||
raw_label = " ".join(label_parts)
|
||||
if normalize_raw_label(raw_label) not in allowed:
|
||||
raise OpenVocabularySemanticError("TAO semantic label escaped its prompt set")
|
||||
try:
|
||||
values = tuple(float(item) for item in numeric_parts)
|
||||
except ValueError as exc:
|
||||
raise OpenVocabularySemanticError("TAO semantic label row is not numeric") from exc
|
||||
if any(not math.isfinite(item) for item in values):
|
||||
raise OpenVocabularySemanticError("TAO semantic label row is not finite")
|
||||
x_min, y_min, x_max, y_max = values[3:7]
|
||||
if x_min < 0.0 or y_min < 0.0 or x_max > local_width or y_max > local_height:
|
||||
raise OpenVocabularySemanticError(
|
||||
"TAO semantic box is not in the declared raw image coordinate space"
|
||||
)
|
||||
detections.append(
|
||||
OpenVocabularyDetection(
|
||||
detection_id=(f"{scope_id}:{prompt_set_id}:label-{line_number:04d}"),
|
||||
source_id=source_id,
|
||||
frame_id=frame_id,
|
||||
prompt_set_id=prompt_set_id,
|
||||
raw_label=raw_label,
|
||||
confidence=values[-1],
|
||||
region=BoundingRegion2D(
|
||||
x_min + offset_x,
|
||||
y_min + offset_y,
|
||||
x_max + offset_x,
|
||||
y_max + offset_y,
|
||||
),
|
||||
)
|
||||
)
|
||||
return tuple(detections)
|
||||
|
||||
|
||||
def fuse_open_vocabulary_detections(
|
||||
detections: tuple[OpenVocabularyDetection, ...],
|
||||
*,
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
vocabulary: ObjectSemanticVocabulary,
|
||||
valid_fov_mask: NDArray[np.bool_],
|
||||
) -> SemanticFusionResult:
|
||||
"""Fuse prompt collisions spatially before the geometry provider owns points."""
|
||||
|
||||
if vocabulary.vocabulary_id != profile.vocabulary_id:
|
||||
raise OpenVocabularySemanticError("semantic profile and vocabulary disagree")
|
||||
if profile.max_hypotheses > vocabulary.max_hypotheses:
|
||||
raise OpenVocabularySemanticError("semantic profile exceeds vocabulary top-k")
|
||||
if not detections:
|
||||
return SemanticFusionResult((), 0, 0, 0, 0)
|
||||
if valid_fov_mask.shape != (profile.height, profile.width) or valid_fov_mask.dtype != np.bool_:
|
||||
raise OpenVocabularySemanticError("semantic valid-FOV mask is incompatible")
|
||||
source_ids = {item.source_id for item in detections}
|
||||
frame_ids = {item.frame_id for item in detections}
|
||||
if source_ids != {profile.source_id} or len(frame_ids) != 1:
|
||||
raise OpenVocabularySemanticError("semantic detections escaped one source frame")
|
||||
below_confidence = 0
|
||||
invalid_area = 0
|
||||
outside_valid_fov = 0
|
||||
retained: list[OpenVocabularyDetection] = []
|
||||
raster_area = float(profile.width * profile.height)
|
||||
for detection in detections:
|
||||
profile.prompt_group(detection.prompt_set_id)
|
||||
if vocabulary.resolve_label(detection.raw_label) is None:
|
||||
raise OpenVocabularySemanticError("semantic label is not in the vocabulary")
|
||||
if detection.confidence < profile.minimum_input_confidence:
|
||||
below_confidence += 1
|
||||
continue
|
||||
area_fraction = _area(detection.region) / raster_area
|
||||
if not (
|
||||
profile.minimum_box_area_fraction <= area_fraction <= profile.maximum_box_area_fraction
|
||||
):
|
||||
invalid_area += 1
|
||||
continue
|
||||
valid_fraction, center_inside = _valid_fov_support(detection.region, valid_fov_mask)
|
||||
if valid_fraction < profile.minimum_valid_fov_fraction or (
|
||||
profile.require_center_inside_valid_fov and not center_inside
|
||||
):
|
||||
outside_valid_fov += 1
|
||||
continue
|
||||
retained.append(detection)
|
||||
retained.sort(key=_detection_priority)
|
||||
clusters: list[list[OpenVocabularyDetection]] = []
|
||||
for detection in retained:
|
||||
spatial_candidates = tuple(
|
||||
(index, max(_iou(detection.region, item.region) for item in cluster))
|
||||
for index, cluster in enumerate(clusters)
|
||||
)
|
||||
match = max(spatial_candidates, key=lambda item: (item[1], -item[0]), default=None)
|
||||
if match is not None and match[1] >= profile.fusion_iou_threshold:
|
||||
clusters[match[0]].append(detection)
|
||||
else:
|
||||
clusters.append([detection])
|
||||
frame_id = next(iter(frame_ids))
|
||||
bindings: list[SemanticProposalBinding] = []
|
||||
for ordinal, cluster in enumerate(clusters):
|
||||
ordered = tuple(sorted(cluster, key=_detection_priority))
|
||||
semantic_candidates = _canonical_candidates(ordered, vocabulary, profile.max_hypotheses)
|
||||
if not semantic_candidates:
|
||||
raise OpenVocabularySemanticError("semantic cluster lost all hypotheses")
|
||||
anchor = ordered[0]
|
||||
proposal_id = f"semantic-{frame_id}-{ordinal:04d}"
|
||||
proposal = ObjectProposal2D(
|
||||
proposal_id=proposal_id,
|
||||
source_id=anchor.source_id,
|
||||
frame_id=anchor.frame_id,
|
||||
region=anchor.region,
|
||||
objectness=anchor.confidence,
|
||||
provider_id=profile.provider_id,
|
||||
model_id=profile.model_id,
|
||||
preprocess_id=profile.preprocess_id,
|
||||
semantic_hint=semantic_candidates[0][0],
|
||||
)
|
||||
bindings.append(SemanticProposalBinding(proposal=proposal, detections=ordered))
|
||||
return SemanticFusionResult(
|
||||
bindings=tuple(bindings),
|
||||
input_detection_count=len(detections),
|
||||
below_confidence_count=below_confidence,
|
||||
invalid_area_count=invalid_area,
|
||||
outside_valid_fov_count=outside_valid_fov,
|
||||
)
|
||||
|
||||
|
||||
def bind_object_understandings(
|
||||
observations: tuple[ObstacleObservation, ...],
|
||||
*,
|
||||
bindings: tuple[SemanticProposalBinding, ...],
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
vocabulary: ObjectSemanticVocabulary,
|
||||
generated_monotonic_ns: int,
|
||||
) -> tuple[ObjectUnderstanding, ...]:
|
||||
"""Compose semantic hypotheses around geometry-owned obstacle observations."""
|
||||
|
||||
by_proposal = {item.proposal.proposal_id: item for item in bindings}
|
||||
if len(by_proposal) != len(bindings):
|
||||
raise OpenVocabularySemanticError("semantic proposal bindings are duplicated")
|
||||
understandings: list[ObjectUnderstanding] = []
|
||||
seen_proposals: set[str] = set()
|
||||
for observation in observations:
|
||||
if not observation.proposal_ids:
|
||||
understanding = _unknown_understanding(
|
||||
observation,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
generated_monotonic_ns=generated_monotonic_ns,
|
||||
reason="geometry-only-semantic-evidence-unavailable",
|
||||
)
|
||||
else:
|
||||
if len(observation.proposal_ids) != 1:
|
||||
raise OpenVocabularySemanticError("semantic observation ownership is ambiguous")
|
||||
proposal_id = observation.proposal_ids[0]
|
||||
binding = by_proposal.get(proposal_id)
|
||||
if binding is None:
|
||||
raise OpenVocabularySemanticError(
|
||||
"geometry references an unknown semantic proposal"
|
||||
)
|
||||
seen_proposals.add(proposal_id)
|
||||
understanding = _semantic_understanding(
|
||||
observation,
|
||||
detections=binding.detections,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
generated_monotonic_ns=generated_monotonic_ns,
|
||||
)
|
||||
validate_object_understanding(understanding, vocabulary)
|
||||
understandings.append(understanding)
|
||||
if seen_proposals != set(by_proposal):
|
||||
raise OpenVocabularySemanticError("a semantic proposal has no geometry observation")
|
||||
return tuple(understandings)
|
||||
|
||||
|
||||
def understand_geometry_observation(
|
||||
observation: ObstacleObservation,
|
||||
*,
|
||||
detections: tuple[OpenVocabularyDetection, ...],
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
vocabulary: ObjectSemanticVocabulary,
|
||||
generated_monotonic_ns: int,
|
||||
) -> ObjectUnderstanding:
|
||||
"""Attach ROI semantics directly to geometry without changing its identity."""
|
||||
|
||||
if not detections:
|
||||
result = _unknown_understanding(
|
||||
observation,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
generated_monotonic_ns=generated_monotonic_ns,
|
||||
reason="geometry-roi-semantic-evidence-unavailable",
|
||||
)
|
||||
else:
|
||||
result = _semantic_understanding(
|
||||
observation,
|
||||
detections=detections,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
generated_monotonic_ns=generated_monotonic_ns,
|
||||
)
|
||||
validate_object_understanding(result, vocabulary)
|
||||
return result
|
||||
|
||||
|
||||
def _semantic_understanding(
|
||||
observation: ObstacleObservation,
|
||||
*,
|
||||
detections: tuple[OpenVocabularyDetection, ...],
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
vocabulary: ObjectSemanticVocabulary,
|
||||
generated_monotonic_ns: int,
|
||||
) -> ObjectUnderstanding:
|
||||
candidates = _canonical_candidates(
|
||||
detections,
|
||||
vocabulary,
|
||||
profile.max_hypotheses,
|
||||
)
|
||||
hypotheses: list[SemanticHypothesis] = []
|
||||
referenced: dict[str, OpenVocabularyDetection] = {}
|
||||
for rank, (class_id, raw_label, confidence, evidence) in enumerate(candidates, start=1):
|
||||
for detection in evidence:
|
||||
referenced[detection.evidence_id] = detection
|
||||
hypotheses.append(
|
||||
SemanticHypothesis(
|
||||
rank=rank,
|
||||
class_id=class_id,
|
||||
raw_label=raw_label,
|
||||
confidence=confidence,
|
||||
evidence_ids=tuple(item.evidence_id for item in evidence),
|
||||
)
|
||||
)
|
||||
top = hypotheses[0]
|
||||
runner_up = hypotheses[1] if len(hypotheses) > 1 else None
|
||||
if top.confidence < profile.selected_minimum_confidence:
|
||||
semantic = SemanticDecision(
|
||||
resolution=SemanticResolution.UNRESOLVED,
|
||||
selected_class_id=None,
|
||||
selected_confidence=None,
|
||||
reason_codes=("top-hypothesis-below-selection-threshold",),
|
||||
)
|
||||
elif (
|
||||
runner_up is not None
|
||||
and top.confidence - runner_up.confidence < profile.selected_minimum_margin
|
||||
):
|
||||
semantic = SemanticDecision(
|
||||
resolution=SemanticResolution.AMBIGUOUS,
|
||||
selected_class_id=None,
|
||||
selected_confidence=None,
|
||||
reason_codes=("top-hypothesis-margin-insufficient",),
|
||||
)
|
||||
else:
|
||||
semantic = SemanticDecision(
|
||||
resolution=SemanticResolution.SELECTED,
|
||||
selected_class_id=top.class_id,
|
||||
selected_confidence=top.confidence,
|
||||
reason_codes=("top-hypothesis-qualified",),
|
||||
)
|
||||
if semantic.resolution is SemanticResolution.SELECTED:
|
||||
definition = vocabulary.class_definition(top.class_id)
|
||||
agency = definition.agency_prior
|
||||
if agency is AgencyState.UNKNOWN:
|
||||
agency_basis = StateBasis.UNKNOWN
|
||||
state_evidence: tuple[str, ...] = ()
|
||||
state_reasons: tuple[str, ...] = (
|
||||
"observed-motion-unavailable",
|
||||
"selected-class-has-no-agency-prior",
|
||||
)
|
||||
else:
|
||||
agency_basis = StateBasis.CLASS_PRIOR
|
||||
state_evidence = top.evidence_ids
|
||||
state_reasons = (
|
||||
"observed-motion-unavailable",
|
||||
"agency-from-selected-class-prior",
|
||||
)
|
||||
else:
|
||||
agency = AgencyState.UNKNOWN
|
||||
agency_basis = StateBasis.UNKNOWN
|
||||
state_evidence = ()
|
||||
state_reasons = ("observed-motion-and-resolved-agency-unavailable",)
|
||||
provenance = tuple(
|
||||
EvidenceProvenance(
|
||||
evidence_id=detection.evidence_id,
|
||||
kind=EvidenceKind.DETECTOR,
|
||||
source_id=observation.source_id,
|
||||
frame_id=observation.frame_id,
|
||||
provider_id=profile.provider_id,
|
||||
model_id=profile.model_id,
|
||||
model_revision=profile.model_revision,
|
||||
preprocess_id=profile.preprocess_id,
|
||||
prompt_set_id=detection.prompt_set_id,
|
||||
)
|
||||
for detection in sorted(referenced.values(), key=lambda item: item.evidence_id)
|
||||
)
|
||||
return ObjectUnderstanding(
|
||||
understanding_id=f"{observation.observation_id}:understanding",
|
||||
vocabulary_id=vocabulary.vocabulary_id,
|
||||
generated_monotonic_ns=generated_monotonic_ns,
|
||||
observation=observation,
|
||||
hypotheses=tuple(hypotheses),
|
||||
semantic=semantic,
|
||||
state=ObjectStateEstimate(
|
||||
motion=MotionState.UNKNOWN,
|
||||
motion_confidence=0.0,
|
||||
agency=agency,
|
||||
agency_basis=agency_basis,
|
||||
evidence_ids=state_evidence,
|
||||
reason_codes=state_reasons,
|
||||
),
|
||||
risk=_unknown_risk(profile, "semantic-shadow-has-no-risk-authority"),
|
||||
provenance=provenance,
|
||||
)
|
||||
|
||||
|
||||
def _unknown_understanding(
|
||||
observation: ObstacleObservation,
|
||||
*,
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
vocabulary: ObjectSemanticVocabulary,
|
||||
generated_monotonic_ns: int,
|
||||
reason: str,
|
||||
) -> ObjectUnderstanding:
|
||||
return ObjectUnderstanding(
|
||||
understanding_id=f"{observation.observation_id}:understanding",
|
||||
vocabulary_id=vocabulary.vocabulary_id,
|
||||
generated_monotonic_ns=generated_monotonic_ns,
|
||||
observation=observation,
|
||||
hypotheses=(),
|
||||
semantic=SemanticDecision(
|
||||
resolution=SemanticResolution.UNRESOLVED,
|
||||
selected_class_id=None,
|
||||
selected_confidence=None,
|
||||
reason_codes=(reason,),
|
||||
),
|
||||
state=ObjectStateEstimate(
|
||||
motion=MotionState.UNKNOWN,
|
||||
motion_confidence=0.0,
|
||||
agency=AgencyState.UNKNOWN,
|
||||
agency_basis=StateBasis.UNKNOWN,
|
||||
evidence_ids=(),
|
||||
reason_codes=("state-evidence-unavailable",),
|
||||
),
|
||||
risk=_unknown_risk(profile, "unknown-object-remains-route-around-obstacle"),
|
||||
provenance=(),
|
||||
)
|
||||
|
||||
|
||||
def _unknown_risk(
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
reason: str,
|
||||
) -> AdvisoryRiskAssessment:
|
||||
return AdvisoryRiskAssessment(
|
||||
policy_id=profile.risk_policy_id,
|
||||
level=RiskLevel.UNKNOWN,
|
||||
confidence=0.0,
|
||||
basis=RiskBasis.UNKNOWN,
|
||||
responses=(AdvisoryResponse.ROUTE_AROUND,),
|
||||
evidence_ids=(),
|
||||
reason_codes=(reason,),
|
||||
)
|
||||
|
||||
|
||||
def _canonical_candidates(
|
||||
detections: tuple[OpenVocabularyDetection, ...],
|
||||
vocabulary: ObjectSemanticVocabulary,
|
||||
maximum: int,
|
||||
) -> tuple[tuple[str, str, float, tuple[OpenVocabularyDetection, ...]], ...]:
|
||||
by_class: dict[str, list[OpenVocabularyDetection]] = {}
|
||||
for detection in detections:
|
||||
class_id = vocabulary.resolve_label(detection.raw_label)
|
||||
if class_id is None:
|
||||
raise OpenVocabularySemanticError("semantic label is not in the vocabulary")
|
||||
by_class.setdefault(class_id, []).append(detection)
|
||||
candidates: list[tuple[str, str, float, tuple[OpenVocabularyDetection, ...]]] = []
|
||||
for class_id, evidence in by_class.items():
|
||||
ordered = tuple(sorted(evidence, key=_detection_priority))
|
||||
candidates.append((class_id, ordered[0].raw_label, ordered[0].confidence, ordered))
|
||||
candidates.sort(key=lambda item: (-item[2], item[0], normalize_raw_label(item[1])))
|
||||
return tuple(candidates[:maximum])
|
||||
|
||||
|
||||
def _detection_priority(item: OpenVocabularyDetection) -> tuple[float, float, str]:
|
||||
return (-item.confidence, _area(item.region), item.detection_id)
|
||||
|
||||
|
||||
def _area(region: BoundingRegion2D) -> float:
|
||||
return (region.x_max - region.x_min) * (region.y_max - region.y_min)
|
||||
|
||||
|
||||
def _iou(left: BoundingRegion2D, right: BoundingRegion2D) -> float:
|
||||
x_min = max(left.x_min, right.x_min)
|
||||
y_min = max(left.y_min, right.y_min)
|
||||
x_max = min(left.x_max, right.x_max)
|
||||
y_max = min(left.y_max, right.y_max)
|
||||
intersection = max(0.0, x_max - x_min) * max(0.0, y_max - y_min)
|
||||
union = _area(left) + _area(right) - intersection
|
||||
return 0.0 if union <= 0.0 else intersection / union
|
||||
|
||||
|
||||
def _valid_fov_support(
|
||||
region: BoundingRegion2D,
|
||||
mask: NDArray[np.bool_],
|
||||
) -> tuple[float, bool]:
|
||||
height, width = mask.shape
|
||||
x_min = int(np.clip(math.floor(region.x_min), 0, width))
|
||||
y_min = int(np.clip(math.floor(region.y_min), 0, height))
|
||||
x_max = int(np.clip(math.ceil(region.x_max), 0, width))
|
||||
y_max = int(np.clip(math.ceil(region.y_max), 0, height))
|
||||
area = max(0, x_max - x_min) * max(0, y_max - y_min)
|
||||
if area == 0:
|
||||
return 0.0, False
|
||||
valid_fraction = float(np.count_nonzero(mask[y_min:y_max, x_min:x_max])) / area
|
||||
center_x = int(np.clip(round((region.x_min + region.x_max) / 2.0), 0, width - 1))
|
||||
center_y = int(np.clip(round((region.y_min + region.y_max) / 2.0), 0, height - 1))
|
||||
return valid_fraction, bool(mask[center_y, center_x])
|
||||
|
||||
|
||||
def _false_authority() -> dict[str, bool]:
|
||||
return {
|
||||
"ground_truth": False,
|
||||
"independent_truth": False,
|
||||
"candidate_accepted": False,
|
||||
"commands_enabled": False,
|
||||
"actuation_allowed": False,
|
||||
"navigation_or_safety_accepted": False,
|
||||
}
|
||||
|
||||
|
||||
def _object(value: object, label: str) -> dict[str, object]:
|
||||
if not isinstance(value, dict) or any(not isinstance(key, str) for key in value):
|
||||
raise OpenVocabularySemanticError(f"{label} must be an object")
|
||||
return value
|
||||
|
||||
|
||||
def _string(document: dict[str, object], key: str) -> str:
|
||||
value = document.get(key)
|
||||
if not isinstance(value, str) or not value:
|
||||
raise OpenVocabularySemanticError(f"{key} must be a nonempty string")
|
||||
return value
|
||||
|
||||
|
||||
def _integer(document: dict[str, object], key: str) -> int:
|
||||
value = document.get(key)
|
||||
if not isinstance(value, int) or isinstance(value, bool):
|
||||
raise OpenVocabularySemanticError(f"{key} must be an integer")
|
||||
return value
|
||||
|
||||
|
||||
def _number(document: dict[str, object], key: str) -> float:
|
||||
value = document.get(key)
|
||||
if not isinstance(value, (int, float)) or isinstance(value, bool):
|
||||
raise OpenVocabularySemanticError(f"{key} must be numeric")
|
||||
return float(value)
|
||||
|
||||
|
||||
def _boolean(document: dict[str, object], key: str) -> bool:
|
||||
value = document.get(key)
|
||||
if not isinstance(value, bool):
|
||||
raise OpenVocabularySemanticError(f"{key} must be boolean")
|
||||
return value
|
||||
|
||||
|
||||
__all__ = [
|
||||
"OPEN_VOCABULARY_SEMANTIC_SHADOW_PROFILE_SCHEMA",
|
||||
"OpenVocabularyDetection",
|
||||
"OpenVocabularySemanticError",
|
||||
"OpenVocabularySemanticProfile",
|
||||
"PromptGroup",
|
||||
"SemanticFusionResult",
|
||||
"SemanticProposalBinding",
|
||||
"bind_object_understandings",
|
||||
"fuse_open_vocabulary_detections",
|
||||
"load_open_vocabulary_semantic_profile",
|
||||
"parse_tao_grounding_dino_labels",
|
||||
"understand_geometry_observation",
|
||||
]
|
||||
@@ -0,0 +1,768 @@
|
||||
"""Immutable bounded M48S replay for raw-KB4 semantic object understanding."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import uuid
|
||||
from collections import Counter
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import numpy as np
|
||||
import yaml
|
||||
from numpy.typing import NDArray
|
||||
from PIL import Image
|
||||
|
||||
from .baseline import BASELINE_RECORDED_JOB_ID
|
||||
from .contracts import EvidenceBasis, ObstacleObservation, SourceEnvelope
|
||||
from .detector_replay_result import (
|
||||
read_detector_replay_result,
|
||||
require_m4_detector_replay_acceptance,
|
||||
)
|
||||
from .geometry import (
|
||||
DEFAULT_GEOMETRY_PROFILE_PATH,
|
||||
Ravnoves00GeometryAssociationProvider,
|
||||
RecordedGeometryStore,
|
||||
load_geometry_profile,
|
||||
)
|
||||
from .graph_validation import validate_observations
|
||||
from .object_understanding import (
|
||||
ObjectUnderstanding,
|
||||
SemanticResolution,
|
||||
load_object_semantic_vocabulary,
|
||||
)
|
||||
from .open_vocabulary_semantics import (
|
||||
OpenVocabularyDetection,
|
||||
OpenVocabularySemanticError,
|
||||
OpenVocabularySemanticProfile,
|
||||
bind_object_understandings,
|
||||
fuse_open_vocabulary_detections,
|
||||
load_open_vocabulary_semantic_profile,
|
||||
parse_tao_grounding_dino_labels,
|
||||
)
|
||||
from .providers import SourcePacket
|
||||
from .recorded_source import RECORDED_SOURCE_PACK_ID, RecordedFrameReference
|
||||
from .yolox_object_detector import load_valid_fov_mask
|
||||
|
||||
SEMANTIC_SHADOW_REPLAY_SCHEMA: Final = "missioncore.m48s-semantic-shadow-replay/v0"
|
||||
SEMANTIC_SHADOW_REPORT_SCHEMA: Final = "missioncore.m48s-semantic-shadow-report/v0"
|
||||
SEMANTIC_WORKER_EXECUTION_SCHEMA: Final = "missioncore.m48s-semantic-worker-execution/v0"
|
||||
SEMANTIC_SHADOW_RESULT_PREFIX: Final = "m48s-semantic-shadow-"
|
||||
SEMANTIC_SHADOW_FRAMES_NAME: Final = "frames.jsonl"
|
||||
SEMANTIC_SHADOW_REPORT_NAME: Final = "report.json"
|
||||
SEMANTIC_SHADOW_MANIFEST_NAME: Final = "manifest.json"
|
||||
|
||||
|
||||
class SemanticShadowReplayError(RuntimeError):
|
||||
"""Bounded semantic shadow evidence is incomplete or internally inconsistent."""
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SemanticShadowReplayResult:
|
||||
result_id: str
|
||||
result_root: Path
|
||||
completed: bool
|
||||
accepted: bool
|
||||
metrics: dict[str, object]
|
||||
report: dict[str, object]
|
||||
manifest: dict[str, object]
|
||||
|
||||
|
||||
def build_semantic_shadow_replay(
|
||||
*,
|
||||
repository_root: Path,
|
||||
profile_path: Path,
|
||||
vocabulary_path: Path,
|
||||
detector_result_root: Path,
|
||||
source_frames_root: Path,
|
||||
inference_frames_root: Path,
|
||||
valid_fov_mask_path: Path,
|
||||
worker_result_roots: Mapping[str, Path],
|
||||
worker_identity_path: Path,
|
||||
frame_indices: tuple[int, ...],
|
||||
output_root: Path,
|
||||
) -> SemanticShadowReplayResult:
|
||||
"""Bind exact Worker labels to admitted geometry for a bounded frame set."""
|
||||
|
||||
repository = repository_root.resolve(strict=True)
|
||||
profile = load_open_vocabulary_semantic_profile(profile_path)
|
||||
vocabulary = load_object_semantic_vocabulary(vocabulary_path)
|
||||
if profile.vocabulary_id != vocabulary.vocabulary_id:
|
||||
raise SemanticShadowReplayError("semantic profile and vocabulary disagree")
|
||||
frames = _frame_selection(frame_indices)
|
||||
worker_identity = _validate_worker_identity(
|
||||
worker_identity_path,
|
||||
profile=profile,
|
||||
frame_indices=frames,
|
||||
)
|
||||
worker_artifacts = _validate_worker_results(
|
||||
worker_result_roots,
|
||||
profile=profile,
|
||||
frame_indices=frames,
|
||||
)
|
||||
valid_fov_mask = load_valid_fov_mask(
|
||||
valid_fov_mask_path,
|
||||
expected_sha256=profile.valid_fov_mask_sha256,
|
||||
)
|
||||
source_artifacts = _validate_source_frames(
|
||||
source_frames_root,
|
||||
profile=profile,
|
||||
frame_indices=frames,
|
||||
)
|
||||
inference_artifacts = _validate_inference_frames(
|
||||
inference_frames_root,
|
||||
source_frames_root=source_frames_root,
|
||||
valid_fov_mask=valid_fov_mask,
|
||||
profile=profile,
|
||||
frame_indices=frames,
|
||||
)
|
||||
detector = read_detector_replay_result(detector_result_root)
|
||||
require_m4_detector_replay_acceptance(detector)
|
||||
detector_by_sequence = {item.sequence: item for item in detector.frames}
|
||||
if any(index not in detector_by_sequence for index in frames):
|
||||
raise SemanticShadowReplayError("semantic frame escaped the detector timeline")
|
||||
geometry_profile = load_geometry_profile(repository / DEFAULT_GEOMETRY_PROFILE_PATH)
|
||||
store = RecordedGeometryStore.from_repository(repository, profile=geometry_profile)
|
||||
geometry = Ravnoves00GeometryAssociationProvider(store=store)
|
||||
|
||||
output = output_root.expanduser().absolute()
|
||||
output.mkdir(mode=0o700, parents=True, exist_ok=True)
|
||||
staging = output / f".semantic-shadow.{uuid.uuid4().hex}.tmp"
|
||||
staging.mkdir(mode=0o700, exist_ok=False)
|
||||
class_counts: Counter[str] = Counter()
|
||||
resolution_counts: Counter[str] = Counter()
|
||||
totals: Counter[str] = Counter()
|
||||
try:
|
||||
frames_path = staging / SEMANTIC_SHADOW_FRAMES_NAME
|
||||
with frames_path.open("wb") as ledger:
|
||||
for frame_index in frames:
|
||||
detector_frame = detector_by_sequence[frame_index]
|
||||
if detector_frame.outcome != "completed":
|
||||
raise SemanticShadowReplayError("accepted detector source frame failed")
|
||||
envelope = detector_frame.envelope
|
||||
detections = _frame_detections(
|
||||
worker_result_roots,
|
||||
frame_index=frame_index,
|
||||
envelope=envelope,
|
||||
profile=profile,
|
||||
)
|
||||
fusion = fuse_open_vocabulary_detections(
|
||||
detections,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
valid_fov_mask=valid_fov_mask,
|
||||
)
|
||||
packet = semantic_replay_packet(envelope)
|
||||
observations = geometry.associate(packet, fusion.proposals)
|
||||
validate_observations(packet, fusion.proposals, observations)
|
||||
understandings = bind_object_understandings(
|
||||
observations,
|
||||
bindings=fusion.bindings,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
generated_monotonic_ns=envelope.timestamps.monotonic_ns,
|
||||
)
|
||||
frame_metrics = _frame_metrics(
|
||||
detections=detections,
|
||||
fusion_input_count=fusion.input_detection_count,
|
||||
below_confidence_count=fusion.below_confidence_count,
|
||||
invalid_area_count=fusion.invalid_area_count,
|
||||
outside_valid_fov_count=fusion.outside_valid_fov_count,
|
||||
retained_detection_count=fusion.retained_detection_count,
|
||||
proposal_count=len(fusion.proposals),
|
||||
observations=observations,
|
||||
understandings=understandings,
|
||||
)
|
||||
for key, value in frame_metrics.items():
|
||||
totals[key] += value
|
||||
for understanding in understandings:
|
||||
resolution_counts[understanding.semantic.resolution.value] += 1
|
||||
if (
|
||||
understanding.semantic.resolution is SemanticResolution.SELECTED
|
||||
and understanding.semantic.selected_class_id is not None
|
||||
):
|
||||
class_counts[understanding.semantic.selected_class_id] += 1
|
||||
document = {
|
||||
"schema_version": "missioncore.m48s-semantic-shadow-frame/v0",
|
||||
"sequence": frame_index,
|
||||
"frame_id": envelope.frame_id,
|
||||
"source_envelope": envelope.to_dict(),
|
||||
"source_frame": source_artifacts[frame_index],
|
||||
"inference_frame": inference_artifacts[frame_index],
|
||||
"detections": [_detection_document(item) for item in detections],
|
||||
"proposals": [item.to_dict() for item in fusion.proposals],
|
||||
"understandings": [item.to_dict() for item in understandings],
|
||||
"metrics": frame_metrics,
|
||||
"authority": _false_authority(),
|
||||
}
|
||||
ledger.write(_canonical_json(document) + b"\n")
|
||||
metrics = {
|
||||
"frames": {
|
||||
"requested": len(frames),
|
||||
"completed": len(frames),
|
||||
"source_available": sum(
|
||||
detector_by_sequence[index].envelope.registered_point_increment.available
|
||||
for index in frames
|
||||
),
|
||||
},
|
||||
"detections": {
|
||||
"raw": totals["raw_detection_count"],
|
||||
"below_confidence": totals["below_confidence_count"],
|
||||
"invalid_area": totals["invalid_area_count"],
|
||||
"outside_valid_fov": totals["outside_valid_fov_count"],
|
||||
"retained": totals["retained_detection_count"],
|
||||
"fused_proposals": totals["proposal_count"],
|
||||
},
|
||||
"geometry": {
|
||||
"proposal_observations": totals["proposal_observation_count"],
|
||||
"ranged_proposal_observations": totals["ranged_proposal_observation_count"],
|
||||
"camera_only_proposal_observations": totals[
|
||||
"camera_only_proposal_observation_count"
|
||||
],
|
||||
"conflict_proposal_observations": totals["conflict_proposal_observation_count"],
|
||||
"geometry_only_observations": totals["geometry_only_observation_count"],
|
||||
},
|
||||
"semantics": {
|
||||
"resolution_counts": dict(sorted(resolution_counts.items())),
|
||||
"selected_class_counts": dict(sorted(class_counts.items())),
|
||||
},
|
||||
"authority": _false_authority(),
|
||||
}
|
||||
completed = (
|
||||
metrics["frames"]["completed"] == len(frames) # type: ignore[index]
|
||||
and totals["proposal_observation_count"] == totals["proposal_count"]
|
||||
and sum(resolution_counts.values())
|
||||
== totals["proposal_observation_count"] + totals["geometry_only_observation_count"]
|
||||
)
|
||||
frames_sha256 = _file_sha256(frames_path)
|
||||
detector_identity = _object(detector.manifest.get("identity"), "detector identity")
|
||||
identity = {
|
||||
"schema_version": SEMANTIC_SHADOW_REPLAY_SCHEMA,
|
||||
"profile_id": profile.profile_id,
|
||||
"profile_sha256": _file_sha256(profile_path),
|
||||
"vocabulary_id": vocabulary.vocabulary_id,
|
||||
"vocabulary_sha256": _file_sha256(vocabulary_path),
|
||||
"frame_indices": list(frames),
|
||||
"detector_result_id": detector.result_id,
|
||||
"detector_frames_sha256": detector_identity.get("frames_sha256"),
|
||||
"geometry_profile_id": geometry_profile.profile_id,
|
||||
"geometry_profile_sha256": geometry_profile.profile_sha256,
|
||||
"source_pack_id": geometry_profile.source_pack_id,
|
||||
"source_pack_sha256": geometry_profile.source_pack_sha256,
|
||||
"valid_fov": {
|
||||
"result_id": profile.valid_fov_result_id,
|
||||
"mask_sha256": profile.valid_fov_mask_sha256,
|
||||
"fill_value": profile.valid_fov_fill_value,
|
||||
},
|
||||
"source_frames": [source_artifacts[index] for index in frames],
|
||||
"inference_frames": [inference_artifacts[index] for index in frames],
|
||||
"worker_execution": worker_identity,
|
||||
"worker_execution_sha256": _file_sha256(worker_identity_path),
|
||||
"worker_artifacts": worker_artifacts,
|
||||
"producer_sha256": {
|
||||
name: _file_sha256(repository / "src/k1link/perception" / name)
|
||||
for name in (
|
||||
"object_understanding.py",
|
||||
"open_vocabulary_semantics.py",
|
||||
"semantic_shadow_replay.py",
|
||||
)
|
||||
},
|
||||
"frames_sha256": frames_sha256,
|
||||
"metrics": metrics,
|
||||
"completed": completed,
|
||||
"accepted": False,
|
||||
"authority": _false_authority(),
|
||||
}
|
||||
identity_sha256 = hashlib.sha256(_canonical_json(identity)).hexdigest()
|
||||
result_id = f"{SEMANTIC_SHADOW_RESULT_PREFIX}{identity_sha256}"
|
||||
created = datetime.now(UTC).isoformat(timespec="milliseconds").replace("+00:00", "Z")
|
||||
report = {
|
||||
"schema_version": SEMANTIC_SHADOW_REPORT_SCHEMA,
|
||||
"result_id": result_id,
|
||||
"identity_sha256": identity_sha256,
|
||||
"created_at_utc": created,
|
||||
"status": (
|
||||
"completed-shadow-evidence-not-accepted"
|
||||
if completed
|
||||
else "rejected-incomplete-shadow"
|
||||
),
|
||||
"completed": completed,
|
||||
"accepted": False,
|
||||
"metrics": metrics,
|
||||
"decision": {
|
||||
"raw_kb4_coordinate_binding_completed": completed,
|
||||
"semantic_quality_accepted": False,
|
||||
"risk_policy_accepted": False,
|
||||
"navigation_or_safety_accepted": False,
|
||||
"next_gate": "M48S human-reviewed semantic object quality slice",
|
||||
},
|
||||
"known_limits": [
|
||||
"bounded 11-frame diagnostic slice is not representative route truth",
|
||||
"Grounding DINO confidence is not calibrated class probability",
|
||||
"motion is intentionally unresolved without temporal evidence",
|
||||
"risk remains unknown and route-around advisory only",
|
||||
"TAO performed network metadata requests during tokenizer startup",
|
||||
],
|
||||
"authority": _false_authority(),
|
||||
}
|
||||
report_path = staging / SEMANTIC_SHADOW_REPORT_NAME
|
||||
_write_json(report_path, report)
|
||||
manifest = {
|
||||
"schema_version": SEMANTIC_SHADOW_REPLAY_SCHEMA,
|
||||
"result_id": result_id,
|
||||
"identity_sha256": identity_sha256,
|
||||
"identity": identity,
|
||||
"created_at_utc": created,
|
||||
"completed": completed,
|
||||
"accepted": False,
|
||||
"artifacts": [
|
||||
_artifact(frames_path, "semantic-shadow-frames"),
|
||||
_artifact(report_path, "semantic-shadow-report"),
|
||||
],
|
||||
"authority": _false_authority(),
|
||||
}
|
||||
_write_json(staging / SEMANTIC_SHADOW_MANIFEST_NAME, manifest)
|
||||
destination = output / result_id
|
||||
if destination.exists():
|
||||
shutil.rmtree(staging)
|
||||
return read_semantic_shadow_replay(destination)
|
||||
os.replace(staging, destination)
|
||||
return read_semantic_shadow_replay(destination)
|
||||
except (OpenVocabularySemanticError, OSError, ValueError) as exc:
|
||||
shutil.rmtree(staging, ignore_errors=True)
|
||||
raise SemanticShadowReplayError("semantic shadow replay failed closed") from exc
|
||||
except BaseException:
|
||||
shutil.rmtree(staging, ignore_errors=True)
|
||||
raise
|
||||
|
||||
|
||||
def read_semantic_shadow_replay(root: Path) -> SemanticShadowReplayResult:
|
||||
"""Read and verify one immutable semantic shadow replay result."""
|
||||
|
||||
resolved = root.resolve(strict=True)
|
||||
if resolved.is_symlink() or not resolved.name.startswith(SEMANTIC_SHADOW_RESULT_PREFIX):
|
||||
raise SemanticShadowReplayError("semantic shadow result root is invalid")
|
||||
manifest = _read_json(resolved / SEMANTIC_SHADOW_MANIFEST_NAME)
|
||||
identity = _object(manifest.get("identity"), "semantic shadow identity")
|
||||
identity_sha256 = hashlib.sha256(_canonical_json(identity)).hexdigest()
|
||||
if (
|
||||
manifest.get("schema_version") != SEMANTIC_SHADOW_REPLAY_SCHEMA
|
||||
or manifest.get("result_id") != resolved.name
|
||||
or manifest.get("identity_sha256") != identity_sha256
|
||||
or resolved.name != f"{SEMANTIC_SHADOW_RESULT_PREFIX}{identity_sha256}"
|
||||
or manifest.get("accepted") is not False
|
||||
or manifest.get("authority") != _false_authority()
|
||||
):
|
||||
raise SemanticShadowReplayError("semantic shadow manifest changed")
|
||||
frames_path = resolved / SEMANTIC_SHADOW_FRAMES_NAME
|
||||
report_path = resolved / SEMANTIC_SHADOW_REPORT_NAME
|
||||
if _file_sha256(frames_path) != identity.get("frames_sha256") or not _artifact_matches(
|
||||
manifest.get("artifacts"), frames_path, report_path
|
||||
):
|
||||
raise SemanticShadowReplayError("semantic shadow artifacts changed")
|
||||
report = _read_json(report_path)
|
||||
if (
|
||||
report.get("schema_version") != SEMANTIC_SHADOW_REPORT_SCHEMA
|
||||
or report.get("result_id") != resolved.name
|
||||
or report.get("identity_sha256") != identity_sha256
|
||||
or report.get("metrics") != identity.get("metrics")
|
||||
or report.get("completed") != identity.get("completed")
|
||||
or report.get("accepted") is not False
|
||||
or report.get("authority") != _false_authority()
|
||||
):
|
||||
raise SemanticShadowReplayError("semantic shadow report changed")
|
||||
metrics = _object(identity.get("metrics"), "semantic shadow metrics")
|
||||
return SemanticShadowReplayResult(
|
||||
result_id=resolved.name,
|
||||
result_root=resolved,
|
||||
completed=identity.get("completed") is True,
|
||||
accepted=False,
|
||||
metrics=metrics,
|
||||
report=report,
|
||||
manifest=manifest,
|
||||
)
|
||||
|
||||
|
||||
def _frame_selection(value: tuple[int, ...]) -> tuple[int, ...]:
|
||||
if (
|
||||
not value
|
||||
or tuple(sorted(set(value))) != value
|
||||
or any(isinstance(item, bool) or item < 0 or item >= 4489 for item in value)
|
||||
):
|
||||
raise SemanticShadowReplayError("semantic shadow frame selection is invalid")
|
||||
return value
|
||||
|
||||
|
||||
def _validate_worker_identity(
|
||||
path: Path,
|
||||
*,
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
frame_indices: tuple[int, ...],
|
||||
) -> dict[str, object]:
|
||||
document = _read_json(path.resolve(strict=True))
|
||||
if (
|
||||
document.get("schema_version") != SEMANTIC_WORKER_EXECUTION_SCHEMA
|
||||
or document.get("model_sha256") != profile.model_sha256
|
||||
or document.get("engine_sha256") != profile.engine_sha256
|
||||
or document.get("container_reference") != profile.container_reference
|
||||
or document.get("container_image_id") != profile.container_image_id
|
||||
or document.get("coordinate_space") != "raw-kb4-800x600"
|
||||
or document.get("preprocess_id") != profile.preprocess_id
|
||||
or document.get("valid_fov_mask_sha256") != profile.valid_fov_mask_sha256
|
||||
or document.get("frame_indices") != list(frame_indices)
|
||||
or document.get("authority") != _false_authority()
|
||||
):
|
||||
raise SemanticShadowReplayError("semantic Worker execution identity changed")
|
||||
prompt_runs = _object(document.get("prompt_runs"), "semantic prompt runs")
|
||||
if set(prompt_runs) != {item.prompt_set_id for item in profile.prompt_groups} or any(
|
||||
_object(value, "semantic prompt run").get("status") != "SUCCESS"
|
||||
for value in prompt_runs.values()
|
||||
):
|
||||
raise SemanticShadowReplayError("semantic Worker prompt run failed")
|
||||
return document
|
||||
|
||||
|
||||
def _validate_worker_results(
|
||||
roots: Mapping[str, Path],
|
||||
*,
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
frame_indices: tuple[int, ...],
|
||||
) -> list[dict[str, object]]:
|
||||
if set(roots) != {item.prompt_set_id for item in profile.prompt_groups}:
|
||||
raise SemanticShadowReplayError("semantic Worker result groups changed")
|
||||
expected_names = {f"frame-{index:06d}.txt" for index in frame_indices}
|
||||
result: list[dict[str, object]] = []
|
||||
for group in profile.prompt_groups:
|
||||
root = roots[group.prompt_set_id].resolve(strict=True)
|
||||
if root.is_symlink() or not root.is_dir():
|
||||
raise SemanticShadowReplayError("semantic Worker result root is invalid")
|
||||
status_path = root / "status.json"
|
||||
status_rows = _read_jsonl(status_path)
|
||||
if not status_rows or status_rows[-1].get("status") != "SUCCESS":
|
||||
raise SemanticShadowReplayError("semantic Worker result did not succeed")
|
||||
experiment_path = root / "experiment.yaml"
|
||||
try:
|
||||
experiment = yaml.safe_load(experiment_path.read_text("utf-8"))
|
||||
except (OSError, yaml.YAMLError) as exc:
|
||||
raise SemanticShadowReplayError("semantic Worker experiment cannot be read") from exc
|
||||
experiment_root = _object(experiment, "semantic Worker experiment")
|
||||
dataset = _object(experiment_root.get("dataset"), "semantic Worker dataset")
|
||||
sources = _object(dataset.get("infer_data_sources"), "semantic Worker sources")
|
||||
inference = _object(experiment_root.get("inference"), "semantic Worker inference")
|
||||
if (
|
||||
sources.get("captions") != list(group.captions)
|
||||
or sources.get("image_dir") != ["/workspace/probe/input"]
|
||||
or inference.get("conf_threshold") != profile.minimum_input_confidence
|
||||
or inference.get("input_width") != profile.engine_input_width
|
||||
or inference.get("input_height") != profile.engine_input_height
|
||||
or not str(inference.get("trt_engine", "")).endswith(
|
||||
"/grounding_dino_swin_tiny_commercial_fp16.engine"
|
||||
)
|
||||
):
|
||||
raise SemanticShadowReplayError("semantic Worker experiment drifted")
|
||||
labels_root = root / "labels"
|
||||
label_paths = tuple(sorted(labels_root.glob("*.txt")))
|
||||
if {item.name for item in label_paths} != expected_names or any(
|
||||
item.is_symlink() or not item.is_file() for item in label_paths
|
||||
):
|
||||
raise SemanticShadowReplayError("semantic Worker label accounting changed")
|
||||
result.append(
|
||||
{
|
||||
"prompt_set_id": group.prompt_set_id,
|
||||
"experiment_sha256": _file_sha256(experiment_path),
|
||||
"status_sha256": _file_sha256(status_path),
|
||||
"labels": [
|
||||
{
|
||||
"name": item.name,
|
||||
"byte_length": item.stat().st_size,
|
||||
"sha256": _file_sha256(item),
|
||||
}
|
||||
for item in label_paths
|
||||
],
|
||||
}
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def _validate_source_frames(
|
||||
root: Path,
|
||||
*,
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
frame_indices: tuple[int, ...],
|
||||
) -> dict[int, dict[str, object]]:
|
||||
resolved = root.resolve(strict=True)
|
||||
expected_names = {f"frame-{index:06d}.jpg" for index in frame_indices}
|
||||
paths = tuple(sorted(resolved.glob("*.jpg")))
|
||||
if {item.name for item in paths} != expected_names:
|
||||
raise SemanticShadowReplayError("semantic raw frame accounting changed")
|
||||
result: dict[int, dict[str, object]] = {}
|
||||
for index in frame_indices:
|
||||
path = resolved / f"frame-{index:06d}.jpg"
|
||||
if path.is_symlink() or not path.is_file():
|
||||
raise SemanticShadowReplayError("semantic raw frame is invalid")
|
||||
try:
|
||||
with Image.open(path) as image:
|
||||
image.verify()
|
||||
with Image.open(path) as image:
|
||||
dimensions = image.size
|
||||
image_format = image.format
|
||||
except OSError as exc:
|
||||
raise SemanticShadowReplayError("semantic raw frame cannot be decoded") from exc
|
||||
if dimensions != (profile.width, profile.height) or image_format != "JPEG":
|
||||
raise SemanticShadowReplayError("semantic raw frame coordinate space changed")
|
||||
result[index] = {
|
||||
"source_frame_index": index,
|
||||
"camera_sequence": index + 1,
|
||||
"name": path.name,
|
||||
"width": profile.width,
|
||||
"height": profile.height,
|
||||
"byte_length": path.stat().st_size,
|
||||
"sha256": _file_sha256(path),
|
||||
"coordinate_space": profile.coordinate_space,
|
||||
"exact_source_frame": True,
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def _validate_inference_frames(
|
||||
root: Path,
|
||||
*,
|
||||
source_frames_root: Path,
|
||||
valid_fov_mask: NDArray[np.bool_],
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
frame_indices: tuple[int, ...],
|
||||
) -> dict[int, dict[str, object]]:
|
||||
resolved = root.resolve(strict=True)
|
||||
source_root = source_frames_root.resolve(strict=True)
|
||||
expected_names = {f"frame-{index:06d}.png" for index in frame_indices}
|
||||
paths = tuple(sorted(resolved.glob("*.png")))
|
||||
if {item.name for item in paths} != expected_names:
|
||||
raise SemanticShadowReplayError("semantic inference frame accounting changed")
|
||||
result: dict[int, dict[str, object]] = {}
|
||||
for index in frame_indices:
|
||||
path = resolved / f"frame-{index:06d}.png"
|
||||
source_path = source_root / f"frame-{index:06d}.jpg"
|
||||
if path.is_symlink() or not path.is_file():
|
||||
raise SemanticShadowReplayError("semantic inference frame is invalid")
|
||||
try:
|
||||
with Image.open(path) as opened:
|
||||
image_format = opened.format
|
||||
inference = np.asarray(opened.convert("RGB"), dtype=np.uint8)
|
||||
with Image.open(source_path) as opened:
|
||||
source = np.asarray(opened.convert("RGB"), dtype=np.uint8)
|
||||
except OSError as exc:
|
||||
raise SemanticShadowReplayError("semantic inference frame cannot be decoded") from exc
|
||||
expected = np.where(
|
||||
valid_fov_mask[..., None],
|
||||
source,
|
||||
profile.valid_fov_fill_value,
|
||||
).astype(np.uint8)
|
||||
if (
|
||||
image_format != "PNG"
|
||||
or inference.shape != (profile.height, profile.width, 3)
|
||||
or not np.array_equal(inference, expected)
|
||||
):
|
||||
raise SemanticShadowReplayError(
|
||||
"semantic inference frame valid-FOV preprocessing changed"
|
||||
)
|
||||
result[index] = {
|
||||
"source_frame_index": index,
|
||||
"name": path.name,
|
||||
"width": profile.width,
|
||||
"height": profile.height,
|
||||
"byte_length": path.stat().st_size,
|
||||
"sha256": _file_sha256(path),
|
||||
"coordinate_space": profile.coordinate_space,
|
||||
"preprocess_id": profile.preprocess_id,
|
||||
"valid_fov_mask_sha256": profile.valid_fov_mask_sha256,
|
||||
"valid_fov_fill_value": profile.valid_fov_fill_value,
|
||||
"derived_from_source_sha256": _file_sha256(source_path),
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def _frame_detections(
|
||||
roots: Mapping[str, Path],
|
||||
*,
|
||||
frame_index: int,
|
||||
envelope: SourceEnvelope,
|
||||
profile: OpenVocabularySemanticProfile,
|
||||
) -> tuple[OpenVocabularyDetection, ...]:
|
||||
detections: list[OpenVocabularyDetection] = []
|
||||
for group in profile.prompt_groups:
|
||||
labels = (
|
||||
roots[group.prompt_set_id].resolve(strict=True)
|
||||
/ "labels"
|
||||
/ f"frame-{frame_index:06d}.txt"
|
||||
)
|
||||
detections.extend(
|
||||
parse_tao_grounding_dino_labels(
|
||||
labels,
|
||||
source_id=envelope.source_id,
|
||||
frame_id=envelope.frame_id,
|
||||
prompt_set_id=group.prompt_set_id,
|
||||
profile=profile,
|
||||
)
|
||||
)
|
||||
return tuple(detections)
|
||||
|
||||
|
||||
def semantic_replay_packet(envelope: SourceEnvelope) -> SourcePacket:
|
||||
"""Rebuild the exact digest-bound source packet used by semantic replays."""
|
||||
|
||||
image = RecordedFrameReference(BASELINE_RECORDED_JOB_ID, envelope.sequence)
|
||||
geometry = (
|
||||
RecordedFrameReference(RECORDED_SOURCE_PACK_ID, envelope.sequence)
|
||||
if envelope.registered_point_increment.available
|
||||
else None
|
||||
)
|
||||
return SourcePacket(
|
||||
envelope=envelope,
|
||||
image_payload=image,
|
||||
registered_point_increment_payload=geometry,
|
||||
pose_payload=geometry,
|
||||
)
|
||||
|
||||
|
||||
def _frame_metrics(
|
||||
*,
|
||||
detections: tuple[OpenVocabularyDetection, ...],
|
||||
fusion_input_count: int,
|
||||
below_confidence_count: int,
|
||||
invalid_area_count: int,
|
||||
outside_valid_fov_count: int,
|
||||
retained_detection_count: int,
|
||||
proposal_count: int,
|
||||
observations: tuple[ObstacleObservation, ...],
|
||||
understandings: tuple[ObjectUnderstanding, ...],
|
||||
) -> dict[str, int]:
|
||||
proposal_observations = tuple(item for item in observations if item.proposal_ids)
|
||||
geometry_only = tuple(item for item in observations if not item.proposal_ids)
|
||||
if len(detections) != fusion_input_count or len(understandings) != len(observations):
|
||||
raise SemanticShadowReplayError("semantic frame accounting changed")
|
||||
ranged = sum(item.metric_geometry is not None for item in proposal_observations)
|
||||
conflict = sum(item.basis is EvidenceBasis.CONFLICT for item in proposal_observations)
|
||||
return {
|
||||
"raw_detection_count": len(detections),
|
||||
"below_confidence_count": below_confidence_count,
|
||||
"invalid_area_count": invalid_area_count,
|
||||
"outside_valid_fov_count": outside_valid_fov_count,
|
||||
"retained_detection_count": retained_detection_count,
|
||||
"proposal_count": proposal_count,
|
||||
"proposal_observation_count": len(proposal_observations),
|
||||
"ranged_proposal_observation_count": ranged,
|
||||
"camera_only_proposal_observation_count": (len(proposal_observations) - ranged - conflict),
|
||||
"conflict_proposal_observation_count": conflict,
|
||||
"geometry_only_observation_count": len(geometry_only),
|
||||
}
|
||||
|
||||
|
||||
def _detection_document(value: OpenVocabularyDetection) -> dict[str, object]:
|
||||
return {
|
||||
"detection_id": value.detection_id,
|
||||
"source_id": value.source_id,
|
||||
"frame_id": value.frame_id,
|
||||
"prompt_set_id": value.prompt_set_id,
|
||||
"raw_label": value.raw_label,
|
||||
"confidence": value.confidence,
|
||||
"region": value.region.to_dict(),
|
||||
}
|
||||
|
||||
|
||||
def _artifact(path: Path, role: str) -> dict[str, object]:
|
||||
return {
|
||||
"role": role,
|
||||
"path": path.name,
|
||||
"byte_length": path.stat().st_size,
|
||||
"sha256": _file_sha256(path),
|
||||
}
|
||||
|
||||
|
||||
def _artifact_matches(value: object, frames_path: Path, report_path: Path) -> bool:
|
||||
if not isinstance(value, list) or len(value) != 2:
|
||||
return False
|
||||
expected = {
|
||||
path.name: (path.stat().st_size, _file_sha256(path)) for path in (frames_path, report_path)
|
||||
}
|
||||
seen: set[str] = set()
|
||||
for item in value:
|
||||
if not isinstance(item, dict):
|
||||
return False
|
||||
path = item.get("path")
|
||||
if not isinstance(path, str) or path not in expected:
|
||||
return False
|
||||
seen.add(path)
|
||||
if (item.get("byte_length"), item.get("sha256")) != expected[path]:
|
||||
return False
|
||||
return seen == set(expected)
|
||||
|
||||
|
||||
def _read_json(path: Path) -> dict[str, object]:
|
||||
try:
|
||||
value = json.loads(path.read_text("utf-8"))
|
||||
except (OSError, json.JSONDecodeError) as exc:
|
||||
raise SemanticShadowReplayError(f"cannot read semantic JSON: {path.name}") from exc
|
||||
return _object(value, path.name)
|
||||
|
||||
|
||||
def _read_jsonl(path: Path) -> list[dict[str, object]]:
|
||||
try:
|
||||
rows = [json.loads(line) for line in path.read_text("utf-8").splitlines() if line]
|
||||
except (OSError, json.JSONDecodeError) as exc:
|
||||
raise SemanticShadowReplayError(f"cannot read semantic JSONL: {path.name}") from exc
|
||||
return [_object(item, path.name) for item in rows]
|
||||
|
||||
|
||||
def _write_json(path: Path, value: object) -> None:
|
||||
path.write_bytes(_canonical_json(value) + b"\n")
|
||||
|
||||
|
||||
def _canonical_json(value: object) -> bytes:
|
||||
return json.dumps(
|
||||
value,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
allow_nan=False,
|
||||
).encode("utf-8")
|
||||
|
||||
|
||||
def _file_sha256(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.expanduser().resolve(strict=True).open("rb") as stream:
|
||||
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def _object(value: object, label: str) -> dict[str, object]:
|
||||
if not isinstance(value, dict) or any(not isinstance(key, str) for key in value):
|
||||
raise SemanticShadowReplayError(f"{label} must be an object")
|
||||
return value
|
||||
|
||||
|
||||
def _false_authority() -> dict[str, bool]:
|
||||
return {
|
||||
"ground_truth": False,
|
||||
"independent_truth": False,
|
||||
"candidate_accepted": False,
|
||||
"commands_enabled": False,
|
||||
"actuation_allowed": False,
|
||||
"navigation_or_safety_accepted": False,
|
||||
}
|
||||
|
||||
|
||||
__all__ = [
|
||||
"SEMANTIC_SHADOW_REPLAY_SCHEMA",
|
||||
"SemanticShadowReplayError",
|
||||
"SemanticShadowReplayResult",
|
||||
"build_semantic_shadow_replay",
|
||||
"read_semantic_shadow_replay",
|
||||
"semantic_replay_packet",
|
||||
]
|
||||
@@ -0,0 +1,158 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
from k1link.perception.contracts import (
|
||||
EvidenceBasis,
|
||||
EvidenceCurrentness,
|
||||
MetricGeometry,
|
||||
ObstacleObservation,
|
||||
)
|
||||
from k1link.perception.geometry import GeometryFrame
|
||||
from k1link.perception.geometry_math import Kb4ProjectionProfile
|
||||
from k1link.perception.geometry_semantic_roi import (
|
||||
build_geometry_semantic_rois,
|
||||
materialize_geometry_semantic_crop,
|
||||
select_geometry_roi_detections,
|
||||
)
|
||||
from k1link.perception.object_understanding import load_object_semantic_vocabulary
|
||||
from k1link.perception.open_vocabulary_semantics import (
|
||||
OpenVocabularyDetection,
|
||||
fuse_open_vocabulary_detections,
|
||||
load_open_vocabulary_semantic_profile,
|
||||
)
|
||||
|
||||
REPOSITORY_ROOT = Path(__file__).resolve().parents[1]
|
||||
PROFILE_PATH = REPOSITORY_ROOT / "config/perception/open-vocabulary-semantic-shadow-v0.json"
|
||||
VOCABULARY_PATH = REPOSITORY_ROOT / "config/perception/object-semantic-vocabulary-v0.json"
|
||||
|
||||
|
||||
def _point_for_pixel(u: float, v: float, *, z: float = 5.0) -> tuple[float, float, float]:
|
||||
theta_x = (u - 400.0) / 100.0
|
||||
theta_y = (v - 300.0) / 100.0
|
||||
ray = np.asarray((np.tan(theta_x), np.tan(theta_y), 1.0), dtype=np.float64)
|
||||
return tuple(float(item) for item in ray * z) # type: ignore[return-value]
|
||||
|
||||
|
||||
def _frame() -> GeometryFrame:
|
||||
points = np.asarray(
|
||||
(
|
||||
_point_for_pixel(350.0, 250.0),
|
||||
_point_for_pixel(450.0, 250.0),
|
||||
_point_for_pixel(350.0, 350.0),
|
||||
_point_for_pixel(450.0, 350.0),
|
||||
_point_for_pixel(600.0, 300.0),
|
||||
),
|
||||
dtype=np.float64,
|
||||
)
|
||||
return GeometryFrame(
|
||||
frame_index=121,
|
||||
points_map=points,
|
||||
point_class=np.full(points.shape[0], 2, dtype=np.uint8),
|
||||
sensor_position_map=np.zeros(3, dtype=np.float64),
|
||||
sensor_orientation_xyzw=np.asarray((0.0, 0.0, 0.0, 1.0), dtype=np.float64),
|
||||
projection=Kb4ProjectionProfile(
|
||||
width=800,
|
||||
height=600,
|
||||
intrinsic_fx_fy_cx_cy=(100.0, 100.0, 400.0, 300.0),
|
||||
distortion_kb4=(0.0, 0.0, 0.0, 0.0),
|
||||
t_camera_from_lidar=np.eye(4, dtype=np.float64),
|
||||
),
|
||||
surface_valid=True,
|
||||
)
|
||||
|
||||
|
||||
def _observation(ordinal: int, source_point_ids: tuple[int, ...]) -> ObstacleObservation:
|
||||
return ObstacleObservation(
|
||||
observation_id=f"frame-000121:observation-{ordinal}",
|
||||
occupancy_key=f"frame-000121:occupancy-{ordinal}",
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000121",
|
||||
evidence_time_ns=121,
|
||||
basis=EvidenceBasis.LIDAR,
|
||||
currentness=EvidenceCurrentness.CURRENT,
|
||||
occupied_support=True,
|
||||
source_point_ids=source_point_ids,
|
||||
metric_geometry=MetricGeometry(
|
||||
coordinate_frame="map",
|
||||
centroid_xyz_m=(1.0, 2.0, 0.5),
|
||||
range_m=2.2,
|
||||
covariance_diagonal_m2=(0.1, 0.1, 0.1),
|
||||
),
|
||||
proposal_ids=(),
|
||||
semantic_hint=None,
|
||||
reason_codes=("test-geometry-only",),
|
||||
)
|
||||
|
||||
|
||||
def test_geometry_points_own_crop_and_sparse_observation_stays_unprojected(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
result = build_geometry_semantic_rois(
|
||||
frame=_frame(),
|
||||
observations=(_observation(0, (0, 1, 2, 3)), _observation(1, (4,))),
|
||||
)
|
||||
|
||||
assert len(result.rois) == 1
|
||||
assert result.not_projected_observations[0].observation_id.endswith("observation-1")
|
||||
roi = result.rois[0]
|
||||
assert roi.observation.source_point_ids == (0, 1, 2, 3)
|
||||
assert roi.core_region.x_min < 400.0 < roi.core_region.x_max
|
||||
assert roi.core_region.y_min < 300.0 < roi.core_region.y_max
|
||||
assert roi.crop_region.x_min <= roi.core_region.x_min
|
||||
assert roi.crop_region.y_max >= roi.core_region.y_max
|
||||
|
||||
source = tmp_path / "frame-000121.png"
|
||||
Image.new("RGB", (800, 600), (114, 114, 114)).save(source)
|
||||
destination = tmp_path / roi.crop_name
|
||||
materialize_geometry_semantic_crop(
|
||||
image_path=source,
|
||||
roi=roi,
|
||||
destination=destination,
|
||||
)
|
||||
with Image.open(destination) as crop:
|
||||
assert crop.size == (
|
||||
int(roi.crop_region.x_max - roi.crop_region.x_min),
|
||||
int(roi.crop_region.y_max - roi.crop_region.y_min),
|
||||
)
|
||||
|
||||
|
||||
def test_roi_selection_returns_only_cluster_covering_geometry_core() -> None:
|
||||
profile = load_open_vocabulary_semantic_profile(PROFILE_PATH)
|
||||
vocabulary = load_object_semantic_vocabulary(VOCABULARY_PATH)
|
||||
roi = build_geometry_semantic_rois(
|
||||
frame=_frame(),
|
||||
observations=(_observation(0, (0, 1, 2, 3)),),
|
||||
).rois[0]
|
||||
fusion = fuse_open_vocabulary_detections(
|
||||
(
|
||||
OpenVocabularyDetection(
|
||||
detection_id="inside",
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000121",
|
||||
prompt_set_id="urban-static/v0",
|
||||
raw_label="trash bin",
|
||||
confidence=0.8,
|
||||
region=roi.core_region,
|
||||
),
|
||||
OpenVocabularyDetection(
|
||||
detection_id="outside",
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000121",
|
||||
prompt_set_id="urban-static/v0",
|
||||
raw_label="traffic cone",
|
||||
confidence=0.9,
|
||||
region=type(roi.core_region)(10.0, 10.0, 50.0, 50.0),
|
||||
),
|
||||
),
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
valid_fov_mask=np.ones((600, 800), dtype=np.bool_),
|
||||
)
|
||||
|
||||
selected = select_geometry_roi_detections(roi, fusion)
|
||||
|
||||
assert tuple(item.detection_id for item in selected) == ("inside",)
|
||||
@@ -0,0 +1,53 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from k1link.perception.geometry_semantic_shadow_replay import (
|
||||
read_geometry_semantic_shadow_replay,
|
||||
)
|
||||
|
||||
REPOSITORY_ROOT = Path(__file__).resolve().parents[1]
|
||||
RESULT_ROOT = (
|
||||
REPOSITORY_ROOT
|
||||
/ ".runtime/compute-experiments/m48s-semantic-shadow/geometry-first-results"
|
||||
/ (
|
||||
"m48s-geometry-semantic-shadow-"
|
||||
"1a6c5d04fb3f7aef4385ee28fc67d3e69f4528a77e8e1194b1283530ef73c7c2"
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def test_geometry_first_semantic_shadow_is_complete_but_not_accepted() -> None:
|
||||
result = read_geometry_semantic_shadow_replay(RESULT_ROOT)
|
||||
|
||||
assert result.completed is True
|
||||
assert result.accepted is False
|
||||
assert result.metrics["geometry"] == {
|
||||
"not_projected_observation_count": 15,
|
||||
"observation_count": 73,
|
||||
"roi_count": 58,
|
||||
}
|
||||
semantics = result.metrics["semantics"]
|
||||
assert isinstance(semantics, dict)
|
||||
assert semantics["resolution_counts"] == {
|
||||
"ambiguous": 6,
|
||||
"selected": 11,
|
||||
"unresolved": 56,
|
||||
}
|
||||
assert semantics["selected_class_counts"] == {
|
||||
"human.adult": 1,
|
||||
"static.bollard": 1,
|
||||
"static.concrete-hemisphere": 1,
|
||||
"vehicle.car": 8,
|
||||
}
|
||||
assert result.report["decision"] == {
|
||||
"geometry_first_binding_completed": True,
|
||||
"navigation_or_safety_accepted": False,
|
||||
"reason_codes": [
|
||||
"two-reviewer-independent-truth-unavailable",
|
||||
"open-vocabulary-geometry-roi-remains-experimental-shadow",
|
||||
"unknown-and-unresolved-objects-remain-route-around-obstacles",
|
||||
],
|
||||
"semantic_quality_accepted": False,
|
||||
}
|
||||
assert result.report["authority"]["commands_enabled"] is False # type: ignore[index]
|
||||
@@ -0,0 +1,46 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
REPOSITORY_ROOT = Path(__file__).resolve().parents[1]
|
||||
RESULT_ID = (
|
||||
"m48s-mask-grounding-dino-shadow-"
|
||||
"b0a37f265223b4138754f76d5d7b8d17e7c4f5395481990f169fccc1645baba8"
|
||||
)
|
||||
RESULT_ROOT = (
|
||||
REPOSITORY_ROOT
|
||||
/ ".runtime/compute-experiments/m48s-semantic-shadow/mask-grounding-dino-results"
|
||||
/ RESULT_ID
|
||||
)
|
||||
|
||||
|
||||
def test_mask_grounding_dino_shadow_is_complete_but_semantically_rejected() -> None:
|
||||
manifest = json.loads((RESULT_ROOT / "manifest.json").read_text("utf-8"))
|
||||
report = json.loads((RESULT_ROOT / "report.json").read_text("utf-8"))
|
||||
|
||||
assert manifest["result_id"] == RESULT_ID
|
||||
assert manifest["completed"] is True
|
||||
assert manifest["accepted"] is False
|
||||
assert manifest["metrics"]["frames"] == {"completed": 11, "requested": 11}
|
||||
assert manifest["metrics"]["geometry_observation_count"] == 73
|
||||
assert manifest["metrics"]["raw_detection_count"] == 25
|
||||
assert manifest["metrics"]["mask_instance_count"] == 7
|
||||
assert manifest["metrics"]["geometry_binding_resolution_counts"] == {
|
||||
"ambiguous": 2,
|
||||
"selected": 4,
|
||||
"unresolved": 1,
|
||||
}
|
||||
assert manifest["metrics"]["combined_resolution_counts"] == {
|
||||
"ambiguous": 6,
|
||||
"unresolved": 1,
|
||||
}
|
||||
assert manifest["metrics"]["single_class_agent_probe"] == {
|
||||
"detection_counts": {"adult": 0, "child": 0, "dog": 0},
|
||||
"frame_indices": [253, 1228],
|
||||
"passed": False,
|
||||
}
|
||||
assert report["decision"]["semantic_quality_accepted"] is False
|
||||
assert report["decision"]["agent_semantics_accepted"] is False
|
||||
assert report["authority"]["commands_enabled"] is False
|
||||
assert report["authority"]["navigation_or_safety_accepted"] is False
|
||||
@@ -0,0 +1,175 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from k1link.perception.contracts import (
|
||||
EvidenceBasis,
|
||||
EvidenceCurrentness,
|
||||
MetricGeometry,
|
||||
ObstacleObservation,
|
||||
)
|
||||
from k1link.perception.geometry_math import ProjectedPointCloud
|
||||
from k1link.perception.mask_grounding_semantics import (
|
||||
MASK_GROUNDING_EVIDENCE_SCHEMA,
|
||||
MaskBindingResolution,
|
||||
MaskGroundingDetection,
|
||||
MaskLabelResolution,
|
||||
bind_mask_instances_to_geometry,
|
||||
cluster_mask_instances,
|
||||
load_mask_grounding_evidence,
|
||||
resolve_mask_instance_label,
|
||||
)
|
||||
|
||||
|
||||
def _observation(ordinal: int, source_ids: tuple[int, ...]) -> ObstacleObservation:
|
||||
return ObstacleObservation(
|
||||
observation_id=f"frame-000253:geometry:{ordinal}",
|
||||
occupancy_key=f"frame-000253:occupancy:{ordinal}",
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000253",
|
||||
evidence_time_ns=253,
|
||||
basis=EvidenceBasis.LIDAR,
|
||||
currentness=EvidenceCurrentness.CURRENT,
|
||||
occupied_support=True,
|
||||
source_point_ids=source_ids,
|
||||
metric_geometry=MetricGeometry(
|
||||
coordinate_frame="map",
|
||||
centroid_xyz_m=(2.0 + ordinal, 0.0, 0.5),
|
||||
range_m=2.0 + ordinal,
|
||||
covariance_diagonal_m2=(0.1, 0.1, 0.1),
|
||||
),
|
||||
proposal_ids=(),
|
||||
semantic_hint=None,
|
||||
reason_codes=("test-geometry-only",),
|
||||
)
|
||||
|
||||
|
||||
def _projected() -> ProjectedPointCloud:
|
||||
return ProjectedPointCloud(
|
||||
pixels_xy=np.asarray(
|
||||
(
|
||||
(10.0, 10.0),
|
||||
(11.0, 10.0),
|
||||
(10.0, 11.0),
|
||||
(11.0, 11.0),
|
||||
(30.0, 30.0),
|
||||
(31.0, 30.0),
|
||||
(30.0, 31.0),
|
||||
(31.0, 31.0),
|
||||
),
|
||||
dtype=np.float64,
|
||||
),
|
||||
depths_m=np.ones(8, dtype=np.float64),
|
||||
source_indices=np.arange(8, dtype=np.int64),
|
||||
source_point_count=8,
|
||||
camera_front_point_count=8,
|
||||
)
|
||||
|
||||
|
||||
def _detection(
|
||||
detection_id: str,
|
||||
class_name: str,
|
||||
confidence: float,
|
||||
mask: np.ndarray,
|
||||
) -> MaskGroundingDetection:
|
||||
packed = mask.astype(np.uint8)
|
||||
return MaskGroundingDetection(
|
||||
detection_id=detection_id,
|
||||
prompt_set_id="urban-static/v0",
|
||||
class_id=0,
|
||||
class_name=class_name,
|
||||
confidence=confidence,
|
||||
box_xyxy=(1.0, 1.0, 40.0, 40.0),
|
||||
mask_sha256=hashlib.sha256(packed.tobytes()).hexdigest(),
|
||||
mask=mask,
|
||||
)
|
||||
|
||||
|
||||
def test_empty_early_v0_ledger_is_read_without_inventing_boxes(tmp_path: Path) -> None:
|
||||
path = tmp_path / "frame-000253.npz"
|
||||
np.savez_compressed(
|
||||
path,
|
||||
schema_version=np.asarray(MASK_GROUNDING_EVIDENCE_SCHEMA),
|
||||
source_file_sha256=np.asarray("a" * 64),
|
||||
source_pixel_sha256=np.asarray("b" * 64),
|
||||
class_ids=np.empty(0, dtype=np.int16),
|
||||
class_names=np.empty(0, dtype="U128"),
|
||||
scores=np.empty(0, dtype=np.float32),
|
||||
boxes_xyxy=np.empty(0, dtype=np.float32),
|
||||
masks=np.empty((0, 600, 800), dtype=np.uint8),
|
||||
)
|
||||
|
||||
evidence = load_mask_grounding_evidence(path, prompt_set_id="urban-agents/v0")
|
||||
|
||||
assert evidence.detections == ()
|
||||
assert evidence.source_file_sha256 == "a" * 64
|
||||
|
||||
|
||||
def test_same_mask_becomes_one_instance_and_conflicting_name_stays_ambiguous() -> None:
|
||||
mask = np.zeros((600, 800), dtype=np.bool_)
|
||||
mask[5:20, 5:20] = True
|
||||
instances = cluster_mask_instances(
|
||||
(
|
||||
_detection("trash", "trash bin", 0.43, mask),
|
||||
_detection("cart", "shopping cart", 0.41, mask),
|
||||
)
|
||||
)
|
||||
|
||||
bindings = bind_mask_instances_to_geometry(
|
||||
instances,
|
||||
observations=(_observation(0, (0, 1, 2, 3)), _observation(1, (4, 5, 6, 7))),
|
||||
projected=_projected(),
|
||||
)
|
||||
label = resolve_mask_instance_label(instances[0])
|
||||
|
||||
assert len(instances) == 1
|
||||
assert bindings[0].resolution is MaskBindingResolution.SELECTED
|
||||
assert bindings[0].selected_observation_id == "frame-000253:geometry:0"
|
||||
assert label.resolution is MaskLabelResolution.AMBIGUOUS
|
||||
assert label.selected_label is None
|
||||
|
||||
|
||||
def test_mask_covering_two_obstacles_does_not_claim_either() -> None:
|
||||
mask = np.zeros((600, 800), dtype=np.bool_)
|
||||
mask[5:40, 5:40] = True
|
||||
instance = cluster_mask_instances((_detection("wide", "trash bin", 0.7, mask),))
|
||||
|
||||
binding = bind_mask_instances_to_geometry(
|
||||
instance,
|
||||
observations=(_observation(0, (0, 1, 2, 3)), _observation(1, (4, 5, 6, 7))),
|
||||
projected=_projected(),
|
||||
)[0]
|
||||
|
||||
assert binding.resolution is MaskBindingResolution.AMBIGUOUS
|
||||
assert binding.selected_observation_id is None
|
||||
assert binding.reason_code == "mask-covers-multiple-geometry-observations"
|
||||
|
||||
|
||||
def test_two_distinct_masks_cannot_claim_one_geometry_observation() -> None:
|
||||
narrow = np.zeros((600, 800), dtype=np.bool_)
|
||||
narrow[9:13, 9:13] = True
|
||||
wide = np.zeros((600, 800), dtype=np.bool_)
|
||||
wide[1:25, 1:25] = True
|
||||
instances = cluster_mask_instances(
|
||||
(
|
||||
_detection("narrow", "bollard", 0.8, narrow),
|
||||
_detection("wide", "post", 0.8, wide),
|
||||
)
|
||||
)
|
||||
|
||||
bindings = bind_mask_instances_to_geometry(
|
||||
instances,
|
||||
observations=(_observation(0, (0, 1, 2, 3)),),
|
||||
projected=_projected(),
|
||||
)
|
||||
|
||||
assert len(instances) == 2
|
||||
assert all(item.resolution is MaskBindingResolution.AMBIGUOUS for item in bindings)
|
||||
assert all(item.selected_observation_id is None for item in bindings)
|
||||
assert all(
|
||||
item.reason_code == "observation-claimed-by-multiple-mask-instances"
|
||||
for item in bindings
|
||||
)
|
||||
@@ -0,0 +1,317 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import json
|
||||
from dataclasses import replace
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from k1link.perception.contracts import (
|
||||
EvidenceBasis,
|
||||
EvidenceCurrentness,
|
||||
MetricGeometry,
|
||||
MotionState,
|
||||
ObstacleObservation,
|
||||
)
|
||||
from k1link.perception.object_understanding import (
|
||||
AdvisoryResponse,
|
||||
AdvisoryRiskAssessment,
|
||||
AgencyState,
|
||||
EvidenceKind,
|
||||
EvidenceProvenance,
|
||||
ObjectStateEstimate,
|
||||
ObjectUnderstanding,
|
||||
ObjectUnderstandingError,
|
||||
RiskBasis,
|
||||
RiskLevel,
|
||||
SemanticDecision,
|
||||
SemanticHypothesis,
|
||||
SemanticResolution,
|
||||
StateBasis,
|
||||
load_object_semantic_vocabulary,
|
||||
validate_object_understanding,
|
||||
)
|
||||
|
||||
REPOSITORY_ROOT = Path(__file__).resolve().parents[1]
|
||||
VOCABULARY_PATH = REPOSITORY_ROOT / "config/perception/object-semantic-vocabulary-v0.json"
|
||||
|
||||
|
||||
def _observation(*, semantic_hint: str | None = None) -> ObstacleObservation:
|
||||
return ObstacleObservation(
|
||||
observation_id="observation-vehicle-1",
|
||||
occupancy_key="occupied-component-17",
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000253",
|
||||
evidence_time_ns=35_421_857_292,
|
||||
basis=EvidenceBasis.FUSED,
|
||||
currentness=EvidenceCurrentness.CURRENT,
|
||||
occupied_support=True,
|
||||
source_point_ids=(4, 7, 9),
|
||||
metric_geometry=MetricGeometry(
|
||||
coordinate_frame="map",
|
||||
centroid_xyz_m=(4.0, 1.0, 0.5),
|
||||
range_m=4.15,
|
||||
covariance_diagonal_m2=(0.04, 0.04, 0.09),
|
||||
),
|
||||
proposal_ids=("proposal-vehicle-1",),
|
||||
semantic_hint=semantic_hint,
|
||||
reason_codes=("current-qualified-points",),
|
||||
)
|
||||
|
||||
|
||||
def _detector_evidence(*, frame_id: str = "frame-000253") -> EvidenceProvenance:
|
||||
return EvidenceProvenance(
|
||||
evidence_id="evidence-grounding-dino-1",
|
||||
kind=EvidenceKind.DETECTOR,
|
||||
source_id="RAVNOVES00",
|
||||
frame_id=frame_id,
|
||||
provider_id="grounding-dino-open-vocabulary/v1",
|
||||
model_id="grounding-dino-tensorrt",
|
||||
model_revision="worker-006-probe-20260825",
|
||||
preprocess_id="kb4-rectified-rgb/v1",
|
||||
prompt_set_id="urban-risk-groups/v0",
|
||||
)
|
||||
|
||||
|
||||
def _temporal_evidence() -> EvidenceProvenance:
|
||||
return EvidenceProvenance(
|
||||
evidence_id="evidence-temporal-1",
|
||||
kind=EvidenceKind.TEMPORAL,
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000253",
|
||||
provider_id="temporal-occupied/v1",
|
||||
model_id=None,
|
||||
model_revision=None,
|
||||
preprocess_id=None,
|
||||
)
|
||||
|
||||
|
||||
def _car_understanding() -> ObjectUnderstanding:
|
||||
detector = _detector_evidence()
|
||||
temporal = _temporal_evidence()
|
||||
return ObjectUnderstanding(
|
||||
understanding_id="understanding-vehicle-1",
|
||||
vocabulary_id="missioncore.urban-object-semantics/v0",
|
||||
generated_monotonic_ns=1_020_000,
|
||||
observation=_observation(semantic_hint="car"),
|
||||
hypotheses=(
|
||||
SemanticHypothesis(
|
||||
rank=1,
|
||||
class_id="vehicle.car",
|
||||
raw_label="car",
|
||||
confidence=0.79,
|
||||
evidence_ids=(detector.evidence_id,),
|
||||
),
|
||||
SemanticHypothesis(
|
||||
rank=2,
|
||||
class_id="vehicle.heavy",
|
||||
raw_label="heavy vehicle",
|
||||
confidence=0.12,
|
||||
evidence_ids=(detector.evidence_id,),
|
||||
),
|
||||
),
|
||||
semantic=SemanticDecision(
|
||||
resolution=SemanticResolution.SELECTED,
|
||||
selected_class_id="vehicle.car",
|
||||
selected_confidence=0.79,
|
||||
reason_codes=("top-hypothesis-qualified",),
|
||||
),
|
||||
state=ObjectStateEstimate(
|
||||
motion=MotionState.STATIONARY,
|
||||
motion_confidence=0.85,
|
||||
agency=AgencyState.SELF_PROPELLED,
|
||||
agency_basis=StateBasis.CLASS_PRIOR,
|
||||
evidence_ids=(detector.evidence_id, temporal.evidence_id),
|
||||
reason_codes=("stationary-observed-vehicle-prior-retained",),
|
||||
),
|
||||
risk=AdvisoryRiskAssessment(
|
||||
policy_id="urban-object-risk/v0",
|
||||
level=RiskLevel.ELEVATED,
|
||||
confidence=0.71,
|
||||
basis=RiskBasis.FUSED,
|
||||
responses=(
|
||||
AdvisoryResponse.MONITOR,
|
||||
AdvisoryResponse.REDUCE_SPEED,
|
||||
AdvisoryResponse.ROUTE_AROUND,
|
||||
),
|
||||
evidence_ids=(detector.evidence_id, temporal.evidence_id),
|
||||
reason_codes=("stationary-vehicle-may-start-moving",),
|
||||
),
|
||||
provenance=(detector, temporal),
|
||||
)
|
||||
|
||||
|
||||
def _unknown_understanding() -> ObjectUnderstanding:
|
||||
return ObjectUnderstanding(
|
||||
understanding_id="understanding-unknown-1",
|
||||
vocabulary_id="missioncore.urban-object-semantics/v0",
|
||||
generated_monotonic_ns=1_020_000,
|
||||
observation=_observation(),
|
||||
hypotheses=(),
|
||||
semantic=SemanticDecision(
|
||||
resolution=SemanticResolution.UNRESOLVED,
|
||||
selected_class_id=None,
|
||||
selected_confidence=None,
|
||||
reason_codes=("semantic-evidence-unavailable",),
|
||||
),
|
||||
state=ObjectStateEstimate(
|
||||
motion=MotionState.UNKNOWN,
|
||||
motion_confidence=0.0,
|
||||
agency=AgencyState.UNKNOWN,
|
||||
agency_basis=StateBasis.UNKNOWN,
|
||||
evidence_ids=(),
|
||||
reason_codes=("state-evidence-unavailable",),
|
||||
),
|
||||
risk=AdvisoryRiskAssessment(
|
||||
policy_id="urban-object-risk/v0",
|
||||
level=RiskLevel.UNKNOWN,
|
||||
confidence=0.0,
|
||||
basis=RiskBasis.UNKNOWN,
|
||||
responses=(AdvisoryResponse.ROUTE_AROUND,),
|
||||
evidence_ids=(),
|
||||
reason_codes=("unknown-object-remains-occupied",),
|
||||
),
|
||||
provenance=(),
|
||||
)
|
||||
|
||||
|
||||
def test_object_understanding_round_trip_is_strict_and_keeps_v1_geometry() -> None:
|
||||
value = _car_understanding()
|
||||
observation_before = value.observation.to_dict()
|
||||
document = json.loads(json.dumps(value.to_dict()))
|
||||
|
||||
restored = ObjectUnderstanding.from_dict(document)
|
||||
|
||||
assert restored == value
|
||||
assert restored.observation.to_dict() == observation_before
|
||||
assert restored.occupancy_identity == "occupied-component-17"
|
||||
assert restored.observation.metric_geometry is not None
|
||||
assert restored.observation.metric_geometry.range_m == pytest.approx(4.15)
|
||||
document["unexpected"] = True
|
||||
with pytest.raises(ObjectUnderstandingError, match="fields are incompatible"):
|
||||
ObjectUnderstanding.from_dict(document)
|
||||
|
||||
|
||||
def test_stationary_vehicle_keeps_separate_self_propelled_prior_and_advisory_risk() -> None:
|
||||
value = _car_understanding()
|
||||
|
||||
assert value.state.motion is MotionState.STATIONARY
|
||||
assert value.state.agency is AgencyState.SELF_PROPELLED
|
||||
assert value.state.agency_basis is StateBasis.CLASS_PRIOR
|
||||
assert value.risk.level is RiskLevel.ELEVATED
|
||||
assert AdvisoryResponse.REDUCE_SPEED in value.risk.responses
|
||||
assert value.authority.commands_enabled is False
|
||||
assert value.authority.actuation_allowed is False
|
||||
assert value.authority.navigation_or_safety_accepted is False
|
||||
|
||||
|
||||
def test_unknown_semantics_never_erase_metric_occupancy() -> None:
|
||||
value = _unknown_understanding()
|
||||
|
||||
assert value.semantic.resolution is SemanticResolution.UNRESOLVED
|
||||
assert value.hypotheses == ()
|
||||
assert value.observation.occupied_support is True
|
||||
assert value.observation.source_point_ids == (4, 7, 9)
|
||||
assert value.occupancy_identity == value.observation.occupancy_identity
|
||||
assert value.risk.responses == (AdvisoryResponse.ROUTE_AROUND,)
|
||||
|
||||
|
||||
def test_ranked_hypotheses_and_selected_class_must_agree() -> None:
|
||||
value = _car_understanding()
|
||||
with pytest.raises(ObjectUnderstandingError, match="ordered by confidence"):
|
||||
replace(
|
||||
value,
|
||||
hypotheses=(
|
||||
replace(value.hypotheses[0], confidence=0.10),
|
||||
replace(value.hypotheses[1], confidence=0.90),
|
||||
),
|
||||
)
|
||||
with pytest.raises(ObjectUnderstandingError, match="match one ranked hypothesis"):
|
||||
replace(
|
||||
value,
|
||||
semantic=replace(
|
||||
value.semantic,
|
||||
selected_class_id="animal.dog",
|
||||
selected_confidence=0.79,
|
||||
),
|
||||
)
|
||||
with pytest.raises(ObjectUnderstandingError, match="two hypotheses"):
|
||||
replace(
|
||||
value,
|
||||
hypotheses=value.hypotheses[:1],
|
||||
semantic=SemanticDecision(
|
||||
resolution=SemanticResolution.CONFLICT,
|
||||
selected_class_id=None,
|
||||
selected_confidence=None,
|
||||
reason_codes=("provider-disagreement",),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def test_evidence_cannot_escape_geometry_frame_or_be_fabricated() -> None:
|
||||
value = _car_understanding()
|
||||
with pytest.raises(ObjectUnderstandingError, match="escaped"):
|
||||
replace(value, provenance=(_detector_evidence(frame_id="frame-000254"),))
|
||||
with pytest.raises(ObjectUnderstandingError, match="unknown evidence"):
|
||||
replace(
|
||||
value,
|
||||
hypotheses=(
|
||||
replace(value.hypotheses[0], evidence_ids=("missing-evidence",)),
|
||||
value.hypotheses[1],
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def test_executable_vocabulary_resolves_current_urban_labels_and_hierarchy() -> None:
|
||||
vocabulary = load_object_semantic_vocabulary(VOCABULARY_PATH)
|
||||
|
||||
assert vocabulary.vocabulary_id == "missioncore.urban-object-semantics/v0"
|
||||
assert vocabulary.resolve_label("trash bin") == "static.trash-bin"
|
||||
assert vocabulary.resolve_label("Dog") == "animal.dog"
|
||||
assert vocabulary.resolve_label("sidewalk-curb") == "terrain.curb"
|
||||
assert {
|
||||
label: vocabulary.resolve_label(label)
|
||||
for label in ("car", "person", "bicycle", "road sign")
|
||||
} == {
|
||||
"car": "vehicle.car",
|
||||
"person": "human.unknown",
|
||||
"bicycle": "vehicle.bicycle",
|
||||
"road sign": "static.road-sign",
|
||||
}
|
||||
assert vocabulary.resolve_label("unseen alien object") is None
|
||||
assert vocabulary.ancestors("human.child") == (
|
||||
"human.unknown",
|
||||
"object.unknown",
|
||||
)
|
||||
validate_object_understanding(_car_understanding(), vocabulary)
|
||||
|
||||
|
||||
def test_vocabulary_validation_rejects_undeclared_class_and_authority_change(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
vocabulary = load_object_semantic_vocabulary(VOCABULARY_PATH)
|
||||
value = _car_understanding()
|
||||
with pytest.raises(ObjectUnderstandingError, match="undeclared classes"):
|
||||
validate_object_understanding(
|
||||
replace(
|
||||
value,
|
||||
hypotheses=(
|
||||
replace(value.hypotheses[0], class_id="vehicle.hovercraft"),
|
||||
value.hypotheses[1],
|
||||
),
|
||||
semantic=replace(
|
||||
value.semantic,
|
||||
selected_class_id="vehicle.hovercraft",
|
||||
),
|
||||
),
|
||||
vocabulary,
|
||||
)
|
||||
|
||||
document = json.loads(VOCABULARY_PATH.read_text("utf-8"))
|
||||
incompatible = copy.deepcopy(document)
|
||||
incompatible["policies"]["planner_command_authority"] = True
|
||||
path = tmp_path / "vocabulary.json"
|
||||
path.write_text(json.dumps(incompatible), "utf-8")
|
||||
with pytest.raises(ObjectUnderstandingError, match="authority policy"):
|
||||
load_object_semantic_vocabulary(path)
|
||||
@@ -0,0 +1,281 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
from k1link.perception.contracts import (
|
||||
BoundingRegion2D,
|
||||
EvidenceBasis,
|
||||
EvidenceCurrentness,
|
||||
MetricGeometry,
|
||||
ObstacleObservation,
|
||||
)
|
||||
from k1link.perception.object_understanding import (
|
||||
AdvisoryResponse,
|
||||
AgencyState,
|
||||
SemanticResolution,
|
||||
StateBasis,
|
||||
load_object_semantic_vocabulary,
|
||||
)
|
||||
from k1link.perception.open_vocabulary_semantics import (
|
||||
OpenVocabularyDetection,
|
||||
OpenVocabularySemanticError,
|
||||
bind_object_understandings,
|
||||
fuse_open_vocabulary_detections,
|
||||
load_open_vocabulary_semantic_profile,
|
||||
parse_tao_grounding_dino_labels,
|
||||
)
|
||||
|
||||
REPOSITORY_ROOT = Path(__file__).resolve().parents[1]
|
||||
PROFILE_PATH = REPOSITORY_ROOT / "config/perception/open-vocabulary-semantic-shadow-v0.json"
|
||||
VOCABULARY_PATH = REPOSITORY_ROOT / "config/perception/object-semantic-vocabulary-v0.json"
|
||||
VALID_FOV_MASK = np.ones((600, 800), dtype=np.bool_)
|
||||
|
||||
|
||||
def _detection(
|
||||
detection_id: str,
|
||||
raw_label: str,
|
||||
confidence: float,
|
||||
region: tuple[float, float, float, float],
|
||||
*,
|
||||
prompt_set_id: str = "urban-static/v0",
|
||||
) -> OpenVocabularyDetection:
|
||||
return OpenVocabularyDetection(
|
||||
detection_id=detection_id,
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000121",
|
||||
prompt_set_id=prompt_set_id,
|
||||
raw_label=raw_label,
|
||||
confidence=confidence,
|
||||
region=BoundingRegion2D(*region),
|
||||
)
|
||||
|
||||
|
||||
def _observation(
|
||||
proposal_id: str | None,
|
||||
*,
|
||||
ordinal: int,
|
||||
) -> ObstacleObservation:
|
||||
return ObstacleObservation(
|
||||
observation_id=f"frame-000121:observation-{ordinal}",
|
||||
occupancy_key=f"frame-000121:occupancy-{ordinal}",
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000121",
|
||||
evidence_time_ns=121,
|
||||
basis=EvidenceBasis.FUSED if proposal_id else EvidenceBasis.LIDAR,
|
||||
currentness=EvidenceCurrentness.CURRENT,
|
||||
occupied_support=True,
|
||||
source_point_ids=(ordinal,),
|
||||
metric_geometry=MetricGeometry(
|
||||
coordinate_frame="map",
|
||||
centroid_xyz_m=(1.0, 2.0, 0.5),
|
||||
range_m=2.2,
|
||||
covariance_diagonal_m2=(0.1, 0.1, 0.1),
|
||||
),
|
||||
proposal_ids=(proposal_id,) if proposal_id else (),
|
||||
semantic_hint="static.trash-bin" if proposal_id else None,
|
||||
reason_codes=("test-current-occupied-support",),
|
||||
)
|
||||
|
||||
|
||||
def test_profile_is_raw_kb4_and_preserves_false_authority() -> None:
|
||||
profile = load_open_vocabulary_semantic_profile(PROFILE_PATH)
|
||||
|
||||
assert profile.coordinate_space == "raw-kb4"
|
||||
assert (profile.width, profile.height) == (800, 600)
|
||||
assert profile.provider_id == "nvidia-tao-grounding-dino-trt/v1"
|
||||
assert profile.model_sha256 == (
|
||||
"6895acdc6b588e923f753e37b3bd18869e064256e5ecc1b2b9853e8c51125f94"
|
||||
)
|
||||
assert tuple(item.prompt_set_id for item in profile.prompt_groups) == (
|
||||
"urban-static/v0",
|
||||
"urban-agents/v0",
|
||||
"urban-vehicles/v0",
|
||||
)
|
||||
|
||||
|
||||
def test_tao_parser_keeps_source_coordinates_and_rejects_rectified_boxes(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
profile = load_open_vocabulary_semantic_profile(PROFILE_PATH)
|
||||
labels = tmp_path / "frame-000121.txt"
|
||||
labels.write_text(
|
||||
"trash bin 0.00 0 0.00 336.0 123.0 359.0 160.0 0.00 0.00 0.00 0.00 0.00 0.00 0.00 0.826\n",
|
||||
"utf-8",
|
||||
)
|
||||
|
||||
parsed = parse_tao_grounding_dino_labels(
|
||||
labels,
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000121",
|
||||
prompt_set_id="urban-static/v0",
|
||||
profile=profile,
|
||||
)
|
||||
|
||||
assert len(parsed) == 1
|
||||
assert parsed[0].raw_label == "trash bin"
|
||||
assert parsed[0].region.as_tuple() == (336.0, 123.0, 359.0, 160.0)
|
||||
labels.write_text(
|
||||
"trash bin 0.00 0 0.00 336.0 123.0 900.0 160.0 0.00 0.00 0.00 0.00 0.00 0.00 0.00 0.826\n",
|
||||
"utf-8",
|
||||
)
|
||||
with pytest.raises(OpenVocabularySemanticError, match="raw image coordinate space"):
|
||||
parse_tao_grounding_dino_labels(
|
||||
labels,
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000121",
|
||||
prompt_set_id="urban-static/v0",
|
||||
profile=profile,
|
||||
)
|
||||
|
||||
|
||||
def test_tao_parser_translates_geometry_crop_back_to_raw_coordinates(tmp_path: Path) -> None:
|
||||
profile = load_open_vocabulary_semantic_profile(PROFILE_PATH)
|
||||
labels = tmp_path / "frame-000121-geometry-roi-000.txt"
|
||||
labels.write_text(
|
||||
"dog 0.00 0 0.00 10.0 20.0 50.0 70.0 0.00 0.00 0.00 0.00 0.00 0.00 0.00 0.714\n",
|
||||
"utf-8",
|
||||
)
|
||||
|
||||
parsed = parse_tao_grounding_dino_labels(
|
||||
labels,
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000121",
|
||||
prompt_set_id="urban-agents/v0",
|
||||
profile=profile,
|
||||
image_width=96,
|
||||
image_height=96,
|
||||
offset_x=200.0,
|
||||
offset_y=150.0,
|
||||
detection_scope_id="frame-000121:geometry-roi-000",
|
||||
)
|
||||
|
||||
assert parsed[0].detection_id.startswith("frame-000121:geometry-roi-000")
|
||||
assert parsed[0].region.as_tuple() == (210.0, 170.0, 250.0, 220.0)
|
||||
|
||||
|
||||
def test_prompt_collisions_fuse_before_geometry_and_background_box_is_removed() -> None:
|
||||
profile = load_open_vocabulary_semantic_profile(PROFILE_PATH)
|
||||
vocabulary = load_object_semantic_vocabulary(VOCABULARY_PATH)
|
||||
detections = (
|
||||
_detection("trash", "trash bin", 0.82, (300.0, 100.0, 360.0, 200.0)),
|
||||
_detection("cone", "traffic cone", 0.35, (301.0, 101.0, 361.0, 201.0)),
|
||||
_detection(
|
||||
"dog",
|
||||
"dog",
|
||||
0.71,
|
||||
(100.0, 250.0, 180.0, 340.0),
|
||||
prompt_set_id="urban-agents/v0",
|
||||
),
|
||||
_detection(
|
||||
"background",
|
||||
"concrete hemisphere",
|
||||
0.66,
|
||||
(0.0, 100.0, 800.0, 600.0),
|
||||
),
|
||||
)
|
||||
|
||||
result = fuse_open_vocabulary_detections(
|
||||
detections,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
valid_fov_mask=VALID_FOV_MASK,
|
||||
)
|
||||
|
||||
assert result.input_detection_count == 4
|
||||
assert result.invalid_area_count == 1
|
||||
assert len(result.bindings) == 2
|
||||
assert sum(len(item.detections) for item in result.bindings) == 3
|
||||
trash = next(
|
||||
item for item in result.bindings if item.proposal.semantic_hint == "static.trash-bin"
|
||||
)
|
||||
assert len(trash.detections) == 2
|
||||
assert trash.proposal.objectness == pytest.approx(0.82)
|
||||
|
||||
|
||||
def test_binding_keeps_ranked_semantics_separate_from_state_risk_and_occupancy() -> None:
|
||||
profile = load_open_vocabulary_semantic_profile(PROFILE_PATH)
|
||||
vocabulary = load_object_semantic_vocabulary(VOCABULARY_PATH)
|
||||
fusion = fuse_open_vocabulary_detections(
|
||||
(
|
||||
_detection("trash", "trash bin", 0.82, (300.0, 100.0, 360.0, 200.0)),
|
||||
_detection("cone", "traffic cone", 0.35, (301.0, 101.0, 361.0, 201.0)),
|
||||
),
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
valid_fov_mask=VALID_FOV_MASK,
|
||||
)
|
||||
proposal = fusion.proposals[0]
|
||||
observations = (
|
||||
_observation(proposal.proposal_id, ordinal=1),
|
||||
_observation(None, ordinal=2),
|
||||
)
|
||||
|
||||
understandings = bind_object_understandings(
|
||||
observations,
|
||||
bindings=fusion.bindings,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
generated_monotonic_ns=121,
|
||||
)
|
||||
|
||||
semantic = understandings[0]
|
||||
assert tuple(item.class_id for item in semantic.hypotheses) == (
|
||||
"static.trash-bin",
|
||||
"static.traffic-cone",
|
||||
)
|
||||
assert semantic.semantic.resolution is SemanticResolution.SELECTED
|
||||
assert semantic.semantic.selected_class_id == "static.trash-bin"
|
||||
assert semantic.state.agency is AgencyState.INERT
|
||||
assert semantic.state.agency_basis is StateBasis.CLASS_PRIOR
|
||||
assert semantic.state.motion.value == "unknown"
|
||||
assert semantic.risk.level.value == "unknown"
|
||||
assert semantic.risk.responses == (AdvisoryResponse.ROUTE_AROUND,)
|
||||
assert semantic.authority.navigation_or_safety_accepted is False
|
||||
assert semantic.observation.source_point_ids == (1,)
|
||||
geometry_only = understandings[1]
|
||||
assert geometry_only.semantic.resolution is SemanticResolution.UNRESOLVED
|
||||
assert geometry_only.hypotheses == ()
|
||||
assert geometry_only.observation.source_point_ids == (2,)
|
||||
|
||||
|
||||
def test_close_semantic_scores_remain_ambiguous() -> None:
|
||||
profile = load_open_vocabulary_semantic_profile(PROFILE_PATH)
|
||||
vocabulary = load_object_semantic_vocabulary(VOCABULARY_PATH)
|
||||
fusion = fuse_open_vocabulary_detections(
|
||||
(
|
||||
_detection(
|
||||
"adult",
|
||||
"adult person",
|
||||
0.64,
|
||||
(100.0, 50.0, 200.0, 350.0),
|
||||
prompt_set_id="urban-agents/v0",
|
||||
),
|
||||
_detection(
|
||||
"child",
|
||||
"child",
|
||||
0.59,
|
||||
(101.0, 51.0, 201.0, 351.0),
|
||||
prompt_set_id="urban-agents/v0",
|
||||
),
|
||||
),
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
valid_fov_mask=VALID_FOV_MASK,
|
||||
)
|
||||
proposal = fusion.proposals[0]
|
||||
|
||||
result = bind_object_understandings(
|
||||
(_observation(proposal.proposal_id, ordinal=1),),
|
||||
bindings=fusion.bindings,
|
||||
profile=profile,
|
||||
vocabulary=vocabulary,
|
||||
generated_monotonic_ns=121,
|
||||
)[0]
|
||||
|
||||
assert result.semantic.resolution is SemanticResolution.AMBIGUOUS
|
||||
assert result.semantic.selected_class_id is None
|
||||
assert result.state.agency is AgencyState.UNKNOWN
|
||||
assert result.state.agency_basis is StateBasis.UNKNOWN
|
||||
@@ -0,0 +1,263 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from k1link.laboratory.semantic_object_quality import (
|
||||
SemanticObjectQualityError,
|
||||
SemanticTruthLabel,
|
||||
load_semantic_object_quality_profile,
|
||||
score_semantic_object_quality,
|
||||
)
|
||||
from k1link.perception.contracts import (
|
||||
EvidenceBasis,
|
||||
EvidenceCurrentness,
|
||||
MetricGeometry,
|
||||
MotionState,
|
||||
ObstacleObservation,
|
||||
)
|
||||
from k1link.perception.object_understanding import (
|
||||
AdvisoryResponse,
|
||||
AdvisoryRiskAssessment,
|
||||
AgencyState,
|
||||
EvidenceKind,
|
||||
EvidenceProvenance,
|
||||
ObjectStateEstimate,
|
||||
ObjectUnderstanding,
|
||||
RiskBasis,
|
||||
RiskLevel,
|
||||
SemanticDecision,
|
||||
SemanticHypothesis,
|
||||
SemanticResolution,
|
||||
StateBasis,
|
||||
load_object_semantic_vocabulary,
|
||||
)
|
||||
|
||||
REPOSITORY_ROOT = Path(__file__).resolve().parents[1]
|
||||
VOCABULARY_PATH = REPOSITORY_ROOT / "config/perception/object-semantic-vocabulary-v0.json"
|
||||
PROFILE_PATH = REPOSITORY_ROOT / "config/perception/m48s-semantic-object-quality-v0.json"
|
||||
|
||||
|
||||
def _observation(index: int) -> ObstacleObservation:
|
||||
return ObstacleObservation(
|
||||
observation_id=f"observation-{index}",
|
||||
occupancy_key=f"occupied-{index}",
|
||||
source_id="RAVNOVES00",
|
||||
frame_id=f"frame-{index:06d}",
|
||||
evidence_time_ns=index * 1_000_000,
|
||||
basis=EvidenceBasis.FUSED,
|
||||
currentness=EvidenceCurrentness.CURRENT,
|
||||
occupied_support=True,
|
||||
source_point_ids=(index,),
|
||||
metric_geometry=MetricGeometry(
|
||||
coordinate_frame="map",
|
||||
centroid_xyz_m=(float(index), 0.0, 0.5),
|
||||
range_m=float(index + 1),
|
||||
covariance_diagonal_m2=(0.04, 0.04, 0.09),
|
||||
),
|
||||
proposal_ids=(f"proposal-{index}",),
|
||||
semantic_hint=None,
|
||||
reason_codes=("current-qualified-points",),
|
||||
)
|
||||
|
||||
|
||||
def _prediction(
|
||||
index: int,
|
||||
*,
|
||||
hypotheses: tuple[tuple[str, str, float], ...],
|
||||
selected_class_id: str | None,
|
||||
) -> ObjectUnderstanding:
|
||||
observation = _observation(index)
|
||||
evidence = EvidenceProvenance(
|
||||
evidence_id=f"semantic-evidence-{index}",
|
||||
kind=EvidenceKind.DETECTOR,
|
||||
source_id=observation.source_id,
|
||||
frame_id=observation.frame_id,
|
||||
provider_id="semantic-candidate/v0",
|
||||
model_id="semantic-model",
|
||||
model_revision="candidate-1",
|
||||
preprocess_id="rgb/v1",
|
||||
prompt_set_id="urban-risk-groups/v0",
|
||||
)
|
||||
ranked = tuple(
|
||||
SemanticHypothesis(
|
||||
rank=rank,
|
||||
class_id=class_id,
|
||||
raw_label=raw_label,
|
||||
confidence=confidence,
|
||||
evidence_ids=(evidence.evidence_id,),
|
||||
)
|
||||
for rank, (class_id, raw_label, confidence) in enumerate(hypotheses, start=1)
|
||||
)
|
||||
selected = next(
|
||||
(item for item in ranked if item.class_id == selected_class_id),
|
||||
None,
|
||||
)
|
||||
return ObjectUnderstanding(
|
||||
understanding_id=f"understanding-{index}",
|
||||
vocabulary_id="missioncore.urban-object-semantics/v0",
|
||||
generated_monotonic_ns=index * 1_000_000 + 1,
|
||||
observation=observation,
|
||||
hypotheses=ranked,
|
||||
semantic=SemanticDecision(
|
||||
resolution=(
|
||||
SemanticResolution.SELECTED
|
||||
if selected is not None
|
||||
else SemanticResolution.UNRESOLVED
|
||||
),
|
||||
selected_class_id=None if selected is None else selected.class_id,
|
||||
selected_confidence=None if selected is None else selected.confidence,
|
||||
reason_codes=(
|
||||
"top-hypothesis-qualified"
|
||||
if selected is not None
|
||||
else "semantic-evidence-insufficient",
|
||||
),
|
||||
),
|
||||
state=ObjectStateEstimate(
|
||||
motion=MotionState.UNKNOWN,
|
||||
motion_confidence=0.0,
|
||||
agency=AgencyState.UNKNOWN,
|
||||
agency_basis=StateBasis.UNKNOWN,
|
||||
evidence_ids=(),
|
||||
reason_codes=("motion-evidence-unavailable",),
|
||||
),
|
||||
risk=AdvisoryRiskAssessment(
|
||||
policy_id="urban-object-risk/v0",
|
||||
level=RiskLevel.UNKNOWN,
|
||||
confidence=0.0,
|
||||
basis=RiskBasis.UNKNOWN,
|
||||
responses=(AdvisoryResponse.ROUTE_AROUND,),
|
||||
evidence_ids=(),
|
||||
reason_codes=("semantic-quality-does-not-score-risk",),
|
||||
),
|
||||
provenance=(evidence,),
|
||||
)
|
||||
|
||||
|
||||
def _truth(index: int, class_id: str) -> SemanticTruthLabel:
|
||||
return SemanticTruthLabel(
|
||||
label_id=f"semantic-label-{index}",
|
||||
observation_id=f"observation-{index}",
|
||||
source_id="RAVNOVES00",
|
||||
frame_id=f"frame-{index:06d}",
|
||||
class_id=class_id,
|
||||
reviewer_count=2,
|
||||
adjudicated=True,
|
||||
)
|
||||
|
||||
|
||||
def test_semantic_quality_scores_exact_group_topk_and_unresolved_separately() -> None:
|
||||
vocabulary = load_object_semantic_vocabulary(VOCABULARY_PATH)
|
||||
profile = load_semantic_object_quality_profile(PROFILE_PATH)
|
||||
predictions = (
|
||||
_prediction(
|
||||
1,
|
||||
hypotheses=(
|
||||
("human.unknown", "person", 0.70),
|
||||
("human.child", "child", 0.65),
|
||||
),
|
||||
selected_class_id="human.unknown",
|
||||
),
|
||||
_prediction(
|
||||
2,
|
||||
hypotheses=(("animal.dog", "dog", 0.80),),
|
||||
selected_class_id="animal.dog",
|
||||
),
|
||||
_prediction(3, hypotheses=(), selected_class_id=None),
|
||||
)
|
||||
truth = (
|
||||
_truth(1, "human.child"),
|
||||
_truth(2, "animal.dog"),
|
||||
_truth(3, "vehicle.car"),
|
||||
)
|
||||
|
||||
result = score_semantic_object_quality(
|
||||
predictions=predictions,
|
||||
truth=truth,
|
||||
vocabulary=vocabulary,
|
||||
profile=profile,
|
||||
)
|
||||
|
||||
metrics = result.report["metrics"]
|
||||
assert isinstance(metrics, dict)
|
||||
assert metrics["prediction_coverage"] == pytest.approx(1.0)
|
||||
assert metrics["exact_top1_accuracy"] == pytest.approx(1 / 3)
|
||||
assert metrics["coarse_group_accuracy"] == pytest.approx(2 / 3)
|
||||
assert metrics["exact_top_k_recall"] == pytest.approx(2 / 3)
|
||||
assert metrics["unresolved_fraction"] == pytest.approx(1 / 3)
|
||||
assert result.report["candidate_semantic_gate_passed"] is False
|
||||
assert result.cases[0].exact_top1_correct is False
|
||||
assert result.cases[0].coarse_group_correct is True
|
||||
assert result.cases[0].exact_top_k_hit is True
|
||||
|
||||
|
||||
def test_semantic_quality_reports_missing_projection_without_calling_it_detection() -> None:
|
||||
vocabulary = load_object_semantic_vocabulary(VOCABULARY_PATH)
|
||||
profile = load_semantic_object_quality_profile(PROFILE_PATH)
|
||||
result = score_semantic_object_quality(
|
||||
predictions=(
|
||||
_prediction(
|
||||
1,
|
||||
hypotheses=(("human.child", "child", 0.90),),
|
||||
selected_class_id="human.child",
|
||||
),
|
||||
),
|
||||
truth=(_truth(1, "human.child"), _truth(2, "animal.dog")),
|
||||
vocabulary=vocabulary,
|
||||
profile=profile,
|
||||
)
|
||||
|
||||
metrics = result.report["metrics"]
|
||||
assert isinstance(metrics, dict)
|
||||
assert metrics["prediction_coverage"] == pytest.approx(0.5)
|
||||
assert result.cases[1].resolution == "unavailable"
|
||||
scope = result.report["scope"]
|
||||
assert isinstance(scope, dict)
|
||||
assert scope["object_presence_scored"] is False
|
||||
|
||||
|
||||
def test_semantic_truth_is_strict_and_requires_independent_adjudication() -> None:
|
||||
label = _truth(1, "human.child")
|
||||
document = json.loads(json.dumps(label.to_dict()))
|
||||
assert SemanticTruthLabel.from_dict(document) == label
|
||||
document["category"] = "child"
|
||||
with pytest.raises(SemanticObjectQualityError, match="fields are incompatible"):
|
||||
SemanticTruthLabel.from_dict(document)
|
||||
with pytest.raises(SemanticObjectQualityError, match="two independent reviewers"):
|
||||
SemanticTruthLabel(
|
||||
label_id="semantic-label-1",
|
||||
observation_id="observation-1",
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000001",
|
||||
class_id="human.child",
|
||||
reviewer_count=1,
|
||||
adjudicated=True,
|
||||
)
|
||||
with pytest.raises(SemanticObjectQualityError, match="must be adjudicated"):
|
||||
SemanticTruthLabel(
|
||||
label_id="semantic-label-1",
|
||||
observation_id="observation-1",
|
||||
source_id="RAVNOVES00",
|
||||
frame_id="frame-000001",
|
||||
class_id="human.child",
|
||||
reviewer_count=2,
|
||||
adjudicated=False,
|
||||
)
|
||||
|
||||
|
||||
def test_semantic_quality_profile_stays_separate_from_presence_and_risk(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
profile = load_semantic_object_quality_profile(PROFILE_PATH)
|
||||
assert profile.profile_id == "m48s-urban-semantic-object-quality/v0"
|
||||
assert profile.vocabulary_id == "missioncore.urban-object-semantics/v0"
|
||||
assert profile.top_k == 5
|
||||
|
||||
document = json.loads(PROFILE_PATH.read_text("utf-8"))
|
||||
document["scope"]["object_presence_scored"] = True
|
||||
changed = tmp_path / "m48s-profile-changed.json"
|
||||
changed.write_text(json.dumps(document), "utf-8")
|
||||
with pytest.raises(SemanticObjectQualityError, match="scope changed"):
|
||||
load_semantic_object_quality_profile(changed)
|
||||
@@ -0,0 +1,47 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from k1link.perception.semantic_shadow_replay import read_semantic_shadow_replay
|
||||
|
||||
REPOSITORY_ROOT = Path(__file__).resolve().parents[1]
|
||||
RESULT_ROOT = (
|
||||
REPOSITORY_ROOT
|
||||
/ ".runtime/compute-experiments/m48s-semantic-shadow/results"
|
||||
/ "m48s-semantic-shadow-a237a8860bd4655b4fae981ba23a5e4d86a560ed243ee7efdf867d87f5071739"
|
||||
)
|
||||
|
||||
|
||||
def test_bounded_raw_kb4_semantic_shadow_is_complete_but_not_accepted() -> None:
|
||||
result = read_semantic_shadow_replay(RESULT_ROOT)
|
||||
|
||||
assert result.completed is True
|
||||
assert result.accepted is False
|
||||
assert result.metrics["frames"] == {
|
||||
"completed": 11,
|
||||
"requested": 11,
|
||||
"source_available": 11,
|
||||
}
|
||||
assert result.metrics["detections"] == {
|
||||
"below_confidence": 0,
|
||||
"fused_proposals": 164,
|
||||
"invalid_area": 16,
|
||||
"outside_valid_fov": 0,
|
||||
"raw": 310,
|
||||
"retained": 294,
|
||||
}
|
||||
semantics = result.metrics["semantics"]
|
||||
assert isinstance(semantics, dict)
|
||||
assert semantics["resolution_counts"] == {
|
||||
"ambiguous": 4,
|
||||
"selected": 40,
|
||||
"unresolved": 165,
|
||||
}
|
||||
assert result.report["decision"] == {
|
||||
"navigation_or_safety_accepted": False,
|
||||
"next_gate": "M48S human-reviewed semantic object quality slice",
|
||||
"raw_kb4_coordinate_binding_completed": True,
|
||||
"risk_policy_accepted": False,
|
||||
"semantic_quality_accepted": False,
|
||||
}
|
||||
assert result.report["authority"]["commands_enabled"] is False
|
||||
Reference in New Issue
Block a user