feat(perception): add frozen yolox provider
This commit is contained in:
@@ -26,6 +26,11 @@
|
|||||||
"module": "k1link.compute.pipeline_telemetry",
|
"module": "k1link.compute.pipeline_telemetry",
|
||||||
"role": "stage, queue and terminal-outcome telemetry",
|
"role": "stage, queue and terminal-outcome telemetry",
|
||||||
"admission": "reuse"
|
"admission": "reuse"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"module": "k1link.compute.yolox_object_detector",
|
||||||
|
"role": "frozen source-neutral YOLOX preprocess, Triton transport and postprocess",
|
||||||
|
"admission": "adapt-behind-provider"
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"historical_wrappers": [
|
"historical_wrappers": [
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
Date: 2026-08-05
|
Date: 2026-08-05
|
||||||
|
|
||||||
Status: in progress; M4.0–M4.2 implemented
|
Status: in progress; M4.0–M4.2 implemented, M4.3 runtime gate open
|
||||||
|
|
||||||
Audit base: `1b3e0b3` on `feat/simulation-polygon-s1`
|
Audit base: `1b3e0b3` on `feat/simulation-polygon-s1`
|
||||||
|
|
||||||
@@ -679,6 +679,36 @@ geometry/temporal/degradation/telemetry tests and the complete Python suite
|
|||||||
`src/k1link/perception`. The canonical GUI remained on `127.0.0.1:8000`; no
|
`src/k1link/perception`. The canonical GUI remained on `127.0.0.1:8000`; no
|
||||||
duplicate development server or Worker 006 mutation was introduced.
|
duplicate development server or Worker 006 mutation was introduced.
|
||||||
|
|
||||||
|
### 2026-08-05 — M4.3 provider increment, runtime gate remains open
|
||||||
|
|
||||||
|
The product detector seam is implemented without importing the immutable E46J
|
||||||
|
wrapper:
|
||||||
|
|
||||||
|
- `k1link.compute.yolox_object_detector` contains only the frozen raw-KB4
|
||||||
|
valid-FOV fill, top-left bilinear letterbox, Triton V2 binary HTTP transport,
|
||||||
|
standard YOLOX decode, fixed score/class-wise NMS and FOV/box validation;
|
||||||
|
- `k1link.perception.detector.FrozenYoloxDetectorProvider` turns exactly one
|
||||||
|
decoded image into class-agnostic `ObjectProposal2D` rows; the COCO label is
|
||||||
|
retained only as optional `semantic_hint`, and no tracklet is published;
|
||||||
|
- model, config, valid-FOV, score, NMS, target classes and raster identities are
|
||||||
|
frozen against in-place tuning;
|
||||||
|
- zero-proposal, pathological/non-finite and provider-failure frames are counted
|
||||||
|
explicitly; graph stage telemetry remains the common latency path;
|
||||||
|
- the recorded source now has a separate sequential PyAV decode seam, so image
|
||||||
|
decoding remains source work and no detector logic enters the adapter;
|
||||||
|
- all 4,489 accepted immutable E46J frame documents and 15,499 detections map to
|
||||||
|
the new proposal contract with exact frame accounting.
|
||||||
|
|
||||||
|
This increment does **not** claim a new 4,489-frame Triton execution. The existing
|
||||||
|
E46J 47.840 FPS result remains the baseline evidence. A fresh provider execution
|
||||||
|
requires a digest-bound shadow package; ad-hoc executable staging on Worker 006
|
||||||
|
is prohibited by the deployment canon. Therefore M4.3 runtime/capacity exit and
|
||||||
|
its final checker remain open, and M4.4 does not start yet.
|
||||||
|
|
||||||
|
Validation at this increment: 76 focused-and-related tests and the complete
|
||||||
|
Python suite (`1221 passed, 1 skipped`). Scoped Ruff and strict mypy pass for the
|
||||||
|
new compute primitive and complete `src/k1link/perception` package.
|
||||||
|
|
||||||
## Implementation order
|
## Implementation order
|
||||||
|
|
||||||
The implementation sequence is intentionally strict:
|
The implementation sequence is intentionally strict:
|
||||||
|
|||||||
@@ -0,0 +1,420 @@
|
|||||||
|
"""Source-neutral frozen YOLOX preprocessing, inference transport and postprocessing."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
import http.client
|
||||||
|
import importlib
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import urllib.parse
|
||||||
|
from collections import Counter
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Final, Protocol, cast
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
from numpy.typing import NDArray
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
YOLOX_MODEL_ID: Final = "yolox_s"
|
||||||
|
YOLOX_MODEL_VERSION: Final = 1
|
||||||
|
YOLOX_MODEL_SHA256: Final = (
|
||||||
|
"c5c2d13e59ae883e6af3b45daea64af4833a4951c92d116ec270d9ddbe998063"
|
||||||
|
)
|
||||||
|
YOLOX_CONFIG_SHA256: Final = (
|
||||||
|
"5795c737a7935a655961b069e8404d336d891f9762fb6dffb93956a076479604"
|
||||||
|
)
|
||||||
|
YOLOX_VALID_FOV_SHA256: Final = (
|
||||||
|
"a40cee06b7c6f69b6a09a11563dcfd237f3de833b1ccd31459e66692e528ba63"
|
||||||
|
)
|
||||||
|
COCO_CLASSES: Final = (
|
||||||
|
"person", "bicycle", "car", "motorcycle", "airplane", "bus", "train",
|
||||||
|
"truck", "boat", "traffic light", "fire hydrant", "stop sign",
|
||||||
|
"parking meter", "bench", "bird", "cat", "dog", "horse", "sheep",
|
||||||
|
"cow", "elephant", "bear", "zebra", "giraffe", "backpack", "umbrella",
|
||||||
|
"handbag", "tie", "suitcase", "frisbee", "skis", "snowboard",
|
||||||
|
"sports ball", "kite", "baseball bat", "baseball glove", "skateboard",
|
||||||
|
"surfboard", "tennis racket", "bottle", "wine glass", "cup", "fork",
|
||||||
|
"knife", "spoon", "bowl", "banana", "apple", "sandwich", "orange",
|
||||||
|
"broccoli", "carrot", "hot dog", "pizza", "donut", "cake", "chair",
|
||||||
|
"couch", "potted plant", "bed", "dining table", "toilet", "tv",
|
||||||
|
"laptop", "mouse", "remote", "keyboard", "cell phone", "microwave",
|
||||||
|
"oven", "toaster", "sink", "refrigerator", "book", "clock", "vase",
|
||||||
|
"scissors", "teddy bear", "hair drier", "toothbrush",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class YoloxDetectorError(RuntimeError):
|
||||||
|
"""The frozen detector profile, tensor or inference response is incompatible."""
|
||||||
|
|
||||||
|
|
||||||
|
class InferenceBackend(Protocol):
|
||||||
|
def infer(self, tensor: NDArray[np.float32]) -> NDArray[np.float32]: ...
|
||||||
|
|
||||||
|
|
||||||
|
class ImageResizer(Protocol):
|
||||||
|
def resize(self, image: NDArray[np.uint8], width: int, height: int) -> NDArray[np.uint8]: ...
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class FrozenYoloxConfig:
|
||||||
|
source_width: int = 800
|
||||||
|
source_height: int = 600
|
||||||
|
input_width: int = 640
|
||||||
|
input_height: int = 640
|
||||||
|
fill_value: int = 114
|
||||||
|
minimum_score: float = 0.5
|
||||||
|
nms_iou_threshold: float = 0.45
|
||||||
|
target_class_ids: tuple[int, ...] = (0, 1, 2, 3, 5, 7)
|
||||||
|
minimum_box_area_pixels: float = 64.0
|
||||||
|
maximum_box_area_fraction: float = 0.5
|
||||||
|
minimum_valid_fov_fraction: float = 0.5
|
||||||
|
require_center_inside_valid_fov: bool = True
|
||||||
|
|
||||||
|
def __post_init__(self) -> None:
|
||||||
|
if (
|
||||||
|
self.source_width,
|
||||||
|
self.source_height,
|
||||||
|
self.input_width,
|
||||||
|
self.input_height,
|
||||||
|
self.fill_value,
|
||||||
|
self.minimum_score,
|
||||||
|
self.nms_iou_threshold,
|
||||||
|
self.target_class_ids,
|
||||||
|
self.minimum_box_area_pixels,
|
||||||
|
self.maximum_box_area_fraction,
|
||||||
|
self.minimum_valid_fov_fraction,
|
||||||
|
self.require_center_inside_valid_fov,
|
||||||
|
) != (800, 600, 640, 640, 114, 0.5, 0.45, (0, 1, 2, 3, 5, 7), 64.0, 0.5, 0.5, True):
|
||||||
|
raise YoloxDetectorError("frozen YOLOX detector profile cannot be tuned in place")
|
||||||
|
|
||||||
|
|
||||||
|
FROZEN_YOLOX_CONFIG: Final = FrozenYoloxConfig()
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class YoloxDetection:
|
||||||
|
class_id: int
|
||||||
|
label: str
|
||||||
|
score: float
|
||||||
|
bbox_xyxy: tuple[float, float, float, float]
|
||||||
|
valid_fov_fraction: float
|
||||||
|
|
||||||
|
def __post_init__(self) -> None:
|
||||||
|
if not 0 <= self.class_id < len(COCO_CLASSES) or self.label != COCO_CLASSES[self.class_id]:
|
||||||
|
raise YoloxDetectorError("YOLOX semantic diagnostic is incompatible")
|
||||||
|
if not math.isfinite(self.score) or not 0.0 <= self.score <= 1.0:
|
||||||
|
raise YoloxDetectorError("YOLOX score is invalid")
|
||||||
|
x1, y1, x2, y2 = self.bbox_xyxy
|
||||||
|
if not all(math.isfinite(value) for value in self.bbox_xyxy) or not (
|
||||||
|
0.0 <= x1 < x2 <= 800.0 and 0.0 <= y1 < y2 <= 600.0
|
||||||
|
):
|
||||||
|
raise YoloxDetectorError("YOLOX source bounding box is invalid")
|
||||||
|
if not 0.0 <= self.valid_fov_fraction <= 1.0:
|
||||||
|
raise YoloxDetectorError("YOLOX valid-FOV fraction is invalid")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class YoloxPostprocessResult:
|
||||||
|
detections: tuple[YoloxDetection, ...]
|
||||||
|
rejected: tuple[tuple[str, int], ...]
|
||||||
|
|
||||||
|
|
||||||
|
class OpenCvBilinearResizer:
|
||||||
|
"""Lazy exact E46J resize backend; importing this module does not require OpenCV."""
|
||||||
|
|
||||||
|
def resize(
|
||||||
|
self,
|
||||||
|
image: NDArray[np.uint8],
|
||||||
|
width: int,
|
||||||
|
height: int,
|
||||||
|
) -> NDArray[np.uint8]:
|
||||||
|
try:
|
||||||
|
cv2 = importlib.import_module("cv2")
|
||||||
|
except ModuleNotFoundError as exc:
|
||||||
|
raise YoloxDetectorError("OpenCV resize backend is unavailable") from exc
|
||||||
|
resized = cv2.resize(image, (width, height), interpolation=cv2.INTER_LINEAR)
|
||||||
|
return np.asarray(resized, dtype=np.uint8)
|
||||||
|
|
||||||
|
|
||||||
|
class TritonHttpInferenceBackend:
|
||||||
|
"""Persistent, single-stream Triton V2 binary HTTP inference transport."""
|
||||||
|
|
||||||
|
def __init__(self, endpoint: str, *, timeout_seconds: float = 60.0) -> None:
|
||||||
|
parsed = urllib.parse.urlsplit(endpoint)
|
||||||
|
if parsed.scheme != "http" or not parsed.hostname or parsed.query or parsed.fragment:
|
||||||
|
raise YoloxDetectorError("Triton endpoint must be an explicit HTTP origin")
|
||||||
|
if not math.isfinite(timeout_seconds) or timeout_seconds <= 0:
|
||||||
|
raise YoloxDetectorError("Triton timeout must be positive")
|
||||||
|
self.path = f"{parsed.path.rstrip('/')}/v2/models/{YOLOX_MODEL_ID}/infer"
|
||||||
|
self.connection = http.client.HTTPConnection(
|
||||||
|
parsed.hostname,
|
||||||
|
parsed.port or 80,
|
||||||
|
timeout=timeout_seconds,
|
||||||
|
)
|
||||||
|
|
||||||
|
def close(self) -> None:
|
||||||
|
self.connection.close()
|
||||||
|
|
||||||
|
def infer(self, tensor: NDArray[np.float32]) -> NDArray[np.float32]:
|
||||||
|
contiguous = np.ascontiguousarray(tensor, dtype=np.float32)
|
||||||
|
if contiguous.shape != (1, 3, 640, 640) or not np.isfinite(contiguous).all():
|
||||||
|
raise YoloxDetectorError("Triton input tensor is incompatible")
|
||||||
|
binary = contiguous.tobytes()
|
||||||
|
header = {
|
||||||
|
"inputs": [{
|
||||||
|
"name": "images",
|
||||||
|
"shape": [1, 3, 640, 640],
|
||||||
|
"datatype": "FP32",
|
||||||
|
"parameters": {"binary_data_size": len(binary)},
|
||||||
|
}],
|
||||||
|
"outputs": [{"name": "output", "parameters": {"binary_data": True}}],
|
||||||
|
}
|
||||||
|
encoded = json.dumps(header, sort_keys=True, separators=(",", ":")).encode()
|
||||||
|
self.connection.request(
|
||||||
|
"POST",
|
||||||
|
self.path,
|
||||||
|
body=encoded + binary,
|
||||||
|
headers={
|
||||||
|
"Content-Type": "application/octet-stream",
|
||||||
|
"Inference-Header-Content-Length": str(len(encoded)),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
response = self.connection.getresponse()
|
||||||
|
payload = response.read()
|
||||||
|
if response.status != 200:
|
||||||
|
raise YoloxDetectorError(f"Triton inference failed with HTTP {response.status}")
|
||||||
|
header_value = response.getheader("Inference-Header-Content-Length")
|
||||||
|
try:
|
||||||
|
header_length = int(header_value or "")
|
||||||
|
descriptor = json.loads(payload[:header_length])["outputs"][0]
|
||||||
|
shape = tuple(int(value) for value in descriptor["shape"])
|
||||||
|
except (KeyError, TypeError, ValueError, json.JSONDecodeError) as exc:
|
||||||
|
raise YoloxDetectorError("Triton output descriptor is invalid") from exc
|
||||||
|
if descriptor.get("name") != "output" or descriptor.get("datatype") != "FP32":
|
||||||
|
raise YoloxDetectorError("Triton output identity changed")
|
||||||
|
array = np.frombuffer(payload[header_length:], dtype="<f4")
|
||||||
|
if shape != (1, 8400, 85) or array.size != math.prod(shape):
|
||||||
|
raise YoloxDetectorError("Triton YOLOX output shape changed")
|
||||||
|
output = np.asarray(array.reshape(shape), dtype=np.float32)
|
||||||
|
if not np.isfinite(output).all():
|
||||||
|
raise YoloxDetectorError("Triton YOLOX output contains non-finite values")
|
||||||
|
return output
|
||||||
|
|
||||||
|
|
||||||
|
def load_valid_fov_mask(
|
||||||
|
path: Path,
|
||||||
|
*,
|
||||||
|
expected_sha256: str = YOLOX_VALID_FOV_SHA256,
|
||||||
|
) -> NDArray[np.bool_]:
|
||||||
|
resolved = path.resolve(strict=True)
|
||||||
|
if resolved.is_symlink() or _sha256(resolved) != expected_sha256:
|
||||||
|
raise YoloxDetectorError("valid-FOV mask identity changed")
|
||||||
|
mask = np.asarray(Image.open(resolved).convert("L")) > 0
|
||||||
|
if mask.shape != (600, 800) or not np.any(mask):
|
||||||
|
raise YoloxDetectorError("valid-FOV mask geometry changed")
|
||||||
|
return np.asarray(mask, dtype=np.bool_)
|
||||||
|
|
||||||
|
|
||||||
|
def preprocess_raw_kb4(
|
||||||
|
image_bgr: NDArray[np.uint8],
|
||||||
|
mask: NDArray[np.bool_],
|
||||||
|
*,
|
||||||
|
config: FrozenYoloxConfig = FROZEN_YOLOX_CONFIG,
|
||||||
|
resizer: ImageResizer | None = None,
|
||||||
|
) -> NDArray[np.float32]:
|
||||||
|
if image_bgr.shape != (config.source_height, config.source_width, 3):
|
||||||
|
raise YoloxDetectorError("raw KB4 image raster changed")
|
||||||
|
if image_bgr.dtype != np.uint8 or mask.shape != image_bgr.shape[:2] or mask.dtype != np.bool_:
|
||||||
|
raise YoloxDetectorError("raw KB4 image or valid-FOV mask type changed")
|
||||||
|
ratio = min(
|
||||||
|
config.input_height / config.source_height,
|
||||||
|
config.input_width / config.source_width,
|
||||||
|
)
|
||||||
|
resized_width = int(config.source_width * ratio)
|
||||||
|
resized_height = int(config.source_height * ratio)
|
||||||
|
masked = np.where(mask[..., None], image_bgr, config.fill_value).astype(np.uint8)
|
||||||
|
resized = (resizer or OpenCvBilinearResizer()).resize(masked, resized_width, resized_height)
|
||||||
|
if resized.shape != (resized_height, resized_width, 3):
|
||||||
|
raise YoloxDetectorError("resize backend returned an incompatible raster")
|
||||||
|
canvas = np.full(
|
||||||
|
(config.input_height, config.input_width, 3),
|
||||||
|
config.fill_value,
|
||||||
|
dtype=np.uint8,
|
||||||
|
)
|
||||||
|
canvas[:resized_height, :resized_width] = resized
|
||||||
|
return np.ascontiguousarray(canvas.transpose(2, 0, 1), dtype=np.float32)[None]
|
||||||
|
|
||||||
|
|
||||||
|
def postprocess_yolox(
|
||||||
|
output: NDArray[np.float32],
|
||||||
|
mask: NDArray[np.bool_],
|
||||||
|
*,
|
||||||
|
config: FrozenYoloxConfig = FROZEN_YOLOX_CONFIG,
|
||||||
|
) -> YoloxPostprocessResult:
|
||||||
|
if output.shape != (1, 8400, 85) or not np.isfinite(output).all():
|
||||||
|
raise YoloxDetectorError("YOLOX output tensor is incompatible")
|
||||||
|
if mask.shape != (config.source_height, config.source_width) or mask.dtype != np.bool_:
|
||||||
|
raise YoloxDetectorError("valid-FOV mask is incompatible")
|
||||||
|
prediction = _decode_yolox(output)[0]
|
||||||
|
boxes = prediction[:, :4]
|
||||||
|
boxes_xyxy = np.empty_like(boxes)
|
||||||
|
boxes_xyxy[:, 0] = boxes[:, 0] - boxes[:, 2] / 2.0
|
||||||
|
boxes_xyxy[:, 1] = boxes[:, 1] - boxes[:, 3] / 2.0
|
||||||
|
boxes_xyxy[:, 2] = boxes[:, 0] + boxes[:, 2] / 2.0
|
||||||
|
boxes_xyxy[:, 3] = boxes[:, 1] + boxes[:, 3] / 2.0
|
||||||
|
ratio = min(
|
||||||
|
config.input_height / config.source_height,
|
||||||
|
config.input_width / config.source_width,
|
||||||
|
)
|
||||||
|
boxes_xyxy /= ratio
|
||||||
|
class_scores = prediction[:, 4:5] * prediction[:, 5:]
|
||||||
|
class_ids = class_scores.argmax(axis=1)
|
||||||
|
scores = class_scores[np.arange(class_scores.shape[0]), class_ids]
|
||||||
|
candidate_mask = np.logical_and(
|
||||||
|
scores >= config.minimum_score,
|
||||||
|
np.isin(class_ids, config.target_class_ids),
|
||||||
|
)
|
||||||
|
candidate_boxes = boxes_xyxy[candidate_mask]
|
||||||
|
candidate_scores = scores[candidate_mask]
|
||||||
|
candidate_classes = class_ids[candidate_mask]
|
||||||
|
finite = np.logical_and(np.isfinite(candidate_boxes).all(axis=1), np.isfinite(candidate_scores))
|
||||||
|
rejected: Counter[str] = Counter()
|
||||||
|
nonfinite_count = int((~finite).sum())
|
||||||
|
if nonfinite_count:
|
||||||
|
rejected["nonfinite"] = nonfinite_count
|
||||||
|
candidate_boxes = candidate_boxes[finite]
|
||||||
|
candidate_scores = candidate_scores[finite]
|
||||||
|
candidate_classes = candidate_classes[finite]
|
||||||
|
integral = np.pad(mask.astype(np.int64), ((1, 0), (1, 0))).cumsum(0).cumsum(1)
|
||||||
|
result: list[YoloxDetection] = []
|
||||||
|
for class_id in config.target_class_ids:
|
||||||
|
indices = np.where(candidate_classes == class_id)[0]
|
||||||
|
if not indices.size:
|
||||||
|
continue
|
||||||
|
keep = _nms(candidate_boxes[indices], candidate_scores[indices], config.nms_iou_threshold)
|
||||||
|
for selected in indices[keep]:
|
||||||
|
box = candidate_boxes[selected].copy()
|
||||||
|
box[[0, 2]] = np.clip(box[[0, 2]], 0, config.source_width)
|
||||||
|
box[[1, 3]] = np.clip(box[[1, 3]], 0, config.source_height)
|
||||||
|
fraction, center_inside, area = _valid_fraction(box, integral)
|
||||||
|
if area < config.minimum_box_area_pixels:
|
||||||
|
rejected["small-box"] += 1
|
||||||
|
continue
|
||||||
|
if (
|
||||||
|
area / (config.source_width * config.source_height)
|
||||||
|
> config.maximum_box_area_fraction
|
||||||
|
):
|
||||||
|
rejected["large-box"] += 1
|
||||||
|
continue
|
||||||
|
if fraction < config.minimum_valid_fov_fraction:
|
||||||
|
rejected["outside-valid-fov"] += 1
|
||||||
|
continue
|
||||||
|
if config.require_center_inside_valid_fov and not center_inside:
|
||||||
|
rejected["center-outside-valid-fov"] += 1
|
||||||
|
continue
|
||||||
|
result.append(YoloxDetection(
|
||||||
|
class_id=int(class_id),
|
||||||
|
label=COCO_CLASSES[int(class_id)],
|
||||||
|
score=round(float(candidate_scores[selected]), 9),
|
||||||
|
bbox_xyxy=(
|
||||||
|
round(float(box[0]), 6),
|
||||||
|
round(float(box[1]), 6),
|
||||||
|
round(float(box[2]), 6),
|
||||||
|
round(float(box[3]), 6),
|
||||||
|
),
|
||||||
|
valid_fov_fraction=round(fraction, 6),
|
||||||
|
))
|
||||||
|
result.sort(key=lambda item: (-item.score, item.class_id))
|
||||||
|
return YoloxPostprocessResult(tuple(result), tuple(sorted(rejected.items())))
|
||||||
|
|
||||||
|
|
||||||
|
def _decode_yolox(output: NDArray[np.float32]) -> NDArray[np.float32]:
|
||||||
|
predictions = output.copy()
|
||||||
|
grids: list[NDArray[np.int64]] = []
|
||||||
|
strides: list[NDArray[np.int64]] = []
|
||||||
|
for stride in (8, 16, 32):
|
||||||
|
height = 640 // stride
|
||||||
|
width = 640 // stride
|
||||||
|
yv, xv = np.meshgrid(np.arange(height), np.arange(width), indexing="ij")
|
||||||
|
grids.append(np.stack((xv, yv), axis=2).reshape(1, -1, 2))
|
||||||
|
strides.append(np.full((1, height * width, 1), stride, dtype=np.int64))
|
||||||
|
grid = np.concatenate(grids, axis=1)
|
||||||
|
expanded_strides = np.concatenate(strides, axis=1)
|
||||||
|
predictions[..., :2] = (predictions[..., :2] + grid) * expanded_strides
|
||||||
|
with np.errstate(over="ignore", invalid="ignore"):
|
||||||
|
predictions[..., 2:4] = np.exp(predictions[..., 2:4]) * expanded_strides
|
||||||
|
return predictions
|
||||||
|
|
||||||
|
|
||||||
|
def _nms(boxes: NDArray[np.float32], scores: NDArray[np.float32], threshold: float) -> list[int]:
|
||||||
|
order = scores.argsort()[::-1]
|
||||||
|
keep: list[int] = []
|
||||||
|
while order.size:
|
||||||
|
index = int(order[0])
|
||||||
|
keep.append(index)
|
||||||
|
overlaps = _box_iou(boxes[index], boxes[order[1:]])
|
||||||
|
order = order[np.where(overlaps <= threshold)[0] + 1]
|
||||||
|
return keep
|
||||||
|
|
||||||
|
|
||||||
|
def _box_iou(one: NDArray[np.float32], many: NDArray[np.float32]) -> NDArray[np.float32]:
|
||||||
|
if many.size == 0:
|
||||||
|
return np.zeros((0,), dtype=np.float32)
|
||||||
|
top_left = np.maximum(one[:2], many[:, :2])
|
||||||
|
bottom_right = np.minimum(one[2:], many[:, 2:])
|
||||||
|
intersection = np.prod(np.maximum(0.0, bottom_right - top_left), axis=1)
|
||||||
|
one_area = max(0.0, float(one[2] - one[0])) * max(0.0, float(one[3] - one[1]))
|
||||||
|
many_area = np.maximum(0.0, many[:, 2] - many[:, 0]) * np.maximum(
|
||||||
|
0.0, many[:, 3] - many[:, 1]
|
||||||
|
)
|
||||||
|
union = one_area + many_area - intersection
|
||||||
|
return cast(
|
||||||
|
NDArray[np.float32],
|
||||||
|
np.divide(intersection, union, out=np.zeros_like(intersection), where=union > 0),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _valid_fraction(
|
||||||
|
box: NDArray[np.float32],
|
||||||
|
integral: NDArray[np.int64],
|
||||||
|
) -> tuple[float, bool, float]:
|
||||||
|
height = integral.shape[0] - 1
|
||||||
|
width = integral.shape[1] - 1
|
||||||
|
x1 = int(np.clip(math.floor(float(box[0])), 0, width))
|
||||||
|
y1 = int(np.clip(math.floor(float(box[1])), 0, height))
|
||||||
|
x2 = int(np.clip(math.ceil(float(box[2])), 0, width))
|
||||||
|
y2 = int(np.clip(math.ceil(float(box[3])), 0, height))
|
||||||
|
area = float(max(0, x2 - x1) * max(0, y2 - y1))
|
||||||
|
if area <= 0:
|
||||||
|
return 0.0, False, 0.0
|
||||||
|
inside = integral[y2, x2] - integral[y1, x2] - integral[y2, x1] + integral[y1, x1]
|
||||||
|
center_x = int(np.clip(round((float(box[0]) + float(box[2])) / 2.0), 0, width - 1))
|
||||||
|
center_y = int(np.clip(round((float(box[1]) + float(box[3])) / 2.0), 0, height - 1))
|
||||||
|
center_inside = bool(
|
||||||
|
integral[center_y + 1, center_x + 1]
|
||||||
|
- integral[center_y, center_x + 1]
|
||||||
|
- integral[center_y + 1, center_x]
|
||||||
|
+ integral[center_y, center_x]
|
||||||
|
)
|
||||||
|
return float(inside) / area, center_inside, area
|
||||||
|
|
||||||
|
|
||||||
|
def _sha256(path: Path) -> str:
|
||||||
|
digest = hashlib.sha256()
|
||||||
|
with path.open("rb") as stream:
|
||||||
|
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
|
||||||
|
digest.update(chunk)
|
||||||
|
return digest.hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"YOLOX_CONFIG_SHA256", "YOLOX_MODEL_ID", "YOLOX_MODEL_SHA256",
|
||||||
|
"YOLOX_MODEL_VERSION", "YOLOX_VALID_FOV_SHA256", "FROZEN_YOLOX_CONFIG",
|
||||||
|
"FrozenYoloxConfig",
|
||||||
|
"ImageResizer", "InferenceBackend", "OpenCvBilinearResizer",
|
||||||
|
"TritonHttpInferenceBackend", "YoloxDetection", "YoloxDetectorError",
|
||||||
|
"YoloxPostprocessResult", "load_valid_fov_mask", "postprocess_yolox",
|
||||||
|
"preprocess_raw_kb4",
|
||||||
|
]
|
||||||
@@ -49,6 +49,15 @@ from .contracts import (
|
|||||||
TimestampBundle,
|
TimestampBundle,
|
||||||
validate_exclusive_point_ownership,
|
validate_exclusive_point_ownership,
|
||||||
)
|
)
|
||||||
|
from .detector import (
|
||||||
|
FROZEN_YOLOX_MODEL_ID,
|
||||||
|
FROZEN_YOLOX_PREPROCESS_ID,
|
||||||
|
FROZEN_YOLOX_PROVIDER_ID,
|
||||||
|
DetectorProviderError,
|
||||||
|
DetectorProviderSnapshot,
|
||||||
|
FrozenYoloxDetectorProvider,
|
||||||
|
proposals_from_detections,
|
||||||
|
)
|
||||||
from .graph import (
|
from .graph import (
|
||||||
GRAPH_RESULT_SCHEMA,
|
GRAPH_RESULT_SCHEMA,
|
||||||
REFERENCE_GRAPH_ID,
|
REFERENCE_GRAPH_ID,
|
||||||
@@ -78,8 +87,11 @@ from .providers import (
|
|||||||
ThreatProvider,
|
ThreatProvider,
|
||||||
)
|
)
|
||||||
from .recorded_source import (
|
from .recorded_source import (
|
||||||
|
DecodedRecordedSource,
|
||||||
LiveSourceAdapter,
|
LiveSourceAdapter,
|
||||||
|
PyAvRecordedImageDecoder,
|
||||||
RecordedFrameReference,
|
RecordedFrameReference,
|
||||||
|
RecordedImageDecoder,
|
||||||
RecordedRavnoves00Source,
|
RecordedRavnoves00Source,
|
||||||
RecordedSourceError,
|
RecordedSourceError,
|
||||||
ReplayPacing,
|
ReplayPacing,
|
||||||
@@ -129,6 +141,13 @@ __all__ = [
|
|||||||
"ThreatDecision",
|
"ThreatDecision",
|
||||||
"TimestampBundle",
|
"TimestampBundle",
|
||||||
"validate_exclusive_point_ownership",
|
"validate_exclusive_point_ownership",
|
||||||
|
"FROZEN_YOLOX_MODEL_ID",
|
||||||
|
"FROZEN_YOLOX_PREPROCESS_ID",
|
||||||
|
"FROZEN_YOLOX_PROVIDER_ID",
|
||||||
|
"DetectorProviderError",
|
||||||
|
"DetectorProviderSnapshot",
|
||||||
|
"FrozenYoloxDetectorProvider",
|
||||||
|
"proposals_from_detections",
|
||||||
"REFERENCE_GRAPH_CONFIG_SCHEMA",
|
"REFERENCE_GRAPH_CONFIG_SCHEMA",
|
||||||
"DetectorProvider",
|
"DetectorProvider",
|
||||||
"GeometryAssociationProvider",
|
"GeometryAssociationProvider",
|
||||||
@@ -154,7 +173,10 @@ __all__ = [
|
|||||||
"TerminalOutcome",
|
"TerminalOutcome",
|
||||||
"TerminalOutcomeType",
|
"TerminalOutcomeType",
|
||||||
"LiveSourceAdapter",
|
"LiveSourceAdapter",
|
||||||
|
"DecodedRecordedSource",
|
||||||
|
"PyAvRecordedImageDecoder",
|
||||||
"RecordedFrameReference",
|
"RecordedFrameReference",
|
||||||
|
"RecordedImageDecoder",
|
||||||
"RecordedRavnoves00Source",
|
"RecordedRavnoves00Source",
|
||||||
"RecordedSourceError",
|
"RecordedSourceError",
|
||||||
"ReplayPacing",
|
"ReplayPacing",
|
||||||
|
|||||||
@@ -21,6 +21,9 @@ BASELINE_SOURCE_ID: Final = "RAVNOVES00"
|
|||||||
BASELINE_SESSION_ID: Final = "20260720T065719Z_viewer_live"
|
BASELINE_SESSION_ID: Final = "20260720T065719Z_viewer_live"
|
||||||
BASELINE_CAMERA_SOURCE_ID: Final = "sensor.camera.right"
|
BASELINE_CAMERA_SOURCE_ID: Final = "sensor.camera.right"
|
||||||
BASELINE_RECORDED_JOB_ID: Final = "recorded-camera-602ac89026ed12978619801d"
|
BASELINE_RECORDED_JOB_ID: Final = "recorded-camera-602ac89026ed12978619801d"
|
||||||
|
BASELINE_CAMERA_STREAM_SHA256: Final = (
|
||||||
|
"cadd1696ff000904eb78633a0a8418104b8024f178b91f3421789021ccb160e8"
|
||||||
|
)
|
||||||
BASELINE_SOURCE_PACK_ID: Final = (
|
BASELINE_SOURCE_PACK_ID: Final = (
|
||||||
"e10-lidar-pack-576c994a6c814e2592dd6240ace3902a5db94843312c759a73ba0c9166157d2b"
|
"e10-lidar-pack-576c994a6c814e2592dd6240ace3902a5db94843312c759a73ba0c9166157d2b"
|
||||||
)
|
)
|
||||||
@@ -124,7 +127,8 @@ def load_m4_baseline(path: Path) -> BaselineProfile:
|
|||||||
raise BaselineContractError("M4 source pack identity changed")
|
raise BaselineContractError("M4 source pack identity changed")
|
||||||
if source.get("source_pack_artifact_sha256") != BASELINE_SOURCE_PACK_SHA256:
|
if source.get("source_pack_artifact_sha256") != BASELINE_SOURCE_PACK_SHA256:
|
||||||
raise BaselineContractError("M4 source pack artifact identity changed")
|
raise BaselineContractError("M4 source pack artifact identity changed")
|
||||||
_digest(source.get("camera_stream_sha256"), "camera stream digest")
|
if source.get("camera_stream_sha256") != BASELINE_CAMERA_STREAM_SHA256:
|
||||||
|
raise BaselineContractError("M4 camera stream identity changed")
|
||||||
modalities = _string_array(source.get("modalities"), "source modalities")
|
modalities = _string_array(source.get("modalities"), "source modalities")
|
||||||
if set(modalities) != {"image", "registered-point-increment", "pose"}:
|
if set(modalities) != {"image", "registered-point-increment", "pose"}:
|
||||||
raise BaselineContractError("baseline source must bind image, points and pose")
|
raise BaselineContractError("baseline source must bind image, points and pose")
|
||||||
|
|||||||
@@ -0,0 +1,156 @@
|
|||||||
|
"""Frozen raw-KB4 YOLOX provider for class-agnostic object proposals."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import time
|
||||||
|
from collections import Counter
|
||||||
|
from collections.abc import Callable
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from threading import Lock
|
||||||
|
from typing import Final
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
from numpy.typing import NDArray
|
||||||
|
|
||||||
|
from k1link.compute.yolox_object_detector import (
|
||||||
|
FROZEN_YOLOX_CONFIG,
|
||||||
|
YOLOX_MODEL_ID,
|
||||||
|
YOLOX_MODEL_VERSION,
|
||||||
|
FrozenYoloxConfig,
|
||||||
|
ImageResizer,
|
||||||
|
InferenceBackend,
|
||||||
|
YoloxDetection,
|
||||||
|
postprocess_yolox,
|
||||||
|
preprocess_raw_kb4,
|
||||||
|
)
|
||||||
|
|
||||||
|
from .contracts import BoundingRegion2D, ObjectProposal2D
|
||||||
|
from .providers import SourcePacket
|
||||||
|
|
||||||
|
FROZEN_YOLOX_PROVIDER_ID: Final = "triton-yolox-s-raw-kb4/v1"
|
||||||
|
FROZEN_YOLOX_MODEL_ID: Final = f"{YOLOX_MODEL_ID}:{YOLOX_MODEL_VERSION}"
|
||||||
|
FROZEN_YOLOX_PREPROCESS_ID: Final = "raw-kb4-valid-fov-letterbox/v1"
|
||||||
|
|
||||||
|
|
||||||
|
class DetectorProviderError(RuntimeError):
|
||||||
|
"""The detector input, frozen inference or proposal output is incompatible."""
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True, slots=True)
|
||||||
|
class DetectorProviderSnapshot:
|
||||||
|
input_frames: int
|
||||||
|
completed_frames: int
|
||||||
|
failed_frames: int
|
||||||
|
zero_proposal_frames: int
|
||||||
|
proposal_count: int
|
||||||
|
rejected: tuple[tuple[str, int], ...]
|
||||||
|
core_duration_ns: int
|
||||||
|
|
||||||
|
|
||||||
|
class FrozenYoloxDetectorProvider:
|
||||||
|
"""One image payload produces one frozen inference request and proposal tuple."""
|
||||||
|
|
||||||
|
provider_id: str = FROZEN_YOLOX_PROVIDER_ID
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
mask: NDArray[np.bool_],
|
||||||
|
backend: InferenceBackend,
|
||||||
|
resizer: ImageResizer | None = None,
|
||||||
|
config: FrozenYoloxConfig = FROZEN_YOLOX_CONFIG,
|
||||||
|
clock_ns: Callable[[], int] = time.perf_counter_ns,
|
||||||
|
) -> None:
|
||||||
|
if mask.shape != (600, 800) or mask.dtype != np.bool_ or not np.any(mask):
|
||||||
|
raise DetectorProviderError("frozen valid-FOV mask is incompatible")
|
||||||
|
self.mask = np.asarray(mask, dtype=np.bool_)
|
||||||
|
self.backend = backend
|
||||||
|
self.resizer = resizer
|
||||||
|
self.config = config
|
||||||
|
self._clock_ns = clock_ns
|
||||||
|
self._lock = Lock()
|
||||||
|
self._input_frames = 0
|
||||||
|
self._completed_frames = 0
|
||||||
|
self._failed_frames = 0
|
||||||
|
self._zero_proposal_frames = 0
|
||||||
|
self._proposal_count = 0
|
||||||
|
self._rejected: Counter[str] = Counter()
|
||||||
|
self._core_duration_ns = 0
|
||||||
|
|
||||||
|
def detect(self, packet: SourcePacket) -> tuple[ObjectProposal2D, ...]:
|
||||||
|
payload = packet.image_payload
|
||||||
|
with self._lock:
|
||||||
|
self._input_frames += 1
|
||||||
|
started_ns = int(self._clock_ns())
|
||||||
|
try:
|
||||||
|
if not isinstance(payload, np.ndarray):
|
||||||
|
raise DetectorProviderError("detector requires a decoded BGR image payload")
|
||||||
|
image = np.asarray(payload)
|
||||||
|
if image.dtype != np.uint8:
|
||||||
|
raise DetectorProviderError("decoded BGR image must be uint8")
|
||||||
|
tensor = preprocess_raw_kb4(
|
||||||
|
image,
|
||||||
|
self.mask,
|
||||||
|
config=self.config,
|
||||||
|
resizer=self.resizer,
|
||||||
|
)
|
||||||
|
output = self.backend.infer(tensor)
|
||||||
|
postprocessed = postprocess_yolox(output, self.mask, config=self.config)
|
||||||
|
proposals = proposals_from_detections(packet, postprocessed.detections)
|
||||||
|
except Exception:
|
||||||
|
with self._lock:
|
||||||
|
self._failed_frames += 1
|
||||||
|
self._core_duration_ns += max(0, int(self._clock_ns()) - started_ns)
|
||||||
|
raise
|
||||||
|
with self._lock:
|
||||||
|
self._completed_frames += 1
|
||||||
|
self._proposal_count += len(proposals)
|
||||||
|
self._zero_proposal_frames += not proposals
|
||||||
|
self._rejected.update(dict(postprocessed.rejected))
|
||||||
|
self._core_duration_ns += max(0, int(self._clock_ns()) - started_ns)
|
||||||
|
return proposals
|
||||||
|
|
||||||
|
def snapshot(self) -> DetectorProviderSnapshot:
|
||||||
|
with self._lock:
|
||||||
|
return DetectorProviderSnapshot(
|
||||||
|
input_frames=self._input_frames,
|
||||||
|
completed_frames=self._completed_frames,
|
||||||
|
failed_frames=self._failed_frames,
|
||||||
|
zero_proposal_frames=self._zero_proposal_frames,
|
||||||
|
proposal_count=self._proposal_count,
|
||||||
|
rejected=tuple(sorted(self._rejected.items())),
|
||||||
|
core_duration_ns=self._core_duration_ns,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def proposals_from_detections(
|
||||||
|
packet: SourcePacket,
|
||||||
|
detections: tuple[YoloxDetection, ...],
|
||||||
|
) -> tuple[ObjectProposal2D, ...]:
|
||||||
|
envelope = packet.envelope
|
||||||
|
return tuple(
|
||||||
|
ObjectProposal2D(
|
||||||
|
proposal_id=f"proposal-{envelope.sequence}-{index}",
|
||||||
|
source_id=envelope.source_id,
|
||||||
|
frame_id=envelope.frame_id,
|
||||||
|
region=BoundingRegion2D(*detection.bbox_xyxy),
|
||||||
|
objectness=detection.score,
|
||||||
|
provider_id=FROZEN_YOLOX_PROVIDER_ID,
|
||||||
|
model_id=FROZEN_YOLOX_MODEL_ID,
|
||||||
|
preprocess_id=FROZEN_YOLOX_PREPROCESS_ID,
|
||||||
|
semantic_hint=detection.label,
|
||||||
|
provider_tracklet=None,
|
||||||
|
)
|
||||||
|
for index, detection in enumerate(detections)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"FROZEN_YOLOX_MODEL_ID",
|
||||||
|
"FROZEN_YOLOX_PREPROCESS_ID",
|
||||||
|
"FROZEN_YOLOX_PROVIDER_ID",
|
||||||
|
"DetectorProviderError",
|
||||||
|
"DetectorProviderSnapshot",
|
||||||
|
"FrozenYoloxDetectorProvider",
|
||||||
|
"proposals_from_detections",
|
||||||
|
]
|
||||||
@@ -3,18 +3,21 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import hashlib
|
import hashlib
|
||||||
|
import importlib
|
||||||
import json
|
import json
|
||||||
import time
|
import time
|
||||||
from collections.abc import Callable, Iterator
|
from collections.abc import Callable, Iterator
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass, replace
|
||||||
from enum import StrEnum
|
from enum import StrEnum
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from threading import Event
|
from threading import Event
|
||||||
from typing import Final, Protocol
|
from typing import Any, Final, Protocol, cast
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
from numpy.typing import NDArray
|
||||||
|
|
||||||
from .baseline import (
|
from .baseline import (
|
||||||
|
BASELINE_CAMERA_STREAM_SHA256,
|
||||||
BASELINE_PROFILE_ID,
|
BASELINE_PROFILE_ID,
|
||||||
BASELINE_RECORDED_JOB_ID,
|
BASELINE_RECORDED_JOB_ID,
|
||||||
BASELINE_SESSION_ID,
|
BASELINE_SESSION_ID,
|
||||||
@@ -75,6 +78,10 @@ class LiveSourceAdapter(Protocol):
|
|||||||
def packets(self, stop_event: Event) -> Iterator[SourcePacket]: ...
|
def packets(self, stop_event: Event) -> Iterator[SourcePacket]: ...
|
||||||
|
|
||||||
|
|
||||||
|
class RecordedImageDecoder(Protocol):
|
||||||
|
def frames(self, stop_event: Event) -> Iterator[NDArray[np.uint8]]: ...
|
||||||
|
|
||||||
|
|
||||||
WaitFunction = Callable[[Event, float], bool]
|
WaitFunction = Callable[[Event, float], bool]
|
||||||
|
|
||||||
|
|
||||||
@@ -165,6 +172,72 @@ class RecordedRavnoves00Source:
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
class DecodedRecordedSource:
|
||||||
|
"""Attach decoded BGR frames without introducing detector logic into the source."""
|
||||||
|
|
||||||
|
provider_id: str = RECORDED_SOURCE_PROVIDER_ID
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
source: RecordedRavnoves00Source,
|
||||||
|
decoder: RecordedImageDecoder,
|
||||||
|
) -> None:
|
||||||
|
self.source = source
|
||||||
|
self.decoder = decoder
|
||||||
|
|
||||||
|
def packets(self, stop_event: Event) -> Iterator[SourcePacket]:
|
||||||
|
images = self.decoder.frames(stop_event)
|
||||||
|
for packet in self.source.packets(stop_event):
|
||||||
|
try:
|
||||||
|
image = next(images)
|
||||||
|
except StopIteration as exc:
|
||||||
|
raise RecordedSourceError(
|
||||||
|
"decoded image stream ended before source timeline"
|
||||||
|
) from exc
|
||||||
|
if image.shape != (600, 800, 3) or image.dtype != np.uint8:
|
||||||
|
raise RecordedSourceError("decoded image raster is incompatible")
|
||||||
|
yield replace(packet, image_payload=image)
|
||||||
|
if not stop_event.is_set():
|
||||||
|
try:
|
||||||
|
next(images)
|
||||||
|
except StopIteration:
|
||||||
|
return
|
||||||
|
raise RecordedSourceError("decoded image stream exceeds source timeline")
|
||||||
|
|
||||||
|
|
||||||
|
class PyAvRecordedImageDecoder:
|
||||||
|
"""Sequential full-video decoder used by the Worker 006 recorded source adapter."""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
path: Path,
|
||||||
|
*,
|
||||||
|
expected_sha256: str = BASELINE_CAMERA_STREAM_SHA256,
|
||||||
|
) -> None:
|
||||||
|
self.path = path.resolve(strict=True)
|
||||||
|
if self.path.is_symlink() or _file_sha256(self.path) != expected_sha256:
|
||||||
|
raise RecordedSourceError("recorded camera video identity changed")
|
||||||
|
|
||||||
|
def frames(self, stop_event: Event) -> Iterator[NDArray[np.uint8]]:
|
||||||
|
try:
|
||||||
|
av: Any = importlib.import_module("av")
|
||||||
|
except ModuleNotFoundError as exc:
|
||||||
|
raise RecordedSourceError("PyAV decoder is unavailable") from exc
|
||||||
|
container: Any = av.open(str(self.path))
|
||||||
|
try:
|
||||||
|
streams = container.streams.video
|
||||||
|
if len(streams) != 1:
|
||||||
|
raise RecordedSourceError("recorded camera video stream count changed")
|
||||||
|
for decoded in container.decode(streams[0]):
|
||||||
|
if stop_event.is_set():
|
||||||
|
return
|
||||||
|
image = cast(NDArray[np.uint8], decoded.to_ndarray(format="bgr24"))
|
||||||
|
yield np.asarray(image, dtype=np.uint8)
|
||||||
|
finally:
|
||||||
|
container.close()
|
||||||
|
|
||||||
|
|
||||||
def _packet(
|
def _packet(
|
||||||
frame_index: int,
|
frame_index: int,
|
||||||
camera: dict[str, object],
|
camera: dict[str, object],
|
||||||
@@ -316,8 +389,11 @@ def _event_wait(stop_event: Event, timeout_seconds: float) -> bool:
|
|||||||
__all__ = [
|
__all__ = [
|
||||||
"BASELINE_PROFILE_ID",
|
"BASELINE_PROFILE_ID",
|
||||||
"LiveSourceAdapter",
|
"LiveSourceAdapter",
|
||||||
|
"DecodedRecordedSource",
|
||||||
|
"PyAvRecordedImageDecoder",
|
||||||
"RECORDED_SOURCE_PROVIDER_ID",
|
"RECORDED_SOURCE_PROVIDER_ID",
|
||||||
"RecordedFrameReference",
|
"RecordedFrameReference",
|
||||||
|
"RecordedImageDecoder",
|
||||||
"RecordedRavnoves00Source",
|
"RecordedRavnoves00Source",
|
||||||
"RecordedSourceError",
|
"RecordedSourceError",
|
||||||
"ReplayPacing",
|
"ReplayPacing",
|
||||||
|
|||||||
@@ -84,16 +84,19 @@ def test_new_perception_boundary_has_no_experiment_specific_imports() -> None:
|
|||||||
assert violations == {}
|
assert violations == {}
|
||||||
|
|
||||||
|
|
||||||
def test_reference_graph_imports_only_an_admitted_compute_primitive() -> None:
|
def test_product_perception_imports_only_admitted_compute_primitives() -> None:
|
||||||
inventory = validate_reuse_inventory(REUSE_PATH)
|
inventory = validate_reuse_inventory(REUSE_PATH)
|
||||||
admitted = {
|
admitted = {
|
||||||
item["module"]
|
item["module"]
|
||||||
for item in inventory["reusable_primitives"]
|
for item in inventory["reusable_primitives"]
|
||||||
if isinstance(item, dict) and isinstance(item.get("module"), str)
|
if isinstance(item, dict) and isinstance(item.get("module"), str)
|
||||||
}
|
}
|
||||||
|
violations: dict[str, set[str]] = {}
|
||||||
|
for path in PERCEPTION_ROOT.glob("*.py"):
|
||||||
compute_imports = {
|
compute_imports = {
|
||||||
module
|
module for module in _imports(path) if module.startswith("k1link.compute")
|
||||||
for module in _imports(PERCEPTION_ROOT / "graph.py")
|
|
||||||
if module.startswith("k1link.compute")
|
|
||||||
}
|
}
|
||||||
assert compute_imports <= admitted
|
unadmitted = compute_imports - admitted
|
||||||
|
if unadmitted:
|
||||||
|
violations[path.name] = unadmitted
|
||||||
|
assert violations == {}
|
||||||
|
|||||||
@@ -50,6 +50,7 @@ from k1link.perception.providers import (
|
|||||||
SourcePacket,
|
SourcePacket,
|
||||||
)
|
)
|
||||||
from k1link.perception.recorded_source import (
|
from k1link.perception.recorded_source import (
|
||||||
|
DecodedRecordedSource,
|
||||||
RecordedRavnoves00Source,
|
RecordedRavnoves00Source,
|
||||||
RecordedSourceError,
|
RecordedSourceError,
|
||||||
ReplayPacing,
|
ReplayPacing,
|
||||||
@@ -529,6 +530,29 @@ def test_recorded_source_rejects_timeline_mismatch(tmp_path: Path) -> None:
|
|||||||
list(source.packets(Event()))
|
list(source.packets(Event()))
|
||||||
|
|
||||||
|
|
||||||
|
def test_decoded_recorded_source_attaches_images_without_detector_logic(tmp_path: Path) -> None:
|
||||||
|
camera_path, timeline_path = _write_recorded_fixture(tmp_path)
|
||||||
|
source = RecordedRavnoves00Source(
|
||||||
|
camera_index_path=camera_path,
|
||||||
|
source_pack_path=timeline_path,
|
||||||
|
expected_frame_count=2,
|
||||||
|
expected_source_pack_sha256=None,
|
||||||
|
)
|
||||||
|
|
||||||
|
class Decoder:
|
||||||
|
def frames(self, stop_event: Event) -> Iterator[np.ndarray]:
|
||||||
|
for value in (3, 7):
|
||||||
|
if stop_event.is_set():
|
||||||
|
return
|
||||||
|
yield np.full((600, 800, 3), value, dtype=np.uint8)
|
||||||
|
|
||||||
|
packets = list(DecodedRecordedSource(source=source, decoder=Decoder()).packets(Event()))
|
||||||
|
assert len(packets) == 2
|
||||||
|
assert isinstance(packets[0].image_payload, np.ndarray)
|
||||||
|
assert int(packets[0].image_payload[0, 0, 0]) == 3
|
||||||
|
assert int(packets[1].image_payload[0, 0, 0]) == 7
|
||||||
|
|
||||||
|
|
||||||
def test_camera_only_path_never_invents_metric_occupancy_or_free_space() -> None:
|
def test_camera_only_path_never_invents_metric_occupancy_or_free_space() -> None:
|
||||||
result = _graph(_Source((_packet(0, lidar=False),))).run()
|
result = _graph(_Source((_packet(0, lidar=False),))).run()
|
||||||
delivery = result.deliveries[0]
|
delivery = result.deliveries[0]
|
||||||
|
|||||||
@@ -0,0 +1,214 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
from pathlib import Path
|
||||||
|
from threading import Event
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import pytest
|
||||||
|
from numpy.typing import NDArray
|
||||||
|
|
||||||
|
from k1link.compute.yolox_object_detector import (
|
||||||
|
FrozenYoloxConfig,
|
||||||
|
YoloxDetection,
|
||||||
|
YoloxDetectorError,
|
||||||
|
postprocess_yolox,
|
||||||
|
preprocess_raw_kb4,
|
||||||
|
)
|
||||||
|
from k1link.perception.contracts import (
|
||||||
|
ClockBasis,
|
||||||
|
ModalityOutcome,
|
||||||
|
ModalityStatus,
|
||||||
|
SourceEnvelope,
|
||||||
|
TimestampBundle,
|
||||||
|
)
|
||||||
|
from k1link.perception.detector import (
|
||||||
|
FROZEN_YOLOX_PROVIDER_ID,
|
||||||
|
DetectorProviderError,
|
||||||
|
FrozenYoloxDetectorProvider,
|
||||||
|
proposals_from_detections,
|
||||||
|
)
|
||||||
|
from k1link.perception.providers import SourcePacket
|
||||||
|
|
||||||
|
REPOSITORY_ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
E46J_FRAMES = (
|
||||||
|
REPOSITORY_ROOT
|
||||||
|
/ ".runtime/compute-experiments/e46j/results"
|
||||||
|
/ "e46j-raw-fisheye-realtime-7119ce4344438eaa0e748db65aa044e9f9f4a0a226e5eea037a7180d0bc7ace7"
|
||||||
|
/ "frames.jsonl"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _status() -> ModalityStatus:
|
||||||
|
return ModalityStatus(True, ModalityOutcome.AVAILABLE, "test-available")
|
||||||
|
|
||||||
|
|
||||||
|
def _packet(sequence: int, image: object) -> SourcePacket:
|
||||||
|
return SourcePacket(
|
||||||
|
envelope=SourceEnvelope(
|
||||||
|
source_id="RAVNOVES00",
|
||||||
|
session_id="20260720T065719Z_viewer_live",
|
||||||
|
frame_id=f"frame-{sequence:06d}",
|
||||||
|
sequence=sequence,
|
||||||
|
timestamps=TimestampBundle(
|
||||||
|
utc_ns=1_000 + sequence,
|
||||||
|
monotonic_ns=2_000 + sequence,
|
||||||
|
source_ns=3_000 + sequence,
|
||||||
|
clock_basis=ClockBasis.RECORDED_HOST,
|
||||||
|
),
|
||||||
|
source_age_ns=0,
|
||||||
|
binding_reason="test-recorded-source",
|
||||||
|
calibration_id="camera-1-kb4-test",
|
||||||
|
representation_id="registered-map-increment-v1",
|
||||||
|
image=_status(),
|
||||||
|
registered_point_increment=_status(),
|
||||||
|
pose=_status(),
|
||||||
|
),
|
||||||
|
image_payload=image,
|
||||||
|
registered_point_increment_payload=("points", sequence),
|
||||||
|
pose_payload=("pose", sequence),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class _Resizer:
|
||||||
|
def resize(
|
||||||
|
self,
|
||||||
|
image: NDArray[np.uint8],
|
||||||
|
width: int,
|
||||||
|
height: int,
|
||||||
|
) -> NDArray[np.uint8]:
|
||||||
|
assert image.shape == (600, 800, 3)
|
||||||
|
return np.zeros((height, width, 3), dtype=np.uint8)
|
||||||
|
|
||||||
|
|
||||||
|
class _Backend:
|
||||||
|
def __init__(self, output: NDArray[np.float32]) -> None:
|
||||||
|
self.output = output
|
||||||
|
self.calls = 0
|
||||||
|
|
||||||
|
def infer(self, tensor: NDArray[np.float32]) -> NDArray[np.float32]:
|
||||||
|
assert tensor.shape == (1, 3, 640, 640)
|
||||||
|
self.calls += 1
|
||||||
|
return self.output
|
||||||
|
|
||||||
|
|
||||||
|
def _one_person_output() -> NDArray[np.float32]:
|
||||||
|
output = np.zeros((1, 8400, 85), dtype=np.float32)
|
||||||
|
output[0, 0, :4] = [40.0, 30.0, math.log(10.0), math.log(10.0)]
|
||||||
|
output[0, 0, 4] = 0.9
|
||||||
|
output[0, 0, 5] = 0.9
|
||||||
|
return output
|
||||||
|
|
||||||
|
|
||||||
|
def test_frozen_preprocess_and_postprocess_match_the_e46j_contract() -> None:
|
||||||
|
image = np.full((600, 800, 3), 7, dtype=np.uint8)
|
||||||
|
mask = np.ones((600, 800), dtype=np.bool_)
|
||||||
|
tensor = preprocess_raw_kb4(image, mask, resizer=_Resizer())
|
||||||
|
result = postprocess_yolox(_one_person_output(), mask)
|
||||||
|
|
||||||
|
assert tensor.shape == (1, 3, 640, 640)
|
||||||
|
assert np.all(tensor[:, :, 480:, :] == 114)
|
||||||
|
assert result.rejected == ()
|
||||||
|
assert len(result.detections) == 1
|
||||||
|
assert result.detections[0].label == "person"
|
||||||
|
assert result.detections[0].score == pytest.approx(0.81)
|
||||||
|
assert result.detections[0].bbox_xyxy == pytest.approx((350.0, 250.0, 450.0, 350.0))
|
||||||
|
|
||||||
|
|
||||||
|
def test_provider_emits_class_optional_product_proposals_and_metrics() -> None:
|
||||||
|
backend = _Backend(_one_person_output())
|
||||||
|
provider = FrozenYoloxDetectorProvider(
|
||||||
|
mask=np.ones((600, 800), dtype=np.bool_),
|
||||||
|
backend=backend,
|
||||||
|
resizer=_Resizer(),
|
||||||
|
clock_ns=iter((10, 20)).__next__,
|
||||||
|
)
|
||||||
|
packet = _packet(7, np.zeros((600, 800, 3), dtype=np.uint8))
|
||||||
|
|
||||||
|
proposals = provider.detect(packet)
|
||||||
|
|
||||||
|
assert backend.calls == 1
|
||||||
|
assert len(proposals) == 1
|
||||||
|
assert proposals[0].proposal_id == "proposal-7-0"
|
||||||
|
assert proposals[0].provider_id == FROZEN_YOLOX_PROVIDER_ID
|
||||||
|
assert proposals[0].semantic_hint == "person"
|
||||||
|
assert proposals[0].provider_tracklet is None
|
||||||
|
assert provider.snapshot().proposal_count == 1
|
||||||
|
assert provider.snapshot().core_duration_ns == 10
|
||||||
|
|
||||||
|
|
||||||
|
def test_provider_accounts_zero_pathological_and_failed_frames() -> None:
|
||||||
|
mask = np.ones((600, 800), dtype=np.bool_)
|
||||||
|
zero = FrozenYoloxDetectorProvider(
|
||||||
|
mask=mask,
|
||||||
|
backend=_Backend(np.zeros((1, 8400, 85), dtype=np.float32)),
|
||||||
|
resizer=_Resizer(),
|
||||||
|
)
|
||||||
|
assert zero.detect(_packet(0, np.zeros((600, 800, 3), dtype=np.uint8))) == ()
|
||||||
|
assert zero.snapshot().zero_proposal_frames == 1
|
||||||
|
|
||||||
|
pathological_output = _one_person_output()
|
||||||
|
pathological_output[0, 0, 2:4] = 1000.0
|
||||||
|
pathological = FrozenYoloxDetectorProvider(
|
||||||
|
mask=mask,
|
||||||
|
backend=_Backend(pathological_output),
|
||||||
|
resizer=_Resizer(),
|
||||||
|
)
|
||||||
|
assert pathological.detect(_packet(1, np.zeros((600, 800, 3), dtype=np.uint8))) == ()
|
||||||
|
assert dict(pathological.snapshot().rejected)["nonfinite"] == 1
|
||||||
|
|
||||||
|
failed = FrozenYoloxDetectorProvider(
|
||||||
|
mask=mask,
|
||||||
|
backend=_Backend(_one_person_output()),
|
||||||
|
resizer=_Resizer(),
|
||||||
|
)
|
||||||
|
with pytest.raises(DetectorProviderError, match="decoded BGR"):
|
||||||
|
failed.detect(_packet(2, "opaque-reference"))
|
||||||
|
assert failed.snapshot().failed_frames == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_frozen_profile_rejects_in_place_threshold_tuning() -> None:
|
||||||
|
with pytest.raises(YoloxDetectorError, match="cannot be tuned"):
|
||||||
|
FrozenYoloxConfig(minimum_score=0.51)
|
||||||
|
|
||||||
|
|
||||||
|
def test_all_4489_accepted_e46j_frames_map_to_product_contract_without_class_routing() -> None:
|
||||||
|
assert E46J_FRAMES.is_file()
|
||||||
|
image = np.zeros((600, 800, 3), dtype=np.uint8)
|
||||||
|
frame_count = 0
|
||||||
|
proposal_count = 0
|
||||||
|
zero_frames = 0
|
||||||
|
for line in E46J_FRAMES.read_text("utf-8").splitlines():
|
||||||
|
row = json.loads(line)
|
||||||
|
detections = tuple(
|
||||||
|
YoloxDetection(
|
||||||
|
class_id=item["class_id"],
|
||||||
|
label=item["label"],
|
||||||
|
score=item["score"],
|
||||||
|
bbox_xyxy=tuple(item["bbox_xyxy"]),
|
||||||
|
valid_fov_fraction=item["valid_fov_fraction"],
|
||||||
|
)
|
||||||
|
for item in row["detections"]
|
||||||
|
)
|
||||||
|
proposals = proposals_from_detections(_packet(frame_count, image), detections)
|
||||||
|
assert all(proposal.source_id == "RAVNOVES00" for proposal in proposals)
|
||||||
|
assert all(proposal.provider_tracklet is None for proposal in proposals)
|
||||||
|
assert len({proposal.proposal_id for proposal in proposals}) == len(proposals)
|
||||||
|
proposal_count += len(proposals)
|
||||||
|
zero_frames += not proposals
|
||||||
|
frame_count += 1
|
||||||
|
|
||||||
|
assert frame_count == 4489
|
||||||
|
assert proposal_count == 15499
|
||||||
|
assert zero_frames == 181
|
||||||
|
|
||||||
|
|
||||||
|
def test_source_packet_requires_a_decoded_image_before_real_provider_execution() -> None:
|
||||||
|
provider = FrozenYoloxDetectorProvider(
|
||||||
|
mask=np.ones((600, 800), dtype=np.bool_),
|
||||||
|
backend=_Backend(_one_person_output()),
|
||||||
|
resizer=_Resizer(),
|
||||||
|
)
|
||||||
|
with pytest.raises(DetectorProviderError, match="decoded BGR"):
|
||||||
|
provider.detect(_packet(0, Event()))
|
||||||
Reference in New Issue
Block a user