Files
parking_solution/apps/vision/vision_service/vehicle.py
T
julian e67f0ccef0 feat(carwash): review outbox, booth side — plate-blurred vehicle crop + the operator's choice, queued for a trusted remote reviewer
The operator's category choice is a hypothesis, not truth (user, 2026-09-06): each wash
order with a vehicle read queues a package for a trusted reviewer over the private overlay
(Netbird); the verdict becomes the phase-B training label and the per-operator error rate.
wiki/concepts/vision-review-outbox.md.

- Boxes: the vision service returns the vehicle bbox; snapshot.ts stores the vehicle and
  plate boxes on the read as FRACTIONS of the analysed frame (the stored snapshot is a
  downscaled copy); vehicleForIdentity() returns them.
- carwash_review_outbox (migration 0031) + review-outbox.ts: crop = detector box + 8 %
  margin, ≤ 640 px, plate blurred in place from the plate box; payload carries a
  pseudonymous booth id and a keyed operator hash — no site name, no plate, no OSD, no
  bystanders; multipart POST with a per-booth bearer; 2xx → sent (image dropped);
  400/404/413/415/422 → abandoned; anything else → backoff 1 min·2^n capped 6 h; voided
  orders and items older than 14 days abandoned unsent. Nothing queued while unconfigured.
- Enqueue is fire-and-forget off the intake path in createOrder; the loop runs every
  CARWASH_REVIEW_INTERVAL_SEC (60) and stops on close.
- GET /api/carwash/review/status (site:read) + a "Remote review" line in Setup → Car wash.
- Env CARWASH_REVIEW_URL / _TOKEN / _BOOTH_ID (all three or off) documented in
  .env.example and forwarded by compose.
- Tests: review-outbox.test.ts (crop + blur on a synthetic frame, config/pseudonyms,
  queue/drain/backoff/abandon, through the app). Wiki: new concept page, index,
  venue-modules As built, log. The collector is not built.

Claude-Session: https://claude.ai/code/session_01FWncR69HgGPuei1dLrW3cU
2026-09-06 22:33:43 +02:00

252 lines
9.5 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Vehicle stage: a COCO object detector beside the plate recognizer (Job 2, phase A).
Answers "what KIND of vehicle is in this entry frame?" for the Car Wash desk's category
suggestion (wiki/decisions/venue-modules.md §Vehicle category from vision). ADVISORY by
design: the Node server records it next to the plate, the desk pre-selects the site
category it maps to, the operator decides, a confident downgrade is flagged. Nothing is
ever gated on it, so a wrong or missing detection costs nothing but a suggestion.
Model: YOLOX (Megvii, Apache-2.0) as an ONNX graph on the ONNX Runtime the plate stage
already uses — the licence rule that keeps Ultralytics (AGPL) out. COCO's vehicle classes
are car / motorcycle / bus / truck: enough to tell a van or a truck from a car, NOT enough
for SUV vs sedan — that is phase B (a body-type classifier on the pilot's own frames).
The detector's vehicle box is also the crop phase B will classify.
Pure numpy/cv2 pre/post-processing, no torch: letterbox to the model's square input
(pad 114, no normalisation — YOLOX's exported graphs take raw 0–255 BGR), decode the
stride grids, class-agnostic NMS, map COCO ids to the shared vocabulary, pick ONE
vehicle: the one whose box holds the plate (when a plate was read), else the largest.
"""
from __future__ import annotations
import time
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Protocol
from .schemas import BBox, VehicleResult
# COCO-80 class index → the shared VEHICLE_CLASSES vocabulary (packages/shared).
COCO_VEHICLE_CLASSES: dict[int, str] = {2: "car", 3: "motorcycle", 5: "bus", 7: "truck"}
# YOLOX feature strides; grids are input/stride per level (8400 anchors at 640).
_STRIDES = (8, 16, 32)
@dataclass(frozen=True)
class Detection:
body_type: str
confidence: float
x1: float
y1: float
x2: float
y2: float
@property
def area(self) -> float:
return max(0.0, self.x2 - self.x1) * max(0.0, self.y2 - self.y1)
def contains(self, x: float, y: float) -> bool:
return self.x1 <= x <= self.x2 and self.y1 <= y <= self.y2
class VehicleDetector(Protocol):
"""What the recognizer composition needs: frame bytes (+ the plate box) → a class."""
@property
def model_version(self) -> str: ...
def detect(self, image_bytes: bytes, plate: BBox | None) -> VehicleResult | None: ...
# ----------------------------------------------------------------------------------
# Pre/post-processing (pure functions — unit-tested on synthetic tensors)
# ----------------------------------------------------------------------------------
def letterbox(frame: Any, size: int) -> tuple[Any, float]:
"""Resize keeping aspect, pad bottom/right with 114 to size×size. Returns the CHW
float32 tensor (batch dim added) and the scale to map boxes back."""
import cv2
import numpy as np
h, w = frame.shape[:2]
r = min(size / h, size / w)
nh, nw = int(round(h * r)), int(round(w * r))
resized = cv2.resize(frame, (nw, nh), interpolation=cv2.INTER_LINEAR)
padded = np.full((size, size, 3), 114, dtype=np.uint8)
padded[:nh, :nw] = resized
tensor = padded.transpose(2, 0, 1)[None].astype(np.float32)
return np.ascontiguousarray(tensor), r
def decode(raw: Any, size: int) -> Any:
"""YOLOX raw output [N, 5+classes] (batch squeezed) → same shape with xywh decoded
into pixel units of the letterboxed input. Rows are ordered stride 8, 16, 32."""
import numpy as np
out = raw.astype(np.float32).copy()
grids = []
strides = []
for s in _STRIDES:
n = size // s
ys, xs = np.meshgrid(np.arange(n), np.arange(n), indexing="ij")
grids.append(np.stack((xs, ys), axis=-1).reshape(-1, 2))
strides.append(np.full((n * n, 1), s, dtype=np.float32))
grid = np.concatenate(grids, axis=0).astype(np.float32)
stride = np.concatenate(strides, axis=0)
if out.shape[0] != grid.shape[0]:
raise ValueError(f"unexpected output rows {out.shape[0]} for input {size} (want {grid.shape[0]})")
out[:, :2] = (out[:, :2] + grid) * stride
out[:, 2:4] = np.exp(out[:, 2:4]) * stride
return out
def nms(boxes: Any, scores: Any, iou_threshold: float) -> list[int]:
"""Greedy class-agnostic non-max suppression over xyxy boxes; returns kept indices."""
import numpy as np
if len(boxes) == 0:
return []
order = scores.argsort()[::-1]
x1, y1, x2, y2 = boxes[:, 0], boxes[:, 1], boxes[:, 2], boxes[:, 3]
areas = np.clip(x2 - x1, 0, None) * np.clip(y2 - y1, 0, None)
keep: list[int] = []
while order.size > 0:
i = int(order[0])
keep.append(i)
if order.size == 1:
break
rest = order[1:]
xx1 = np.maximum(x1[i], x1[rest])
yy1 = np.maximum(y1[i], y1[rest])
xx2 = np.minimum(x2[i], x2[rest])
yy2 = np.minimum(y2[i], y2[rest])
inter = np.clip(xx2 - xx1, 0, None) * np.clip(yy2 - yy1, 0, None)
iou = inter / (areas[i] + areas[rest] - inter + 1e-9)
order = rest[iou <= iou_threshold]
return keep
def vehicles_from_output(
raw: Any, size: int, scale: float, min_confidence: float, iou_threshold: float = 0.45
) -> list[Detection]:
"""Full post-processing: decode → vehicle classes only → confidence floor → NMS →
boxes in ORIGINAL frame pixels."""
import numpy as np
dec = decode(raw, size)
cls_scores = dec[:, 5:]
cls_idx = cls_scores.argmax(axis=1)
score = dec[:, 4] * cls_scores[np.arange(len(dec)), cls_idx]
wanted = np.isin(cls_idx, list(COCO_VEHICLE_CLASSES)) & (score >= min_confidence)
if not wanted.any():
return []
d = dec[wanted]
s = score[wanted]
c = cls_idx[wanted]
boxes = np.stack(
(d[:, 0] - d[:, 2] / 2, d[:, 1] - d[:, 3] / 2, d[:, 0] + d[:, 2] / 2, d[:, 1] + d[:, 3] / 2), axis=1
)
keep = nms(boxes, s, iou_threshold)
out: list[Detection] = []
for i in keep:
b = boxes[i] / scale
out.append(
Detection(
body_type=COCO_VEHICLE_CLASSES[int(c[i])],
confidence=float(s[i]),
x1=float(b[0]),
y1=float(b[1]),
x2=float(b[2]),
y2=float(b[3]),
)
)
return out
def pick_vehicle(detections: list[Detection], plate: BBox | None) -> Detection | None:
"""ONE vehicle per frame: the box holding the plate's centre (the car that was read —
a lane frame can show the car behind too), else the largest box (nearest the camera)."""
if not detections:
return None
if plate is not None:
cx = (plate.x1 + plate.x2) / 2
cy = (plate.y1 + plate.y2) / 2
holders = [d for d in detections if d.contains(cx, cy)]
if holders:
return min(holders, key=lambda d: d.area) # the tightest box around the plate
return max(detections, key=lambda d: d.area)
# ----------------------------------------------------------------------------------
# The ONNX Runtime detector
# ----------------------------------------------------------------------------------
class YoloxVehicleDetector:
"""YOLOX ONNX on onnxruntime (CPU). Loads once; a load failure is surfaced through
`error` and the stage simply yields no vehicle (never breaks the plate path)."""
def __init__(self, model_path: str, input_size: int = 640, min_confidence: float = 0.4) -> None:
self._path = Path(model_path)
self._size = input_size
self._min_confidence = min_confidence
self._session = None
self._input_name = "images"
self._error: str | None = None
try:
import onnxruntime as ort
opts = ort.SessionOptions()
opts.intra_op_num_threads = 2 # one frame per entry; leave cores to the lane
self._session = ort.InferenceSession(
str(self._path), sess_options=opts, providers=["CPUExecutionProvider"]
)
self._input_name = self._session.get_inputs()[0].name
except Exception as exc: # noqa: BLE001 - not-ready, never fatal
self._error = f"{type(exc).__name__}: {exc}"
@property
def model_version(self) -> str:
return f"yolox:{self._path.name}@{self._size}"
@property
def ready(self) -> bool:
return self._session is not None
@property
def error(self) -> str | None:
return self._error
def detect(self, image_bytes: bytes, plate: BBox | None) -> VehicleResult | None:
if self._session is None:
return None
import cv2
import numpy as np
frame = cv2.imdecode(np.frombuffer(image_bytes, dtype=np.uint8), cv2.IMREAD_COLOR)
if frame is None:
return None
tensor, scale = letterbox(frame, self._size)
raw = self._session.run(None, {self._input_name: tensor})[0][0]
found = vehicles_from_output(raw, self._size, scale, self._min_confidence)
best = pick_vehicle(found, plate)
if best is None:
return None
h, w = frame.shape[:2]
box = BBox(
x1=max(0, int(best.x1)), y1=max(0, int(best.y1)), x2=min(w, int(best.x2)), y2=min(h, int(best.y2))
)
return VehicleResult(body_type=best.body_type, confidence=round(best.confidence, 4), bbox=box)
def time_detect(
detector: VehicleDetector, image_bytes: bytes, plate: BBox | None
) -> tuple[VehicleResult | None, float]:
"""detect() with wall time in ms (for logs/benchmarks)."""
started = time.perf_counter()
result = detector.detect(image_bytes, plate)
return result, (time.perf_counter() - started) * 1000.0