feat(vision): vehicle stage, phase A — YOLOX-S (Apache-2.0 ONNX) beside the plate recognizer

Fills /analyze vehicle.body_type + confidence (car / motorcycle / bus / truck from COCO,
mapped to the shared vocabulary) for the Car Wash desk's category suggestion
(venue-modules.md §Vehicle category from vision). Advisory: the operator decides, a
confident downgrade is flagged, nothing is gated on it.

- vision_service/vehicle.py: pure numpy/cv2 letterbox (pad 114, raw BGR), stride-grid
  decode, class-agnostic NMS, one vehicle per frame (the box holding the plate's centre,
  else the largest); YoloxVehicleDetector on onnxruntime CPU, 2 intra-op threads.
- recognizer.py: WithVehicle composes the stage over any plate recognizer (stub included);
  a failing stage yields vehicle=null + a "vehicle: …" note in /health.detail — never
  costs the plate read. model_version reads "<plate>+yolox:yolox_s.onnx@640".
- settings: VISION_VEHICLE_MODEL_PATH (unset = off), _INPUT_SIZE (640), _MIN_CONFIDENCE
  (0.4, the detector's floor; the flag threshold is site config).
- Dockerfile bakes yolox_s.onnx (best-effort curl at build; no network → stage off) and
  sets the path; compose forwards it (empty = off); .env.example documents it.
- Measured on four real dev entry frames (DS-2CD1047G3H, 2560×1440): car at 0.83–0.88 in
  ~240–330 ms; empty lane with a person → none.
- tests/test_vehicle.py: decode/NMS/pick/letterbox on synthetic tensors, the composition,
  and a missing-model /health. Wiki: opencv-anpr-service, venue-modules, log.

Claude-Session: https://claude.ai/code/session_01FWncR69HgGPuei1dLrW3cU
This commit is contained in:
2026-09-06 19:53:11 +02:00
parent 5e1395db18
commit 20a3cb3e80
10 changed files with 521 additions and 15 deletions
+53 -3
View File
@@ -19,6 +19,7 @@ from typing import Protocol
from .schemas import AnalyzeResponse, BBox, PlateResult
from .settings import Settings
from .vehicle import VehicleDetector, YoloxVehicleDetector
class Recognizer(Protocol):
@@ -165,10 +166,59 @@ class FastAlprRecognizer:
)
class WithVehicle:
"""Composition: any plate recognizer + the vehicle stage. Runs the plate stage first
(its box picks WHICH vehicle), then fills `vehicle`. A failing vehicle stage is
logged into `error` and yields null — it must never cost the plate read."""
def __init__(self, inner: Recognizer, detector: VehicleDetector) -> None:
self._inner = inner
self._detector = detector
self.vehicle_error: str | None = None
@property
def model_version(self) -> str:
return f"{self._inner.model_version}+{self._detector.model_version}"
@property
def ready(self) -> bool:
return bool(self._inner.ready)
@property
def error(self) -> str | None:
inner = getattr(self._inner, "error", None)
det = getattr(self._detector, "error", None) or self.vehicle_error
parts = [p for p in (inner, f"vehicle: {det}" if det else None) if p]
return "; ".join(parts) if parts else None
def analyze(self, image_bytes: bytes) -> AnalyzeResponse:
started = time.perf_counter()
res = self._inner.analyze(image_bytes)
try:
vehicle = self._detector.detect(image_bytes, res.plate.bbox if res.plate else None)
except Exception as exc: # noqa: BLE001 - advisory stage, never fatal
self.vehicle_error = f"{type(exc).__name__}: {exc}"
vehicle = None
took_ms = (time.perf_counter() - started) * 1000.0
return res.model_copy(
update={"vehicle": vehicle, "model_version": self.model_version, "took_ms": took_ms}
)
def build_recognizer(settings: Settings) -> Recognizer:
"""Factory: pick the recognizer from settings. Falls back to the stub if the real
one can't load, so the service always comes up (with ready=False surfaced)."""
one can't load, so the service always comes up (with ready=False surfaced). The
vehicle stage wraps whichever recognizer runs when a model path is configured."""
rec: Recognizer
if settings.recognizer == "fast_alpr":
rec = FastAlprRecognizer(settings)
return rec
return StubRecognizer(settings)
else:
rec = StubRecognizer(settings)
if settings.vehicle_model_path:
detector = YoloxVehicleDetector(
settings.vehicle_model_path,
input_size=settings.vehicle_input_size,
min_confidence=settings.vehicle_min_confidence,
)
return WithVehicle(rec, detector)
return rec
+10
View File
@@ -32,6 +32,16 @@ class Settings(BaseSettings):
# Node side can fall back to the ticket path rather than trust it.
min_confidence: float = 0.5
# Vehicle stage (phase A — venue-modules.md §Vehicle category from vision): a YOLOX
# ONNX graph (Apache-2.0) run beside the plate recognizer. Unset = stage off (the
# response's `vehicle` stays null). Bake the file into the image (models/), never a
# path an operator can write (vision-service-hardening.md).
vehicle_model_path: str | None = None
vehicle_input_size: int = 640
# Detection score floor for a vehicle box to count at all (the Node side applies the
# site's own, stricter threshold before it FLAGS anything).
vehicle_min_confidence: float = 0.4
def get_settings() -> Settings:
return Settings()
+247
View File
@@ -0,0 +1,247 @@
"""Vehicle stage: a COCO object detector beside the plate recognizer (Job 2, phase A).
Answers "what KIND of vehicle is in this entry frame?" for the Car Wash desk's category
suggestion (wiki/decisions/venue-modules.md §Vehicle category from vision). ADVISORY by
design: the Node server records it next to the plate, the desk pre-selects the site
category it maps to, the operator decides, a confident downgrade is flagged. Nothing is
ever gated on it, so a wrong or missing detection costs nothing but a suggestion.
Model: YOLOX (Megvii, Apache-2.0) as an ONNX graph on the ONNX Runtime the plate stage
already uses — the licence rule that keeps Ultralytics (AGPL) out. COCO's vehicle classes
are car / motorcycle / bus / truck: enough to tell a van or a truck from a car, NOT enough
for SUV vs sedan — that is phase B (a body-type classifier on the pilot's own frames).
The detector's vehicle box is also the crop phase B will classify.
Pure numpy/cv2 pre/post-processing, no torch: letterbox to the model's square input
(pad 114, no normalisation — YOLOX's exported graphs take raw 0–255 BGR), decode the
stride grids, class-agnostic NMS, map COCO ids to the shared vocabulary, pick ONE
vehicle: the one whose box holds the plate (when a plate was read), else the largest.
"""
from __future__ import annotations
import time
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Protocol
from .schemas import BBox, VehicleResult
# COCO-80 class index → the shared VEHICLE_CLASSES vocabulary (packages/shared).
COCO_VEHICLE_CLASSES: dict[int, str] = {2: "car", 3: "motorcycle", 5: "bus", 7: "truck"}
# YOLOX feature strides; grids are input/stride per level (8400 anchors at 640).
_STRIDES = (8, 16, 32)
@dataclass(frozen=True)
class Detection:
body_type: str
confidence: float
x1: float
y1: float
x2: float
y2: float
@property
def area(self) -> float:
return max(0.0, self.x2 - self.x1) * max(0.0, self.y2 - self.y1)
def contains(self, x: float, y: float) -> bool:
return self.x1 <= x <= self.x2 and self.y1 <= y <= self.y2
class VehicleDetector(Protocol):
"""What the recognizer composition needs: frame bytes (+ the plate box) → a class."""
@property
def model_version(self) -> str: ...
def detect(self, image_bytes: bytes, plate: BBox | None) -> VehicleResult | None: ...
# ----------------------------------------------------------------------------------
# Pre/post-processing (pure functions — unit-tested on synthetic tensors)
# ----------------------------------------------------------------------------------
def letterbox(frame: Any, size: int) -> tuple[Any, float]:
"""Resize keeping aspect, pad bottom/right with 114 to size×size. Returns the CHW
float32 tensor (batch dim added) and the scale to map boxes back."""
import cv2
import numpy as np
h, w = frame.shape[:2]
r = min(size / h, size / w)
nh, nw = int(round(h * r)), int(round(w * r))
resized = cv2.resize(frame, (nw, nh), interpolation=cv2.INTER_LINEAR)
padded = np.full((size, size, 3), 114, dtype=np.uint8)
padded[:nh, :nw] = resized
tensor = padded.transpose(2, 0, 1)[None].astype(np.float32)
return np.ascontiguousarray(tensor), r
def decode(raw: Any, size: int) -> Any:
"""YOLOX raw output [N, 5+classes] (batch squeezed) → same shape with xywh decoded
into pixel units of the letterboxed input. Rows are ordered stride 8, 16, 32."""
import numpy as np
out = raw.astype(np.float32).copy()
grids = []
strides = []
for s in _STRIDES:
n = size // s
ys, xs = np.meshgrid(np.arange(n), np.arange(n), indexing="ij")
grids.append(np.stack((xs, ys), axis=-1).reshape(-1, 2))
strides.append(np.full((n * n, 1), s, dtype=np.float32))
grid = np.concatenate(grids, axis=0).astype(np.float32)
stride = np.concatenate(strides, axis=0)
if out.shape[0] != grid.shape[0]:
raise ValueError(f"unexpected output rows {out.shape[0]} for input {size} (want {grid.shape[0]})")
out[:, :2] = (out[:, :2] + grid) * stride
out[:, 2:4] = np.exp(out[:, 2:4]) * stride
return out
def nms(boxes: Any, scores: Any, iou_threshold: float) -> list[int]:
"""Greedy class-agnostic non-max suppression over xyxy boxes; returns kept indices."""
import numpy as np
if len(boxes) == 0:
return []
order = scores.argsort()[::-1]
x1, y1, x2, y2 = boxes[:, 0], boxes[:, 1], boxes[:, 2], boxes[:, 3]
areas = np.clip(x2 - x1, 0, None) * np.clip(y2 - y1, 0, None)
keep: list[int] = []
while order.size > 0:
i = int(order[0])
keep.append(i)
if order.size == 1:
break
rest = order[1:]
xx1 = np.maximum(x1[i], x1[rest])
yy1 = np.maximum(y1[i], y1[rest])
xx2 = np.minimum(x2[i], x2[rest])
yy2 = np.minimum(y2[i], y2[rest])
inter = np.clip(xx2 - xx1, 0, None) * np.clip(yy2 - yy1, 0, None)
iou = inter / (areas[i] + areas[rest] - inter + 1e-9)
order = rest[iou <= iou_threshold]
return keep
def vehicles_from_output(
raw: Any, size: int, scale: float, min_confidence: float, iou_threshold: float = 0.45
) -> list[Detection]:
"""Full post-processing: decode → vehicle classes only → confidence floor → NMS →
boxes in ORIGINAL frame pixels."""
import numpy as np
dec = decode(raw, size)
cls_scores = dec[:, 5:]
cls_idx = cls_scores.argmax(axis=1)
score = dec[:, 4] * cls_scores[np.arange(len(dec)), cls_idx]
wanted = np.isin(cls_idx, list(COCO_VEHICLE_CLASSES)) & (score >= min_confidence)
if not wanted.any():
return []
d = dec[wanted]
s = score[wanted]
c = cls_idx[wanted]
boxes = np.stack(
(d[:, 0] - d[:, 2] / 2, d[:, 1] - d[:, 3] / 2, d[:, 0] + d[:, 2] / 2, d[:, 1] + d[:, 3] / 2), axis=1
)
keep = nms(boxes, s, iou_threshold)
out: list[Detection] = []
for i in keep:
b = boxes[i] / scale
out.append(
Detection(
body_type=COCO_VEHICLE_CLASSES[int(c[i])],
confidence=float(s[i]),
x1=float(b[0]),
y1=float(b[1]),
x2=float(b[2]),
y2=float(b[3]),
)
)
return out
def pick_vehicle(detections: list[Detection], plate: BBox | None) -> Detection | None:
"""ONE vehicle per frame: the box holding the plate's centre (the car that was read —
a lane frame can show the car behind too), else the largest box (nearest the camera)."""
if not detections:
return None
if plate is not None:
cx = (plate.x1 + plate.x2) / 2
cy = (plate.y1 + plate.y2) / 2
holders = [d for d in detections if d.contains(cx, cy)]
if holders:
return min(holders, key=lambda d: d.area) # the tightest box around the plate
return max(detections, key=lambda d: d.area)
# ----------------------------------------------------------------------------------
# The ONNX Runtime detector
# ----------------------------------------------------------------------------------
class YoloxVehicleDetector:
"""YOLOX ONNX on onnxruntime (CPU). Loads once; a load failure is surfaced through
`error` and the stage simply yields no vehicle (never breaks the plate path)."""
def __init__(self, model_path: str, input_size: int = 640, min_confidence: float = 0.4) -> None:
self._path = Path(model_path)
self._size = input_size
self._min_confidence = min_confidence
self._session = None
self._input_name = "images"
self._error: str | None = None
try:
import onnxruntime as ort
opts = ort.SessionOptions()
opts.intra_op_num_threads = 2 # one frame per entry; leave cores to the lane
self._session = ort.InferenceSession(
str(self._path), sess_options=opts, providers=["CPUExecutionProvider"]
)
self._input_name = self._session.get_inputs()[0].name
except Exception as exc: # noqa: BLE001 - not-ready, never fatal
self._error = f"{type(exc).__name__}: {exc}"
@property
def model_version(self) -> str:
return f"yolox:{self._path.name}@{self._size}"
@property
def ready(self) -> bool:
return self._session is not None
@property
def error(self) -> str | None:
return self._error
def detect(self, image_bytes: bytes, plate: BBox | None) -> VehicleResult | None:
if self._session is None:
return None
import cv2
import numpy as np
frame = cv2.imdecode(np.frombuffer(image_bytes, dtype=np.uint8), cv2.IMREAD_COLOR)
if frame is None:
return None
tensor, scale = letterbox(frame, self._size)
raw = self._session.run(None, {self._input_name: tensor})[0][0]
found = vehicles_from_output(raw, self._size, scale, self._min_confidence)
best = pick_vehicle(found, plate)
if best is None:
return None
return VehicleResult(body_type=best.body_type, confidence=round(best.confidence, 4))
def time_detect(
detector: VehicleDetector, image_bytes: bytes, plate: BBox | None
) -> tuple[VehicleResult | None, float]:
"""detect() with wall time in ms (for logs/benchmarks)."""
started = time.perf_counter()
result = detector.detect(image_bytes, plate)
return result, (time.perf_counter() - started) * 1000.0