Compare commits
12 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 29594f8bad | |||
| 4ff31557a8 | |||
| 3e77a4ad7c | |||
| 3f16925fe0 | |||
| f7a262ac9a | |||
| f9cb973fe9 | |||
| 1a0fe59488 | |||
| 8bfc29db2a | |||
| dbbb051ebd | |||
| 3e57af5abc | |||
| ef55d1c6a9 | |||
| 0411b71c2d |
@@ -1,6 +1,6 @@
|
||||
name: Build & push images
|
||||
|
||||
# Build the SERVER (API + SPA), COLLECTOR (wash review) and VISION (ANPR) container images and push them to the
|
||||
# Build the SERVER (API + SPA), COLLECTOR (wash review), VISION (ANPR) and TRAINER (phase-B job) container images and push them to the
|
||||
# house Gitea registry, tagged by BRANCH + short SHA (branch-aware: dev→:dev, stage→:stage,
|
||||
# main→:main). Separate from ci.yml (checks-only) and release.yml (tag-only desktop bundle).
|
||||
# Mirrors the house pattern (cf. trm/processor build.yml). See
|
||||
@@ -14,6 +14,7 @@ on:
|
||||
- 'apps/web/**'
|
||||
- 'apps/vision/**'
|
||||
- 'apps/collector/**'
|
||||
- 'apps/trainer/**'
|
||||
- 'packages/**'
|
||||
- 'package.json'
|
||||
- 'pnpm-lock.yaml'
|
||||
@@ -61,6 +62,11 @@ jobs:
|
||||
working-directory: apps/vision
|
||||
run: uv sync --frozen
|
||||
|
||||
- name: Sync trainer deps
|
||||
# Light core only — NOT the `train` extra (CPU torch, ~200 MB); the torch tests skip.
|
||||
working-directory: apps/trainer
|
||||
run: uv sync --frozen
|
||||
|
||||
# Don't publish a broken image — run the same checks as ci.yml first.
|
||||
- name: Build + lint + test (Turbo)
|
||||
run: pnpm turbo run build lint test
|
||||
@@ -119,12 +125,29 @@ jobs:
|
||||
context: apps/vision
|
||||
file: apps/vision/Dockerfile
|
||||
push: true
|
||||
# The phase-B body-type classifier is fetched from the Gitea generic package registry
|
||||
# at build when apps/vision/models/bodytype.version pins a version (empty = none). The
|
||||
# registry user's credentials double as the fetch auth (BuildKit secret, never a layer).
|
||||
secrets: |
|
||||
bodytype_auth=${{ secrets.REGISTRY_USERNAME }}:${{ secrets.REGISTRY_PASSWORD }}
|
||||
tags: |
|
||||
${{ env.REGISTRY }}/parking-vision:${{ steps.meta.outputs.branch }}
|
||||
${{ env.REGISTRY }}/parking-vision:${{ steps.meta.outputs.branch }}-${{ steps.meta.outputs.sha }}
|
||||
cache-from: type=registry,ref=${{ env.REGISTRY }}/parking-vision:buildcache
|
||||
cache-to: type=registry,ref=${{ env.REGISTRY }}/parking-vision:buildcache,mode=max
|
||||
|
||||
- name: Build & push TRAINER (phase-B job)
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
context: apps/trainer
|
||||
file: apps/trainer/Dockerfile
|
||||
push: true
|
||||
tags: |
|
||||
${{ env.REGISTRY }}/parking-trainer:${{ steps.meta.outputs.branch }}
|
||||
${{ env.REGISTRY }}/parking-trainer:${{ steps.meta.outputs.branch }}-${{ steps.meta.outputs.sha }}
|
||||
cache-from: type=registry,ref=${{ env.REGISTRY }}/parking-trainer:buildcache
|
||||
cache-to: type=registry,ref=${{ env.REGISTRY }}/parking-trainer:buildcache,mode=max
|
||||
|
||||
# Optional: trigger a Komodo stack redeploy (cf. trm/processor). Enable by setting the
|
||||
# KOMODO_* secrets; left guarded so it no-ops until the parking stack is wired.
|
||||
- name: Trigger Komodo redeploy
|
||||
|
||||
@@ -48,6 +48,11 @@ jobs:
|
||||
working-directory: apps/vision
|
||||
run: uv sync --frozen
|
||||
|
||||
- name: Sync trainer deps
|
||||
# Same rule: light core only, not the `train` extra (CPU torch); torch tests skip.
|
||||
working-directory: apps/trainer
|
||||
run: uv sync --frozen
|
||||
|
||||
- name: Build + lint (Turbo)
|
||||
# Covers tsc typecheck, vite build, i18n catalog type-parity (a missing sq/en
|
||||
# key fails the build), AND the vision service's ruff lint via uv.
|
||||
|
||||
@@ -16,7 +16,7 @@ const basic = "Basic " + Buffer.from(`${REVIEWER.user}:${REVIEWER.pass}`).toStri
|
||||
|
||||
beforeEach(async () => {
|
||||
dir = await mkdtemp(path.join(tmpdir(), "collector-"));
|
||||
app = await buildCollector({ host: "127.0.0.1", port: 0, dataDir: dir, boothTokens: TOKENS, reviewer: REVIEWER }, { dbFile: ":memory:" });
|
||||
app = await buildCollector({ host: "127.0.0.1", port: 0, dataDir: dir, boothTokens: TOKENS, reviewer: REVIEWER, trainerUrl: null }, { dbFile: ":memory:" });
|
||||
await app.ready();
|
||||
});
|
||||
afterEach(async () => {
|
||||
@@ -109,8 +109,8 @@ describe("review + export", () => {
|
||||
|
||||
const stats = (await app.inject({ method: "GET", url: "/api/stats", headers: { authorization: basic } })).json();
|
||||
expect(stats.booths).toEqual([
|
||||
{ booth: "booth-7", received: 2, pending: 0, reviewed: 2 },
|
||||
{ booth: "booth-9", received: 1, pending: 0, reviewed: 1 },
|
||||
{ booth: "booth-7", received: 2, pending: 0, reviewed: 2, entries: 0 },
|
||||
{ booth: "booth-9", received: 1, pending: 0, reviewed: 1, entries: 0 },
|
||||
]);
|
||||
expect(stats.operators).toEqual([
|
||||
{ booth: "booth-7", operatorRef: "ab12cd34ef56ab12", reviewed: 2, agree: 1, disagree: 1, unusable: 0 },
|
||||
@@ -120,9 +120,21 @@ describe("review + export", () => {
|
||||
const csv = await app.inject({ method: "GET", url: "/export/labels.csv", headers: { authorization: basic } });
|
||||
expect(csv.statusCode).toBe(200);
|
||||
const lines = csv.body.trim().split("\n");
|
||||
expect(lines[0]).toBe("item,booth,path,label,operator_category,operator_classes,vision_class,vision_confidence,downgraded,at,reviewed_at");
|
||||
expect(lines[0]).toBe("item,booth,kind,path,label,operator_category,operator_classes,vision_class,vision_confidence,downgraded,at,reviewed_at");
|
||||
expect(lines).toHaveLength(3); // header + 2 usable labels; the unusable one is left out
|
||||
expect(lines[1]).toContain('"item-1","booth-7","crops/booth-7/item-1.jpg","suv","Vetura","car|sedan|hatchback","suv"');
|
||||
expect(lines[1]).toContain('"item-1","booth-7","wash","crops/booth-7/item-1.jpg","suv","Vetura","car|sedan|hatchback","suv"');
|
||||
|
||||
// An ENTRY sample: no order, no operator — accepted, reviewable, in the export, and
|
||||
// never counted in any operator's agreement.
|
||||
const entry = await ingest({ v: 1, kind: "entry", booth: "booth-7", item: "entry-1", at: "2026-09-06T11:00:00.000Z", vision: { class: "car", confidence: 0.7 }, image: { width: 300, height: 180, plateBlurred: true } });
|
||||
expect(entry.statusCode).toBe(201);
|
||||
expect((await ingest({ v: 1, kind: "entry", booth: "booth-7", item: "entry-2", at: "x", vision: { class: "car", confidence: 0.7 }, image: { width: 1, height: 1, plateBlurred: true } })).statusCode).toBe(422);
|
||||
expect((await post("entry-1", "suv")).statusCode).toBe(200);
|
||||
const stats2 = (await app.inject({ method: "GET", url: "/api/stats", headers: { authorization: basic } })).json();
|
||||
expect(stats2.booths[0]).toEqual({ booth: "booth-7", received: 3, pending: 0, reviewed: 3, entries: 1 });
|
||||
expect(stats2.operators.find((o: { booth: string }) => o.booth === "booth-7")).toMatchObject({ reviewed: 2, agree: 1, disagree: 1 });
|
||||
const csv3 = (await app.inject({ method: "GET", url: "/export/labels.csv", headers: { authorization: basic } })).body;
|
||||
expect(csv3).toContain('"entry-1","booth-7","entry","crops/booth-7/entry-1.jpg","suv","","","car"');
|
||||
|
||||
// A booth-supplied name that looks like a spreadsheet formula is neutralised in the export.
|
||||
await ingest(meta({ item: "item-4", operatorCategory: { id: "x", name: "=HYPERLINK(\"http://evil\")", classes: ["car"] } }));
|
||||
|
||||
+82
-22
@@ -15,20 +15,25 @@ import { reviewPage } from "./review-page.js";
|
||||
// /review + /api/* the reviewer's screen (HTTP Basic, one login)
|
||||
// GET /export/labels.csv the training set: reviewed, usable rows (crops sit beside it on
|
||||
// the volume, so the trainer on this host reads them directly)
|
||||
// /api/training/* the Training section: a thin proxy to the trainer's job API on
|
||||
// the compose network (never published), behind the reviewer login
|
||||
// It deliberately has no fleet features and no path back into a booth.
|
||||
|
||||
/** The package's `meta` part, as the booth sends it (review-outbox.ts). */
|
||||
interface IngestMeta {
|
||||
v: number;
|
||||
/** "wash" (default when absent) = a desk decision; "entry" = a sampled entry read with
|
||||
* no order and no operator — crop + the camera's class only. */
|
||||
kind?: "wash" | "entry";
|
||||
booth: string;
|
||||
item: string;
|
||||
order: string;
|
||||
order?: string;
|
||||
at: string;
|
||||
operator: string;
|
||||
operatorCategory: { id: string; name: string; classes?: string[] };
|
||||
service: string;
|
||||
vision: { class: string; confidence: number; categoryId: string | null };
|
||||
downgraded: boolean;
|
||||
operator?: string;
|
||||
operatorCategory?: { id: string; name: string; classes?: string[] };
|
||||
service?: string;
|
||||
vision: { class: string; confidence: number; categoryId?: string | null };
|
||||
downgraded?: boolean;
|
||||
image: { width: number; height: number; plateBlurred: boolean };
|
||||
}
|
||||
|
||||
@@ -46,17 +51,21 @@ function checkMeta(m: unknown, booth: string): { ok: true; meta: IngestMeta } |
|
||||
if (x.v !== 1) return { ok: false, why: "unsupported meta version" };
|
||||
if (x.booth !== booth) return { ok: false, why: "meta.booth does not match the token's booth" };
|
||||
if (!str(x.item, 64) || !ID_RE.test(x.item as string)) return { ok: false, why: "bad item id" };
|
||||
if (!str(x.order, 64)) return { ok: false, why: "bad order ref" };
|
||||
if (!str(x.at, 40) || Number.isNaN(Date.parse(x.at as string))) return { ok: false, why: "bad timestamp" };
|
||||
if (!str(x.operator, 64)) return { ok: false, why: "bad operator ref" };
|
||||
const oc = x.operatorCategory as Record<string, unknown> | undefined;
|
||||
if (!oc || !str(oc.id, 64) || !str(oc.name, 120)) return { ok: false, why: "bad operatorCategory" };
|
||||
if (oc.classes !== undefined && (!Array.isArray(oc.classes) || !oc.classes.every(isVehicleClass))) return { ok: false, why: "bad operatorCategory.classes" };
|
||||
if (!str(x.service, 120)) return { ok: false, why: "bad service" };
|
||||
const kind = x.kind === undefined ? "wash" : x.kind;
|
||||
if (kind !== "wash" && kind !== "entry") return { ok: false, why: "bad kind" };
|
||||
const v = x.vision as Record<string, unknown> | undefined;
|
||||
if (!v || !isVehicleClass(v.class) || typeof v.confidence !== "number" || v.confidence < 0 || v.confidence > 1) return { ok: false, why: "bad vision read" };
|
||||
if (v.categoryId != null && !str(v.categoryId, 64)) return { ok: false, why: "bad vision.categoryId" };
|
||||
if (typeof x.downgraded !== "boolean") return { ok: false, why: "bad downgraded" };
|
||||
if (kind === "wash") {
|
||||
if (!str(x.order, 64)) return { ok: false, why: "bad order ref" };
|
||||
if (!str(x.operator, 64)) return { ok: false, why: "bad operator ref" };
|
||||
const oc = x.operatorCategory as Record<string, unknown> | undefined;
|
||||
if (!oc || !str(oc.id, 64) || !str(oc.name, 120)) return { ok: false, why: "bad operatorCategory" };
|
||||
if (oc.classes !== undefined && (!Array.isArray(oc.classes) || !oc.classes.every(isVehicleClass))) return { ok: false, why: "bad operatorCategory.classes" };
|
||||
if (!str(x.service, 120)) return { ok: false, why: "bad service" };
|
||||
if (typeof x.downgraded !== "boolean") return { ok: false, why: "bad downgraded" };
|
||||
}
|
||||
const im = x.image as Record<string, unknown> | undefined;
|
||||
if (!im || typeof im.width !== "number" || typeof im.height !== "number" || typeof im.plateBlurred !== "boolean") return { ok: false, why: "bad image meta" };
|
||||
return { ok: true, meta: x as unknown as IngestMeta };
|
||||
@@ -148,16 +157,18 @@ export async function buildCollector(cfg: CollectorConfig, opts: { dbFile?: stri
|
||||
const rel = path.posix.join("crops", booth, `${meta.item}.jpg`);
|
||||
await mkdir(path.join(cfg.dataDir, "crops", booth), { recursive: true });
|
||||
await writeFile(path.join(cfg.dataDir, rel), image);
|
||||
const kind = meta.kind ?? "wash";
|
||||
db.insert({
|
||||
id: meta.item,
|
||||
booth,
|
||||
orderRef: meta.order,
|
||||
kind,
|
||||
orderRef: meta.order ?? "",
|
||||
at: meta.at,
|
||||
operatorRef: meta.operator,
|
||||
operatorCategoryId: meta.operatorCategory.id,
|
||||
operatorCategoryName: meta.operatorCategory.name,
|
||||
operatorClasses: JSON.stringify(meta.operatorCategory.classes ?? []),
|
||||
service: meta.service,
|
||||
operatorRef: meta.operator ?? "",
|
||||
operatorCategoryId: meta.operatorCategory?.id ?? "",
|
||||
operatorCategoryName: meta.operatorCategory?.name ?? "",
|
||||
operatorClasses: JSON.stringify(meta.operatorCategory?.classes ?? []),
|
||||
service: meta.service ?? "",
|
||||
visionClass: meta.vision.class,
|
||||
visionConfidence: meta.vision.confidence,
|
||||
visionCategoryId: meta.vision.categoryId ?? null,
|
||||
@@ -168,7 +179,7 @@ export async function buildCollector(cfg: CollectorConfig, opts: { dbFile?: stri
|
||||
imagePath: rel,
|
||||
receivedAt: new Date().toISOString(),
|
||||
});
|
||||
req.log.info(`ingest: ${booth} item ${meta.item} (${meta.vision.class} → ${meta.operatorCategory.name})`);
|
||||
req.log.info(`ingest: ${booth} ${kind} ${meta.item} (${meta.vision.class}${kind === "wash" ? ` → ${meta.operatorCategory!.name}` : ""})`);
|
||||
return reply.code(201).send({ ok: true });
|
||||
});
|
||||
|
||||
@@ -214,13 +225,62 @@ export async function buildCollector(cfg: CollectorConfig, opts: { dbFile?: stri
|
||||
if (/^[=+\-@\t\r]/.test(v)) v = `'${v}`;
|
||||
return `"${v.replace(/"/g, '""')}"`;
|
||||
};
|
||||
const head = "item,booth,path,label,operator_category,operator_classes,vision_class,vision_confidence,downgraded,at,reviewed_at";
|
||||
const head = "item,booth,kind,path,label,operator_category,operator_classes,vision_class,vision_confidence,downgraded,at,reviewed_at";
|
||||
const lines = rows.map((r) =>
|
||||
[r.id, r.booth, r.imagePath, r.reviewLabel, r.operatorCategoryName, JSON.parse(r.operatorClasses).join("|"), r.visionClass, r.visionConfidence, r.downgraded, r.at, r.reviewedAt].map(q).join(","),
|
||||
[r.id, r.booth, r.kind, r.imagePath, r.reviewLabel, r.operatorCategoryName, JSON.parse(r.operatorClasses).join("|"), r.visionClass, r.visionConfidence, r.downgraded, r.at, r.reviewedAt].map(q).join(","),
|
||||
);
|
||||
return reply.type("text/csv; charset=utf-8").header("content-disposition", 'attachment; filename="labels.csv"').send([head, ...lines].join("\n") + "\n");
|
||||
});
|
||||
|
||||
// --- Training (proxy to the trainer's job API) ----------------------------------------
|
||||
// The trainer is a sibling container reading the same volume; it is reachable only on the
|
||||
// compose network, so the reviewer's login here is the only gate. The proxy forwards a
|
||||
// fixed set of paths and passes the trainer's status codes through (409 = a job runs).
|
||||
const trainer = cfg.trainerUrl;
|
||||
async function viaTrainer(reply: FastifyReply, tpath: string, init?: RequestInit): Promise<unknown> {
|
||||
if (!trainer) return reply.code(503).send({ error: "trainer not configured" });
|
||||
let r: Response;
|
||||
try {
|
||||
r = await fetch(trainer + tpath, { ...init, signal: AbortSignal.timeout(15_000) });
|
||||
} catch (err) {
|
||||
return reply.code(502).send({ error: `trainer unreachable: ${(err as Error).message}` });
|
||||
}
|
||||
const ctype = r.headers.get("content-type") ?? "application/json";
|
||||
return reply.code(r.status).type(ctype).send(Buffer.from(await r.arrayBuffer()));
|
||||
}
|
||||
app.get("/api/training/status", { preHandler: requireReviewer }, async (_req, reply) => {
|
||||
if (!trainer) return { configured: false };
|
||||
try {
|
||||
const get = async (p: string) => {
|
||||
const r = await fetch(trainer + p, { signal: AbortSignal.timeout(15_000) });
|
||||
if (!r.ok) throw new Error(`${p} → HTTP ${r.status}`);
|
||||
return r.json() as Promise<Record<string, unknown>>;
|
||||
};
|
||||
const [health, readiness, versions, jobs] = await Promise.all([get("/health"), get("/readiness"), get("/versions"), get("/jobs")]);
|
||||
return { configured: true, reachable: true, health, readiness, versions: versions.versions, jobs: jobs.jobs, current: jobs.current };
|
||||
} catch (err) {
|
||||
return reply.code(200).send({ configured: true, reachable: false, error: (err as Error).message });
|
||||
}
|
||||
});
|
||||
app.post<{ Body: Record<string, unknown> }>("/api/training/jobs", { preHandler: requireReviewer }, async (req, reply) => {
|
||||
const b = req.body && typeof req.body === "object" ? req.body : {};
|
||||
const kind = b.kind;
|
||||
if (kind !== "train" && kind !== "evaluate" && kind !== "publish") return reply.code(400).send({ error: "kind must be train, evaluate or publish" });
|
||||
// Only the knobs the UI offers cross over; the trainer validates their values.
|
||||
const allowed = ["kind", "mode", "backbone", "minAccuracy", "minPerClass", "epochs", "version"];
|
||||
const body: Record<string, unknown> = {};
|
||||
for (const k of allowed) if (b[k] !== undefined) body[k] = b[k];
|
||||
return viaTrainer(reply, "/jobs", { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify(body) });
|
||||
});
|
||||
app.get<{ Params: { id: string } }>("/api/training/jobs/:id", { preHandler: requireReviewer }, async (req, reply) => {
|
||||
if (!ID_RE.test(req.params.id)) return reply.code(400).send({ error: "bad job id" });
|
||||
return viaTrainer(reply, `/jobs/${encodeURIComponent(req.params.id)}`);
|
||||
});
|
||||
app.get<{ Params: { v: string } }>("/api/training/versions/:v/report", { preHandler: requireReviewer }, async (req, reply) => {
|
||||
if (!ID_RE.test(req.params.v)) return reply.code(400).send({ error: "bad version" });
|
||||
return viaTrainer(reply, `/versions/${encodeURIComponent(req.params.v)}/report`);
|
||||
});
|
||||
|
||||
return app;
|
||||
}
|
||||
|
||||
|
||||
@@ -6,6 +6,9 @@ export interface CollectorConfig {
|
||||
readonly boothTokens: ReadonlyMap<string, string>;
|
||||
/** The single reviewer login; null = review screen and export refuse (503). */
|
||||
readonly reviewer: { readonly user: string; readonly pass: string } | null;
|
||||
/** The trainer's job API on the compose network (http://trainer:8091); null = the
|
||||
* Training section is hidden and /api/training/* answers 503. */
|
||||
readonly trainerUrl: string | null;
|
||||
}
|
||||
|
||||
/** "booth-7:abc,booth-9:def" (commas, whitespace or newlines between pairs). */
|
||||
@@ -32,5 +35,6 @@ export function configFromEnv(env: NodeJS.ProcessEnv = process.env): CollectorCo
|
||||
dataDir: env.COLLECTOR_DATA_DIR ?? "/data",
|
||||
boothTokens: parseBoothTokens(env.COLLECTOR_BOOTH_TOKENS ?? ""),
|
||||
reviewer: user && pass.length >= 8 ? { user, pass } : null,
|
||||
trainerUrl: (env.COLLECTOR_TRAINER_URL ?? "").trim().replace(/\/+$/, "") || null,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -9,6 +9,9 @@ import type { VehicleClass } from "@parking/shared";
|
||||
export interface ItemRow {
|
||||
id: string;
|
||||
booth: string;
|
||||
/** "wash" = a desk decision (operator fields set); "entry" = a sampled entry read (pure
|
||||
* training material: crop + the camera's class, operator fields empty). */
|
||||
kind: "wash" | "entry";
|
||||
orderRef: string;
|
||||
at: string;
|
||||
operatorRef: string;
|
||||
@@ -44,11 +47,12 @@ export class CollectorDb {
|
||||
CREATE TABLE IF NOT EXISTS items (
|
||||
id TEXT PRIMARY KEY,
|
||||
booth TEXT NOT NULL,
|
||||
kind TEXT NOT NULL DEFAULT 'wash',
|
||||
order_ref TEXT NOT NULL,
|
||||
at TEXT NOT NULL,
|
||||
operator_ref TEXT NOT NULL,
|
||||
operator_category_id TEXT NOT NULL,
|
||||
operator_category_name TEXT NOT NULL,
|
||||
operator_ref TEXT NOT NULL DEFAULT '',
|
||||
operator_category_id TEXT NOT NULL DEFAULT '',
|
||||
operator_category_name TEXT NOT NULL DEFAULT '',
|
||||
operator_classes TEXT NOT NULL DEFAULT '[]',
|
||||
service TEXT NOT NULL,
|
||||
vision_class TEXT NOT NULL,
|
||||
@@ -77,6 +81,7 @@ export class CollectorDb {
|
||||
return {
|
||||
id: r.id as string,
|
||||
booth: r.booth as string,
|
||||
kind: r.kind === "entry" ? "entry" : "wash",
|
||||
orderRef: r.order_ref as string,
|
||||
at: r.at as string,
|
||||
operatorRef: r.operator_ref as string,
|
||||
@@ -107,10 +112,10 @@ export class CollectorDb {
|
||||
insert(row: Omit<ItemRow, "reviewLabel" | "reviewedAt" | "reviewer">): void {
|
||||
this.#db
|
||||
.prepare(
|
||||
`INSERT INTO items (id, booth, order_ref, at, operator_ref, operator_category_id, operator_category_name,
|
||||
`INSERT INTO items (id, booth, kind, order_ref, at, operator_ref, operator_category_id, operator_category_name,
|
||||
operator_classes, service, vision_class, vision_confidence, vision_category_id, downgraded,
|
||||
image_width, image_height, plate_blurred, image_path, received_at)
|
||||
VALUES (@id, @booth, @orderRef, @at, @operatorRef, @operatorCategoryId, @operatorCategoryName,
|
||||
VALUES (@id, @booth, @kind, @orderRef, @at, @operatorRef, @operatorCategoryId, @operatorCategoryName,
|
||||
@operatorClasses, @service, @visionClass, @visionConfidence, @visionCategoryId, @downgraded,
|
||||
@imageWidth, @imageHeight, @plateBlurred, @imagePath, @receivedAt)`,
|
||||
)
|
||||
@@ -142,19 +147,21 @@ export class CollectorDb {
|
||||
* reviewer's class fell inside the operator's chosen category (agree) or outside
|
||||
* (disagree) — the honest-mistake / fraud rate the outbox exists for. */
|
||||
stats(): {
|
||||
booths: { booth: string; received: number; pending: number; reviewed: number }[];
|
||||
booths: { booth: string; received: number; pending: number; reviewed: number; entries: number }[];
|
||||
operators: { booth: string; operatorRef: string; reviewed: number; agree: number; disagree: number; unusable: number }[];
|
||||
} {
|
||||
const booths = this.#db
|
||||
.prepare(
|
||||
`SELECT booth, COUNT(*) AS received,
|
||||
SUM(CASE WHEN reviewed_at IS NULL THEN 1 ELSE 0 END) AS pending,
|
||||
SUM(CASE WHEN reviewed_at IS NOT NULL THEN 1 ELSE 0 END) AS reviewed
|
||||
SUM(CASE WHEN reviewed_at IS NOT NULL THEN 1 ELSE 0 END) AS reviewed,
|
||||
SUM(CASE WHEN kind = 'entry' THEN 1 ELSE 0 END) AS entries
|
||||
FROM items GROUP BY booth ORDER BY booth`,
|
||||
)
|
||||
.all() as { booth: string; received: number; pending: number; reviewed: number }[];
|
||||
.all() as { booth: string; received: number; pending: number; reviewed: number; entries: number }[];
|
||||
// Operator agreement is a WASH thing — an entry sample has no operator decision.
|
||||
const reviewed = this.#db
|
||||
.prepare("SELECT booth, operator_ref, operator_classes, review_label FROM items WHERE reviewed_at IS NOT NULL")
|
||||
.prepare("SELECT booth, operator_ref, operator_classes, review_label FROM items WHERE reviewed_at IS NOT NULL AND kind = 'wash'")
|
||||
.all() as { booth: string; operator_ref: string; operator_classes: string; review_label: string }[];
|
||||
const ops = new Map<string, { booth: string; operatorRef: string; reviewed: number; agree: number; disagree: number; unusable: number }>();
|
||||
for (const r of reviewed) {
|
||||
|
||||
@@ -5,7 +5,7 @@ const cfg = configFromEnv();
|
||||
const app = await buildCollector(cfg);
|
||||
if (cfg.boothTokens.size === 0) app.log.warn("COLLECTOR_BOOTH_TOKENS is empty — no booth can ingest");
|
||||
if (!cfg.reviewer) app.log.warn("COLLECTOR_REVIEWER_USER/PASS not set — the review screen and export refuse");
|
||||
app.log.info(`collector: ${cfg.boothTokens.size} booth token(s), data in ${cfg.dataDir}`);
|
||||
app.log.info(`collector: ${cfg.boothTokens.size} booth token(s), data in ${cfg.dataDir}, trainer ${cfg.trainerUrl ?? "not configured"}`);
|
||||
await app.listen({ host: cfg.host, port: cfg.port });
|
||||
|
||||
const stop = async () => {
|
||||
|
||||
@@ -35,6 +35,16 @@ export function reviewPage(): string {
|
||||
td, th { text-align:left; padding:.2rem .5rem; border-bottom:1px solid #2a2a2a; }
|
||||
th { color:var(--muted); font-weight:normal; font-size:.75rem; text-transform:uppercase; letter-spacing:.06em; }
|
||||
kbd { background:#2a2a2a; border:1px solid #444; border-radius:3px; padding:0 .3rem; font-size:.75rem; }
|
||||
h2 { font-size:.8rem; letter-spacing:.08em; text-transform:uppercase; color:var(--amber); margin:0 0 .6rem; }
|
||||
.row { display:flex; flex-wrap:wrap; gap:.6rem; align-items:center; }
|
||||
select, input { background:#2a2a2a; color:var(--text); border:1px solid #444; border-radius:4px; padding:.4rem .5rem; font:inherit; }
|
||||
input[type=number] { width:5rem; }
|
||||
label { color:var(--muted); font-size:.8rem; }
|
||||
pre { background:#0d0d0d; border:1px solid #2a2a2a; border-radius:4px; padding:.6rem; max-height:22rem; overflow:auto; font-size:.75rem; white-space:pre-wrap; margin:.6rem 0 0; }
|
||||
.ok { color:var(--green); }
|
||||
.bad { color:var(--red); }
|
||||
button:disabled { opacity:.45; cursor:not-allowed; }
|
||||
button.small { padding:.25rem .5rem; font-size:.75rem; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
@@ -46,6 +56,10 @@ export function reviewPage(): string {
|
||||
<section class="card">
|
||||
<table id="stats"><thead><tr><th>booth</th><th>operator</th><th>reviewed</th><th>agree</th><th>disagree</th><th>unusable</th></tr></thead><tbody></tbody></table>
|
||||
</section>
|
||||
<section class="card" id="training" hidden>
|
||||
<h2>Training</h2>
|
||||
<div id="tr-body"></div>
|
||||
</section>
|
||||
<p class="muted">Keys: <kbd>1</kbd>–<kbd>9</kbd>, <kbd>0</kbd> pick a class in order · <kbd>u</kbd> unusable · <kbd>s</kbd> skip. Skipped items come back after a reload. Your verdict is the training label; the operator's pick is only compared against it.</p>
|
||||
</main>
|
||||
<script>
|
||||
@@ -80,10 +94,13 @@ async function next() {
|
||||
el.innerHTML =
|
||||
'<img src="/api/items/' + encodeURIComponent(it.id) + '/image" alt="">' +
|
||||
'<dl style="margin-top:.8rem">' +
|
||||
'<dt>operator chose</dt><dd><b>' + esc(it.operatorCategoryName) + '</b> <span class="muted">(' + esc(opClasses.join(', ') || 'no classes mapped') + ')</span></dd>' +
|
||||
(it.kind === 'entry'
|
||||
? '<dt>sample</dt><dd><span class="muted">entry stream — no wash, no operator decision; label the vehicle</span></dd>'
|
||||
: '<dt>operator chose</dt><dd><b>' + esc(it.operatorCategoryName) + '</b> <span class="muted">(' + esc(opClasses.join(', ') || 'no classes mapped') + ')</span></dd>') +
|
||||
'<dt>camera saw</dt><dd class="mono">' + esc(it.visionClass) + ' <span class="muted">' + Math.round(it.visionConfidence * 100) + '%</span>' + (it.downgraded ? ' <span class="warn">flagged downgrade at the booth</span>' : '') + '</dd>' +
|
||||
'<dt>service</dt><dd>' + esc(it.service) + '</dd>' +
|
||||
'<dt>booth · operator</dt><dd class="mono">' + esc(it.booth) + ' · ' + esc(it.operatorRef) + '</dd>' +
|
||||
(it.kind === 'entry' ? '<dt>booth</dt><dd class="mono">' + esc(it.booth) + '</dd>' :
|
||||
'<dt>service</dt><dd>' + esc(it.service) + '</dd>' +
|
||||
'<dt>booth · operator</dt><dd class="mono">' + esc(it.booth) + ' · ' + esc(it.operatorRef) + '</dd>') +
|
||||
'<dt>at</dt><dd>' + esc(it.at) + '</dd>' +
|
||||
'</dl>' +
|
||||
'<div class="buttons" style="margin-top:.8rem">' +
|
||||
@@ -111,6 +128,106 @@ document.addEventListener('keydown', e => {
|
||||
|
||||
next().catch(e => { document.getElementById('item').innerHTML = '<p class="warn">' + esc(e.message) + '</p>'; });
|
||||
loadStats().catch(() => {});
|
||||
|
||||
// ---- Training: the trainer's job API, proxied by the collector -------------------------
|
||||
// Readiness (labels per class vs the minimum), one job at a time with a live log, the
|
||||
// versions a run produced (written or refused) with Report / Evaluate / Publish. Pinning a
|
||||
// published version into the vision image stays a git commit — that is the deploy control.
|
||||
let trPoll = null;
|
||||
let trShownReport = null;
|
||||
const trDefaults = { mode: 'features', backbone: 'resnet18', minAccuracy: 0.85 };
|
||||
|
||||
function pct(x) { return x == null ? '—' : Math.round(x * 100) + ' %'; }
|
||||
|
||||
async function training() {
|
||||
const box = document.getElementById('training');
|
||||
const el = document.getElementById('tr-body');
|
||||
let s;
|
||||
try { s = await api('/api/training/status'); } catch (e) { box.hidden = false; el.innerHTML = '<p class="warn">' + esc(e.message) + '</p>'; return; }
|
||||
if (!s.configured) { box.hidden = true; return; }
|
||||
box.hidden = false;
|
||||
if (!s.reachable) { el.innerHTML = '<p class="warn">trainer not reachable: ' + esc(s.error || '') + '</p>'; schedule(true); return; }
|
||||
const r = s.readiness, run = r.run || {}, minPer = run.minPerClass || 20;
|
||||
const byClass = (r.labelled && r.labelled.byClass) || {};
|
||||
const classes = Object.keys(byClass);
|
||||
const cur = s.current;
|
||||
const readyLine = r.ready
|
||||
? '<span class="ok">enough labels to train</span> — classes this run: ' + esc((run.classes || []).join(', '))
|
||||
: '<span class="warn">not enough labels yet</span> — a class needs ' + minPer + ' reviewed crops; two classes must clear it';
|
||||
let html = '<p>' + readyLine + ' <span class="muted">(' + (r.labelled ? r.labelled.total : 0) + ' labelled, ' + (r.missingCrops || 0) + ' missing crop files)</span></p>';
|
||||
html += '<table><thead><tr><th>class</th><th>reviewed</th><th>train</th><th>val</th><th></th></tr></thead><tbody>' +
|
||||
(classes.map(c => '<tr><td class="mono">' + esc(c) + '</td><td>' + byClass[c] + '</td><td>' + ((run.train || {})[c] ?? '—') + '</td><td>' + ((run.val || {})[c] ?? '—') + '</td><td class="muted">' + (byClass[c] < minPer ? 'below ' + minPer + ' — dropped' : '') + '</td></tr>').join('') || '<tr><td colspan="5" class="muted">no labels yet — review crops above</td></tr>') +
|
||||
'</tbody></table>';
|
||||
const d = Object.assign({}, trDefaults, r.defaults || {});
|
||||
html += '<div class="row" style="margin-top:.8rem">' +
|
||||
'<label>mode <select id="tr-mode">' + (r.modes || ['features', 'finetune']).map(m => '<option' + (m === d.mode ? ' selected' : '') + '>' + m + '</option>').join('') + '</select></label>' +
|
||||
'<label>backbone <select id="tr-backbone">' + (r.backbones || ['resnet18']).map(b => '<option' + (b === d.backbone ? ' selected' : '') + '>' + b + '</option>').join('') + '</select></label>' +
|
||||
'<label>floor <input id="tr-floor" type="number" min="0" max="1" step="0.01" value="' + d.minAccuracy + '"></label>' +
|
||||
'<button id="tr-train"' + (r.ready && !cur ? '' : ' disabled') + '>Train</button>' +
|
||||
(cur ? '<span class="warn">running: ' + esc(cur.kind) + ' ' + esc(cur.id) + '</span>' : '') +
|
||||
'</div>';
|
||||
const last = cur || (s.jobs && s.jobs[0]);
|
||||
if (last) {
|
||||
const cls = last.status === 'done' ? 'ok' : last.status === 'running' ? 'warn' : 'bad';
|
||||
html += '<p style="margin:.8rem 0 0"><span class="' + cls + '">' + esc(last.status) + '</span> <span class="mono">' + esc(last.kind) + ' ' + esc(last.id) + '</span> <span class="muted">' + esc(last.startedAt || '') + (last.exitCode != null ? ' · exit ' + last.exitCode : '') + '</span> <button class="small" data-job="' + esc(last.id) + '">log</button></p>' +
|
||||
'<pre id="tr-log" hidden></pre>';
|
||||
}
|
||||
const vs = s.versions || [];
|
||||
html += '<h2 style="margin-top:1rem">Versions</h2>';
|
||||
html += vs.length
|
||||
? '<table><thead><tr><th>version</th><th>model</th><th>accuracy</th><th>classes</th><th>mode</th><th></th></tr></thead><tbody>' +
|
||||
vs.map(v => '<tr><td class="mono">' + esc(v.version) + '</td><td>' + (v.written ? '<span class="ok">written</span>' : '<span class="bad">refused</span>') + '</td><td>' + pct(v.accuracy) + (v.floor != null ? ' <span class="muted">/ floor ' + pct(v.floor) + '</span>' : '') + '</td><td class="muted">' + esc((v.classes || []).join(', ')) + '</td><td class="muted">' + esc(v.mode || '') + '</td><td>' +
|
||||
'<button class="small" data-report="' + esc(v.version) + '">report</button> ' +
|
||||
(v.written ? '<button class="small" data-eval="' + esc(v.version) + '"' + (cur ? ' disabled' : '') + '>evaluate</button> <button class="small" data-publish="' + esc(v.version) + '"' + (cur ? ' disabled' : '') + '>publish</button>' : '') +
|
||||
'</td></tr>').join('') + '</tbody></table>'
|
||||
: '<p class="muted">no runs yet</p>';
|
||||
html += '<pre id="tr-report" hidden></pre>';
|
||||
html += '<p class="muted" style="margin:.8rem 0 0">A written model is only a file here. To put it on a booth: publish, then pin the version in <span class="mono">apps/vision/models/bodytype.version</span>, commit, and bump the TAG of the booth.</p>';
|
||||
el.innerHTML = html;
|
||||
|
||||
const trainBtn = document.getElementById('tr-train');
|
||||
if (trainBtn) trainBtn.addEventListener('click', () => startJob({ kind: 'train', mode: document.getElementById('tr-mode').value, backbone: document.getElementById('tr-backbone').value, minAccuracy: Number(document.getElementById('tr-floor').value) }));
|
||||
el.querySelectorAll('button[data-eval]').forEach(b => b.addEventListener('click', () => startJob({ kind: 'evaluate', version: b.dataset.eval })));
|
||||
el.querySelectorAll('button[data-publish]').forEach(b => b.addEventListener('click', () => { if (confirm('Publish ' + b.dataset.publish + ' to the package registry?')) startJob({ kind: 'publish', version: b.dataset.publish }); }));
|
||||
el.querySelectorAll('button[data-job]').forEach(b => b.addEventListener('click', () => showLog(b.dataset.job)));
|
||||
el.querySelectorAll('button[data-report]').forEach(b => b.addEventListener('click', () => showReport(b.dataset.report)));
|
||||
if (cur) showLog(cur.id).catch(() => {});
|
||||
if (trShownReport) showReport(trShownReport).catch(() => {});
|
||||
schedule(!!cur);
|
||||
}
|
||||
|
||||
function schedule(soon) {
|
||||
if (trPoll) clearTimeout(trPoll);
|
||||
trPoll = setTimeout(() => training().catch(() => {}), soon ? 4000 : 60000);
|
||||
}
|
||||
|
||||
async function startJob(body) {
|
||||
try {
|
||||
const r = await fetch('/api/training/jobs', { method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body) });
|
||||
if (!r.ok) { const e = await r.json().catch(() => ({})); alert('trainer: ' + (e.error || ('HTTP ' + r.status))); }
|
||||
} catch (e) { alert(e.message); }
|
||||
training().catch(() => {});
|
||||
}
|
||||
|
||||
async function showLog(id) {
|
||||
const j = await api('/api/training/jobs/' + encodeURIComponent(id));
|
||||
const pre = document.getElementById('tr-log');
|
||||
if (!pre) return;
|
||||
pre.hidden = false;
|
||||
pre.textContent = j.log || '(no output yet)';
|
||||
pre.scrollTop = pre.scrollHeight;
|
||||
}
|
||||
|
||||
async function showReport(v) {
|
||||
const r = await fetch('/api/training/versions/' + encodeURIComponent(v) + '/report');
|
||||
const pre = document.getElementById('tr-report');
|
||||
if (!pre) return;
|
||||
trShownReport = v;
|
||||
pre.hidden = false;
|
||||
pre.textContent = r.ok ? await r.text() : 'no report for ' + v + ' (HTTP ' + r.status + ')';
|
||||
}
|
||||
|
||||
training().catch(() => {});
|
||||
</script>
|
||||
</body>
|
||||
</html>`;
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
import { afterEach, beforeEach, describe, expect, it } from "vitest";
|
||||
import { createServer, type IncomingMessage, type Server, type ServerResponse } from "node:http";
|
||||
import { mkdtemp, rm } from "node:fs/promises";
|
||||
import { tmpdir } from "node:os";
|
||||
import path from "node:path";
|
||||
import { buildCollector, type CollectorApp } from "./app.js";
|
||||
|
||||
// The Training section's proxy: reviewer-gated, forwards a fixed set of paths to the
|
||||
// trainer's job API, passes its status codes through, and degrades cleanly when the trainer
|
||||
// is not configured or not reachable. The trainer is faked with a bare node http server.
|
||||
|
||||
const REVIEWER = { user: "julian", pass: "review-pass-123" };
|
||||
const basic = "Basic " + Buffer.from(`${REVIEWER.user}:${REVIEWER.pass}`).toString("base64");
|
||||
|
||||
let dir: string;
|
||||
let fake: Server;
|
||||
let fakeUrl: string;
|
||||
let seen: { method: string; url: string; body: string }[];
|
||||
let app: CollectorApp;
|
||||
|
||||
async function start(trainerUrl: string | null): Promise<void> {
|
||||
app = await buildCollector({ host: "127.0.0.1", port: 0, dataDir: dir, boothTokens: new Map(), reviewer: REVIEWER, trainerUrl }, { dbFile: ":memory:" });
|
||||
await app.ready();
|
||||
}
|
||||
|
||||
beforeEach(async () => {
|
||||
dir = await mkdtemp(path.join(tmpdir(), "collector-"));
|
||||
seen = [];
|
||||
fake = createServer((req: IncomingMessage, res: ServerResponse) => {
|
||||
let body = "";
|
||||
req.on("data", (c) => (body += c));
|
||||
req.on("end", () => {
|
||||
seen.push({ method: req.method ?? "", url: req.url ?? "", body });
|
||||
const json = (code: number, obj: unknown) => {
|
||||
res.writeHead(code, { "content-type": "application/json" });
|
||||
res.end(JSON.stringify(obj));
|
||||
};
|
||||
if (req.url === "/health") return json(200, { ok: true, busy: false });
|
||||
if (req.url === "/readiness") return json(200, { ready: false, labelled: { total: 3 } });
|
||||
if (req.url === "/versions") return json(200, { versions: [{ version: "v1", written: true }] });
|
||||
if (req.url === "/jobs" && req.method === "GET") return json(200, { jobs: [{ id: "j1" }], current: null });
|
||||
if (req.url === "/jobs" && req.method === "POST") return body.includes('"busy"') ? json(409, { error: "a job is already running" }) : json(202, { id: "j2", status: "running" });
|
||||
if (req.url === "/jobs/j1") return json(200, { id: "j1", status: "done", log: "ok" });
|
||||
if (req.url === "/versions/v1/report") {
|
||||
res.writeHead(200, { "content-type": "text/markdown; charset=utf-8" });
|
||||
return res.end("# Body-type classifier v1\n");
|
||||
}
|
||||
return json(404, { error: "not found" });
|
||||
});
|
||||
});
|
||||
await new Promise<void>((r) => fake.listen(0, "127.0.0.1", r));
|
||||
const a = fake.address() as { port: number };
|
||||
fakeUrl = `http://127.0.0.1:${a.port}`;
|
||||
});
|
||||
afterEach(async () => {
|
||||
await app?.close();
|
||||
await new Promise<void>((r) => fake.close(() => r()));
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
describe("review page script", () => {
|
||||
it("parses as JavaScript (an apostrophe in a template literal once broke the whole page)", async () => {
|
||||
const { reviewPage } = await import("./review-page.js");
|
||||
const html = reviewPage();
|
||||
const script = html.slice(html.indexOf("<script>") + 8, html.lastIndexOf("</script>"));
|
||||
expect(() => new Function(script)).not.toThrow();
|
||||
});
|
||||
});
|
||||
|
||||
describe("training proxy", () => {
|
||||
it("is hidden when no trainer is configured", async () => {
|
||||
await start(null);
|
||||
const s = await app.inject({ method: "GET", url: "/api/training/status", headers: { authorization: basic } });
|
||||
expect(s.json()).toEqual({ configured: false });
|
||||
const j = await app.inject({ method: "POST", url: "/api/training/jobs", headers: { authorization: basic }, payload: { kind: "train" } });
|
||||
expect(j.statusCode).toBe(503);
|
||||
});
|
||||
|
||||
it("aggregates status and forwards jobs and reports behind the reviewer login", async () => {
|
||||
await start(fakeUrl);
|
||||
expect((await app.inject({ method: "GET", url: "/api/training/status" })).statusCode).toBe(401);
|
||||
const s = await app.inject({ method: "GET", url: "/api/training/status", headers: { authorization: basic } });
|
||||
expect(s.statusCode).toBe(200);
|
||||
const body = s.json();
|
||||
expect(body.configured).toBe(true);
|
||||
expect(body.reachable).toBe(true);
|
||||
expect(body.readiness.labelled.total).toBe(3);
|
||||
expect(body.versions[0].version).toBe("v1");
|
||||
expect(body.jobs[0].id).toBe("j1");
|
||||
|
||||
const j = await app.inject({
|
||||
method: "POST",
|
||||
url: "/api/training/jobs",
|
||||
headers: { authorization: basic },
|
||||
payload: { kind: "train", mode: "features", minAccuracy: 0.9, secret: "nope", version: "v2" },
|
||||
});
|
||||
expect(j.statusCode).toBe(202);
|
||||
expect(j.json().id).toBe("j2");
|
||||
const posted = seen.find((r) => r.method === "POST")!;
|
||||
expect(JSON.parse(posted.body)).toEqual({ kind: "train", mode: "features", minAccuracy: 0.9, version: "v2" }); // unknown keys dropped
|
||||
|
||||
const busy = await app.inject({ method: "POST", url: "/api/training/jobs", headers: { authorization: basic }, payload: { kind: "evaluate", version: "busy" } });
|
||||
expect(busy.statusCode).toBe(409); // the trainer's answer passes through
|
||||
|
||||
const bad = await app.inject({ method: "POST", url: "/api/training/jobs", headers: { authorization: basic }, payload: { kind: "rm-rf" } });
|
||||
expect(bad.statusCode).toBe(400);
|
||||
|
||||
const one = await app.inject({ method: "GET", url: "/api/training/jobs/j1", headers: { authorization: basic } });
|
||||
expect(one.json().status).toBe("done");
|
||||
expect((await app.inject({ method: "GET", url: "/api/training/jobs/..%2Fx", headers: { authorization: basic } })).statusCode).toBe(400);
|
||||
|
||||
const rep = await app.inject({ method: "GET", url: "/api/training/versions/v1/report", headers: { authorization: basic } });
|
||||
expect(rep.statusCode).toBe(200);
|
||||
expect(rep.headers["content-type"]).toContain("text/markdown");
|
||||
expect(rep.body).toContain("# Body-type classifier v1");
|
||||
});
|
||||
|
||||
it("reports an unreachable trainer without failing the page", async () => {
|
||||
await start("http://127.0.0.1:9"); // nothing listens on the discard port
|
||||
const s = await app.inject({ method: "GET", url: "/api/training/status", headers: { authorization: basic } });
|
||||
expect(s.statusCode).toBe(200);
|
||||
expect(s.json().reachable).toBe(false);
|
||||
const j = await app.inject({ method: "POST", url: "/api/training/jobs", headers: { authorization: basic }, payload: { kind: "train" } });
|
||||
expect(j.statusCode).toBe(502);
|
||||
});
|
||||
});
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"$schema": "https://schema.tauri.app/config/2",
|
||||
"productName": "Parking System",
|
||||
"version": "0.1.0",
|
||||
"version": "0.2.0",
|
||||
"identifier": "com.parking.desktop",
|
||||
"build": {
|
||||
"devUrl": "http://localhost:5173",
|
||||
|
||||
@@ -99,3 +99,8 @@ WS_ALLOWED_ORIGINS=http://localhost:5173,tauri://localhost,http://tauri.localhos
|
||||
# CARWASH_REVIEW_TOKEN=
|
||||
# CARWASH_REVIEW_BOOTH_ID=
|
||||
# CARWASH_REVIEW_INTERVAL_SEC=60
|
||||
# Entry-stream sampling: also queue one in N ENTRY vehicle reads (no wash, no operator) as
|
||||
# pure training material in the gate view — many times the wash stream, zero domain shift.
|
||||
# 1 = every entry (the reviewer labels what they have time for; the rest waits and stays
|
||||
# useful), N = one in N, 0/unset = off. Needs the three settings above.
|
||||
# CARWASH_REVIEW_ENTRY_SAMPLE=1
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import { EventEmitter } from "node:events";
|
||||
import type { PrinterStatus } from "@parking/devices";
|
||||
import type { LedgerEventRow } from "@parking/db";
|
||||
import type { VehicleRead } from "@parking/shared";
|
||||
|
||||
// Internal event bus for device-originated events (button presses, etc.).
|
||||
// Hardware drivers / inbound device pushes emit here; business logic (entry
|
||||
@@ -105,6 +106,17 @@ export interface PlateRecognizedEvent {
|
||||
readonly direction: "entry" | "exit";
|
||||
}
|
||||
|
||||
/** Emitted when vision classified the vehicle in an entry/exit frame (advisory; stored on
|
||||
* the read row like the plate). A module may sample these — the Car Wash review outbox
|
||||
* queues one in N ENTRY reads for the remote reviewer, in the gate view the classifier
|
||||
* will be trained on (wiki/concepts/vision-review-outbox.md). The core emits; it never
|
||||
* knows who listens. */
|
||||
export interface VehicleReadEvent {
|
||||
readonly identity: string;
|
||||
readonly direction: "entry" | "exit";
|
||||
readonly read: VehicleRead;
|
||||
}
|
||||
|
||||
/** Per-lane RADAR presence — a vehicle-presence INPUT (loop/radar) is shorted at the
|
||||
* entry/exit barrier, i.e. "something is in the lane vicinity" BEFORE the camera has
|
||||
* confirmed a vehicle. Same signal that makes the physical button lamp (relay 3) blink:
|
||||
@@ -197,6 +209,13 @@ class DeviceEventBus extends EventEmitter {
|
||||
this.on("plate-recognized", cb);
|
||||
return () => this.off("plate-recognized", cb);
|
||||
}
|
||||
emitVehicleRead(event: VehicleReadEvent): void {
|
||||
this.emit("vehicle-read", event);
|
||||
}
|
||||
onVehicleRead(cb: (event: VehicleReadEvent) => void): () => void {
|
||||
this.on("vehicle-read", cb);
|
||||
return () => this.off("vehicle-read", cb);
|
||||
}
|
||||
}
|
||||
|
||||
/** Process-wide device event bus. */
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { deviceEvents } from "../../device-events.js";
|
||||
import type { ServerModule } from "../index.js";
|
||||
import { ReviewOutbox, reviewUploadConfigFromEnv } from "./review-outbox.js";
|
||||
import { carwashRoutes } from "./routes.js";
|
||||
@@ -18,6 +19,13 @@ export const carwashModule: ServerModule = {
|
||||
const outbox = new ReviewOutbox(deps.db, app.log, cfg);
|
||||
app.log.info(cfg ? `carwash review upload: on → ${new URL(cfg.url).host} as ${cfg.boothId}` : "carwash review upload: off");
|
||||
outbox.start();
|
||||
// Entry-stream sampling: one in N entry vehicle reads goes to the reviewer as pure
|
||||
// training material (the gate view, no order attached). The core announces the read;
|
||||
// the module decides. Off unless CARWASH_REVIEW_ENTRY_SAMPLE is set.
|
||||
const offVehicleRead = deviceEvents.onVehicleRead((e) => {
|
||||
if (e.direction === "entry" && outbox.sampleEntry()) void outbox.enqueueEntry(e.read);
|
||||
});
|
||||
app.addHook("onClose", async () => offVehicleRead());
|
||||
app.addHook("onClose", async () => outbox.stop());
|
||||
const service = new CarwashService(deps, app.log, outbox);
|
||||
// A wash ordered with payAt = "booth" is a charge line on the parking settlement;
|
||||
|
||||
@@ -3,6 +3,7 @@ import sharp from "sharp";
|
||||
import { createTestDb } from "@parking/db/testing";
|
||||
import { carwashOrders, carwashReviewOutbox, deviceEvents, snapshots, type Db } from "@parking/db";
|
||||
import type { FastifyInstance } from "fastify";
|
||||
import { deviceEvents as deviceEventBus } from "../../device-events.js";
|
||||
import { buildServer } from "../../server.js";
|
||||
import { login, makeLog, minutesAgo, seedTariff, seedUser, silentLogger } from "../../test-helpers.js";
|
||||
import { EXPIRE_DAYS, ReviewOutbox, makeReviewCrop, operatorRef, reviewUploadConfigFromEnv } from "./review-outbox.js";
|
||||
@@ -81,7 +82,7 @@ describe("queue + drain", () => {
|
||||
});
|
||||
afterEach(() => close());
|
||||
|
||||
const cfg = { url: "https://collector.overlay/ingest", token: "secret-1", boothId: "booth-7", intervalSec: 60 };
|
||||
const cfg = { url: "https://collector.overlay/ingest", token: "secret-1", boothId: "booth-7", intervalSec: 60, entrySample: 0 };
|
||||
const read = { bodyType: "car" as const, confidence: 0.86, snapshotId: "snap-1", box: CAR, plateBox: PLATE };
|
||||
const item = { orderId: "o-1", createdAt: "2026-09-06T10:00:00.000Z", createdBy: "lavazhier", categoryId: "car", categoryName: "Vetura", categoryClasses: ["car", "sedan"], serviceName: "Standard", visionCategoryId: "car", downgraded: false };
|
||||
|
||||
@@ -105,7 +106,7 @@ describe("queue + drain", () => {
|
||||
const row = db.select().from(carwashReviewOutbox).all()[0]!;
|
||||
expect(row.status).toBe("queued");
|
||||
expect(row.image!.length).toBeGreaterThan(500);
|
||||
expect(row.payload).toMatchObject({ v: 1, booth: "booth-7", order: "o-1", operatorCategory: { id: "car", name: "Vetura", classes: ["car", "sedan"] }, vision: { class: "car", confidence: 0.86 }, downgraded: false, image: { plateBlurred: true } });
|
||||
expect(row.payload).toMatchObject({ v: 1, kind: "wash", booth: "booth-7", order: "o-1", operatorCategory: { id: "car", name: "Vetura", classes: ["car", "sedan"] }, vision: { class: "car", confidence: 0.86 }, downgraded: false, image: { plateBlurred: true } });
|
||||
expect(JSON.stringify(row.payload)).not.toContain("lavazhier");
|
||||
|
||||
expect(await ob.drain()).toEqual({ sent: 1, failed: 0, deferred: 0 });
|
||||
@@ -159,6 +160,18 @@ describe("queue + drain", () => {
|
||||
expect(fetchFn.mock.calls.length).toBe(before);
|
||||
expect(ob.status().failed).toBe(3);
|
||||
|
||||
// Entry sampling: one in N entry reads becomes a package with the crop and the
|
||||
// camera's class only — no order, no operator, no category.
|
||||
const sampler = new ReviewOutbox(db, silentLogger(), { ...cfg, entrySample: 3 }, fetchFn);
|
||||
expect([sampler.sampleEntry(), sampler.sampleEntry(), sampler.sampleEntry(), sampler.sampleEntry()]).toEqual([false, false, true, false]);
|
||||
expect(ob.sampleEntry()).toBe(false); // entrySample 0 = off
|
||||
expect(await sampler.enqueueEntry(read)).toBe(true);
|
||||
const entryRow = db.select().from(carwashReviewOutbox).where(eq(carwashReviewOutbox.orderId, "entry:snap-1")).get()!;
|
||||
expect(entryRow.payload).toMatchObject({ v: 1, kind: "entry", booth: "booth-7", vision: { class: "car", confidence: 0.86 }, image: { plateBlurred: true } });
|
||||
expect(entryRow.payload).not.toHaveProperty("operator");
|
||||
expect(entryRow.payload).not.toHaveProperty("operatorCategory");
|
||||
expect(entryRow.image!.length).toBeGreaterThan(500);
|
||||
|
||||
// No vehicle box, no snapshot, or upload off → nothing queued.
|
||||
expect(await ob.enqueue(item, { ...read, box: null })).toBe(false);
|
||||
expect(await ob.enqueue(item, { ...read, snapshotId: "gone" })).toBe(false);
|
||||
@@ -178,6 +191,7 @@ describe("through the app", () => {
|
||||
process.env.CARWASH_REVIEW_URL = "https://collector.overlay/ingest";
|
||||
process.env.CARWASH_REVIEW_TOKEN = "tok";
|
||||
process.env.CARWASH_REVIEW_BOOTH_ID = "booth-9";
|
||||
process.env.CARWASH_REVIEW_ENTRY_SAMPLE = "1";
|
||||
const t = createTestDb();
|
||||
db = t.db;
|
||||
close = t.close;
|
||||
@@ -187,7 +201,7 @@ describe("through the app", () => {
|
||||
afterEach(async () => {
|
||||
await app.close();
|
||||
close();
|
||||
for (const k of ["CARWASH_REVIEW_URL", "CARWASH_REVIEW_TOKEN", "CARWASH_REVIEW_BOOTH_ID"]) {
|
||||
for (const k of ["CARWASH_REVIEW_URL", "CARWASH_REVIEW_TOKEN", "CARWASH_REVIEW_BOOTH_ID", "CARWASH_REVIEW_ENTRY_SAMPLE"]) {
|
||||
if (saved[k] === undefined) delete process.env[k];
|
||||
else process.env[k] = saved[k];
|
||||
}
|
||||
@@ -214,6 +228,14 @@ describe("through the app", () => {
|
||||
// Enqueue is fire-and-forget: give the crop a moment.
|
||||
await vi.waitFor(() => expect(db.select().from(carwashReviewOutbox).all()).toHaveLength(1));
|
||||
const status = (await app.inject({ method: "GET", url: "/api/carwash/review/status", headers: { cookie: a.cookie } })).json();
|
||||
expect(status).toMatchObject({ enabled: true, boothId: "booth-9", queued: 1, sent: 0 });
|
||||
expect(status).toMatchObject({ enabled: true, boothId: "booth-9", queued: 1, sent: 0, entrySample: 1 });
|
||||
|
||||
// An ENTRY vehicle read announced by the core (snapshot.ts) is sampled by the module
|
||||
// (1 in 1 here) into an entry package; an exit read is not.
|
||||
deviceEventBus.emitVehicleRead({ identity: "T-X", direction: "exit", read: { bodyType: "car", confidence: 0.8, snapshotId: "snap-r", box: CAR, plateBox: PLATE } });
|
||||
deviceEventBus.emitVehicleRead({ identity: "T-R", direction: "entry", read: { bodyType: "car", confidence: 0.8, snapshotId: "snap-r", box: CAR, plateBox: PLATE } });
|
||||
await vi.waitFor(() => expect(db.select().from(carwashReviewOutbox).all()).toHaveLength(2));
|
||||
const rows = db.select().from(carwashReviewOutbox).all();
|
||||
expect(rows.map((r) => (r.payload as { kind: string }).kind).sort()).toEqual(["entry", "wash"]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -34,6 +34,10 @@ export interface ReviewUploadConfig {
|
||||
/** Pseudonymous booth id — a label the reviewer maps to a site; never the site name. */
|
||||
readonly boothId: string;
|
||||
readonly intervalSec: number;
|
||||
/** Queue one in N ENTRY vehicle reads (no order attached) for the reviewer — the gate
|
||||
* view is exactly what the classifier is trained on, and the entry stream is many times
|
||||
* the wash stream. 0 = off. */
|
||||
readonly entrySample: number;
|
||||
}
|
||||
|
||||
/** From the server env (Komodo stack env). All three of URL, token and booth id, or off. */
|
||||
@@ -43,7 +47,12 @@ export function reviewUploadConfigFromEnv(env: NodeJS.ProcessEnv = process.env):
|
||||
const boothId = (env.CARWASH_REVIEW_BOOTH_ID ?? "").trim();
|
||||
if (!url || !token || !boothId) return null;
|
||||
const raw = Number(env.CARWASH_REVIEW_INTERVAL_SEC ?? 60);
|
||||
return { url, token, boothId, intervalSec: Number.isFinite(raw) && raw >= 10 ? raw : 60 };
|
||||
const sample = Number(env.CARWASH_REVIEW_ENTRY_SAMPLE ?? 0);
|
||||
return {
|
||||
url, token, boothId,
|
||||
intervalSec: Number.isFinite(raw) && raw >= 10 ? raw : 60,
|
||||
entrySample: Number.isInteger(sample) && sample > 0 ? sample : 0,
|
||||
};
|
||||
}
|
||||
|
||||
/** The crop's longest edge, in pixels — enough for a reviewer and a classifier, small
|
||||
@@ -145,6 +154,8 @@ export interface OutboxStatus {
|
||||
readonly failed: number;
|
||||
readonly lastSentAt: string | null;
|
||||
readonly lastError: string | null;
|
||||
/** 0 = entry sampling off; N = one in N entry reads is queued. */
|
||||
readonly entrySample: number;
|
||||
}
|
||||
|
||||
export class ReviewOutbox {
|
||||
@@ -154,6 +165,7 @@ export class ReviewOutbox {
|
||||
readonly #fetch: FetchLike;
|
||||
#timer: NodeJS.Timeout | null = null;
|
||||
#draining = false;
|
||||
#entrySeen = 0;
|
||||
|
||||
constructor(db: Db, logger: FastifyBaseLogger, cfg: ReviewUploadConfig | null, fetchFn?: FetchLike) {
|
||||
this.#db = db;
|
||||
@@ -172,35 +184,68 @@ export class ReviewOutbox {
|
||||
* sample) or when upload is not configured (an unbounded queue nobody drains). */
|
||||
async enqueue(item: ReviewItemInput, read: VehicleRead): Promise<boolean> {
|
||||
if (!this.#cfg) return false;
|
||||
return this.#queue(item.orderId, read, (id, crop) => ({
|
||||
v: 1,
|
||||
kind: "wash",
|
||||
booth: this.#cfg!.boothId,
|
||||
item: id,
|
||||
order: item.orderId,
|
||||
at: item.createdAt,
|
||||
operator: operatorRef(this.#cfg!.boothId, item.createdBy),
|
||||
operatorCategory: { id: item.categoryId, name: item.categoryName, classes: [...item.categoryClasses] },
|
||||
service: item.serviceName,
|
||||
vision: { class: read.bodyType, confidence: read.confidence, categoryId: item.visionCategoryId },
|
||||
downgraded: item.downgraded,
|
||||
image: crop,
|
||||
}));
|
||||
}
|
||||
|
||||
/** Every Nth entry read is a sample (N = entrySample); the caller queues it. Counted
|
||||
* in-process, so "1 in 5" is exactly that across a booth's day. */
|
||||
sampleEntry(): boolean {
|
||||
const n = this.#cfg?.entrySample ?? 0;
|
||||
if (n <= 0) return false;
|
||||
this.#entrySeen += 1;
|
||||
return this.#entrySeen % n === 0;
|
||||
}
|
||||
|
||||
/** Queue an ENTRY sample: the crop and the camera's class only — no order, no operator,
|
||||
* no category. Pure training material in the gate view; the reviewer labels it. */
|
||||
async enqueueEntry(read: VehicleRead): Promise<boolean> {
|
||||
if (!this.#cfg) return false;
|
||||
return this.#queue(`entry:${read.snapshotId ?? "?"}`, read, (id, crop) => ({
|
||||
v: 1,
|
||||
kind: "entry",
|
||||
booth: this.#cfg!.boothId,
|
||||
item: id,
|
||||
at: new Date().toISOString(),
|
||||
vision: { class: read.bodyType, confidence: read.confidence },
|
||||
image: crop,
|
||||
}));
|
||||
}
|
||||
|
||||
async #queue(
|
||||
ref: string,
|
||||
read: VehicleRead,
|
||||
build: (id: string, image: { width: number; height: number; plateBlurred: boolean }) => Record<string, unknown>,
|
||||
): Promise<boolean> {
|
||||
if (!read.box || !read.snapshotId) return false;
|
||||
try {
|
||||
const snap = this.#db.select().from(snapshots).where(eq(snapshots.id, read.snapshotId)).get();
|
||||
if (!snap) {
|
||||
this.#logger.info(`carwash review: snapshot ${read.snapshotId} gone (pruned) — order ${item.orderId} not queued`);
|
||||
this.#logger.info(`carwash review: snapshot ${read.snapshotId} gone (pruned) — ${ref} not queued`);
|
||||
return false;
|
||||
}
|
||||
const crop = await makeReviewCrop(snap.bytes, read.box, read.plateBox);
|
||||
const id = randomUUID();
|
||||
const payload = {
|
||||
v: 1,
|
||||
booth: this.#cfg.boothId,
|
||||
item: id,
|
||||
order: item.orderId,
|
||||
at: item.createdAt,
|
||||
operator: operatorRef(this.#cfg.boothId, item.createdBy),
|
||||
operatorCategory: { id: item.categoryId, name: item.categoryName, classes: [...item.categoryClasses] },
|
||||
service: item.serviceName,
|
||||
vision: { class: read.bodyType, confidence: read.confidence, categoryId: item.visionCategoryId },
|
||||
downgraded: item.downgraded,
|
||||
image: { width: crop.width, height: crop.height, plateBlurred: crop.plateBlurred },
|
||||
};
|
||||
const payload = build(id, { width: crop.width, height: crop.height, plateBlurred: crop.plateBlurred });
|
||||
this.#db
|
||||
.insert(carwashReviewOutbox)
|
||||
.values({ id, orderId: item.orderId, createdAt: new Date().toISOString(), status: "queued", attempts: 0, nextAttemptAt: null, image: crop.bytes, payload })
|
||||
.values({ id, orderId: ref, createdAt: new Date().toISOString(), status: "queued", attempts: 0, nextAttemptAt: null, image: crop.bytes, payload })
|
||||
.run();
|
||||
return true;
|
||||
} catch (err) {
|
||||
this.#logger.warn(`carwash review: could not queue order ${item.orderId}: ${(err as Error).message}`);
|
||||
this.#logger.warn(`carwash review: could not queue ${ref}: ${(err as Error).message}`);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -320,6 +365,7 @@ export class ReviewOutbox {
|
||||
return {
|
||||
enabled: this.enabled,
|
||||
boothId: this.#cfg?.boothId ?? null,
|
||||
entrySample: this.#cfg?.entrySample ?? 0,
|
||||
queued: count("queued"),
|
||||
sent: count("sent"),
|
||||
failed: count("failed"),
|
||||
|
||||
@@ -42,7 +42,7 @@ export async function carwashRoutes(app: FastifyInstance, deps: ServerModuleDeps
|
||||
// The review outbox's health (Setup → Car wash): how many decisions wait for the
|
||||
// reviewer, how many went, the last error. Site admin's read.
|
||||
app.get("/api/carwash/review/status", { preHandler: settingsRead }, async () =>
|
||||
outbox?.status() ?? { enabled: false, boothId: null, queued: 0, sent: 0, failed: 0, lastSentAt: null, lastError: null },
|
||||
outbox?.status() ?? { enabled: false, boothId: null, queued: 0, sent: 0, failed: 0, lastSentAt: null, lastError: null, entrySample: 0 },
|
||||
);
|
||||
|
||||
app.put<{ Body: SettingsBody }>("/api/carwash/settings", { preHandler: settingsWrite }, async (req, reply) => {
|
||||
|
||||
@@ -210,7 +210,16 @@ async function recognizePlate(
|
||||
occurredAt: new Date().toISOString(),
|
||||
})
|
||||
.run();
|
||||
if (result.vehicle) logger.info(`vision vehicle '${result.vehicle.bodyType}' (${result.vehicle.confidence.toFixed(3)}) for ${identity}`);
|
||||
if (result.vehicle) {
|
||||
logger.info(`vision vehicle '${result.vehicle.bodyType}' (${result.vehicle.confidence.toFixed(3)}) for ${identity}`);
|
||||
if (vehicleBox) {
|
||||
deviceEvents.emitVehicleRead({
|
||||
identity,
|
||||
direction,
|
||||
read: { bodyType: result.vehicle.bodyType, confidence: result.vehicle.confidence, snapshotId, box: vehicleBox, plateBox },
|
||||
});
|
||||
}
|
||||
}
|
||||
if (!plate) return;
|
||||
logger.info(`anpr plate '${plate}' (${result.plate!.confidence.toFixed(3)}) for ${identity}`);
|
||||
// The session's entry/exit event already shipped without this (async) plate — tell the
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
.venv/
|
||||
**/__pycache__/
|
||||
.pytest_cache/
|
||||
.mypy_cache/
|
||||
.ruff_cache/
|
||||
out/
|
||||
.env
|
||||
@@ -0,0 +1,12 @@
|
||||
# Python
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
.venv/
|
||||
.mypy_cache/
|
||||
.pytest_cache/
|
||||
.ruff_cache/
|
||||
|
||||
# Model weights (fetched at deploy / first run, never committed — can be large + license-scoped)
|
||||
out/
|
||||
|
||||
*.onnx
|
||||
@@ -0,0 +1 @@
|
||||
3.12
|
||||
@@ -0,0 +1,49 @@
|
||||
# syntax=docker/dockerfile:1.7
|
||||
# Parking TRAINER image: the phase-B body-type classifier. Build CONTEXT is apps/trainer
|
||||
# (self-contained Python package). Runs on the reviewer's host (art-docker-station) beside
|
||||
# the collector, never on a booth: by default it SERVES the job API the collector's Training
|
||||
# section drives (`serve`); the same image runs the CLI one-off (`train`, `inspect`, …). It
|
||||
# reads the wash collector's volume (collector.sqlite + crops/) and writes versioned model
|
||||
# folders. CPU-only PyTorch — the host has no usable GPU and a few thousand crops train in
|
||||
# minutes/an hour on four Xeon cores.
|
||||
# See wiki/decisions/bodytype-classifier-training.md.
|
||||
|
||||
FROM ghcr.io/astral-sh/uv:python3.12-bookworm-slim AS base
|
||||
WORKDIR /app
|
||||
ENV UV_LINK_MODE=copy \
|
||||
UV_COMPILE_BYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends libgl1 libglib2.0-0 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY pyproject.toml uv.lock .python-version ./
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen --no-install-project --no-dev --extra train
|
||||
|
||||
COPY trainer/ ./trainer/
|
||||
COPY README.md ./
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen --no-dev --extra train
|
||||
|
||||
# Pre-warm the ImageNet backbone weights INTO the image so a run needs no network (the
|
||||
# host has one, but a job that fetches at run time is a job that fails at 2 am). Best-effort:
|
||||
# without network at build time torchvision fetches lazily on the first run.
|
||||
ENV TORCH_HOME=/app/torch-home
|
||||
RUN uv run python -c "import torchvision.models as m; m.resnet18(weights=m.ResNet18_Weights.IMAGENET1K_V1); m.mobilenet_v3_small(weights=m.MobileNet_V3_Small_Weights.IMAGENET1K_V1)" \
|
||||
|| echo "[build] backbone weights not pre-warmed (no network) — fetched on first run"
|
||||
|
||||
RUN useradd --system --create-home --uid 999 trainer \
|
||||
&& mkdir -p /data /out && chown -R trainer:trainer /app /out
|
||||
USER trainer
|
||||
|
||||
ENV TRAINER_DATA_DIR=/data \
|
||||
TRAINER_OUT_DIR=/out \
|
||||
TRAINER_PORT=8091
|
||||
VOLUME ["/out"]
|
||||
EXPOSE 8091
|
||||
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \
|
||||
CMD python -c "import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://localhost:8091/health').status==200 else 1)" || exit 1
|
||||
ENTRYPOINT ["uv", "run", "--no-sync", "parking-trainer"]
|
||||
CMD ["serve"]
|
||||
@@ -0,0 +1,36 @@
|
||||
# parking-trainer
|
||||
|
||||
The phase-B **body-type classifier** job. Reads the wash collector's volume
|
||||
(`collector.sqlite` + `crops/`), trains a classifier on the reviewer's labels, and writes a
|
||||
versioned model folder the vision image bakes in — or refuses when validation is below the
|
||||
floor. Design and decisions: `wiki/decisions/bodytype-classifier-training.md`.
|
||||
|
||||
```
|
||||
parking-trainer inspect --data /data # what a run would train on
|
||||
parking-trainer train --data /data --out /out # features mode (minutes)
|
||||
parking-trainer train --mode finetune --epochs 12 ... # full fine-tune (about an hour on 4 cores)
|
||||
parking-trainer evaluate --model /out/<version>/bodytype.onnx --data /data
|
||||
parking-trainer publish /out/<version> --url https://git.infra.msai.al/api/packages/mca/generic/parking-bodytype
|
||||
```
|
||||
|
||||
Exit codes: `0` model written · `2` not enough labels · `3` below the floor (report written,
|
||||
no model) · `1` other.
|
||||
|
||||
A passing run writes `<out>/<version>/`:
|
||||
|
||||
| file | what |
|
||||
| --- | --- |
|
||||
| `bodytype.onnx` | the classifier; input `image` = RGB float32 0–255 `[N,3,S,S]`, output `logits` `[N,K]`; normalisation is inside the graph |
|
||||
| `bodytype.json` | sidecar: version, class list (in vocabulary order), input size, crop margin, backbone, mode, label counts, validation metrics |
|
||||
| `report.md` | the human report: accuracy, per-class recall/precision, confusion matrix, dropped classes, loss weights |
|
||||
| `metrics.json` | the same numbers, machine-readable |
|
||||
|
||||
On the reviewer's host the image runs `serve` as the `trainer` service of the
|
||||
`wash-collector` stack: a job API (`/health`, `/readiness`, `/versions`, `/jobs`) on the compose
|
||||
network that the collector's **Training section** (`/review`) drives — readiness, Train /
|
||||
Evaluate / Publish, reports and logs. Jobs run as subprocesses of the CLI, one at a time; state
|
||||
and logs persist under `/out/jobs/`. The CLI stays for debugging:
|
||||
`docker compose -f docker-compose.collector.yml exec trainer parking-trainer inspect`.
|
||||
|
||||
Local dev: `uv sync --extra train` (CPU torch, ~200 MB), `uv run pytest -q`. The test suite
|
||||
runs without the extra (torch tests skip), matching CI.
|
||||
@@ -0,0 +1,13 @@
|
||||
{
|
||||
"name": "@parking/trainer",
|
||||
"version": "0.0.0",
|
||||
"private": true,
|
||||
"//": "Thin shim so this Python job is a node in the Turbo task graph (NOT a JS package — deps are managed by uv/pyproject.toml). It is a one-off job image, never a booth service: see wiki/decisions/bodytype-classifier-training.md.",
|
||||
"scripts": {
|
||||
"lint": "uv run ruff check .",
|
||||
"format": "uv run ruff format .",
|
||||
"typecheck": "uv run mypy trainer",
|
||||
"test": "uv run pytest -q",
|
||||
"build": "echo 'no build step (Python job; see Dockerfile)'"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
[project]
|
||||
name = "parking-trainer"
|
||||
version = "0.0.0"
|
||||
description = "Phase-B body-type classifier trainer: reviewer labels + crops off the wash collector's volume → an ONNX classifier the vision image bakes in."
|
||||
requires-python = ">=3.10,<4.0"
|
||||
# Core deps are LIGHT on purpose (same rule as the vision service): `inspect`, `evaluate`
|
||||
# and the data/report code run with only these, so `uv sync` and the test suite work
|
||||
# in CI without the PyTorch stack. Training itself needs the `train` extra.
|
||||
# See wiki/decisions/bodytype-classifier-training.md.
|
||||
dependencies = [
|
||||
"numpy>=1.26",
|
||||
# OpenCV does the decode + resize on BOTH sides (trainer and vision service): same
|
||||
# library, same interpolation, same pixels — the preprocessing contract (preprocess.py).
|
||||
"opencv-python-headless>=4.10",
|
||||
"onnxruntime>=1.19",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
parking-trainer = "trainer.cli:main"
|
||||
|
||||
[project.optional-dependencies]
|
||||
# The training stack. CPU-only PyTorch (the reviewer's host has no usable GPU — the
|
||||
# decision is recorded in the wiki page above): resolved from PyTorch's CPU wheel index,
|
||||
# ~200 MB instead of the ~5 GB CUDA build. Install with: uv sync --extra train
|
||||
# torch / torchvision are BSD-3; the ImageNet backbone weights ship under the same
|
||||
# licence (the licence rule applies to weights as much as code).
|
||||
train = [
|
||||
"torch>=2.4",
|
||||
"torchvision>=0.19",
|
||||
"onnx>=1.16",
|
||||
"onnxscript>=0.3", # the torch.export-based ONNX exporter (MIT)
|
||||
]
|
||||
|
||||
[dependency-groups]
|
||||
dev = [
|
||||
"ruff>=0.8",
|
||||
"pytest>=8.3",
|
||||
"mypy>=1.13",
|
||||
]
|
||||
|
||||
[tool.uv]
|
||||
# Pick the CPU wheels for torch/torchvision from PyTorch's own index; everything else
|
||||
# from PyPI. `explicit = true` keeps the index from shadowing PyPI for other packages.
|
||||
[[tool.uv.index]]
|
||||
name = "pytorch-cpu"
|
||||
url = "https://download.pytorch.org/whl/cpu"
|
||||
explicit = true
|
||||
|
||||
[tool.uv.sources]
|
||||
torch = [{ index = "pytorch-cpu" }]
|
||||
torchvision = [{ index = "pytorch-cpu" }]
|
||||
|
||||
[tool.ruff]
|
||||
line-length = 110
|
||||
target-version = "py310"
|
||||
|
||||
[tool.ruff.lint]
|
||||
select = ["E", "F", "I", "B", "UP"]
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
testpaths = ["tests"]
|
||||
|
||||
[tool.mypy]
|
||||
python_version = "3.12"
|
||||
strict = true
|
||||
ignore_missing_imports = true
|
||||
|
||||
[build-system]
|
||||
requires = ["hatchling"]
|
||||
build-backend = "hatchling.build"
|
||||
|
||||
[tool.hatch.build.targets.wheel]
|
||||
packages = ["trainer"]
|
||||
@@ -0,0 +1,102 @@
|
||||
"""A synthetic collector volume: the collector's `items` table (same DDL as apps/collector
|
||||
src/db.ts) + JPEG crops. Classes are told apart by COLOUR so even a random-init backbone's
|
||||
features separate them — the tests check the plumbing (split, floor, export, sidecar),
|
||||
not accuracy on real cars."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sqlite3
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
DDL = """
|
||||
CREATE TABLE IF NOT EXISTS items (
|
||||
id TEXT PRIMARY KEY, booth TEXT NOT NULL, kind TEXT NOT NULL DEFAULT 'wash',
|
||||
order_ref TEXT NOT NULL, at TEXT NOT NULL, operator_ref TEXT NOT NULL DEFAULT '',
|
||||
operator_category_id TEXT NOT NULL DEFAULT '', operator_category_name TEXT NOT NULL DEFAULT '',
|
||||
operator_classes TEXT NOT NULL DEFAULT '[]', service TEXT NOT NULL, vision_class TEXT NOT NULL,
|
||||
vision_confidence REAL NOT NULL, vision_category_id TEXT, downgraded INTEGER NOT NULL DEFAULT 0,
|
||||
image_width INTEGER NOT NULL, image_height INTEGER NOT NULL, plate_blurred INTEGER NOT NULL,
|
||||
image_path TEXT NOT NULL, received_at TEXT NOT NULL, review_label TEXT, reviewed_at TEXT, reviewer TEXT
|
||||
);
|
||||
"""
|
||||
|
||||
COLOURS = {"sedan": (200, 40, 40), "suv": (40, 200, 40), "van": (40, 40, 200), "truck": (200, 200, 40)}
|
||||
|
||||
|
||||
def write_jpeg(path: Path, colour: tuple[int, int, int], rng: np.random.Generator) -> None:
|
||||
import cv2
|
||||
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
h, w = int(rng.integers(120, 200)), int(rng.integers(160, 260))
|
||||
img = np.empty((h, w, 3), np.uint8)
|
||||
img[:] = colour[::-1] # BGR
|
||||
noise = rng.integers(-20, 20, size=img.shape, dtype=np.int16)
|
||||
img = np.clip(img.astype(np.int16) + noise, 0, 255).astype(np.uint8)
|
||||
cv2.imwrite(str(path), img, [cv2.IMWRITE_JPEG_QUALITY, 85])
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def collector_dir(tmp_path: Path) -> Path:
|
||||
"""40 labelled crops per class for sedan/suv/van, 5 for truck (below the minimum), a few
|
||||
unusable, a few pending, one labelled row whose file is missing."""
|
||||
rng = np.random.default_rng(1)
|
||||
con = sqlite3.connect(tmp_path / "collector.sqlite")
|
||||
con.executescript(DDL)
|
||||
t0 = datetime(2026, 9, 1, tzinfo=timezone.utc)
|
||||
n = 0
|
||||
|
||||
def add(label: str | None, reviewed: bool, kind: str = "wash", missing: bool = False) -> None:
|
||||
nonlocal n
|
||||
n += 1
|
||||
item = f"item-{n:04d}"
|
||||
rel = f"crops/booth-2/{item}.jpg"
|
||||
colour = COLOURS.get(label or "sedan", (128, 128, 128))
|
||||
if not missing:
|
||||
write_jpeg(tmp_path / rel, colour, rng)
|
||||
at = (t0 + timedelta(minutes=10 * n)).isoformat().replace("+00:00", "Z")
|
||||
reviewed_at = (
|
||||
(t0 + timedelta(days=1, minutes=n)).isoformat().replace("+00:00", "Z") if reviewed else None
|
||||
)
|
||||
con.execute(
|
||||
"INSERT INTO items (id, booth, kind, order_ref, at, service, vision_class, vision_confidence, "
|
||||
"image_width, image_height, plate_blurred, image_path, received_at, review_label, "
|
||||
"reviewed_at, reviewer) "
|
||||
"VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
|
||||
(
|
||||
item,
|
||||
"booth-2",
|
||||
kind,
|
||||
"o",
|
||||
at,
|
||||
"wash",
|
||||
"car" if label != "truck" else "truck",
|
||||
0.9,
|
||||
200,
|
||||
150,
|
||||
1,
|
||||
rel,
|
||||
at,
|
||||
label if reviewed else None,
|
||||
reviewed_at,
|
||||
"reviewer" if reviewed else None,
|
||||
),
|
||||
)
|
||||
|
||||
# Interleaved in time so every class exists on both sides of the time split.
|
||||
for i in range(40):
|
||||
for label in ("sedan", "suv", "van"):
|
||||
add(label, True)
|
||||
if i % 8 == 0:
|
||||
add("truck", True)
|
||||
add("unusable", True)
|
||||
add("unusable", True)
|
||||
add("sedan", True, missing=True)
|
||||
for _ in range(6):
|
||||
add(None, False, kind="entry")
|
||||
con.commit()
|
||||
con.close()
|
||||
return tmp_path
|
||||
@@ -0,0 +1,98 @@
|
||||
"""Data rules, torch-free: labels, the time split, thin classes, weights, the report."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from trainer.cli import main
|
||||
from trainer.data import (
|
||||
class_weights,
|
||||
load_labelled,
|
||||
load_reviewed_since,
|
||||
load_unlabelled,
|
||||
make_split,
|
||||
summarise,
|
||||
)
|
||||
from trainer.preprocess import CROP_MARGIN, Sidecar, load_input
|
||||
from trainer.report import compute_metrics, render_report
|
||||
|
||||
|
||||
def test_loads_only_reviewed_usable_rows_with_a_crop_on_disk(collector_dir: Path) -> None:
|
||||
samples, missing = load_labelled(collector_dir)
|
||||
assert missing == 1 # the labelled row whose file is gone
|
||||
assert len(samples) == 125 # 3×40 + 5 trucks; unusable and pending excluded
|
||||
assert all(s.path.is_file() for s in samples)
|
||||
assert {s.label for s in samples} == {"sedan", "suv", "van", "truck"}
|
||||
assert summarise(samples)["byClass"] == {"sedan": 40, "suv": 40, "van": 40, "truck": 5}
|
||||
assert len(load_unlabelled(collector_dir)) == 6
|
||||
assert len(load_reviewed_since(collector_dir, "2026-09-02T00:00:00Z")) == 125
|
||||
assert load_reviewed_since(collector_dir, "2030-01-01T00:00:00Z") == []
|
||||
|
||||
|
||||
def test_split_is_by_time_and_drops_thin_classes(collector_dir: Path) -> None:
|
||||
samples, _ = load_labelled(collector_dir)
|
||||
split = make_split(samples, val_fraction=0.2, min_per_class=20)
|
||||
assert split.classes == ("sedan", "suv", "van") # canonical order, truck dropped
|
||||
assert split.dropped == {"truck": 5}
|
||||
assert len(split.train) + len(split.val) == 120
|
||||
assert len(split.val) == 24
|
||||
assert max(s.at for s in split.train) < min(s.at for s in split.val) # newest = validation
|
||||
assert all(v > 0 for v in split.counts("val").values())
|
||||
|
||||
|
||||
def test_class_weights_lean_against_imbalance_but_gently(collector_dir: Path) -> None:
|
||||
samples, _ = load_labelled(collector_dir)
|
||||
vans = [s for s in samples if s.label == "van"]
|
||||
keep = set(vans[::10]) # 4 of 40 vans survive
|
||||
split = make_split([s for s in samples if s.label != "van" or s in keep], 0.2, 3)
|
||||
w = dict(zip(split.classes, class_weights(split), strict=True))
|
||||
assert w["van"] > w["sedan"] > 0 # the rare class weighs more
|
||||
assert w["van"] / w["sedan"] < 4 # but not the full inverse ratio (damped)
|
||||
assert abs(sum(w.values()) / len(w) - 1.0) < 1e-9
|
||||
|
||||
|
||||
def test_metrics_and_report() -> None:
|
||||
classes = ("sedan", "suv")
|
||||
m = compute_metrics(classes, [0, 0, 1, 1], [0, 1, 1, 1], camera=["car"] * 4)
|
||||
assert m.accuracy == 0.75
|
||||
assert m.per_class["sedan"].recall == 0.5 and m.per_class["suv"].precision == 2 / 3
|
||||
assert m.confusion == [[1, 1], [0, 2]]
|
||||
assert m.camera_agreement == 0.0
|
||||
text = render_report(
|
||||
version="v1",
|
||||
trained_at="t",
|
||||
mode="features",
|
||||
backbone="resnet18",
|
||||
epochs=3,
|
||||
classes=classes,
|
||||
train_counts={"sedan": 10, "suv": 8},
|
||||
val_counts={"sedan": 2, "suv": 2},
|
||||
dropped={"truck": 2},
|
||||
missing_files=1,
|
||||
weights=[0.9, 1.1],
|
||||
metrics=m,
|
||||
min_accuracy=0.85,
|
||||
written=False,
|
||||
)
|
||||
assert "MODEL NOT WRITTEN" in text and "| **sedan** | 1 | 1 |" in text and "truck (2)" in text
|
||||
|
||||
|
||||
def test_preprocess_contract(tmp_path: Path, collector_dir: Path) -> None:
|
||||
samples, _ = load_labelled(collector_dir)
|
||||
x = load_input(samples[0].path, 32)
|
||||
assert x.shape == (3, 32, 32) and x.dtype.name == "float32" and 0 <= x.min() and x.max() <= 255
|
||||
assert x[0].mean() > x[2].mean() # a sedan crop is red: RGB order, not BGR
|
||||
assert load_input(tmp_path / "nope.jpg", 32) is None
|
||||
side = Sidecar(version="v1", classes=["sedan", "suv"])
|
||||
side.write(tmp_path / "s.json")
|
||||
back = Sidecar.read(tmp_path / "s.json")
|
||||
assert back == side and back.crop_margin == CROP_MARGIN == 0.08 and back.normalization == "in-graph"
|
||||
|
||||
|
||||
def test_inspect_prints_the_run_shape(collector_dir: Path, capsys) -> None: # type: ignore[no-untyped-def]
|
||||
assert main(["inspect", "--data", str(collector_dir)]) == 0
|
||||
out = json.loads(capsys.readouterr().out)
|
||||
assert out["ready"] is True and out["run"]["classes"] == ["sedan", "suv", "van"]
|
||||
assert out["run"]["dropped"] == {"truck": 5} and out["missingCrops"] == 1
|
||||
assert main(["inspect", "--data", str(collector_dir), "--min-per-class", "100"]) == 2
|
||||
@@ -0,0 +1,111 @@
|
||||
"""The job API: readiness, one job at a time, subprocess jobs with persisted logs, versions."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import threading
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from http.server import ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from trainer.server import Handler, Jobs, readiness, versions, wait_idle
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def api(collector_dir: Path, tmp_path: Path): # type: ignore[no-untyped-def]
|
||||
out = tmp_path / "out"
|
||||
Handler.jobs = Jobs(collector_dir, out, "https://example.invalid/pkg", "tok")
|
||||
httpd = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
|
||||
t = threading.Thread(target=httpd.serve_forever, daemon=True)
|
||||
t.start()
|
||||
base = f"http://127.0.0.1:{httpd.server_address[1]}"
|
||||
|
||||
def call(method: str, path: str, body: dict | None = None): # type: ignore[no-untyped-def]
|
||||
req = urllib.request.Request(base + path, method=method)
|
||||
data = None
|
||||
if body is not None:
|
||||
data = json.dumps(body).encode()
|
||||
req.add_header("Content-Type", "application/json")
|
||||
try:
|
||||
with urllib.request.urlopen(req, data=data, timeout=10) as r:
|
||||
raw = r.read()
|
||||
return r.status, (
|
||||
json.loads(raw) if r.headers.get_content_type() == "application/json" else raw.decode()
|
||||
)
|
||||
except urllib.error.HTTPError as e:
|
||||
return e.code, json.loads(e.read() or b"{}")
|
||||
|
||||
yield call, out
|
||||
httpd.shutdown()
|
||||
httpd.server_close()
|
||||
|
||||
|
||||
def test_readiness_and_empty_versions(api) -> None: # type: ignore[no-untyped-def]
|
||||
call, _ = api
|
||||
code, r = call("GET", "/readiness")
|
||||
assert code == 200 and r["ready"] is True and r["run"]["classes"] == ["sedan", "suv", "van"]
|
||||
assert r["defaults"]["minAccuracy"] == 0.85 and "finetune" in r["modes"]
|
||||
assert call("GET", "/versions") == (200, {"versions": []})
|
||||
assert call("GET", "/health")[1]["busy"] is False
|
||||
assert readiness(Path("/nonexistent"))["ready"] is False
|
||||
|
||||
|
||||
def test_evaluate_job_runs_as_a_subprocess_and_is_recorded(api) -> None: # type: ignore[no-untyped-def]
|
||||
call, out = api
|
||||
code, job = call("POST", "/jobs", {"kind": "evaluate", "version": "nope"})
|
||||
assert code == 202 and job["status"] == "running" and job["kind"] == "evaluate"
|
||||
wait_idle(Handler.jobs)
|
||||
code, j = call("GET", f"/jobs/{job['id']}")
|
||||
assert code == 200 and j["status"] == "failed" and j["exitCode"] == 1
|
||||
assert "evaluate --data" in j["log"] and "nope" in j["log"]
|
||||
assert (out / "jobs" / f"{job['id']}.json").is_file() and (out / "jobs" / f"{job['id']}.log").is_file()
|
||||
code, lst = call("GET", "/jobs")
|
||||
assert code == 200 and lst["jobs"][0]["id"] == job["id"] and lst["current"] is None
|
||||
|
||||
|
||||
def test_bad_requests(api) -> None: # type: ignore[no-untyped-def]
|
||||
call, _ = api
|
||||
assert call("POST", "/jobs", {"kind": "nuke"})[0] == 400
|
||||
assert call("POST", "/jobs", {"kind": "train", "mode": "magic"})[0] == 400
|
||||
assert call("POST", "/jobs", {"kind": "evaluate", "version": "../etc"})[0] == 400
|
||||
assert call("POST", "/jobs", {"kind": "publish", "version": "v1", "url": "ftp://x"})[0] == 400
|
||||
assert call("GET", "/versions/../x/report")[0] == 400
|
||||
assert call("GET", "/versions/v9/report")[0] == 404
|
||||
assert call("GET", "/jobs/nope")[0] == 404
|
||||
assert call("GET", "/nothing")[0] == 404
|
||||
|
||||
|
||||
def test_train_job_then_versions_and_report(api) -> None: # type: ignore[no-untyped-def]
|
||||
pytest.importorskip("torch")
|
||||
call, out = api
|
||||
body = {"kind": "train", "mode": "features", "minAccuracy": 0.0, "epochs": 100, "version": "vapi"}
|
||||
# The test-only flags are not offered by the API; inject them via the CLI args the runner builds.
|
||||
orig = Jobs._argv
|
||||
|
||||
def patched(self, kind, a): # type: ignore[no-untyped-def]
|
||||
argv = orig(self, kind, a)
|
||||
return argv + ["--no-pretrained", "--input-size", "64", "--no-cache"] if kind == "train" else argv
|
||||
|
||||
Jobs._argv = patched # type: ignore[method-assign]
|
||||
try:
|
||||
code, job = call("POST", "/jobs", body)
|
||||
assert code == 202
|
||||
assert call("POST", "/jobs", {"kind": "evaluate", "version": "vapi"})[0] == 409 # one at a time
|
||||
wait_idle(Handler.jobs, 120)
|
||||
finally:
|
||||
Jobs._argv = orig # type: ignore[method-assign]
|
||||
code, j = call("GET", f"/jobs/{job['id']}")
|
||||
assert j["status"] == "done" and "MODEL WRITTEN" in j["log"]
|
||||
code, v = call("GET", "/versions")
|
||||
assert code == 200 and v["versions"][0]["version"] == "vapi" and v["versions"][0]["written"] is True
|
||||
assert v["versions"][0]["classes"] == ["sedan", "suv", "van"] and v["versions"][0]["accuracy"] >= 0.9
|
||||
code, report = call("GET", "/versions/vapi/report")
|
||||
assert code == 200 and report.startswith("# Body-type classifier vapi")
|
||||
assert versions(out)[0]["floor"] == 0.0
|
||||
# evaluate on the written model now succeeds
|
||||
code, job2 = call("POST", "/jobs", {"kind": "evaluate", "version": "vapi"})
|
||||
wait_idle(Handler.jobs)
|
||||
assert call("GET", f"/jobs/{job2['id']}")[1]["status"] == "done"
|
||||
@@ -0,0 +1,155 @@
|
||||
"""The training job end to end on the synthetic volume — needs the `train` extra (torch);
|
||||
skipped where it is not installed (CI syncs without it, like the vision service)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
torch = pytest.importorskip("torch")
|
||||
|
||||
from trainer.cli import main # noqa: E402
|
||||
from trainer.infer import OnnxClassifier # noqa: E402
|
||||
from trainer.preprocess import Sidecar # noqa: E402
|
||||
|
||||
COMMON = ["--no-pretrained", "--input-size", "64", "--no-cache", "--seed", "3"]
|
||||
|
||||
|
||||
def test_features_run_writes_model_sidecar_report_and_evaluates(
|
||||
collector_dir: Path, tmp_path: Path, capsys
|
||||
) -> None: # type: ignore[no-untyped-def]
|
||||
out = tmp_path / "out"
|
||||
rc = main(
|
||||
[
|
||||
"train",
|
||||
"--data",
|
||||
str(collector_dir),
|
||||
"--out",
|
||||
str(out),
|
||||
"--version",
|
||||
"vtest",
|
||||
"--mode",
|
||||
"features",
|
||||
"--epochs",
|
||||
"150",
|
||||
"--min-accuracy",
|
||||
"0.0",
|
||||
*COMMON,
|
||||
]
|
||||
)
|
||||
assert rc == 0
|
||||
d = out / "vtest"
|
||||
assert {p.name for p in d.iterdir()} == {"bodytype.onnx", "bodytype.json", "report.md", "metrics.json"}
|
||||
side = Sidecar.read(d / "bodytype.json")
|
||||
assert side.classes == ["sedan", "suv", "van"] and side.input_size == 64 and side.mode == "features"
|
||||
assert side.labels == {"train": 96, "val": 24} and side.metrics["floor"] == 0.0
|
||||
metrics = json.loads((d / "metrics.json").read_text())
|
||||
assert metrics["n"] == 24 and metrics["onnx_agreement"] == 1.0
|
||||
# Colour-coded classes: even a random backbone's pooled features separate them.
|
||||
assert metrics["accuracy"] >= 0.9
|
||||
report = (d / "report.md").read_text()
|
||||
assert (
|
||||
"MODEL WRITTEN" in report
|
||||
and "truck (5)" in report
|
||||
and "crop is missing on disk (skipped): 1" in report
|
||||
)
|
||||
|
||||
# The exported graph takes raw 0–255 RGB and answers by itself.
|
||||
clf = OnnxClassifier(d / "bodytype.onnx")
|
||||
probs, kept = clf.predict_files(
|
||||
[s for s in sorted((collector_dir / "crops" / "booth-2").glob("*.jpg"))][:6]
|
||||
)
|
||||
assert probs.shape == (6, 3) and kept == [0, 1, 2, 3, 4, 5]
|
||||
assert np.allclose(probs.sum(axis=1), 1.0, atol=1e-4)
|
||||
|
||||
# evaluate: labels reviewed after training (none — the fixture's reviews predate it) and the
|
||||
# unlabelled pile (6 entry samples).
|
||||
capsys.readouterr()
|
||||
assert main(["evaluate", "--data", str(collector_dir), "--model", str(d / "bodytype.onnx")]) == 0
|
||||
res = json.loads(capsys.readouterr().out)
|
||||
assert res["model"] == "vtest" and res["reviewedSince"] is None
|
||||
assert res["unlabelled"]["n"] == 6 and sum(res["unlabelled"]["predicted"].values()) == 6
|
||||
assert (
|
||||
main(
|
||||
[
|
||||
"evaluate",
|
||||
"--data",
|
||||
str(collector_dir),
|
||||
"--model",
|
||||
str(d / "bodytype.onnx"),
|
||||
"--since",
|
||||
"2026-09-01T00:00:00Z",
|
||||
]
|
||||
)
|
||||
== 0
|
||||
)
|
||||
res2 = json.loads(capsys.readouterr().out)
|
||||
assert res2["reviewedSince"]["n"] == 125 - 5 # trucks are not a class the model knows
|
||||
|
||||
|
||||
def test_below_the_floor_writes_the_report_but_no_model(collector_dir: Path, tmp_path: Path) -> None:
|
||||
out = tmp_path / "out"
|
||||
rc = main(
|
||||
[
|
||||
"train",
|
||||
"--data",
|
||||
str(collector_dir),
|
||||
"--out",
|
||||
str(out),
|
||||
"--version",
|
||||
"vlow",
|
||||
"--mode",
|
||||
"features",
|
||||
"--epochs",
|
||||
"5",
|
||||
"--min-accuracy",
|
||||
"1.01",
|
||||
*COMMON,
|
||||
]
|
||||
)
|
||||
assert rc == 3
|
||||
d = out / "vlow"
|
||||
assert {p.name for p in d.iterdir()} == {"report.md", "metrics.json"}
|
||||
assert "MODEL NOT WRITTEN" in (d / "report.md").read_text()
|
||||
|
||||
|
||||
def test_not_enough_labels_is_exit_2(collector_dir: Path, tmp_path: Path) -> None:
|
||||
out = tmp_path / "out"
|
||||
rc = main(["train", "--data", str(collector_dir), "--out", str(out), "--min-per-class", "100", *COMMON])
|
||||
assert rc == 2
|
||||
assert not out.exists()
|
||||
|
||||
|
||||
def test_finetune_runs_and_uses_the_feature_cache(collector_dir: Path, tmp_path: Path) -> None:
|
||||
out = tmp_path / "out"
|
||||
args = [
|
||||
"train",
|
||||
"--data",
|
||||
str(collector_dir),
|
||||
"--out",
|
||||
str(out),
|
||||
"--mode",
|
||||
"finetune",
|
||||
"--backbone",
|
||||
"mobilenet_v3_small",
|
||||
"--epochs",
|
||||
"1",
|
||||
"--batch",
|
||||
"16",
|
||||
"--min-accuracy",
|
||||
"0.0",
|
||||
"--no-pretrained",
|
||||
"--input-size",
|
||||
"64",
|
||||
"--seed",
|
||||
"3",
|
||||
]
|
||||
assert main([*args, "--version", "vft"]) == 0
|
||||
cache = out / "cache" / "features-mobilenet_v3_small-64.npz"
|
||||
assert cache.exists()
|
||||
z = np.load(cache)
|
||||
assert len(z["ids"]) == 96 and z["feats"].shape == (96, 576)
|
||||
assert Sidecar.read(out / "vft" / "bodytype.json").mode == "finetune"
|
||||
@@ -0,0 +1,7 @@
|
||||
"""parking-trainer — the phase-B body-type classifier job.
|
||||
|
||||
Reads the wash collector's SQLite + crops straight off its volume, splits by TIME, trains a
|
||||
small classifier on a pretrained backbone, and writes the ONNX model + sidecar + report —
|
||||
or refuses to write the model when validation is below the owner's floor.
|
||||
See wiki/decisions/bodytype-classifier-training.md.
|
||||
"""
|
||||
@@ -0,0 +1,414 @@
|
||||
"""parking-trainer — inspect / train / evaluate / publish.
|
||||
|
||||
parking-trainer inspect --data /data
|
||||
parking-trainer train --data /data --out /out [--mode features|finetune] [--min-accuracy 0.85]
|
||||
parking-trainer evaluate --model /out/<version>/bodytype.onnx --data /data
|
||||
parking-trainer publish /out/<version> --url https://<gitea>/api/packages/<owner>/generic/parking-bodytype
|
||||
|
||||
Exit codes: 0 ok · 2 not enough labels · 3 trained but below the floor (report written, model
|
||||
NOT written) · 1 anything else.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
from .data import (
|
||||
VEHICLE_CLASSES,
|
||||
class_weights,
|
||||
load_labelled,
|
||||
load_reviewed_since,
|
||||
load_unlabelled,
|
||||
make_split,
|
||||
suggested_epochs,
|
||||
summarise,
|
||||
)
|
||||
from .preprocess import CROP_MARGIN, Sidecar
|
||||
from .report import compute_metrics, render_report
|
||||
|
||||
MODEL_FILE = "bodytype.onnx"
|
||||
SIDECAR_FILE = "bodytype.json"
|
||||
REPORT_FILE = "report.md"
|
||||
METRICS_FILE = "metrics.json"
|
||||
|
||||
|
||||
def _log(msg: str) -> None:
|
||||
print(f"[trainer] {msg}", file=sys.stderr, flush=True)
|
||||
|
||||
|
||||
def _now() -> str:
|
||||
return datetime.now(timezone.utc).replace(microsecond=0).isoformat().replace("+00:00", "Z")
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
# inspect
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def cmd_inspect(a: argparse.Namespace) -> int:
|
||||
samples, missing = load_labelled(a.data)
|
||||
split = make_split(samples, a.val_fraction, a.min_per_class)
|
||||
out = {
|
||||
"labelled": summarise(samples),
|
||||
"missingCrops": missing,
|
||||
"run": {
|
||||
"classes": list(split.classes),
|
||||
"train": split.counts("train"),
|
||||
"val": split.counts("val"),
|
||||
"dropped": split.dropped,
|
||||
"minPerClass": a.min_per_class,
|
||||
"valFraction": a.val_fraction,
|
||||
},
|
||||
"ready": len(split.classes) >= 2,
|
||||
}
|
||||
print(json.dumps(out, indent=2))
|
||||
return 0 if out["ready"] else 2
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
# train
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def cmd_train(a: argparse.Namespace) -> int:
|
||||
try:
|
||||
from . import model as M
|
||||
except ImportError as exc: # torch missing
|
||||
_log(f"the training stack is not installed ({exc}); install with: uv sync --extra train")
|
||||
return 1
|
||||
import numpy as np
|
||||
|
||||
samples, missing = load_labelled(a.data)
|
||||
split = make_split(samples, a.val_fraction, a.min_per_class)
|
||||
if len(split.classes) < 2:
|
||||
_log(
|
||||
f"not enough labels: {len(samples)} usable, classes with >= {a.min_per_class}: "
|
||||
f"{list(split.classes)} (dropped {split.dropped}); nothing to train"
|
||||
)
|
||||
return 2
|
||||
version = a.version or datetime.now(timezone.utc).strftime("v%Y%m%d-%H%M")
|
||||
out_dir = a.out / version
|
||||
epochs = a.epochs or suggested_epochs(len(split.train), a.mode)
|
||||
weights = class_weights(split)
|
||||
idx = split.class_index
|
||||
_log(
|
||||
f"{version}: {a.mode} on {a.backbone}, classes {list(split.classes)}, "
|
||||
f"{len(split.train)} train / {len(split.val)} val, {epochs} epochs"
|
||||
)
|
||||
|
||||
t0 = time.perf_counter()
|
||||
x_train, train = M.load_images(split.train, a.input_size)
|
||||
x_val, val = M.load_images(split.val, a.input_size)
|
||||
y_train = [idx[s.label] for s in train]
|
||||
y_val = [idx[s.label] for s in val]
|
||||
_log(f"decoded {len(train)} + {len(val)} crops in {time.perf_counter() - t0:.1f}s")
|
||||
if len(val) == 0 or len(set(y_train)) < 2:
|
||||
_log("not enough decodable crops on both sides of the split")
|
||||
return 2
|
||||
|
||||
backbone = M.build_backbone(a.backbone, pretrained=not a.no_pretrained)
|
||||
cache = (
|
||||
None if a.no_cache else M.FeatureCache(a.out / "cache" / f"features-{a.backbone}-{a.input_size}.npz")
|
||||
)
|
||||
t0 = time.perf_counter()
|
||||
f_train = M.features_for(backbone, train, x_train, cache)
|
||||
_log(
|
||||
f"features for {len(train)} train crops in {time.perf_counter() - t0:.1f}s "
|
||||
f"(cache: {cache.path if cache else 'off'})"
|
||||
)
|
||||
head_epochs = epochs if a.mode == "features" else max(30, epochs * 5)
|
||||
head = M.train_head(f_train, y_train, len(split.classes), weights, head_epochs, seed=a.seed)
|
||||
net = M.Classifier.make(backbone, head)
|
||||
|
||||
if a.mode == "finetune":
|
||||
t0 = time.perf_counter()
|
||||
net = M.finetune(
|
||||
net, x_train, y_train, weights, epochs, batch=a.batch, lr=a.lr, seed=a.seed, log=_log
|
||||
)
|
||||
_log(f"fine-tuned in {(time.perf_counter() - t0) / 60:.1f} min")
|
||||
|
||||
logits = M.predict_logits(net, x_val)
|
||||
y_pred = logits.argmax(axis=1).tolist()
|
||||
metrics = compute_metrics(split.classes, y_val, y_pred, camera=[s.vision_class for s in val])
|
||||
|
||||
# Export and check the graph gives the same answers as the torch model.
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
tmp_model = out_dir / (MODEL_FILE + ".tmp")
|
||||
M.export_onnx(net, a.input_size, tmp_model)
|
||||
from .infer import OnnxClassifier
|
||||
|
||||
sidecar = Sidecar(
|
||||
version=version,
|
||||
classes=list(split.classes),
|
||||
input_size=a.input_size,
|
||||
backbone=a.backbone,
|
||||
mode=a.mode,
|
||||
trained_at=_now(),
|
||||
labels={"train": len(train), "val": len(val)},
|
||||
)
|
||||
tmp_side = out_dir / (SIDECAR_FILE + ".tmp")
|
||||
sidecar.write(tmp_side)
|
||||
onnx_pred = OnnxClassifier(tmp_model, tmp_side).predict_inputs(x_val.astype(np.float32)).argmax(axis=1)
|
||||
metrics.onnx_agreement = float((onnx_pred == np.array(y_pred)).mean()) if len(y_pred) else None
|
||||
sidecar.metrics = {
|
||||
"accuracy": metrics.accuracy,
|
||||
"macroRecall": metrics.macro_recall,
|
||||
"perClass": {c: m.__dict__ for c, m in metrics.per_class.items()},
|
||||
"floor": a.min_accuracy,
|
||||
}
|
||||
|
||||
written = metrics.accuracy >= a.min_accuracy and (metrics.onnx_agreement or 0.0) >= 0.99
|
||||
notes = []
|
||||
if metrics.onnx_agreement is not None and metrics.onnx_agreement < 0.99:
|
||||
notes.append(
|
||||
f"ONNX export disagrees with the torch model ({metrics.onnx_agreement:.3f}); model withheld"
|
||||
)
|
||||
report = render_report(
|
||||
version=version,
|
||||
trained_at=sidecar.trained_at,
|
||||
mode=a.mode,
|
||||
backbone=a.backbone,
|
||||
epochs=epochs,
|
||||
classes=split.classes,
|
||||
train_counts=split.counts("train"),
|
||||
val_counts=split.counts("val"),
|
||||
dropped=split.dropped,
|
||||
missing_files=missing,
|
||||
weights=weights,
|
||||
metrics=metrics,
|
||||
min_accuracy=a.min_accuracy,
|
||||
written=written,
|
||||
notes=notes,
|
||||
)
|
||||
(out_dir / REPORT_FILE).write_text(report)
|
||||
(out_dir / METRICS_FILE).write_text(json.dumps(metrics.to_dict(), indent=2) + "\n")
|
||||
if written:
|
||||
sidecar.write(out_dir / SIDECAR_FILE)
|
||||
tmp_model.replace(out_dir / MODEL_FILE)
|
||||
tmp_side.unlink(missing_ok=True)
|
||||
_log(
|
||||
f"MODEL WRITTEN: {out_dir / MODEL_FILE} "
|
||||
f"(accuracy {metrics.accuracy:.3f} >= floor {a.min_accuracy})"
|
||||
)
|
||||
else:
|
||||
tmp_model.unlink(missing_ok=True)
|
||||
tmp_side.unlink(missing_ok=True)
|
||||
_log(
|
||||
f"MODEL NOT WRITTEN: accuracy {metrics.accuracy:.3f} < floor {a.min_accuracy}; "
|
||||
f"see {out_dir / REPORT_FILE}"
|
||||
)
|
||||
print(report)
|
||||
return 0 if written else 3
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
# evaluate — an existing model against labels that arrived AFTER it was trained, and its
|
||||
# view of the unlabelled pile (the ongoing accuracy check without labelling everything)
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def cmd_evaluate(a: argparse.Namespace) -> int:
|
||||
from .infer import OnnxClassifier
|
||||
|
||||
clf = OnnxClassifier(a.model)
|
||||
since = a.since or clf.sidecar.trained_at
|
||||
classes = clf.classes
|
||||
result: dict[str, object] = {"model": clf.sidecar.version, "classes": classes, "since": since}
|
||||
|
||||
reviewed = [s for s in load_reviewed_since(a.data, since) if s.label in classes]
|
||||
if reviewed:
|
||||
probs, kept = clf.predict_files([s.path for s in reviewed])
|
||||
rows = [reviewed[i] for i in kept]
|
||||
y_true = [classes.index(s.label) for s in rows]
|
||||
y_pred = probs.argmax(axis=1).tolist()
|
||||
m = compute_metrics(classes, y_true, y_pred, camera=[s.vision_class for s in rows])
|
||||
result["reviewedSince"] = m.to_dict()
|
||||
else:
|
||||
result["reviewedSince"] = None
|
||||
|
||||
pending = load_unlabelled(a.data, a.limit)
|
||||
if pending:
|
||||
probs, kept = clf.predict_files([s.path for s in pending])
|
||||
rows = [pending[i] for i in kept]
|
||||
pred = probs.argmax(axis=1)
|
||||
conf = probs.max(axis=1)
|
||||
hist = {c: int((pred == i).sum()) for i, c in enumerate(classes)}
|
||||
result["unlabelled"] = {
|
||||
"n": len(rows),
|
||||
"predicted": hist,
|
||||
"meanConfidence": float(conf.mean()) if len(rows) else None,
|
||||
"belowHalf": int((conf < 0.5).sum()),
|
||||
"agreesWithDetector": float(
|
||||
sum(1 for p, s in zip(pred, rows, strict=True) if classes[int(p)] == s.vision_class)
|
||||
/ len(rows)
|
||||
)
|
||||
if rows
|
||||
else None,
|
||||
}
|
||||
else:
|
||||
result["unlabelled"] = None
|
||||
print(json.dumps(result, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
# publish — the versioned files to a Gitea generic package (weights are not code, they
|
||||
# do not live in git; the vision image fetches them by URL at build)
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def cmd_publish(a: argparse.Namespace) -> int:
|
||||
d: Path = a.dir
|
||||
files = [d / MODEL_FILE, d / SIDECAR_FILE, d / REPORT_FILE, d / METRICS_FILE]
|
||||
for f in files[:2]:
|
||||
if not f.exists():
|
||||
_log(f"{f} missing — nothing to publish (a run below the floor writes no model)")
|
||||
return 1
|
||||
version = Sidecar.read(d / SIDECAR_FILE).version
|
||||
if not a.url:
|
||||
_log("no publish url: pass --url or set TRAINER_PUBLISH_URL")
|
||||
return 1
|
||||
token = a.token or os.environ.get("TRAINER_PUBLISH_TOKEN", "")
|
||||
if not token:
|
||||
_log("no token: pass --token or set TRAINER_PUBLISH_TOKEN")
|
||||
return 1
|
||||
base = a.url.rstrip("/") + "/" + version
|
||||
for f in files:
|
||||
if not f.exists():
|
||||
continue
|
||||
req = urllib.request.Request(f"{base}/{f.name}", data=f.read_bytes(), method="PUT")
|
||||
req.add_header("Authorization", f"token {token}")
|
||||
req.add_header("Content-Type", "application/octet-stream")
|
||||
with urllib.request.urlopen(req, timeout=120) as r:
|
||||
_log(f"PUT {base}/{f.name} → {r.status}")
|
||||
_log(f"published {version}; pin it in apps/vision/models/bodytype.version and rebuild the vision image")
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_serve(a: argparse.Namespace) -> int:
|
||||
from .server import serve
|
||||
|
||||
serve(
|
||||
a.data,
|
||||
a.out,
|
||||
a.host,
|
||||
a.port,
|
||||
os.environ.get("TRAINER_PUBLISH_URL", ""),
|
||||
os.environ.get("TRAINER_PUBLISH_TOKEN", ""),
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
p = argparse.ArgumentParser(
|
||||
prog="parking-trainer", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
|
||||
)
|
||||
sub = p.add_subparsers(dest="cmd", required=True)
|
||||
|
||||
def data_args(sp: argparse.ArgumentParser) -> None:
|
||||
sp.add_argument(
|
||||
"--data",
|
||||
type=Path,
|
||||
default=Path(os.environ.get("TRAINER_DATA_DIR", "/data")),
|
||||
help="collector volume (collector.sqlite + crops/)",
|
||||
)
|
||||
|
||||
def split_args(sp: argparse.ArgumentParser) -> None:
|
||||
sp.add_argument(
|
||||
"--min-per-class",
|
||||
type=int,
|
||||
default=20,
|
||||
help="classes with fewer reviewed crops are dropped from the run",
|
||||
)
|
||||
sp.add_argument(
|
||||
"--val-fraction", type=float, default=0.2, help="newest fraction held out for validation"
|
||||
)
|
||||
|
||||
i = sub.add_parser("inspect", help="what a run would train on")
|
||||
data_args(i)
|
||||
split_args(i)
|
||||
i.set_defaults(fn=cmd_inspect)
|
||||
|
||||
t = sub.add_parser("train", help="train, evaluate, export (or refuse)")
|
||||
data_args(t)
|
||||
split_args(t)
|
||||
t.add_argument(
|
||||
"--out",
|
||||
type=Path,
|
||||
default=Path(os.environ.get("TRAINER_OUT_DIR", "/out")),
|
||||
help="output root; a <version>/ folder is created under it",
|
||||
)
|
||||
t.add_argument("--mode", choices=["features", "finetune"], default="features")
|
||||
t.add_argument(
|
||||
"--backbone", choices=["resnet18", "mobilenet_v3_small", "efficientnet_b0"], default="resnet18"
|
||||
)
|
||||
t.add_argument("--epochs", type=int, default=0, help="0 = pick from the data size")
|
||||
t.add_argument("--batch", type=int, default=32)
|
||||
t.add_argument("--lr", type=float, default=1e-4, help="fine-tune learning rate")
|
||||
t.add_argument("--input-size", type=int, default=224)
|
||||
t.add_argument(
|
||||
"--min-accuracy", type=float, default=0.85, help="validation floor below which NO model is written"
|
||||
)
|
||||
t.add_argument("--version", default="", help="model version (default v<date>-<time>)")
|
||||
t.add_argument("--seed", type=int, default=7)
|
||||
t.add_argument(
|
||||
"--no-pretrained", action="store_true", help="random init (tests only — never for a real run)"
|
||||
)
|
||||
t.add_argument("--no-cache", action="store_true", help="do not read/write the feature cache")
|
||||
t.set_defaults(fn=cmd_train)
|
||||
|
||||
e = sub.add_parser(
|
||||
"evaluate", help="an existing model vs labels reviewed after it was trained + the unlabelled pile"
|
||||
)
|
||||
data_args(e)
|
||||
e.add_argument(
|
||||
"--model", type=Path, required=True, help="path to bodytype.onnx (sidecar .json beside it)"
|
||||
)
|
||||
e.add_argument("--since", default="", help="ISO time; default = the model's trained_at")
|
||||
e.add_argument(
|
||||
"--limit", type=int, default=2000, help="how many unlabelled crops to score (newest first)"
|
||||
)
|
||||
e.set_defaults(fn=cmd_evaluate)
|
||||
|
||||
u = sub.add_parser("publish", help="PUT a version folder to a Gitea generic package")
|
||||
u.add_argument("dir", type=Path, help="the <version>/ folder a passing run wrote")
|
||||
u.add_argument(
|
||||
"--url",
|
||||
default=os.environ.get("TRAINER_PUBLISH_URL", ""),
|
||||
help="https://<gitea>/api/packages/<owner>/generic/parking-bodytype (or TRAINER_PUBLISH_URL)",
|
||||
)
|
||||
u.add_argument("--token", default="", help="Gitea token with package:write (or TRAINER_PUBLISH_TOKEN)")
|
||||
u.set_defaults(fn=cmd_publish)
|
||||
|
||||
s = sub.add_parser("serve", help="the job API the collector's Training section talks to")
|
||||
data_args(s)
|
||||
s.add_argument("--out", type=Path, default=Path(os.environ.get("TRAINER_OUT_DIR", "/out")))
|
||||
s.add_argument("--host", default=os.environ.get("TRAINER_HOST", "0.0.0.0"))
|
||||
s.add_argument("--port", type=int, default=int(os.environ.get("TRAINER_PORT", "8091")))
|
||||
s.set_defaults(fn=cmd_serve)
|
||||
return p
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
a = build_parser().parse_args(argv)
|
||||
try:
|
||||
return int(a.fn(a))
|
||||
except FileNotFoundError as exc:
|
||||
_log(str(exc))
|
||||
return 1
|
||||
|
||||
|
||||
__all__ = ["main", "build_parser", "VEHICLE_CLASSES", "CROP_MARGIN"]
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,179 @@
|
||||
"""The training set: reviewed, usable rows off the collector's SQLite, and how they are
|
||||
split and weighed. Pure Python + sqlite3 — no torch, so `inspect` and the tests run light.
|
||||
|
||||
Rules (wiki/decisions/bodytype-classifier-training.md):
|
||||
- Only rows a REVIEWER labelled count; the operator's pick and the camera's class are
|
||||
never labels. `unusable` rows are dropped.
|
||||
- Split by TIME (`at` = when the vehicle was seen): validation = the newest slice, so
|
||||
the number reflects tomorrow's traffic rather than a random shuffle of the same days.
|
||||
- Classes with too few labels are dropped from the run (and reported), never trained
|
||||
on a handful of examples that would make the softmax confidently wrong.
|
||||
- Class imbalance is weighed in the loss (inverse frequency, damped) and reported.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import sqlite3
|
||||
from collections import Counter
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
# The shared vocabulary (packages/shared VEHICLE_CLASSES) — the only labels a reviewer can
|
||||
# give, and the only classes a model may emit. Order is the canonical one; the model's own
|
||||
# class list (sidecar) is the subset it trained on, in this order.
|
||||
VEHICLE_CLASSES: tuple[str, ...] = (
|
||||
"car",
|
||||
"sedan",
|
||||
"hatchback",
|
||||
"suv",
|
||||
"minivan",
|
||||
"pickup",
|
||||
"van",
|
||||
"truck",
|
||||
"bus",
|
||||
"motorcycle",
|
||||
)
|
||||
|
||||
DB_FILE = "collector.sqlite"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Sample:
|
||||
item: str
|
||||
booth: str
|
||||
kind: str # wash | entry
|
||||
at: str # ISO-8601, when the vehicle was seen (the split key)
|
||||
label: str # the reviewer's class
|
||||
vision_class: str # what the detector said (for the report, never a label)
|
||||
path: Path # the crop on disk
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Split:
|
||||
classes: tuple[str, ...]
|
||||
train: list[Sample]
|
||||
val: list[Sample]
|
||||
dropped: dict[str, int] # class → count, below the per-class minimum
|
||||
missing_files: int # labelled rows whose crop is not on disk
|
||||
|
||||
@property
|
||||
def class_index(self) -> dict[str, int]:
|
||||
return {c: i for i, c in enumerate(self.classes)}
|
||||
|
||||
def counts(self, part: str) -> dict[str, int]:
|
||||
rows = self.train if part == "train" else self.val
|
||||
c = Counter(s.label for s in rows)
|
||||
return {k: c.get(k, 0) for k in self.classes}
|
||||
|
||||
|
||||
_SELECT = "SELECT id, booth, kind, at, review_label, vision_class, image_path FROM items "
|
||||
|
||||
|
||||
def _rows(
|
||||
data_dir: Path, where: str, params: tuple[object, ...], db_file: Path | None = None
|
||||
) -> list[Sample]:
|
||||
db = db_file or data_dir / DB_FILE
|
||||
if not db.exists():
|
||||
raise FileNotFoundError(f"collector database not found: {db}")
|
||||
con = sqlite3.connect(f"file:{db}?mode=ro", uri=True)
|
||||
try:
|
||||
rows = con.execute(_SELECT + where, params).fetchall()
|
||||
finally:
|
||||
con.close()
|
||||
out: list[Sample] = []
|
||||
for item, booth, kind, at, label, vision_class, image_path in rows:
|
||||
out.append(Sample(item, booth, kind, at, label or "", vision_class, data_dir / image_path))
|
||||
return out
|
||||
|
||||
|
||||
def load_labelled(data_dir: Path, db_file: Path | None = None) -> tuple[list[Sample], int]:
|
||||
"""Every reviewed, usable row with its crop path resolved. Returns (samples, missing)
|
||||
where `missing` counts rows whose crop file is gone (pruned/moved) — skipped."""
|
||||
rows = _rows(
|
||||
data_dir,
|
||||
"WHERE reviewed_at IS NOT NULL AND review_label != 'unusable' ORDER BY at, id",
|
||||
(),
|
||||
db_file,
|
||||
)
|
||||
out: list[Sample] = []
|
||||
missing = 0
|
||||
for s in rows:
|
||||
if s.label not in VEHICLE_CLASSES:
|
||||
continue # a label outside the vocabulary can only be a future/foreign row
|
||||
if not s.path.is_file():
|
||||
missing += 1
|
||||
continue
|
||||
out.append(s)
|
||||
return out, missing
|
||||
|
||||
|
||||
def load_reviewed_since(data_dir: Path, since: str, db_file: Path | None = None) -> list[Sample]:
|
||||
"""Usable labels whose REVIEW happened after `since` (ISO) — a clean held-out check for a
|
||||
model trained before then. Rows whose crop is gone are skipped."""
|
||||
rows = _rows(
|
||||
data_dir,
|
||||
"WHERE reviewed_at IS NOT NULL AND review_label != 'unusable' AND reviewed_at > ? "
|
||||
"ORDER BY reviewed_at, id",
|
||||
(since,),
|
||||
db_file,
|
||||
)
|
||||
return [s for s in rows if s.label in VEHICLE_CLASSES and s.path.is_file()]
|
||||
|
||||
|
||||
def load_unlabelled(data_dir: Path, limit: int = 2000, db_file: Path | None = None) -> list[Sample]:
|
||||
"""The pending pile, newest first — what the model would say about traffic nobody has
|
||||
labelled (its class histogram and detector agreement are the cheap drift check)."""
|
||||
rows = _rows(data_dir, "WHERE reviewed_at IS NULL ORDER BY received_at DESC LIMIT ?", (limit,), db_file)
|
||||
return [s for s in rows if s.path.is_file()]
|
||||
|
||||
|
||||
def make_split(samples: list[Sample], val_fraction: float = 0.2, min_per_class: int = 20) -> Split:
|
||||
"""Drop thin classes, then cut by time: the newest `val_fraction` is validation."""
|
||||
if not 0.0 < val_fraction < 1.0:
|
||||
raise ValueError("val_fraction must be in (0, 1)")
|
||||
counts = Counter(s.label for s in samples)
|
||||
kept = tuple(c for c in VEHICLE_CLASSES if counts.get(c, 0) >= min_per_class)
|
||||
dropped = {c: n for c, n in counts.items() if c not in kept}
|
||||
rows = sorted((s for s in samples if s.label in kept), key=lambda s: (s.at, s.item))
|
||||
n_val = int(round(len(rows) * val_fraction))
|
||||
if rows and n_val == 0:
|
||||
n_val = 1
|
||||
cut = len(rows) - n_val
|
||||
return Split(classes=kept, train=rows[:cut], val=rows[cut:], dropped=dropped, missing_files=0)
|
||||
|
||||
|
||||
def class_weights(split: Split, damping: float = 0.5) -> list[float]:
|
||||
"""Inverse-frequency weights for the loss, damped by `damping` (0.5 = square root, so a
|
||||
1:9 imbalance becomes 1:3 rather than 1:9 — full inverse weights over-correct on small
|
||||
sets). Normalised to mean 1 so the learning rate keeps its meaning."""
|
||||
counts = split.counts("train")
|
||||
total = sum(counts.values())
|
||||
k = len(split.classes)
|
||||
raw = [(total / (k * max(1, counts[c]))) ** damping for c in split.classes]
|
||||
mean = sum(raw) / max(1, len(raw))
|
||||
return [w / mean for w in raw]
|
||||
|
||||
|
||||
def summarise(samples: list[Sample]) -> dict[str, object]:
|
||||
"""What `inspect` prints: per-class counts, per-booth counts, time range."""
|
||||
by_class = Counter(s.label for s in samples)
|
||||
by_booth = Counter(s.booth for s in samples)
|
||||
by_kind = Counter(s.kind for s in samples)
|
||||
ats = sorted(s.at for s in samples)
|
||||
return {
|
||||
"total": len(samples),
|
||||
"byClass": {c: by_class.get(c, 0) for c in VEHICLE_CLASSES if by_class.get(c, 0)},
|
||||
"byBooth": dict(sorted(by_booth.items())),
|
||||
"byKind": dict(sorted(by_kind.items())),
|
||||
"from": ats[0] if ats else None,
|
||||
"to": ats[-1] if ats else None,
|
||||
}
|
||||
|
||||
|
||||
def suggested_epochs(n_train: int, mode: str) -> int:
|
||||
"""A sane default when the owner gives none: enough passes for a small set, fewer as
|
||||
it grows. Feature-extraction heads converge fast; fine-tunes need more but cost more."""
|
||||
if mode == "features":
|
||||
return int(min(200, max(30, 4000 / max(1, n_train) * 10)))
|
||||
return int(min(30, max(8, math.ceil(3000 / max(1, n_train)) * 4)))
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Run an exported classifier (ONNX + sidecar) — torch-free. Used by `evaluate` and by
|
||||
the tests; the vision service carries its own, equivalent, reader (vehicle.py)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from .preprocess import Sidecar, load_input, softmax
|
||||
|
||||
|
||||
class OnnxClassifier:
|
||||
def __init__(self, model_path: Path, sidecar_path: Path | None = None) -> None:
|
||||
import onnxruntime as ort
|
||||
|
||||
self.model_path = Path(model_path)
|
||||
self.sidecar = Sidecar.read(sidecar_path or self.model_path.with_suffix(".json"))
|
||||
opts = ort.SessionOptions()
|
||||
opts.intra_op_num_threads = 2
|
||||
self._session = ort.InferenceSession(
|
||||
str(self.model_path), sess_options=opts, providers=["CPUExecutionProvider"]
|
||||
)
|
||||
self._input = self._session.get_inputs()[0].name
|
||||
|
||||
@property
|
||||
def classes(self) -> list[str]:
|
||||
return list(self.sidecar.classes)
|
||||
|
||||
def predict_inputs(self, x: Any, batch: int = 64) -> Any:
|
||||
"""[N,3,S,S] float32 → probabilities [N,K]."""
|
||||
import numpy as np
|
||||
|
||||
outs = []
|
||||
for i in range(0, len(x), batch):
|
||||
logits = self._session.run(None, {self._input: x[i : i + batch]})[0]
|
||||
outs.append(softmax(logits))
|
||||
return np.concatenate(outs, axis=0) if outs else np.zeros((0, len(self.classes)), np.float32)
|
||||
|
||||
def predict_files(self, paths: list[Path], batch: int = 64) -> tuple[Any, list[int]]:
|
||||
"""Decode + classify crop files. Returns (probs, indices of paths that decoded)."""
|
||||
import numpy as np
|
||||
|
||||
xs, kept = [], []
|
||||
for i, p in enumerate(paths):
|
||||
x = load_input(p, self.sidecar.input_size)
|
||||
if x is not None:
|
||||
xs.append(x)
|
||||
kept.append(i)
|
||||
if not xs:
|
||||
return np.zeros((0, len(self.classes)), np.float32), []
|
||||
return self.predict_inputs(np.stack(xs), batch), kept
|
||||
@@ -0,0 +1,305 @@
|
||||
"""The network and the two ways of training it. Imports torch — only `train` reaches here;
|
||||
everything else in the package stays torch-free (the `train` extra is heavy).
|
||||
|
||||
Backbone: a small ImageNet-pretrained torchvision model (BSD-3, weights included), used as
|
||||
a feature extractor; head: one linear layer over its pooled features.
|
||||
|
||||
- "features" mode: the backbone is FROZEN. Every crop goes through it once, the vectors
|
||||
are cached on disk (keyed by item id), and only the head is trained — minutes for a few
|
||||
thousand crops, seconds to retrain when new labels arrive. Expected to carry most of
|
||||
the accuracy on frontal gate views.
|
||||
- "finetune" mode: warm-starts the head the same way, then unfreezes everything and trains
|
||||
end to end with light augmentation. Roughly an hour on four Xeon cores for a few
|
||||
thousand crops with a mobile-sized backbone; the step when the cheap mode plateaus.
|
||||
|
||||
The exported ONNX graph takes raw RGB 0–255 float pixels and normalises INSIDE
|
||||
(see preprocess.py), so the vision service cannot get the constants wrong.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from .data import Sample
|
||||
from .preprocess import IMAGENET_MEAN, IMAGENET_STD, load_input
|
||||
|
||||
BACKBONES: dict[str, int] = {"resnet18": 512, "mobilenet_v3_small": 576, "efficientnet_b0": 1280}
|
||||
|
||||
|
||||
def _torch() -> Any:
|
||||
import torch
|
||||
|
||||
torch.set_num_threads(max(1, os.cpu_count() or 1))
|
||||
return torch
|
||||
|
||||
|
||||
def build_backbone(name: str, pretrained: bool = True) -> Any:
|
||||
"""torchvision model with its classifier removed → pooled feature vector."""
|
||||
torch = _torch()
|
||||
import torchvision.models as tvm
|
||||
|
||||
if name not in BACKBONES:
|
||||
raise ValueError(f"unknown backbone {name!r} (choose from {', '.join(BACKBONES)})")
|
||||
if name == "resnet18":
|
||||
m = tvm.resnet18(weights=tvm.ResNet18_Weights.IMAGENET1K_V1 if pretrained else None)
|
||||
m.fc = torch.nn.Identity()
|
||||
elif name == "mobilenet_v3_small":
|
||||
m = tvm.mobilenet_v3_small(
|
||||
weights=tvm.MobileNet_V3_Small_Weights.IMAGENET1K_V1 if pretrained else None
|
||||
)
|
||||
m.classifier = torch.nn.Identity()
|
||||
else:
|
||||
m = tvm.efficientnet_b0(weights=tvm.EfficientNet_B0_Weights.IMAGENET1K_V1 if pretrained else None)
|
||||
m.classifier = torch.nn.Identity()
|
||||
return m
|
||||
|
||||
|
||||
class Classifier: # a factory, not a Module subclass at import time (torch is lazy)
|
||||
@staticmethod
|
||||
def make(backbone: Any, head: Any) -> Any:
|
||||
torch = _torch()
|
||||
|
||||
class _Net(torch.nn.Module): # type: ignore[misc,name-defined]
|
||||
def __init__(self) -> None:
|
||||
super().__init__()
|
||||
self.backbone = backbone
|
||||
self.head = head
|
||||
self.register_buffer("mean", torch.tensor(IMAGENET_MEAN).view(1, 3, 1, 1))
|
||||
self.register_buffer("std", torch.tensor(IMAGENET_STD).view(1, 3, 1, 1))
|
||||
|
||||
def forward(self, x: Any) -> Any:
|
||||
x = (x / 255.0 - self.mean) / self.std
|
||||
return self.head(self.backbone(x))
|
||||
|
||||
return _Net()
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
# Images in memory (uint8 — a few thousand 224² crops is a few hundred MB; float32 would
|
||||
# be four times that)
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def load_images(samples: list[Sample], input_size: int) -> tuple[Any, list[Sample]]:
|
||||
"""Decode + resize every crop once. Returns (uint8 [N,3,S,S], the samples that decoded)."""
|
||||
import numpy as np
|
||||
|
||||
xs, kept = [], []
|
||||
for s in samples:
|
||||
x = load_input(s.path, input_size)
|
||||
if x is None:
|
||||
continue
|
||||
xs.append(x.astype(np.uint8))
|
||||
kept.append(s)
|
||||
if not xs:
|
||||
return np.zeros((0, 3, input_size, input_size), np.uint8), []
|
||||
return np.stack(xs), kept
|
||||
|
||||
|
||||
def _batches(n: int, batch: int) -> list[slice]:
|
||||
return [slice(i, min(n, i + batch)) for i in range(0, n, batch)]
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
# Feature extraction (+ on-disk cache) and the head
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
@dataclass
|
||||
class FeatureCache:
|
||||
"""`<cache_dir>/features-<backbone>-<size>.npz`: item ids + vectors. Retraining the head
|
||||
after new labels arrive only runs the backbone on the NEW crops."""
|
||||
|
||||
path: Path
|
||||
|
||||
def load(self) -> dict[str, Any]:
|
||||
import numpy as np
|
||||
|
||||
if not self.path.exists():
|
||||
return {}
|
||||
z = np.load(self.path, allow_pickle=False)
|
||||
return dict(zip(z["ids"].tolist(), z["feats"], strict=True))
|
||||
|
||||
def save(self, table: dict[str, Any]) -> None:
|
||||
import numpy as np
|
||||
|
||||
self.path.parent.mkdir(parents=True, exist_ok=True)
|
||||
ids = np.array(list(table), dtype=str)
|
||||
feats = np.stack(list(table.values())) if table else np.zeros((0, 0), np.float32)
|
||||
np.savez(self.path, ids=ids, feats=feats)
|
||||
|
||||
|
||||
def extract_features(backbone: Any, x_u8: Any, batch: int = 64) -> Any:
|
||||
"""Frozen forward pass → [N,D] float32 (normalisation applied here, as in the graph)."""
|
||||
torch = _torch()
|
||||
import numpy as np
|
||||
|
||||
net = Classifier.make(backbone, torch.nn.Identity()).eval()
|
||||
out = []
|
||||
with torch.no_grad():
|
||||
for sl in _batches(len(x_u8), batch):
|
||||
xb = torch.from_numpy(x_u8[sl]).float()
|
||||
out.append(net(xb).numpy())
|
||||
return np.concatenate(out, axis=0) if out else np.zeros((0, 0), np.float32)
|
||||
|
||||
|
||||
def features_for(
|
||||
backbone: Any, samples: list[Sample], x_u8: Any, cache: FeatureCache | None, batch: int = 64
|
||||
) -> Any:
|
||||
"""Feature vectors for `samples` (aligned with x_u8), from the cache where present."""
|
||||
import numpy as np
|
||||
|
||||
table = cache.load() if cache else {}
|
||||
todo = [i for i, s in enumerate(samples) if s.item not in table]
|
||||
if todo:
|
||||
fresh = extract_features(backbone, x_u8[todo], batch)
|
||||
for i, f in zip(todo, fresh, strict=True):
|
||||
table[samples[i].item] = f.astype(np.float32)
|
||||
if cache:
|
||||
cache.save(table)
|
||||
return np.stack([table[s.item] for s in samples]) if samples else np.zeros((0, 0), np.float32)
|
||||
|
||||
|
||||
def train_head(
|
||||
feats: Any,
|
||||
y: list[int],
|
||||
n_classes: int,
|
||||
weights: list[float],
|
||||
epochs: int,
|
||||
lr: float = 1e-3,
|
||||
seed: int = 7,
|
||||
) -> Any:
|
||||
"""Multinomial logistic regression on cached features (full batch, Adam, weighted CE)."""
|
||||
torch = _torch()
|
||||
torch.manual_seed(seed)
|
||||
f = torch.from_numpy(feats).float()
|
||||
t = torch.tensor(y, dtype=torch.long)
|
||||
head = torch.nn.Linear(f.shape[1], n_classes)
|
||||
opt = torch.optim.Adam(head.parameters(), lr=lr, weight_decay=1e-4)
|
||||
loss_fn = torch.nn.CrossEntropyLoss(weight=torch.tensor(weights, dtype=torch.float32))
|
||||
head.train()
|
||||
for _ in range(epochs):
|
||||
opt.zero_grad()
|
||||
loss = loss_fn(head(f), t)
|
||||
loss.backward()
|
||||
opt.step()
|
||||
return head.eval()
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
# Full fine-tune
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _augment(xb: Any) -> Any:
|
||||
"""Light, label-preserving augmentation on a float batch [B,3,S,S] (0–255): horizontal
|
||||
flip (a gate view mirrored is still the same body type), a mild random zoom, and
|
||||
brightness/contrast jitter (dusk, headlights, wet tarmac)."""
|
||||
torch = _torch()
|
||||
b, _, s, _ = xb.shape
|
||||
flip = torch.rand(b) < 0.5
|
||||
xb = torch.where(flip.view(b, 1, 1, 1), xb.flip(-1), xb)
|
||||
# zoom: crop a random 85–100 % window and resize back
|
||||
out = torch.empty_like(xb)
|
||||
for i in range(b):
|
||||
frac = float(torch.empty(1).uniform_(0.85, 1.0))
|
||||
w = max(8, int(s * frac))
|
||||
x0 = int(torch.randint(0, s - w + 1, (1,)))
|
||||
y0 = int(torch.randint(0, s - w + 1, (1,)))
|
||||
crop = xb[i : i + 1, :, y0 : y0 + w, x0 : x0 + w]
|
||||
out[i : i + 1] = torch.nn.functional.interpolate(
|
||||
crop, size=(s, s), mode="bilinear", align_corners=False
|
||||
)
|
||||
bright = torch.empty(b, 1, 1, 1).uniform_(-25, 25)
|
||||
contrast = torch.empty(b, 1, 1, 1).uniform_(0.8, 1.2)
|
||||
mean = out.mean(dim=(1, 2, 3), keepdim=True)
|
||||
out = (out - mean) * contrast + mean + bright
|
||||
return out.clamp_(0, 255)
|
||||
|
||||
|
||||
def finetune(
|
||||
model: Any,
|
||||
x_u8: Any,
|
||||
y: list[int],
|
||||
weights: list[float],
|
||||
epochs: int,
|
||||
batch: int = 32,
|
||||
lr: float = 1e-4,
|
||||
seed: int = 7,
|
||||
log: Any = None,
|
||||
) -> Any:
|
||||
torch = _torch()
|
||||
import numpy as np
|
||||
|
||||
torch.manual_seed(seed)
|
||||
rng = np.random.default_rng(seed)
|
||||
t = torch.tensor(y, dtype=torch.long)
|
||||
opt = torch.optim.AdamW(model.parameters(), lr=lr, weight_decay=1e-2)
|
||||
steps = epochs * max(1, (len(y) + batch - 1) // batch)
|
||||
sched = torch.optim.lr_scheduler.OneCycleLR(opt, max_lr=lr, total_steps=max(1, steps), pct_start=0.15)
|
||||
loss_fn = torch.nn.CrossEntropyLoss(weight=torch.tensor(weights, dtype=torch.float32))
|
||||
for epoch in range(epochs):
|
||||
model.train()
|
||||
order = rng.permutation(len(y))
|
||||
total = 0.0
|
||||
for sl in _batches(len(y), batch):
|
||||
idx = order[sl]
|
||||
xb = _augment(torch.from_numpy(x_u8[idx]).float())
|
||||
opt.zero_grad()
|
||||
loss = loss_fn(model(xb), t[idx])
|
||||
loss.backward()
|
||||
opt.step()
|
||||
sched.step()
|
||||
total += float(loss.detach()) * len(idx)
|
||||
if log:
|
||||
log(f"epoch {epoch + 1}/{epochs} loss {total / max(1, len(y)):.4f}")
|
||||
return model.eval()
|
||||
|
||||
|
||||
def predict_logits(model: Any, x_u8: Any, batch: int = 64) -> Any:
|
||||
torch = _torch()
|
||||
import numpy as np
|
||||
|
||||
model.eval()
|
||||
out = []
|
||||
with torch.no_grad():
|
||||
for sl in _batches(len(x_u8), batch):
|
||||
out.append(model(torch.from_numpy(x_u8[sl]).float()).numpy())
|
||||
return np.concatenate(out, axis=0) if out else np.zeros((0, 0), np.float32)
|
||||
|
||||
|
||||
def export_onnx(model: Any, input_size: int, path: Path) -> None:
|
||||
"""Export with the current (torch.export-based) exporter; fall back to the legacy
|
||||
TorchScript one where the new path is unavailable or trips over an op. Whichever wrote
|
||||
the graph, the job then checks it against the torch model (onnx_agreement) before the
|
||||
file is kept."""
|
||||
torch = _torch()
|
||||
|
||||
model.eval()
|
||||
dummy = torch.zeros(1, 3, input_size, input_size)
|
||||
names: dict[str, Any] = dict(input_names=["image"], output_names=["logits"])
|
||||
try:
|
||||
torch.onnx.export(
|
||||
model,
|
||||
(dummy,),
|
||||
str(path),
|
||||
dynamo=True,
|
||||
dynamic_shapes={"x": {0: "batch"}},
|
||||
opset_version=18,
|
||||
external_data=False, # ONE file: the vision image bakes bodytype.onnx + .json, nothing else
|
||||
verbose=False,
|
||||
**names,
|
||||
)
|
||||
except Exception: # noqa: BLE001 - any failure → the legacy exporter
|
||||
torch.onnx.export(
|
||||
model,
|
||||
dummy,
|
||||
str(path),
|
||||
dynamo=False,
|
||||
dynamic_axes={"image": {0: "batch"}, "logits": {0: "batch"}},
|
||||
opset_version=17,
|
||||
**names,
|
||||
)
|
||||
@@ -0,0 +1,88 @@
|
||||
"""Crop → model input. THE CONTRACT between the trainer and the vision service's classifier
|
||||
stage: what the network sees at training time must be exactly what it sees on the booth.
|
||||
|
||||
The trainer does not share code with the vision service (different packages, different
|
||||
images), so the contract is DATA: every constant here is written into the model's sidecar
|
||||
(`bodytype.json`) and the vision side reads and applies them from there — nothing is
|
||||
assumed on either side. Both use OpenCV with the same interpolation so the pixels match.
|
||||
|
||||
- input: the collector's crop (the detector's vehicle box + margin, plate blurred), or on
|
||||
the booth the same cut made live from the frame (vehicle.py mirrors `makeReviewCrop`).
|
||||
- resize: squash to input_size × input_size with INTER_AREA (the crop IS the vehicle; no
|
||||
centre-crop that would lose a bumper or a roofline — the shape is the signal).
|
||||
- colour: RGB, float32, 0–255. Normalisation (/255, ImageNet mean/std) lives INSIDE the
|
||||
ONNX graph, so a consumer feeds raw pixels and cannot get the constants wrong.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
SIDECAR_FORMAT = "parking-bodytype/1"
|
||||
IMAGENET_MEAN = (0.485, 0.456, 0.406)
|
||||
IMAGENET_STD = (0.229, 0.224, 0.225)
|
||||
CROP_MARGIN = 0.08 # must equal CROP_MARGIN in apps/server review-outbox.ts
|
||||
|
||||
|
||||
@dataclass
|
||||
class Sidecar:
|
||||
"""`bodytype.json` beside `bodytype.onnx`."""
|
||||
|
||||
version: str
|
||||
classes: list[str]
|
||||
input_size: int = 224
|
||||
color: str = "rgb"
|
||||
resize: str = "area"
|
||||
crop_margin: float = CROP_MARGIN
|
||||
normalization: str = "in-graph" # the ONNX divides by 255 and applies mean/std itself
|
||||
mean: list[float] = field(default_factory=lambda: list(IMAGENET_MEAN))
|
||||
std: list[float] = field(default_factory=lambda: list(IMAGENET_STD))
|
||||
backbone: str = ""
|
||||
mode: str = ""
|
||||
trained_at: str = ""
|
||||
labels: dict[str, int] = field(default_factory=dict) # train / val counts
|
||||
metrics: dict[str, Any] = field(default_factory=dict) # accuracy, macro, per-class
|
||||
format: str = SIDECAR_FORMAT
|
||||
|
||||
def write(self, path: Path) -> None:
|
||||
path.write_text(json.dumps(asdict(self), indent=2) + "\n")
|
||||
|
||||
@classmethod
|
||||
def read(cls, path: Path) -> Sidecar:
|
||||
d = json.loads(path.read_text())
|
||||
if d.get("format") != SIDECAR_FORMAT:
|
||||
raise ValueError(f"{path}: unknown sidecar format {d.get('format')!r}")
|
||||
known = {f for f in cls.__dataclass_fields__}
|
||||
return cls(**{k: v for k, v in d.items() if k in known})
|
||||
|
||||
|
||||
def load_input(path: Path, input_size: int) -> Any:
|
||||
"""Decode a crop and produce the network input: RGB float32 CHW, 0–255, squashed to
|
||||
input_size. Returns None when the file cannot be decoded."""
|
||||
import cv2
|
||||
|
||||
img = cv2.imread(str(path), cv2.IMREAD_COLOR)
|
||||
if img is None:
|
||||
return None
|
||||
return array_to_input(img, input_size)
|
||||
|
||||
|
||||
def array_to_input(bgr: Any, input_size: int) -> Any:
|
||||
"""BGR uint8 HWC (OpenCV's native) → RGB float32 CHW 0–255 at input_size."""
|
||||
import cv2
|
||||
import numpy as np
|
||||
|
||||
resized = cv2.resize(bgr, (input_size, input_size), interpolation=cv2.INTER_AREA)
|
||||
rgb = cv2.cvtColor(resized, cv2.COLOR_BGR2RGB)
|
||||
return np.ascontiguousarray(rgb.transpose(2, 0, 1).astype(np.float32))
|
||||
|
||||
|
||||
def softmax(logits: Any) -> Any:
|
||||
import numpy as np
|
||||
|
||||
z = logits - logits.max(axis=-1, keepdims=True)
|
||||
e = np.exp(z)
|
||||
return e / e.sum(axis=-1, keepdims=True)
|
||||
@@ -0,0 +1,152 @@
|
||||
"""Metrics + the human-readable report. The owner reads this BEFORE anything ships; the
|
||||
floor decision is made on these numbers. Pure Python, no torch."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from typing import Any
|
||||
|
||||
|
||||
@dataclass
|
||||
class ClassMetrics:
|
||||
support: int
|
||||
recall: float # of the true members, how many the model caught
|
||||
precision: float # of the model's picks, how many were right
|
||||
|
||||
|
||||
@dataclass
|
||||
class Metrics:
|
||||
n: int
|
||||
accuracy: float
|
||||
macro_recall: float
|
||||
per_class: dict[str, ClassMetrics]
|
||||
confusion: list[list[int]] # rows = true class, cols = predicted, in `classes` order
|
||||
camera_agreement: float | None = None # how often the model equals the detector's class
|
||||
onnx_agreement: float | None = None # exported graph vs the torch model, argmax
|
||||
extra: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
def compute_metrics(
|
||||
classes: tuple[str, ...] | list[str],
|
||||
y_true: list[int],
|
||||
y_pred: list[int],
|
||||
camera: list[str] | None = None,
|
||||
) -> Metrics:
|
||||
k = len(classes)
|
||||
conf = [[0] * k for _ in range(k)]
|
||||
for t, p in zip(y_true, y_pred, strict=True):
|
||||
conf[t][p] += 1
|
||||
per: dict[str, ClassMetrics] = {}
|
||||
recalls: list[float] = []
|
||||
for i, c in enumerate(classes):
|
||||
support = sum(conf[i])
|
||||
tp = conf[i][i]
|
||||
picked = sum(conf[r][i] for r in range(k))
|
||||
recall = tp / support if support else 0.0
|
||||
precision = tp / picked if picked else 0.0
|
||||
per[c] = ClassMetrics(support=support, recall=recall, precision=precision)
|
||||
if support:
|
||||
recalls.append(recall)
|
||||
n = len(y_true)
|
||||
acc = sum(1 for t, p in zip(y_true, y_pred, strict=True) if t == p) / n if n else 0.0
|
||||
agree = None
|
||||
if camera is not None and n:
|
||||
agree = sum(1 for p, cam in zip(y_pred, camera, strict=True) if classes[p] == cam) / n
|
||||
return Metrics(
|
||||
n=n,
|
||||
accuracy=acc,
|
||||
macro_recall=sum(recalls) / len(recalls) if recalls else 0.0,
|
||||
per_class=per,
|
||||
confusion=conf,
|
||||
camera_agreement=agree,
|
||||
)
|
||||
|
||||
|
||||
def _pct(x: float | None) -> str:
|
||||
return "—" if x is None else f"{x * 100:.1f} %"
|
||||
|
||||
|
||||
def render_report(
|
||||
*,
|
||||
version: str,
|
||||
trained_at: str,
|
||||
mode: str,
|
||||
backbone: str,
|
||||
epochs: int,
|
||||
classes: tuple[str, ...] | list[str],
|
||||
train_counts: dict[str, int],
|
||||
val_counts: dict[str, int],
|
||||
dropped: dict[str, int],
|
||||
missing_files: int,
|
||||
weights: list[float],
|
||||
metrics: Metrics,
|
||||
min_accuracy: float,
|
||||
written: bool,
|
||||
notes: list[str] | None = None,
|
||||
) -> str:
|
||||
lines: list[str] = []
|
||||
verdict = "MODEL WRITTEN" if written else "MODEL NOT WRITTEN — below the floor"
|
||||
lines.append(f"# Body-type classifier {version}")
|
||||
lines.append("")
|
||||
lines.append(
|
||||
f"**{verdict}** · validation accuracy {_pct(metrics.accuracy)} vs floor {_pct(min_accuracy)}"
|
||||
)
|
||||
lines.append("")
|
||||
lines.append(f"- trained: {trained_at}")
|
||||
lines.append(f"- mode: {mode} · backbone: {backbone} · epochs: {epochs}")
|
||||
lines.append(f"- classes ({len(classes)}): {', '.join(classes)}")
|
||||
lines.append(
|
||||
f"- labels: {sum(train_counts.values())} train · {sum(val_counts.values())} validation "
|
||||
"(validation = the NEWEST slice, by time seen)"
|
||||
)
|
||||
if dropped:
|
||||
lines.append(
|
||||
"- dropped (too few labels this run): "
|
||||
+ ", ".join(f"{c} ({n})" for c, n in sorted(dropped.items()))
|
||||
)
|
||||
if missing_files:
|
||||
lines.append(f"- labelled rows whose crop is missing on disk (skipped): {missing_files}")
|
||||
lines.append("")
|
||||
lines.append("## Validation")
|
||||
lines.append("")
|
||||
lines.append(f"- accuracy: {_pct(metrics.accuracy)} on {metrics.n} crops")
|
||||
lines.append(f"- macro recall (each class counted equally): {_pct(metrics.macro_recall)}")
|
||||
if metrics.camera_agreement is not None:
|
||||
lines.append(
|
||||
f"- agrees with the detector's coarse class: {_pct(metrics.camera_agreement)} "
|
||||
"(informational — the detector only knows car/truck/bus/motorcycle)"
|
||||
)
|
||||
if metrics.onnx_agreement is not None:
|
||||
lines.append(
|
||||
f"- exported ONNX matches the trained model on validation: {_pct(metrics.onnx_agreement)}"
|
||||
)
|
||||
lines.append("")
|
||||
lines.append("| class | train | val | recall | precision | loss weight |")
|
||||
lines.append("|---|---:|---:|---:|---:|---:|")
|
||||
for c, w in zip(classes, weights, strict=True):
|
||||
m = metrics.per_class[c]
|
||||
lines.append(
|
||||
f"| {c} | {train_counts.get(c, 0)} | {m.support} | {_pct(m.recall) if m.support else '—'} | "
|
||||
f"{_pct(m.precision) if m.support else '—'} | {w:.2f} |"
|
||||
)
|
||||
lines.append("")
|
||||
lines.append("## Confusion (rows = reviewer's label, columns = model)")
|
||||
lines.append("")
|
||||
lines.append("| | " + " | ".join(classes) + " |")
|
||||
lines.append("|---|" + "---:|" * len(classes))
|
||||
for c, row in zip(classes, metrics.confusion, strict=True):
|
||||
lines.append(f"| **{c}** | " + " | ".join(str(n) for n in row) + " |")
|
||||
lines.append("")
|
||||
if notes:
|
||||
lines.append("## Notes")
|
||||
lines.append("")
|
||||
lines.extend(f"- {n}" for n in notes)
|
||||
lines.append("")
|
||||
lines.append(
|
||||
"The flag on the booth records, it never bills: even a model that passes the floor is "
|
||||
"advisory (the site threshold gates the flag)."
|
||||
)
|
||||
return "\n".join(lines) + "\n"
|
||||
@@ -0,0 +1,382 @@
|
||||
"""`parking-trainer serve` — the job API behind the collector's Training section.
|
||||
|
||||
A tiny stdlib HTTP server (no framework, no extra deps) on the compose-internal network,
|
||||
never published: the collector proxies to it behind the reviewer's login. One job at a
|
||||
time; each job is the CLI run as a SUBPROCESS (`python -m trainer.cli …`) with its output
|
||||
captured to a log file — torch's memory goes away with the process, and a crashing job
|
||||
cannot take the service down. Job state + logs persist under `<out>/jobs/` so a restart
|
||||
still shows history.
|
||||
|
||||
GET /health {ok, busy, version}
|
||||
GET /readiness what `inspect` prints (+ the defaults the UI offers)
|
||||
GET /versions every <out>/<version>/ folder: written?, metrics, sidecar
|
||||
GET /versions/<v>/report report.md (text/markdown)
|
||||
GET /jobs recent jobs, newest first
|
||||
GET /jobs/<id> one job incl. the log tail
|
||||
POST /jobs {kind: train|evaluate|publish, …args} → 202 {id} | 409 busy
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from datetime import datetime, timezone
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from .cli import METRICS_FILE, MODEL_FILE, REPORT_FILE, SIDECAR_FILE
|
||||
from .data import load_labelled, make_split, summarise
|
||||
from .preprocess import Sidecar
|
||||
|
||||
_VERSION_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$")
|
||||
BACKBONES = ("resnet18", "mobilenet_v3_small", "efficientnet_b0")
|
||||
MODES = ("features", "finetune")
|
||||
LOG_TAIL_BYTES = 16_000
|
||||
|
||||
|
||||
def _now() -> str:
|
||||
return datetime.now(timezone.utc).replace(microsecond=0).isoformat().replace("+00:00", "Z")
|
||||
|
||||
|
||||
class Jobs:
|
||||
"""The single-slot job runner. `start` refuses while one runs."""
|
||||
|
||||
def __init__(self, data_dir: Path, out_dir: Path, publish_url: str, publish_token: str) -> None:
|
||||
self.data_dir = data_dir
|
||||
self.out_dir = out_dir
|
||||
self.publish_url = publish_url
|
||||
self.publish_token = publish_token
|
||||
self.jobs_dir = out_dir / "jobs"
|
||||
self.jobs_dir.mkdir(parents=True, exist_ok=True)
|
||||
self._lock = threading.Lock()
|
||||
self._current: dict[str, Any] | None = None
|
||||
self._proc: subprocess.Popen[bytes] | None = None
|
||||
|
||||
# ---- state -----------------------------------------------------------------------
|
||||
def _write(self, job: dict[str, Any]) -> None:
|
||||
(self.jobs_dir / f"{job['id']}.json").write_text(json.dumps(job, indent=2))
|
||||
|
||||
def _read(self, job_id: str) -> dict[str, Any] | None:
|
||||
p = self.jobs_dir / f"{job_id}.json"
|
||||
if not p.is_file():
|
||||
return None
|
||||
return json.loads(p.read_text()) # type: ignore[no-any-return]
|
||||
|
||||
def log_tail(self, job_id: str) -> str:
|
||||
p = self.jobs_dir / f"{job_id}.log"
|
||||
if not p.is_file():
|
||||
return ""
|
||||
size = p.stat().st_size
|
||||
with p.open("rb") as f:
|
||||
if size > LOG_TAIL_BYTES:
|
||||
f.seek(size - LOG_TAIL_BYTES)
|
||||
return f.read().decode("utf-8", "replace")
|
||||
|
||||
@property
|
||||
def busy(self) -> bool:
|
||||
return self._current is not None
|
||||
|
||||
def current(self) -> dict[str, Any] | None:
|
||||
return dict(self._current) if self._current else None
|
||||
|
||||
def get(self, job_id: str) -> dict[str, Any] | None:
|
||||
if self._current and self._current["id"] == job_id:
|
||||
job = dict(self._current)
|
||||
else:
|
||||
job = self._read(job_id) or {}
|
||||
if not job:
|
||||
return None
|
||||
job["log"] = self.log_tail(job_id)
|
||||
return job
|
||||
|
||||
def recent(self, limit: int = 20) -> list[dict[str, Any]]:
|
||||
files = sorted(self.jobs_dir.glob("*.json"), key=lambda p: p.stat().st_mtime, reverse=True)
|
||||
out = []
|
||||
for p in files[:limit]:
|
||||
try:
|
||||
out.append(json.loads(p.read_text()))
|
||||
except ValueError:
|
||||
continue
|
||||
if self._current and all(j["id"] != self._current["id"] for j in out):
|
||||
out.insert(0, dict(self._current))
|
||||
out.sort(key=lambda j: j.get("startedAt", ""), reverse=True)
|
||||
return out
|
||||
|
||||
# ---- args → CLI ------------------------------------------------------------------
|
||||
def _argv(self, kind: str, a: dict[str, Any]) -> list[str]:
|
||||
base = [sys.executable, "-m", "trainer.cli"]
|
||||
if kind == "train":
|
||||
mode = a.get("mode", "features")
|
||||
backbone = a.get("backbone", "resnet18")
|
||||
if mode not in MODES or backbone not in BACKBONES:
|
||||
raise ValueError("bad mode/backbone")
|
||||
floor = float(a.get("minAccuracy", 0.85))
|
||||
min_per = int(a.get("minPerClass", 20))
|
||||
epochs = int(a.get("epochs", 0))
|
||||
if not 0.0 <= floor <= 1.0 or min_per < 1 or epochs < 0:
|
||||
raise ValueError("bad numbers")
|
||||
argv = base + [
|
||||
"train",
|
||||
"--data",
|
||||
str(self.data_dir),
|
||||
"--out",
|
||||
str(self.out_dir),
|
||||
"--mode",
|
||||
mode,
|
||||
"--backbone",
|
||||
backbone,
|
||||
"--min-accuracy",
|
||||
str(floor),
|
||||
"--min-per-class",
|
||||
str(min_per),
|
||||
"--epochs",
|
||||
str(epochs),
|
||||
]
|
||||
if a.get("version"):
|
||||
argv += ["--version", self._version(a["version"])]
|
||||
return argv
|
||||
if kind == "evaluate":
|
||||
v = self._version(a.get("version", ""))
|
||||
return base + [
|
||||
"evaluate",
|
||||
"--data",
|
||||
str(self.data_dir),
|
||||
"--model",
|
||||
str(self.out_dir / v / MODEL_FILE),
|
||||
]
|
||||
if kind == "publish":
|
||||
v = self._version(a.get("version", ""))
|
||||
url = str(a.get("url") or self.publish_url)
|
||||
if not url.startswith("https://") and not url.startswith("http://"):
|
||||
raise ValueError("bad publish url")
|
||||
return base + ["publish", str(self.out_dir / v), "--url", url]
|
||||
raise ValueError("kind must be train, evaluate or publish")
|
||||
|
||||
@staticmethod
|
||||
def _version(v: Any) -> str:
|
||||
if not isinstance(v, str) or not _VERSION_RE.match(v) or v in ("cache", "jobs"):
|
||||
raise ValueError("bad version")
|
||||
return v
|
||||
|
||||
# ---- run -------------------------------------------------------------------------
|
||||
def start(self, kind: str, args: dict[str, Any]) -> dict[str, Any]:
|
||||
argv = self._argv(kind, args)
|
||||
with self._lock:
|
||||
if self._current is not None:
|
||||
raise RuntimeError("busy")
|
||||
job_id = f"{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')}-{uuid.uuid4().hex[:6]}"
|
||||
job: dict[str, Any] = {
|
||||
"id": job_id,
|
||||
"kind": kind,
|
||||
"args": {k: v for k, v in args.items() if k != "token"},
|
||||
"status": "running",
|
||||
"startedAt": _now(),
|
||||
"finishedAt": None,
|
||||
"exitCode": None,
|
||||
}
|
||||
env = dict(os.environ)
|
||||
if kind == "publish":
|
||||
env["TRAINER_PUBLISH_TOKEN"] = self.publish_token
|
||||
log = (self.jobs_dir / f"{job_id}.log").open("wb")
|
||||
log.write(f"$ {' '.join(argv[3:])}\n".encode())
|
||||
log.flush()
|
||||
self._proc = subprocess.Popen(argv, stdout=log, stderr=subprocess.STDOUT, env=env)
|
||||
self._current = job
|
||||
self._write(job)
|
||||
threading.Thread(target=self._wait, args=(job, log), daemon=True).start()
|
||||
return dict(job)
|
||||
|
||||
def _wait(self, job: dict[str, Any], log: Any) -> None:
|
||||
assert self._proc is not None
|
||||
code = self._proc.wait()
|
||||
log.close()
|
||||
with self._lock:
|
||||
job["exitCode"] = code
|
||||
job["finishedAt"] = _now()
|
||||
job["status"] = "done" if code == 0 else ("refused" if code in (2, 3) else "failed")
|
||||
self._write(job)
|
||||
self._current = None
|
||||
self._proc = None
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def readiness(data_dir: Path, min_per_class: int = 20, val_fraction: float = 0.2) -> dict[str, Any]:
|
||||
try:
|
||||
samples, missing = load_labelled(data_dir)
|
||||
except FileNotFoundError:
|
||||
return {
|
||||
"ready": False,
|
||||
"labelled": {"total": 0, "byClass": {}},
|
||||
"missingCrops": 0,
|
||||
"error": "no collector database yet",
|
||||
}
|
||||
split = make_split(samples, val_fraction, min_per_class)
|
||||
return {
|
||||
"ready": len(split.classes) >= 2,
|
||||
"labelled": summarise(samples),
|
||||
"missingCrops": missing,
|
||||
"run": {
|
||||
"classes": list(split.classes),
|
||||
"train": split.counts("train"),
|
||||
"val": split.counts("val"),
|
||||
"dropped": split.dropped,
|
||||
"minPerClass": min_per_class,
|
||||
"valFraction": val_fraction,
|
||||
},
|
||||
"defaults": {
|
||||
"mode": "features",
|
||||
"backbone": "resnet18",
|
||||
"minAccuracy": 0.85,
|
||||
"minPerClass": min_per_class,
|
||||
},
|
||||
"modes": list(MODES),
|
||||
"backbones": list(BACKBONES),
|
||||
}
|
||||
|
||||
|
||||
def versions(out_dir: Path) -> list[dict[str, Any]]:
|
||||
out: list[dict[str, Any]] = []
|
||||
if not out_dir.is_dir():
|
||||
return out
|
||||
for d in sorted(out_dir.iterdir(), key=lambda p: p.name, reverse=True):
|
||||
if not d.is_dir() or d.name in ("cache", "jobs"):
|
||||
continue
|
||||
if not (d / REPORT_FILE).is_file() and not (d / METRICS_FILE).is_file():
|
||||
continue
|
||||
entry: dict[str, Any] = {
|
||||
"version": d.name,
|
||||
"written": (d / MODEL_FILE).is_file() and (d / SIDECAR_FILE).is_file(),
|
||||
"hasReport": (d / REPORT_FILE).is_file(),
|
||||
"modifiedAt": datetime.fromtimestamp(d.stat().st_mtime, tz=timezone.utc)
|
||||
.replace(microsecond=0)
|
||||
.isoformat(),
|
||||
}
|
||||
try:
|
||||
m = json.loads((d / METRICS_FILE).read_text())
|
||||
entry["accuracy"] = m.get("accuracy")
|
||||
entry["macroRecall"] = m.get("macro_recall")
|
||||
entry["n"] = m.get("n")
|
||||
except (OSError, ValueError):
|
||||
pass
|
||||
if entry["written"]:
|
||||
try:
|
||||
side = Sidecar.read(d / SIDECAR_FILE)
|
||||
entry["classes"] = side.classes
|
||||
entry["mode"] = side.mode
|
||||
entry["backbone"] = side.backbone
|
||||
entry["trainedAt"] = side.trained_at
|
||||
entry["labels"] = side.labels
|
||||
entry["floor"] = side.metrics.get("floor")
|
||||
except (OSError, ValueError):
|
||||
pass
|
||||
out.append(entry)
|
||||
return out
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
jobs: Jobs # set on the class by serve()
|
||||
server_version = "parking-trainer"
|
||||
|
||||
def log_message(self, fmt: str, *args: Any) -> None: # quieter than the default
|
||||
sys.stderr.write(f"[trainer.serve] {self.address_string()} {fmt % args}\n")
|
||||
|
||||
def _json(self, code: int, body: Any) -> None:
|
||||
raw = json.dumps(body).encode()
|
||||
self.send_response(code)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("Content-Length", str(len(raw)))
|
||||
self.end_headers()
|
||||
self.wfile.write(raw)
|
||||
|
||||
def _text(self, code: int, body: str, ctype: str = "text/markdown; charset=utf-8") -> None:
|
||||
raw = body.encode()
|
||||
self.send_response(code)
|
||||
self.send_header("Content-Type", ctype)
|
||||
self.send_header("Content-Length", str(len(raw)))
|
||||
self.end_headers()
|
||||
self.wfile.write(raw)
|
||||
|
||||
def do_GET(self) -> None: # noqa: N802
|
||||
path = self.path.split("?", 1)[0]
|
||||
j = self.jobs
|
||||
if path == "/health":
|
||||
self._json(200, {"ok": True, "busy": j.busy, "version": "parking-trainer"})
|
||||
elif path == "/readiness":
|
||||
self._json(200, readiness(j.data_dir))
|
||||
elif path == "/versions":
|
||||
self._json(200, {"versions": versions(j.out_dir)})
|
||||
elif path.startswith("/versions/") and path.endswith("/report"):
|
||||
v = path[len("/versions/") : -len("/report")]
|
||||
try:
|
||||
p = j.out_dir / Jobs._version(v) / REPORT_FILE
|
||||
except ValueError:
|
||||
self._json(400, {"error": "bad version"})
|
||||
return
|
||||
if not p.is_file():
|
||||
self._json(404, {"error": "no report"})
|
||||
else:
|
||||
self._text(200, p.read_text())
|
||||
elif path == "/jobs":
|
||||
self._json(200, {"jobs": j.recent(), "current": j.current()})
|
||||
elif path.startswith("/jobs/"):
|
||||
job = j.get(path[len("/jobs/") :])
|
||||
self._json(200, job) if job else self._json(404, {"error": "no such job"})
|
||||
else:
|
||||
self._json(404, {"error": "not found"})
|
||||
|
||||
def do_POST(self) -> None: # noqa: N802
|
||||
if self.path.split("?", 1)[0] != "/jobs":
|
||||
self._json(404, {"error": "not found"})
|
||||
return
|
||||
n = int(self.headers.get("Content-Length") or 0)
|
||||
if n > 64_000:
|
||||
self._json(413, {"error": "too large"})
|
||||
return
|
||||
try:
|
||||
body = json.loads(self.rfile.read(n) or b"{}")
|
||||
if not isinstance(body, dict):
|
||||
raise ValueError("object expected")
|
||||
except ValueError as exc:
|
||||
self._json(400, {"error": f"bad json: {exc}"})
|
||||
return
|
||||
kind = str(body.pop("kind", ""))
|
||||
try:
|
||||
job = self.jobs.start(kind, body)
|
||||
except ValueError as exc:
|
||||
self._json(400, {"error": str(exc)})
|
||||
return
|
||||
except RuntimeError:
|
||||
self._json(409, {"error": "a job is already running", "current": self.jobs.current()})
|
||||
return
|
||||
self._json(202, job)
|
||||
|
||||
|
||||
def serve(data_dir: Path, out_dir: Path, host: str, port: int, publish_url: str, publish_token: str) -> None:
|
||||
Handler.jobs = Jobs(data_dir, out_dir, publish_url, publish_token)
|
||||
httpd = ThreadingHTTPServer((host, port), Handler)
|
||||
sys.stderr.write(f"[trainer.serve] listening on {host}:{port}, data {data_dir}, out {out_dir}\n")
|
||||
try:
|
||||
httpd.serve_forever()
|
||||
except KeyboardInterrupt:
|
||||
pass
|
||||
finally:
|
||||
httpd.server_close()
|
||||
|
||||
|
||||
def wait_idle(jobs: Jobs, timeout: float = 60.0) -> None:
|
||||
"""Test helper: block until no job runs."""
|
||||
t0 = time.monotonic()
|
||||
while jobs.busy and time.monotonic() - t0 < timeout:
|
||||
time.sleep(0.1)
|
||||
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"$schema": "https://turbo.build/schema.json",
|
||||
"extends": ["//"],
|
||||
"tasks": {
|
||||
"build": {
|
||||
"outputs": []
|
||||
}
|
||||
}
|
||||
}
|
||||
Generated
+1444
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,9 @@
|
||||
.venv/
|
||||
**/__pycache__/
|
||||
.pytest_cache/
|
||||
.mypy_cache/
|
||||
.ruff_cache/
|
||||
.env
|
||||
# weights are fetched inside the build (yolox) or pinned by models/bodytype.version
|
||||
models/*
|
||||
!models/bodytype.version
|
||||
@@ -29,3 +29,11 @@ VISION_MIN_CONFIDENCE=0.5
|
||||
# VISION_VEHICLE_MODEL_PATH=models/yolox_s.onnx
|
||||
# VISION_VEHICLE_INPUT_SIZE=640
|
||||
# VISION_VEHICLE_MIN_CONFIDENCE=0.4
|
||||
|
||||
# Phase B: the body-type classifier (sedan/hatchback/suv/… on the detector's crop), trained
|
||||
# by apps/trainer on the reviewer's labels. The Docker image bakes it at
|
||||
# /app/models/bodytype.onnx (+ .json sidecar) when models/bodytype.version pins a published
|
||||
# version; locally copy a trainer output folder's two files into apps/vision/models/.
|
||||
# Path set but no file = stage off (the normal state before the first model).
|
||||
# VISION_VEHICLE_CLASSIFIER_PATH=models/bodytype.onnx
|
||||
# VISION_VEHICLE_CLASSIFIER_MIN_CONFIDENCE=0.6
|
||||
|
||||
@@ -7,5 +7,6 @@ __pycache__/
|
||||
.ruff_cache/
|
||||
|
||||
# Model weights (fetched at deploy / first run, never committed — can be large + license-scoped)
|
||||
models/
|
||||
models/*
|
||||
!models/bodytype.version
|
||||
*.onnx
|
||||
|
||||
+20
-1
@@ -33,6 +33,24 @@ ARG YOLOX_URL=https://github.com/Megvii-BaseDetection/YOLOX/releases/download/0.
|
||||
RUN mkdir -p /app/models \
|
||||
&& (curl -fsSL -o /app/models/yolox_s.onnx "$YOLOX_URL" \
|
||||
|| (echo "[build] yolox weights not fetched (no network) — vehicle stage off" && rm -f /app/models/yolox_s.onnx))
|
||||
# Phase B body-type classifier (apps/trainer output, published to the Gitea generic package
|
||||
# registry — weights are not code, they never live in git). models/bodytype.version PINS the
|
||||
# version this image carries: empty = no classifier (phase B off). A pinned version that
|
||||
# cannot be fetched FAILS the build — the image must carry what git says it carries. The
|
||||
# registry may need auth: pass a BuildKit secret `bodytype_auth` holding "user:token".
|
||||
ARG BODYTYPE_BASE_URL=https://git.infra.msai.al/api/packages/mca/generic/parking-bodytype
|
||||
COPY models/bodytype.version ./models/bodytype.version
|
||||
RUN --mount=type=secret,id=bodytype_auth \
|
||||
v="$(tr -d '[:space:]' < /app/models/bodytype.version)"; \
|
||||
if [ -n "$v" ]; then \
|
||||
cfg=/tmp/curl.cfg; : > "$cfg"; \
|
||||
[ -f /run/secrets/bodytype_auth ] && printf 'user = "%s"\n' "$(cat /run/secrets/bodytype_auth)" > "$cfg"; \
|
||||
curl -fsSL -K "$cfg" -o /app/models/bodytype.onnx "$BODYTYPE_BASE_URL/$v/bodytype.onnx" \
|
||||
&& curl -fsSL -K "$cfg" -o /app/models/bodytype.json "$BODYTYPE_BASE_URL/$v/bodytype.json" \
|
||||
&& echo "[build] bodytype classifier $v baked" \
|
||||
|| { echo "[build] bodytype classifier $v could not be fetched"; rm -f "$cfg"; exit 1; }; \
|
||||
rm -f "$cfg"; \
|
||||
else echo "[build] no bodytype version pinned — phase B off"; fi
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen --extra alpr
|
||||
|
||||
@@ -57,7 +75,8 @@ RUN uv run python -c "from fast_alpr import ALPR; ALPR()" \
|
||||
ENV VISION_RECOGNIZER=stub \
|
||||
VISION_HOST=0.0.0.0 \
|
||||
VISION_PORT=8089 \
|
||||
VISION_VEHICLE_MODEL_PATH=/app/models/yolox_s.onnx
|
||||
VISION_VEHICLE_MODEL_PATH=/app/models/yolox_s.onnx \
|
||||
VISION_VEHICLE_CLASSIFIER_PATH=/app/models/bodytype.onnx
|
||||
EXPOSE 8089
|
||||
HEALTHCHECK --interval=30s --timeout=5s --start-period=20s --retries=3 \
|
||||
CMD python -c "import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://localhost:8089/health').status==200 else 1)" || exit 1
|
||||
|
||||
@@ -35,6 +35,9 @@ dev = [
|
||||
"pytest>=8.3",
|
||||
"httpx>=0.27", # FastAPI TestClient transport
|
||||
"mypy>=1.13",
|
||||
# The vehicle stage's post-processing tests run on synthetic tensors without the model
|
||||
# stack (CI syncs WITHOUT the alpr extra); the service itself imports numpy lazily.
|
||||
"numpy>=1.26",
|
||||
]
|
||||
|
||||
[tool.ruff]
|
||||
|
||||
@@ -6,6 +6,7 @@ from __future__ import annotations
|
||||
import os
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from vision_service.schemas import BBox, VehicleResult
|
||||
@@ -82,6 +83,7 @@ def test_pick_prefers_the_box_holding_the_plate_else_the_largest() -> None:
|
||||
|
||||
|
||||
def test_letterbox_keeps_aspect_and_pads_with_114() -> None:
|
||||
pytest.importorskip("cv2") # letterbox resizes with OpenCV — only present with the alpr extra
|
||||
frame = np.zeros((30, 60, 3), dtype=np.uint8)
|
||||
tensor, scale = letterbox(frame, 64)
|
||||
assert tensor.shape == (1, 3, 64, 64) and tensor.dtype == np.float32
|
||||
@@ -141,3 +143,135 @@ def test_app_reports_a_missing_model_file_and_keeps_serving() -> None:
|
||||
assert res.json()["vehicle"] is None
|
||||
finally:
|
||||
os.environ.pop("VISION_VEHICLE_MODEL_PATH", None)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
# Phase B: the classifier stage over the detector
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_crop_vehicle_adds_the_margin_clamps_and_blurs_the_plate() -> None:
|
||||
cv2 = pytest.importorskip("cv2")
|
||||
from vision_service.vehicle import crop_vehicle
|
||||
|
||||
frame = np.zeros((100, 200, 3), dtype=np.uint8)
|
||||
frame[40:50, 90:110] = (0, 255, 0) # a green "plate"
|
||||
box = BBox(x1=50, y1=20, x2=150, y2=80) # 100×60 → 8 % margin = 8 / 5 px
|
||||
crop = crop_vehicle(frame, box, None, 0.08)
|
||||
assert crop.shape == (70, 116, 3)
|
||||
edge = crop_vehicle(frame, BBox(x1=0, y1=0, x2=100, y2=60), None, 0.08)
|
||||
assert edge.shape == (65, 108, 3) # clamped at the frame's top-left
|
||||
assert crop_vehicle(frame, BBox(x1=10, y1=10, x2=12, y2=12), None, 0.08) is None
|
||||
blurred = crop_vehicle(frame, box, BBox(x1=90, y1=40, x2=110, y2=50), 0.08)
|
||||
strip = blurred[40 - 20 + 5 : 50 - 20 + 5, 90 - 50 + 8 : 110 - 50 + 8, 1] # plate strip, green channel
|
||||
assert strip.mean() < 200 and crop[20 + 5 : 30 + 5, 40 + 8 : 60 + 8, 1].mean() == 255
|
||||
assert cv2 is not None
|
||||
|
||||
|
||||
class FakeClassifier:
|
||||
ready = True
|
||||
error = None
|
||||
min_confidence = 0.6
|
||||
model_version = "bodytype:vfake"
|
||||
|
||||
def __init__(self, classes: list[str], answer: tuple[str, float] | None) -> None:
|
||||
self.classes = classes
|
||||
self.answer = answer
|
||||
self.calls = 0
|
||||
|
||||
def classify(self, frame, box, plate): # type: ignore[no-untyped-def]
|
||||
self.calls += 1
|
||||
if isinstance(self.answer, Exception):
|
||||
raise self.answer
|
||||
return self.answer
|
||||
|
||||
|
||||
class FrameDetector:
|
||||
"""A detector that answers on decoded frames (like YOLOX) with a fixed result."""
|
||||
|
||||
model_version = "det"
|
||||
ready = True
|
||||
error = None
|
||||
|
||||
def __init__(self, result: VehicleResult | None) -> None:
|
||||
self.result = result
|
||||
|
||||
def detect_frame(self, frame, plate): # type: ignore[no-untyped-def]
|
||||
return self.result
|
||||
|
||||
def detect(self, image_bytes: bytes, plate: BBox | None) -> VehicleResult | None:
|
||||
raise AssertionError("the refined stage should share the decoded frame")
|
||||
|
||||
|
||||
def _jpeg() -> bytes:
|
||||
cv2 = pytest.importorskip("cv2")
|
||||
ok, buf = cv2.imencode(".jpg", np.zeros((60, 80, 3), dtype=np.uint8))
|
||||
assert ok
|
||||
return bytes(buf)
|
||||
|
||||
|
||||
def test_refined_detector_replaces_car_when_confident_else_keeps_the_detector() -> None:
|
||||
from vision_service.vehicle import RefinedVehicleDetector
|
||||
|
||||
car = VehicleResult(body_type="car", confidence=0.85, bbox=BBox(x1=10, y1=10, x2=70, y2=50))
|
||||
sure = FakeClassifier(["sedan", "suv"], ("suv", 0.91))
|
||||
res = RefinedVehicleDetector(FrameDetector(car), sure).detect(_jpeg(), None)
|
||||
assert (
|
||||
res is not None and res.body_type == "suv" and res.confidence == 0.91 and res.detector_class == "car"
|
||||
)
|
||||
assert res.bbox == car.bbox
|
||||
|
||||
unsure = FakeClassifier(["sedan", "suv"], ("suv", 0.4))
|
||||
res2 = RefinedVehicleDetector(FrameDetector(car), unsure).detect(_jpeg(), None)
|
||||
assert (
|
||||
res2 is not None
|
||||
and res2.body_type == "car"
|
||||
and res2.confidence == 0.85
|
||||
and res2.detector_class == "car"
|
||||
)
|
||||
|
||||
# A class the classifier never trained on is left alone (its softmax means nothing there).
|
||||
bus = VehicleResult(body_type="bus", confidence=0.9, bbox=car.bbox)
|
||||
skip = FakeClassifier(["sedan", "suv"], ("suv", 0.99))
|
||||
res3 = RefinedVehicleDetector(FrameDetector(bus), skip).detect(_jpeg(), None)
|
||||
assert res3 == bus and skip.calls == 0
|
||||
# …unless it was: a classifier that knows trucks may override a truck.
|
||||
knows = FakeClassifier(["sedan", "truck", "van"], ("van", 0.8))
|
||||
truck = VehicleResult(body_type="truck", confidence=0.7, bbox=car.bbox)
|
||||
res4 = RefinedVehicleDetector(FrameDetector(truck), knows).detect(_jpeg(), None)
|
||||
assert res4 is not None and res4.body_type == "van" and res4.detector_class == "truck"
|
||||
|
||||
|
||||
def test_refined_detector_survives_a_broken_classifier_and_reports_it() -> None:
|
||||
from vision_service.vehicle import RefinedVehicleDetector
|
||||
|
||||
car = VehicleResult(body_type="car", confidence=0.85, bbox=BBox(x1=10, y1=10, x2=70, y2=50))
|
||||
boom = FakeClassifier(["sedan"], RuntimeError("bad graph")) # type: ignore[arg-type]
|
||||
ref = RefinedVehicleDetector(FrameDetector(car), boom)
|
||||
assert ref.detect(_jpeg(), None) == car
|
||||
assert ref.error == "classifier: RuntimeError: bad graph"
|
||||
assert ref.ready is True and ref.model_version == "det+bodytype:vfake"
|
||||
# No box, or a detector that found nothing → nothing to classify.
|
||||
assert RefinedVehicleDetector(FrameDetector(None), boom).detect(_jpeg(), None) is None
|
||||
boxless = VehicleResult(body_type="car", confidence=0.85)
|
||||
assert RefinedVehicleDetector(FrameDetector(boxless), boom).detect(_jpeg(), None) == boxless
|
||||
|
||||
|
||||
def test_classifier_without_files_is_not_ready_and_the_factory_skips_a_missing_model(tmp_path) -> None: # type: ignore[no-untyped-def]
|
||||
from vision_service.recognizer import WithVehicle, build_recognizer
|
||||
from vision_service.settings import Settings
|
||||
from vision_service.vehicle import BodyTypeClassifier, RefinedVehicleDetector
|
||||
|
||||
clf = BodyTypeClassifier(str(tmp_path / "bodytype.onnx"))
|
||||
assert clf.ready is False and "FileNotFoundError" in (clf.error or "")
|
||||
(tmp_path / "bodytype.json").write_text('{"format": "other"}')
|
||||
assert "unknown sidecar format" in (BodyTypeClassifier(str(tmp_path / "bodytype.onnx")).error or "")
|
||||
|
||||
# A path with no file = the normal pre-model state: phase A only, no error in health.
|
||||
s = Settings(
|
||||
vehicle_model_path="/nonexistent/yolox.onnx", vehicle_classifier_path=str(tmp_path / "none.onnx")
|
||||
)
|
||||
rec = build_recognizer(s)
|
||||
assert isinstance(rec, WithVehicle)
|
||||
assert not isinstance(rec._detector, RefinedVehicleDetector) # noqa: SLF001
|
||||
assert "classifier" not in (rec.error or "")
|
||||
|
||||
Generated
+3
@@ -760,6 +760,8 @@ alpr = [
|
||||
dev = [
|
||||
{ name = "httpx" },
|
||||
{ name = "mypy" },
|
||||
{ name = "numpy", version = "2.2.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" },
|
||||
{ name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.11'" },
|
||||
{ name = "pytest" },
|
||||
{ name = "ruff" },
|
||||
]
|
||||
@@ -779,6 +781,7 @@ provides-extras = ["alpr"]
|
||||
dev = [
|
||||
{ name = "httpx", specifier = ">=0.27" },
|
||||
{ name = "mypy", specifier = ">=1.13" },
|
||||
{ name = "numpy", specifier = ">=1.26" },
|
||||
{ name = "pytest", specifier = ">=8.3" },
|
||||
{ name = "ruff", specifier = ">=0.8" },
|
||||
]
|
||||
|
||||
@@ -14,12 +14,16 @@ Adding a recognizer (e.g. a fine-tuned YOLO + PaddleOCR) = a new class here, no
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Protocol
|
||||
|
||||
from .schemas import AnalyzeResponse, BBox, PlateResult
|
||||
from .settings import Settings
|
||||
from .vehicle import VehicleDetector, YoloxVehicleDetector
|
||||
from .vehicle import BodyTypeClassifier, RefinedVehicleDetector, VehicleDetector, YoloxVehicleDetector
|
||||
|
||||
log = logging.getLogger("vision")
|
||||
|
||||
|
||||
class Recognizer(Protocol):
|
||||
@@ -215,10 +219,21 @@ def build_recognizer(settings: Settings) -> Recognizer:
|
||||
else:
|
||||
rec = StubRecognizer(settings)
|
||||
if settings.vehicle_model_path:
|
||||
detector = YoloxVehicleDetector(
|
||||
detector: VehicleDetector = YoloxVehicleDetector(
|
||||
settings.vehicle_model_path,
|
||||
input_size=settings.vehicle_input_size,
|
||||
min_confidence=settings.vehicle_min_confidence,
|
||||
)
|
||||
if settings.vehicle_classifier_path:
|
||||
if Path(settings.vehicle_classifier_path).is_file():
|
||||
detector = RefinedVehicleDetector(
|
||||
detector,
|
||||
BodyTypeClassifier(
|
||||
settings.vehicle_classifier_path,
|
||||
min_confidence=settings.vehicle_classifier_min_confidence,
|
||||
),
|
||||
)
|
||||
else:
|
||||
log.info("no body-type classifier at %s — phase B off", settings.vehicle_classifier_path)
|
||||
return WithVehicle(rec, detector)
|
||||
return rec
|
||||
|
||||
@@ -45,6 +45,9 @@ class VehicleResult(BaseModel):
|
||||
confidence: float | None = Field(default=None, ge=0.0, le=1.0)
|
||||
# The vehicle's box in frame pixels — the crop a reviewer sees / a classifier eats.
|
||||
bbox: BBox | None = None
|
||||
# Phase B: the detector's coarse class when the body-type classifier ran on this crop
|
||||
# (body_type is then the classifier's answer if confident, else the detector's).
|
||||
detector_class: str | None = None
|
||||
make: str | None = None
|
||||
model: str | None = None
|
||||
|
||||
|
||||
@@ -42,6 +42,14 @@ class Settings(BaseSettings):
|
||||
# site's own, stricter threshold before it FLAGS anything).
|
||||
vehicle_min_confidence: float = 0.4
|
||||
|
||||
# Phase B — the body-type classifier on the detector's crop (bodytype.onnx + its .json
|
||||
# sidecar, produced by apps/trainer, baked into the image when
|
||||
# models/bodytype.version pins a published version). Path set but NO file = the normal
|
||||
# state before the first model ships: the stage is simply off (logged, not an error).
|
||||
vehicle_classifier_path: str | None = None
|
||||
# Below this probability the classifier's answer is dropped and the detector's stands.
|
||||
vehicle_classifier_min_confidence: float = 0.6
|
||||
|
||||
|
||||
def get_settings() -> Settings:
|
||||
return Settings()
|
||||
|
||||
@@ -229,6 +229,13 @@ class YoloxVehicleDetector:
|
||||
frame = cv2.imdecode(np.frombuffer(image_bytes, dtype=np.uint8), cv2.IMREAD_COLOR)
|
||||
if frame is None:
|
||||
return None
|
||||
return self.detect_frame(frame, plate)
|
||||
|
||||
def detect_frame(self, frame: Any, plate: BBox | None) -> VehicleResult | None:
|
||||
"""Same as detect() on an already-decoded BGR frame (the classifier stage decodes
|
||||
once and shares it)."""
|
||||
if self._session is None:
|
||||
return None
|
||||
tensor, scale = letterbox(frame, self._size)
|
||||
raw = self._session.run(None, {self._input_name: tensor})[0][0]
|
||||
found = vehicles_from_output(raw, self._size, scale, self._min_confidence)
|
||||
@@ -242,6 +249,169 @@ class YoloxVehicleDetector:
|
||||
return VehicleResult(body_type=best.body_type, confidence=round(best.confidence, 4), bbox=box)
|
||||
|
||||
|
||||
# ----------------------------------------------------------------------------------
|
||||
# Phase B: the body-type classifier on the detector's crop
|
||||
# ----------------------------------------------------------------------------------
|
||||
|
||||
SIDECAR_FORMAT = "parking-bodytype/1"
|
||||
|
||||
|
||||
def crop_vehicle(frame: Any, box: BBox, plate: BBox | None, margin: float) -> Any:
|
||||
"""The detector's box + margin, plate blurred — the SAME cut the collector stores
|
||||
(apps/server review-outbox.ts makeReviewCrop), so the classifier sees at the booth
|
||||
what it was trained on. Returns a BGR array, or None when the box is degenerate."""
|
||||
import cv2
|
||||
|
||||
h, w = frame.shape[:2]
|
||||
mw = round((box.x2 - box.x1) * margin)
|
||||
mh = round((box.y2 - box.y1) * margin)
|
||||
left, top = max(0, box.x1 - mw), max(0, box.y1 - mh)
|
||||
right, bottom = min(w, box.x2 + mw), min(h, box.y2 + mh)
|
||||
if right - left < 8 or bottom - top < 8:
|
||||
return None
|
||||
crop = frame[top:bottom, left:right].copy()
|
||||
if plate is not None:
|
||||
pad = round(max(plate.x2 - plate.x1, plate.y2 - plate.y1) * 0.25)
|
||||
pl, pt = max(0, plate.x1 - pad - left), max(0, plate.y1 - pad - top)
|
||||
pr, pb = min(right - left, plate.x2 + pad - left), min(bottom - top, plate.y2 + pad - top)
|
||||
if pr - pl >= 2 and pb - pt >= 2:
|
||||
sigma = max(6, round((pr - pl) / 6))
|
||||
crop[pt:pb, pl:pr] = cv2.GaussianBlur(crop[pt:pb, pl:pr], (0, 0), sigma)
|
||||
return crop
|
||||
|
||||
|
||||
class BodyTypeClassifier:
|
||||
"""`bodytype.onnx` + its `bodytype.json` sidecar (written by apps/trainer). The sidecar
|
||||
carries the preprocessing contract — class list, input size, crop margin — and the graph
|
||||
normalises internally, so this side only cuts, resizes (INTER_AREA, like the trainer)
|
||||
and feeds raw RGB 0–255. Load failure → `error`, the stage yields nothing."""
|
||||
|
||||
def __init__(self, model_path: str, min_confidence: float = 0.6) -> None:
|
||||
import json
|
||||
|
||||
self._path = Path(model_path)
|
||||
self.min_confidence = min_confidence
|
||||
self._session = None
|
||||
self._input_name = "image"
|
||||
self._error: str | None = None
|
||||
self.classes: list[str] = []
|
||||
self.version = "?"
|
||||
self.input_size = 224
|
||||
self.crop_margin = 0.08
|
||||
try:
|
||||
side = json.loads(self._path.with_suffix(".json").read_text())
|
||||
if side.get("format") != SIDECAR_FORMAT:
|
||||
raise ValueError(f"unknown sidecar format {side.get('format')!r}")
|
||||
self.classes = [str(c) for c in side["classes"]]
|
||||
self.version = str(side.get("version", "?"))
|
||||
self.input_size = int(side.get("input_size", 224))
|
||||
self.crop_margin = float(side.get("crop_margin", 0.08))
|
||||
import onnxruntime as ort
|
||||
|
||||
opts = ort.SessionOptions()
|
||||
opts.intra_op_num_threads = 2
|
||||
self._session = ort.InferenceSession(
|
||||
str(self._path), sess_options=opts, providers=["CPUExecutionProvider"]
|
||||
)
|
||||
self._input_name = self._session.get_inputs()[0].name
|
||||
except Exception as exc: # noqa: BLE001 - not-ready, never fatal
|
||||
self._error = f"{type(exc).__name__}: {exc}"
|
||||
|
||||
@property
|
||||
def model_version(self) -> str:
|
||||
return f"bodytype:{self.version}"
|
||||
|
||||
@property
|
||||
def ready(self) -> bool:
|
||||
return self._session is not None
|
||||
|
||||
@property
|
||||
def error(self) -> str | None:
|
||||
return self._error
|
||||
|
||||
def classify(self, frame: Any, box: BBox, plate: BBox | None) -> tuple[str, float] | None:
|
||||
"""(class, probability) for the vehicle in `box`, or None when nothing could be cut."""
|
||||
if self._session is None:
|
||||
return None
|
||||
import cv2
|
||||
import numpy as np
|
||||
|
||||
crop = crop_vehicle(frame, box, plate, self.crop_margin)
|
||||
if crop is None:
|
||||
return None
|
||||
resized = cv2.resize(crop, (self.input_size, self.input_size), interpolation=cv2.INTER_AREA)
|
||||
rgb = cv2.cvtColor(resized, cv2.COLOR_BGR2RGB)
|
||||
x = np.ascontiguousarray(rgb.transpose(2, 0, 1)[None].astype(np.float32))
|
||||
logits = self._session.run(None, {self._input_name: x})[0][0]
|
||||
z = logits - logits.max()
|
||||
p = np.exp(z) / np.exp(z).sum()
|
||||
i = int(p.argmax())
|
||||
return self.classes[i], float(p[i])
|
||||
|
||||
|
||||
class RefinedVehicleDetector:
|
||||
"""Detector + classifier. The detector finds the vehicle (and picks WHICH one); when its
|
||||
class is `car` — or one the classifier was trained on — the classifier's answer replaces
|
||||
it if confident enough, else the detector's stands. A truck or bus the classifier has
|
||||
never seen is left alone: its softmax on an unknown thing means nothing."""
|
||||
|
||||
def __init__(self, detector: Any, classifier: BodyTypeClassifier) -> None:
|
||||
self._detector = detector
|
||||
self._classifier = classifier
|
||||
self.stage_error: str | None = None
|
||||
|
||||
@property
|
||||
def model_version(self) -> str:
|
||||
return f"{self._detector.model_version}+{self._classifier.model_version}"
|
||||
|
||||
@property
|
||||
def ready(self) -> bool:
|
||||
return bool(getattr(self._detector, "ready", True))
|
||||
|
||||
@property
|
||||
def error(self) -> str | None:
|
||||
parts = [
|
||||
getattr(self._detector, "error", None),
|
||||
f"classifier: {self._classifier.error}" if self._classifier.error else None,
|
||||
f"classifier: {self.stage_error}" if self.stage_error else None,
|
||||
]
|
||||
kept = [p for p in parts if p]
|
||||
return "; ".join(kept) if kept else None
|
||||
|
||||
def detect(self, image_bytes: bytes, plate: BBox | None) -> VehicleResult | None:
|
||||
import cv2
|
||||
import numpy as np
|
||||
|
||||
frame = cv2.imdecode(np.frombuffer(image_bytes, dtype=np.uint8), cv2.IMREAD_COLOR)
|
||||
if frame is None:
|
||||
return None
|
||||
detect_frame = getattr(self._detector, "detect_frame", None)
|
||||
base: VehicleResult | None = (
|
||||
detect_frame(frame, plate) if detect_frame else self._detector.detect(image_bytes, plate)
|
||||
)
|
||||
if base is None or base.bbox is None or not self._classifier.ready:
|
||||
return base
|
||||
if not (base.body_type == "car" or base.body_type in self._classifier.classes):
|
||||
return base
|
||||
try:
|
||||
out = self._classifier.classify(frame, base.bbox, plate)
|
||||
except Exception as exc: # noqa: BLE001 - advisory stage, never fatal
|
||||
self.stage_error = f"{type(exc).__name__}: {exc}"
|
||||
return base
|
||||
if out is None:
|
||||
return base
|
||||
body_type, confidence = out
|
||||
if confidence < self._classifier.min_confidence:
|
||||
return base.model_copy(update={"detector_class": base.body_type})
|
||||
return base.model_copy(
|
||||
update={
|
||||
"body_type": body_type,
|
||||
"confidence": round(confidence, 4),
|
||||
"detector_class": base.body_type,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def time_detect(
|
||||
detector: VehicleDetector, image_bytes: bytes, plate: BBox | None
|
||||
) -> tuple[VehicleResult | None, float]:
|
||||
|
||||
@@ -143,6 +143,7 @@ export const en: Catalog = {
|
||||
reviewTitle: "Remote review",
|
||||
reviewOff: "off — no collector configured for this booth",
|
||||
reviewCounts: "{{queued}} waiting · {{sent}} delivered · {{failed}} abandoned",
|
||||
reviewEntrySample: "1 in {{n}} entries sampled",
|
||||
reviewHint: "Each wash order sends the vehicle crop (plate blurred) and the chosen category to a trusted reviewer over the private network. One-way; nothing that names this site leaves.",
|
||||
visionClassesHint: "The camera's fixed vocabulary (set in code, not here). Tick the classes this category covers.",
|
||||
visionThreshold: "Camera confidence to flag a downgrade",
|
||||
|
||||
@@ -145,6 +145,7 @@ export const sq = {
|
||||
reviewTitle: "Shqyrtim në distancë",
|
||||
reviewOff: "joaktiv — asnjë mbledhës i konfiguruar për këtë kabinë",
|
||||
reviewCounts: "{{queued}} në pritje · {{sent}} të dërguara · {{failed}} të braktisura",
|
||||
reviewEntrySample: "1 në {{n}} hyrje merret mostër",
|
||||
reviewHint: "Çdo porosi lavazhi dërgon prerjen e mjetit (targa e turbulluar) dhe kategorinë e zgjedhur te një shqyrtues i besuar përmes rrjetit privat. Njëkahësh; asgjë që emërton këtë vend nuk del.",
|
||||
visionClassesHint: "Fjalori i fiksuar i kamerës (vendoset në kod, jo këtu). Shëno klasat që mbulon kjo kategori.",
|
||||
visionThreshold: "Siguria e kamerës për të shënuar një ulje kategorie",
|
||||
|
||||
@@ -199,8 +199,11 @@ export function CarWashSetup({ canEdit }: { canEdit: boolean }) {
|
||||
const currency = settings?.currency ?? "";
|
||||
|
||||
return (
|
||||
<div className="mt-6 flex flex-wrap items-start gap-6">
|
||||
<section className="card w-full max-w-2xl p-4">
|
||||
// Two columns on a wide screen: the master data (categories with their camera-class
|
||||
// chips, services, the price matrix, policies) takes the room it needs; the sponsorship
|
||||
// program keeps a fixed, readable width beside it. Stacks on narrow screens.
|
||||
<div className="mt-6 grid items-start gap-6 lg:grid-cols-[minmax(0,1fr)_minmax(20rem,26rem)]">
|
||||
<section className="card min-w-0 p-4">
|
||||
<div className="grid gap-4">
|
||||
<ListEditor title={t("wash.categories")} items={categories} onChange={setCategories} addLabel={t("wash.addCategory")} visionMap />
|
||||
<ListEditor title={t("wash.services")} items={services} onChange={setServices} addLabel={t("wash.addService")} />
|
||||
@@ -274,6 +277,7 @@ export function CarWashSetup({ canEdit }: { canEdit: boolean }) {
|
||||
{review.enabled
|
||||
? t("wash.reviewCounts", { queued: review.queued, sent: review.sent, failed: review.failed })
|
||||
: t("wash.reviewOff")}
|
||||
{review.enabled && review.entrySample > 0 && <span className="ml-2 text-term-muted">· {t("wash.reviewEntrySample", { n: review.entrySample })}</span>}
|
||||
{review.enabled && review.lastError && <span className="ml-2 text-term-amber">{review.lastError}</span>}
|
||||
</span>
|
||||
<span className="hint">{t("wash.reviewHint")}</span>
|
||||
@@ -290,7 +294,7 @@ export function CarWashSetup({ canEdit }: { canEdit: boolean }) {
|
||||
</section>
|
||||
|
||||
{canEdit && program && (
|
||||
<section className="card w-full max-w-md p-4">
|
||||
<section className="card min-w-0 p-4">
|
||||
<div className="text-[0.6875rem] uppercase tracking-wider text-term-muted">{t("wash.sponsorship")}</div>
|
||||
<span className="hint">{t("wash.sponsorshipHint")}</span>
|
||||
<div className="mt-2">
|
||||
|
||||
@@ -40,6 +40,8 @@ export interface CarwashReviewStatus {
|
||||
failed: number;
|
||||
lastSentAt: string | null;
|
||||
lastError: string | null;
|
||||
/** 0 = entry sampling off; N = one in N entry reads is queued as training material. */
|
||||
entrySample: number;
|
||||
}
|
||||
export function fetchCarwashReviewStatus(): Promise<CarwashReviewStatus> {
|
||||
return apiFetch("/api/carwash/review/status");
|
||||
|
||||
@@ -22,26 +22,30 @@ services:
|
||||
COLLECTOR_REVIEWER_USER: ${COLLECTOR_REVIEWER_USER:-reviewer}
|
||||
COLLECTOR_REVIEWER_PASS: ${COLLECTOR_REVIEWER_PASS:?set COLLECTOR_REVIEWER_PASS in the stack env}
|
||||
LOG_LEVEL: ${LOG_LEVEL:-info}
|
||||
# The trainer's job API (sibling service above). Unset = no Training section.
|
||||
COLLECTOR_TRAINER_URL: ${COLLECTOR_TRAINER_URL-http://trainer:8091}
|
||||
volumes:
|
||||
- collector-data:/data
|
||||
|
||||
# Phase B trainer — a one-off job on this host's GPU, NOT a service (profile "train": it
|
||||
# only runs when asked: `docker compose --profile train run --rm trainer`). Reads the
|
||||
# collector's export + crops straight off the same volume; writes the ONNX classifier the
|
||||
# vision image then bakes in. The image/script are the next increment; this is the seam.
|
||||
# trainer:
|
||||
# image: ${REGISTRY:-git.infra.msai.al/mca/parking_solution}/parking-trainer:${TAG:-dev}
|
||||
# profiles: ["train"]
|
||||
# deploy:
|
||||
# resources:
|
||||
# reservations:
|
||||
# devices:
|
||||
# - driver: nvidia
|
||||
# count: all
|
||||
# capabilities: [gpu]
|
||||
# volumes:
|
||||
# - collector-data:/data:ro
|
||||
# - ./models:/out
|
||||
# Phase B trainer — a small always-on job service beside the collector (CPU-only torch;
|
||||
# idle it is a tiny Python HTTP server, torch loads only when a job runs). It reads the
|
||||
# collector's SQLite + crops off the same volume (read-only) and keeps models, reports and
|
||||
# job logs in its own volume. NOT published: only the collector reaches it, on this compose
|
||||
# network, and the reviewer's login on the collector is the gate. The Training section of
|
||||
# /review is its UI (readiness, Train / Evaluate / Publish, reports, logs).
|
||||
# See wiki/decisions/bodytype-classifier-training.md.
|
||||
trainer:
|
||||
image: ${REGISTRY:-git.infra.msai.al/mca/parking_solution}/parking-trainer:${TAG:-dev}
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
# Where `publish` PUTs a passing model (a Gitea generic package) and the token it uses
|
||||
# (package:write). Only publishing needs the token; training runs without it.
|
||||
TRAINER_PUBLISH_URL: ${TRAINER_PUBLISH_URL:-https://git.infra.msai.al/api/packages/mca/generic/parking-bodytype}
|
||||
TRAINER_PUBLISH_TOKEN: ${TRAINER_PUBLISH_TOKEN:-}
|
||||
volumes:
|
||||
- collector-data:/data:ro
|
||||
- trainer-out:/out
|
||||
|
||||
volumes:
|
||||
collector-data:
|
||||
trainer-out:
|
||||
|
||||
@@ -33,6 +33,7 @@ services:
|
||||
CARWASH_REVIEW_URL: ${CARWASH_REVIEW_URL:-}
|
||||
CARWASH_REVIEW_TOKEN: ${CARWASH_REVIEW_TOKEN:-}
|
||||
CARWASH_REVIEW_BOOTH_ID: ${CARWASH_REVIEW_BOOTH_ID:-}
|
||||
CARWASH_REVIEW_ENTRY_SAMPLE: ${CARWASH_REVIEW_ENTRY_SAMPLE:-0}
|
||||
# CRITICAL on the plain-HTTP booth LAN: cookies are Secure (HTTPS-only) by DEFAULT,
|
||||
# so without COOKIE_SECURE=0 the auth cookie is never sent over http and operators
|
||||
# CANNOT LOG IN. Leave unset only behind TLS. See disk-os-hardening "deploy-time runbook".
|
||||
|
||||
+12
-5
@@ -85,7 +85,7 @@ REGISTRY=git.infra.msai.al/mca/parking_solution
|
||||
# Staging booth: pinned immutable stage-<sha>. After each promotion (merge dev → stage, CI builds
|
||||
# :stage-<sha>), bump this to the new sha and re-sync/deploy from Core. The moving `:stage` tag
|
||||
# exists as the pointer; we deploy the sha, not the mover.
|
||||
TAG=stage-2aa1045
|
||||
TAG=stage-dbbb051
|
||||
COOKIE_SECURE=0
|
||||
# Venue modules this site is ENTITLED to (vendor decision; the site admin activates within
|
||||
# this set in Setup → Site). Unset = every registered module. See wiki/decisions/venue-modules.md.
|
||||
@@ -94,9 +94,12 @@ MODULES_ENTITLED=parking,carwash
|
||||
# Car Wash review outbox (wiki/concepts/vision-review-outbox.md): the collector's ingest URL
|
||||
# on the Netbird overlay, this booth's pseudonymous id, and its token — the SAME secret the
|
||||
# wash-collector stack lists under that id. Leave all three unset to keep the outbox off.
|
||||
#CARWASH_REVIEW_URL=http://docker-station.nb.infra:8090/ingest
|
||||
#CARWASH_REVIEW_BOOTH_ID=booth-2
|
||||
#CARWASH_REVIEW_TOKEN=[[wash_review_token_booth_2]]
|
||||
CARWASH_REVIEW_URL=http://docker-station.nb.infra:8090/ingest
|
||||
CARWASH_REVIEW_BOOTH_ID=booth-2
|
||||
CARWASH_REVIEW_TOKEN=[[wash_review_token_booth_2]]
|
||||
# Also send ENTRY reads as training material (gate view, no wash): 1 = every entry (storage
|
||||
# and bandwidth are not the limit; review what you have time for). N = one in N. 0 = off.
|
||||
CARWASH_REVIEW_ENTRY_SAMPLE=1
|
||||
VISION_ENABLED=1
|
||||
# Desktop app WS handshake: Origin is tauri://localhost (set explicitly by
|
||||
# platform-ws.ts, since the native WS plugin has no page context to auto-attach
|
||||
@@ -131,7 +134,7 @@ registry_account = "komodo"
|
||||
environment = """
|
||||
REGISTRY=git.infra.msai.al/mca/parking_solution
|
||||
# Pinned like the booths: bump to the stage-<sha> that carries the collector.
|
||||
TAG=stage-REPLACE
|
||||
TAG=stage-f7a262a
|
||||
# The host's NETBIRD address (an IP: Docker port bindings take no hostname) — the ingest port
|
||||
# is published on the overlay only. Booths reach it by its Netbird DNS name.
|
||||
COLLECTOR_BIND=100.75.184.156
|
||||
@@ -140,6 +143,10 @@ COLLECTOR_BIND=100.75.184.156
|
||||
# to keep in sync, and rotating a booth touches one secret. The booth id is the booth's
|
||||
# pseudonymous CARWASH_REVIEW_BOOTH_ID, never a site name. Add a pair per booth.
|
||||
COLLECTOR_BOOTH_TOKENS=booth-2:[[wash_review_token_booth_2]]
|
||||
# Phase-B trainer (the `trainer` service beside the collector; the Training section of /review
|
||||
# is its UI). Only `publish` needs this: a Gitea token with package:write for the model's generic
|
||||
# package. Uncomment when the first model is to be published.
|
||||
#TRAINER_PUBLISH_TOKEN=[[gitea_package_write_token]]
|
||||
COLLECTOR_REVIEWER_USER=reviewer
|
||||
COLLECTOR_REVIEWER_PASS=[[wash_collector_reviewer_pass]]
|
||||
"""
|
||||
|
||||
Generated
+2
@@ -117,6 +117,8 @@ importers:
|
||||
specifier: ^4.1.9
|
||||
version: 4.1.9(@types/node@25.9.3)(jsdom@25.0.1)(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4))
|
||||
|
||||
apps/trainer: {}
|
||||
|
||||
apps/vision: {}
|
||||
|
||||
apps/web:
|
||||
|
||||
@@ -39,6 +39,29 @@ locked-down collector without exposing anything to the open internet.
|
||||
image is dropped from the row once delivered; a voided order is abandoned unsent; anything
|
||||
older than 14 days is abandoned ("expired") rather than resurfacing a fortnight in a burst.
|
||||
|
||||
## The entry stream — the real accelerator (built 2026-09-07)
|
||||
|
||||
The wash stream is small; the **entry camera photographs every car**, in exactly the view the
|
||||
classifier is trained on, with zero domain shift. So the booth can also queue **one in N entry
|
||||
vehicle reads** as pure training material: the crop and the camera's class, *no* order, *no*
|
||||
operator, *no* category — same crop-and-blur pipeline, same one-way path, same privacy
|
||||
properties. `CARWASH_REVIEW_ENTRY_SAMPLE=N` = one in N entries; **`1` = every entry, and that is
|
||||
the setting park-2 runs** (user, 2026-09-07: 4 TB on the collector host, bandwidth not an issue —
|
||||
the only limit was ever the reviewer's time; the reviewer labels what they have time for, the
|
||||
rest waits and stays useful once a first model exists, as the unlabelled pile it is measured on).
|
||||
0/unset = off; needs the three upload settings. A washed car arrives twice, as an entry sample
|
||||
and as the wash decision — intended, the `kind` keeps them apart.
|
||||
Seam: the core announces every vehicle read (`deviceEvents.emitVehicleRead`, snapshot.ts, entry
|
||||
and exit) and the Car Wash module decides — it samples entry reads in-process (`sampleEntry()`,
|
||||
exactly one in N) and calls `enqueueEntry()`; the core never imports the module. Packages carry
|
||||
`kind: "wash" | "entry"`; the collector stores the kind, the review screen shows an entry sample
|
||||
as "entry stream — label the vehicle", the export carries a `kind` column, and **operator
|
||||
agreement is computed from wash items only** (an entry sample has no operator decision).
|
||||
|
||||
An internet feed was considered the same day and kept OUT of the collector's ingest: licensed
|
||||
sets only, in a separate folder with provenance, used as warm-up and weighted down, and never
|
||||
the judge of accuracy — the evaluation set is gate crops only.
|
||||
|
||||
## The package
|
||||
|
||||
`multipart/form-data`: `meta` (JSON) + `image` (JPEG). Meta = `{ v, booth, item, order, at,
|
||||
@@ -60,6 +83,24 @@ Setup → Car wash show queued / delivered / abandoned + the last error.
|
||||
is off and **nothing is queued** (an unbounded queue nobody drains is worse than none). Set per
|
||||
booth in the Komodo stack env; compose forwards them.
|
||||
|
||||
- **URL by Netbird DNS name** (`http://docker-station.nb.infra:8090/ingest`): the server container
|
||||
runs on the host network in prod, so it uses the booth's resolver and Netbird's DNS answers
|
||||
`*.nb.infra`; a collector that moves address costs no booth change. A failed lookup behaves like
|
||||
a collector outage (defer, backoff). The collector's own `COLLECTOR_BIND` must be the raw overlay
|
||||
**IP** — Docker port bindings take no hostname.
|
||||
- **Secrets: one per booth, two consumers.** `wash_review_token_booth_2` is referenced by the
|
||||
booth's stack as its `CARWASH_REVIEW_TOKEN` *and* by the collector's stack inside
|
||||
`COLLECTOR_BOOTH_TOKENS=booth-2:[[wash_review_token_booth_2]],booth-3:[[…]]` — one value, nothing
|
||||
to keep in sync, rotating a booth touches one secret. (A first cut had one combined secret for
|
||||
the whole list; replaced the same day — rotation was all-or-nothing and the value lived twice.)
|
||||
Token format: opaque, `openssl rand -hex 32`; the collector only demands ≥ 16 chars and the list
|
||||
splits on commas/whitespace, which hex never contains. Never share a token between booths — it
|
||||
is what names the booth. Total Komodo secrets for one booth + the collector: two (the booth's
|
||||
token, the reviewer's password).
|
||||
- **The operator hash needs no variable**: `sha256(boothId + ":" + username)[:16]`, computed on
|
||||
the booth from values already set; the owner recomputes it from the booth's usernames to map a
|
||||
hash back, the collector never can.
|
||||
|
||||
## The collector — skeleton built 2026-09-06 (`apps/collector`)
|
||||
|
||||
A deliberately small Fastify + SQLite service **in this monorepo** (so it imports the payload
|
||||
@@ -87,12 +128,32 @@ Three surfaces, nothing else — it must not grow into a fleet console:
|
||||
/ fraud rate.
|
||||
- **`GET /export/labels.csv`** — reviewed, usable rows: item, booth, crop path, the reviewer's
|
||||
label, the operator's category + classes, the camera's class + confidence, downgraded, at.
|
||||
Crops are not packaged: the phase-B trainer runs **on the same host** (its GPU) and reads them
|
||||
off the volume — `docker-compose.collector.yml` carries the `trainer` seam as a commented
|
||||
`profiles: [train]` one-off job (next increment).
|
||||
Crops are not packaged: the phase-B trainer runs **on the same host** and reads the SQLite
|
||||
+ crops straight off the volume, read-only ([[bodytype-classifier-training]]: CPU-only, the
|
||||
Xeon is enough) — the `trainer` service beside the collector in
|
||||
`docker-compose.collector.yml` (the CSV export stays for a human with a spreadsheet).
|
||||
- **Training section on `/review`** (+ `/api/training/status|jobs|jobs/:id|versions/:v/report`)
|
||||
— a thin proxy, behind the same reviewer login, to the trainer's job API on the compose
|
||||
network (`COLLECTOR_TRAINER_URL`, unset = hidden): labels per class vs the minimum, Train
|
||||
(mode / backbone / floor), the running job's log, the versions with Report / Evaluate /
|
||||
Publish. The collector forwards only a fixed set of paths and knobs; the trainer validates
|
||||
values and answers 409 while a job runs.
|
||||
|
||||
**Where the data lives.** The collector writes to `/data` in its container: `collector.sqlite`
|
||||
and one JPEG per item at `crops/<booth-id>/<item-id>.jpg`. `/data` is the named Docker volume
|
||||
`collector-data` (compose), on the host under Docker's volume directory — normally
|
||||
`/var/lib/docker/volumes/wash-collector_collector-data/_data/` (`docker volume inspect
|
||||
wash-collector_collector-data` confirms). The trainer mounts the same volume read-only at its
|
||||
own `/data`; nothing is copied or exported for training.
|
||||
|
||||
**Deploy notes.** Bind the published port to the host's **Netbird address** (`COLLECTOR_BIND`),
|
||||
never `0.0.0.0` on a host with a public interface; Netbird policy: booths → this host:8090 and
|
||||
nothing else. The host must be onboarded as a Komodo server like the booths. `TAG` is pinned
|
||||
and promoted with the booths (one sha for all stacks) — fine while the collector stays small;
|
||||
its own repo the day it needs its own cadence.
|
||||
its own repo the day it needs its own cadence. Deploy the collector BEFORE a booth that sends a
|
||||
package kind it does not know (a 422 is abandoned, not retried). The export neutralises cells
|
||||
that start like a spreadsheet formula (category/service names are booth-supplied text).
|
||||
|
||||
**Status (2026-09-07).** Live: the collector runs on `art-docker-station` and park-2 is wired to
|
||||
it (`stage-dbbb051` on both stacks, every entry sampled). The review screen at
|
||||
`http://docker-station.nb.infra:8090/review` is filling; no labels reviewed yet.
|
||||
|
||||
@@ -0,0 +1,193 @@
|
||||
---
|
||||
title: Body-type classifier (phase B) — training path and hardware
|
||||
type: decision
|
||||
status: decided 2026-09-07; BUILT 2026-09-07 (trainer + vision stage); first real run waits for labels
|
||||
related: [vision-review-outbox, opencv-anpr-service, venue-modules, fleet-deployment-komodo, vision-service-packaging, technology-stack]
|
||||
---
|
||||
|
||||
# Body-type classifier (phase B) — training path and hardware
|
||||
|
||||
The Car Wash category suggestion needs SUV vs sedan, which the phase-A COCO detector cannot give
|
||||
([[opencv-anpr-service]] §Vehicle body type). Phase B is a **classifier over the detector's crop**,
|
||||
trained on the reviewer's labels gathered through the [[vision-review-outbox]]. Decided with the
|
||||
user on 2026-09-07 (morning), **built the same day** once the user said "build the trainer for the
|
||||
Xeon". This page is the loop as built; what is still outstanding is at the end.
|
||||
|
||||
## The loop (as built)
|
||||
|
||||
Each step is a place where a person decides. Nothing here runs on its own.
|
||||
|
||||
1. **Train** — `apps/trainer` (`parking-trainer`, Python/uv like the vision service; its own
|
||||
image `parking-trainer`, the `trainer` service beside the collector on the reviewer's host —
|
||||
never a booth service). **Started from the collector's UI:** the Training section of
|
||||
`/review` (readiness, a Train button with mode / backbone / floor, the live log, the
|
||||
versions with Report / Evaluate / Publish) drives a small job API the trainer serves on the
|
||||
compose network (`serve`; stdlib HTTP, one job at a time, each job the CLI as a subprocess
|
||||
with its log persisted under `/out/jobs/`). The collector proxies it behind the reviewer's
|
||||
login; the trainer is never published. `train` reads the collector's `collector.sqlite` and `crops/` **straight off the volume**
|
||||
(read-only), takes only reviewed, usable rows (the operator's pick and the camera's class are
|
||||
never labels), **splits by TIME** (validation = the newest 20 % by *time seen*, so the number
|
||||
reflects tomorrow's traffic), drops classes with fewer than `--min-per-class` (20) labels from
|
||||
that run and reports them, weighs the loss by damped inverse frequency (√, mean 1 — full
|
||||
inverse over-corrects on small sets), trains, and exports. Two modes:
|
||||
- `--mode features` (default): the ImageNet backbone is **frozen**; every crop's feature
|
||||
vector is cached on disk (`<out>/cache/features-<backbone>-<size>.npz`, keyed by item id),
|
||||
and only a linear head is trained — minutes for thousands of crops, **seconds** to retrain
|
||||
when labels arrive (only new crops go through the backbone).
|
||||
- `--mode finetune`: warm-starts the head the same way, then unfreezes everything with light
|
||||
label-preserving augmentation (flip, mild zoom, brightness/contrast) — the step when the
|
||||
cheap mode plateaus.
|
||||
Backbones: `resnet18` (default), `mobilenet_v3_small`, `efficientnet_b0` — torchvision,
|
||||
BSD-3, and the ImageNet weights ship under the same licence (the licence rule applies to
|
||||
weights as much as code). CPU-only PyTorch from PyTorch's own wheel index (`tool.uv.index`).
|
||||
2. **Evaluate before anything ships** — every run writes `report.md` (accuracy, macro recall,
|
||||
per-class recall/precision, confusion matrix, dropped classes, loss weights, agreement with the
|
||||
detector's coarse class) and `metrics.json`. The job **refuses to write the model** below
|
||||
`--min-accuracy` (default 0.85) — exit 3, report still written — and also withholds it if the
|
||||
exported ONNX disagrees with the torch model on validation (< 99 % argmax agreement). Exit 2 =
|
||||
not enough labels (fewer than two classes clear the minimum). Later, `evaluate --model …`
|
||||
scores a shipped model against labels **reviewed after it was trained** (a clean held-out
|
||||
check) and prints its class histogram + detector agreement over the **unlabelled** pile — the
|
||||
ongoing drift check without labelling everything, which is why every entry is sent
|
||||
([[vision-review-outbox]] §The entry stream). 85–95 % on frontal gate views is the
|
||||
expectation once tuned; enough to *flag*, never to *bill*.
|
||||
3. **Publish** — weights are not code and do not live in git. `publish <version dir> --url
|
||||
https://git.infra.msai.al/api/packages/mca/generic/parking-bodytype` PUTs the four files as a
|
||||
**Gitea generic package** version (token with `package:write`, `TRAINER_PUBLISH_TOKEN`).
|
||||
4. **Bake** — `apps/vision/models/bodytype.version` (tracked in git, empty today) **pins** the
|
||||
version the vision image carries. The Dockerfile fetches `bodytype.onnx` + `bodytype.json`
|
||||
from the package registry at build (auth via a BuildKit secret `bodytype_auth` = the
|
||||
registry user's credentials, never a layer); **a pinned version that cannot be fetched fails
|
||||
the build** — the image must carry what git says it carries; empty pin = no classifier, phase
|
||||
B off, build passes. The pin is a normal commit: reviewable, revertible.
|
||||
5. **Deploy** — a TAG bump on the booth's stack. **A booth gets a model the way it gets code**: a
|
||||
pinned release you can see and roll back. No runtime model fetch (air-gapped appliance,
|
||||
read-only model path — [[vision-service-hardening]]).
|
||||
|
||||
Retrain when the labels have grown meaningfully (every few hundred new verdicts at first). First
|
||||
run needs roughly **200 reviewed crops per class that matters** (Vetura and SUV at least) — until
|
||||
then `inspect` says `ready: false` and `train` exits 2.
|
||||
|
||||
## The contract between trainer and booth
|
||||
|
||||
The trainer and the vision service share **no code** (different packages, different images), so
|
||||
the preprocessing contract is **data**: the `bodytype.json` sidecar (`format:
|
||||
parking-bodytype/1`) carries version, the class list (a subset of the shared vocabulary, in
|
||||
vocabulary order), `input_size` (224), `crop_margin` (0.08 — the same as the outbox's
|
||||
`makeReviewCrop`), colour order, resize method, backbone, mode, label counts and the validation
|
||||
metrics. Both sides cut the detector's box + margin, blur the plate strip, squash-resize with
|
||||
OpenCV `INTER_AREA` (the crop *is* the vehicle; no centre-crop that loses a bumper), and feed raw
|
||||
RGB 0–255 float; **normalisation lives inside the ONNX graph**, so a consumer cannot get the
|
||||
constants wrong. Verified on 2026-09-07: a trainer-produced model loaded by the vision service's
|
||||
`BodyTypeClassifier` gives identical classes and probabilities (< 1e-4) to the trainer's own
|
||||
`OnnxClassifier` on the same crops.
|
||||
|
||||
On the booth ([[opencv-anpr-service]] §Phase B): `RefinedVehicleDetector` runs the classifier
|
||||
only when the detector said `car` **or** a class the classifier trained on; a truck or bus it
|
||||
never saw is left alone (its softmax on an unknown thing means nothing). Below
|
||||
`VISION_VEHICLE_CLASSIFIER_MIN_CONFIDENCE` (0.6) the detector's class stands. `vehicle.
|
||||
detector_class` records the coarse class whenever the stage ran; `model_version` reads
|
||||
`…+yolox:…+bodytype:<version>`. The flag on the desk, the mapping chips, the threshold — nothing
|
||||
downstream changed: the vocabulary already held sedan/hatchback/suv/minivan/pickup.
|
||||
|
||||
## One model for the fleet, not one per site (user asked, 2026-09-07)
|
||||
|
||||
The trainer pools **every booth's** reviewed labels into one training set (no per-booth filter),
|
||||
and one `bodytype.version` pin bakes one model into the one vision image every booth runs. Body
|
||||
type is a property of the car, not the site; pooling is what makes 200 crops per class reachable;
|
||||
and a single pinned version is the whole "a booth gets a model the way it gets code" idea. What
|
||||
*is* per site stays in Setup: the class→category mapping and the flag threshold — the model says
|
||||
"suv", the site decides what an SUV costs and when a downgrade is worth flagging.
|
||||
|
||||
Where a site can still differ is the **camera** (mount height, angle, lens), not the cars. Every
|
||||
crop carries its booth id, so the report can break accuracy down per booth — not in the report
|
||||
today; add it once a second site sends labels. A booth filter in the trainer and a second pin
|
||||
would only be built on evidence that a site's view needs its own model.
|
||||
|
||||
## Secrets and access (2026-09-07)
|
||||
|
||||
- **`TRAINER_PUBLISH_TOKEN`** — a Gitea access token with the `write:package` scope, used by the
|
||||
`publish` command and nothing else, to PUT a passing model's files into the generic package
|
||||
`mca/parking-bodytype`. Training, `inspect` and `evaluate` need no token; leave it unset until
|
||||
the first model passes the floor. Create it under a user who can write packages in the `mca`
|
||||
org, store it as the Komodo secret `gitea_package_write_token`, uncomment the line in the
|
||||
`wash-collector` stack.
|
||||
- **Read side** — the CI build fetches the pinned version with the existing registry
|
||||
credentials (`REGISTRY_USERNAME:PASSWORD` as the BuildKit secret `bodytype_auth`); if the
|
||||
package is org-private that user needs package read, which the Docker-registry user already
|
||||
has in Gitea.
|
||||
|
||||
## Hardware (decided 2026-09-07)
|
||||
|
||||
What the owner has: an **NVIDIA Quadro FX 3800** (in hand, not installed), and in
|
||||
`art-docker-station` an **Intel Xeon E3-1225 v5** (4 Skylake cores, AVX2, no AVX-512) with the
|
||||
**Intel HD P530** iGPU.
|
||||
|
||||
- **Quadro FX 3800 — stays in the drawer.** 2009, GT200, compute capability 1.3, 1 GB. CUDA dropped
|
||||
that generation in 2015; no PyTorch build of the last decade can use it. Installing it buys a
|
||||
heater and a driver problem.
|
||||
- **HD P530 — not for training.** Usable for *inference* via OpenVINO, irrelevant here: inference
|
||||
runs on the booths' CPUs, which already do YOLOX in ~250 ms.
|
||||
- **The Xeon does the job.** The problem is small (a few thousand 224-px crops, ten classes, a
|
||||
small pretrained backbone): features mode in minutes, a full fine-tune in roughly an hour with a
|
||||
mobile-sized backbone. Training is occasional and unattended, and the data is already on that
|
||||
host, so nothing moves.
|
||||
- **Consequences for the build (done):** the trainer image is **CPU-only PyTorch** (torch
|
||||
2.14+cpu, ~200 MB of wheels, not the ~5 GB CUDA build); the `trainer` service in
|
||||
`docker-compose.collector.yml` is real — always on, serving the job API, no device
|
||||
reservation (one block to add if a modern card ever lands; the trainer would pick up CUDA),
|
||||
the collector volume mounted read-only, models/reports/logs in its own `trainer-out` volume.
|
||||
- **If faster is ever wanted:** a used mid-range card of the last few generations (~€200) turns
|
||||
the hour into a minute, given a slot and a PSU. **Renting a cloud GPU is rejected**: the crops
|
||||
would leave the premises, and even scrubbed of plates and site that runs against the whole
|
||||
privacy design of the outbox.
|
||||
|
||||
## Running it
|
||||
|
||||
From the collector's `/review` page, Training section: **Train** (mode, backbone, floor) when
|
||||
readiness says enough labels; watch the log; read the report under Versions; **Evaluate** a
|
||||
written version against labels reviewed since; **Publish** it (needs `TRAINER_PUBLISH_TOKEN`
|
||||
in the `wash-collector` stack — commented until the first publish). Then, in git: write the
|
||||
version into `apps/vision/models/bodytype.version`, commit, let the build produce the image,
|
||||
bump the booth's `TAG`. The pin stays a commit on purpose — it is the deploy control.
|
||||
|
||||
The CLI is still there for debugging, inside the running container:
|
||||
`docker compose -f docker-compose.collector.yml exec trainer parking-trainer inspect`.
|
||||
|
||||
## Operating notes (2026-09-07)
|
||||
|
||||
- **First deploy (`stage-f7a262a`) shipped the trainer as a compose *profile*** — a one-off
|
||||
job the owner had to start by hand with `docker compose … --profile train run …` from
|
||||
wherever Komodo's periphery had cloned the repo (`/etc/komodo/stacks/wash-collector/`).
|
||||
The user rightly called that "not so smart": the host runs a periphery, and the reviewer is
|
||||
already in the collector's UI. **Superseded the same day:** the trainer is now a
|
||||
**service** (`restart: unless-stopped`, the `serve` command) and the collector's
|
||||
`/review` page carries the Training section. A deploy starts both containers; `docker ps`
|
||||
shows two.
|
||||
- **Why not a Docker socket in the collector** (the other way to a button): it would hand
|
||||
root on the host to a service that accepts uploads from booths — the party the
|
||||
[[threat-model]] distrusts. The job API keeps the trainer a normal container with a
|
||||
read-only data mount and its own `trainer-out` volume.
|
||||
- **park-2 does not need a bump** until a model is pinned: the vision image ships with an
|
||||
empty `bodytype.version`, phase B off, nothing for a booth to gain.
|
||||
- **Reviewing is the bottleneck**: the Training section shows labels per class against the
|
||||
minimum and keeps Train disabled until two classes clear it.
|
||||
|
||||
## Packaging rule (same as the vision service)
|
||||
|
||||
Core deps are light (numpy, opencv-headless, onnxruntime): `inspect`, `evaluate`, the data and
|
||||
report code, and the tests run with `uv sync --frozen` alone — **CI syncs without the `train`
|
||||
extra** ([[vision-service-packaging]]); the torch tests `importorskip`. The image bakes
|
||||
`--extra train` and pre-warms the resnet18 + mobilenet_v3_small ImageNet weights so a run needs
|
||||
no network. `pnpm turbo run lint test` covers `@parking/trainer` through the same package.json
|
||||
shim pattern (workspace count 7→8).
|
||||
|
||||
## Outstanding
|
||||
|
||||
- **The first real run** — waits for ~200 reviewed crops per class on the collector (reviewing
|
||||
is the bottleneck now, not code).
|
||||
- **Secrets on the reviewer's host** — a Gitea token with `package:write`
|
||||
(`gitea_package_write_token`) for `publish`; the CI registry user must be able to *read* the
|
||||
generic package (it passes its credentials as the build secret).
|
||||
- **Tuning knobs after the first report** — the floor, `--min-per-class`, whether finetune beats
|
||||
features on this camera. The report decides, not a guess.
|
||||
@@ -484,3 +484,16 @@ booth, `pkexec dpkg -i`, polkit dialog, relaunch, badge shows 0.1.7). The prompt
|
||||
- **Operator-facing consequence:** the in-app prompt now says the install needs the
|
||||
administrator password (i18n `update.prompt`, en + sq). An operator who accepts and can't
|
||||
authenticate simply stays on the current version; nothing breaks, and the failure is logged.
|
||||
|
||||
### v0.2.0 — the first feature release of the desktop bundle (2026-09-07)
|
||||
|
||||
Every tag from v0.1.0 to v0.1.7 was a desktop-shell fix (origins, cookies, WS tickets, the
|
||||
updater manifest). Since v0.1.7 the SPA the bundle carries (`frontendDist: ../../web/dist`)
|
||||
gained the venue-module registry, the Car Wash module with per-till shifts and the wash-desk
|
||||
printer role, roles that remember their jobs with signed edits, the advisory vehicle category
|
||||
from the entry camera, the review outbox status in Setup, and the two-column Car Wash setup —
|
||||
31 commits, none of them shell fixes. Under 0.x that is a **minor** bump, not a patch: **v0.2.0**.
|
||||
`tauri.conf.json` now says 0.2.0 too (the release workflow still rewrites it from the tag, so
|
||||
the file only matters for local bundles). The README's release gate — run the real bundle, LIVE,
|
||||
one mutation, a frontend log row — is still the step between the tag and the push of the tag.
|
||||
|
||||
|
||||
@@ -158,6 +158,12 @@ collector ([[vision-review-outbox]]) runs on the reviewer's GPU host as its own
|
||||
(`wash-collector`, `server = "art-docker-station"`, `file_paths = ["docker-compose.collector.yml"]`).
|
||||
Same repo, branch and pinned `TAG` promotion, its own secret references, and — because a stack
|
||||
names its compose files — nothing booth-side lands on that host and nothing of it on a booth.
|
||||
The same stack carries the phase-B **trainer** as a second service ([[bodytype-classifier-training]]):
|
||||
a deploy starts both, `docker ps` shows two containers, and the trainer is driven from the
|
||||
collector's UI, never from the host's shell (a first cut as a compose *profile* run by hand was
|
||||
replaced the same day — the host runs a periphery, nobody should be typing compose there). Its two env lines (`TRAINER_OUT`, the
|
||||
`TRAINER_PUBLISH_TOKEN` secret reference) stay commented in `resources.toml` until the first
|
||||
publish.
|
||||
|
||||
## Open / not yet done
|
||||
|
||||
|
||||
@@ -87,6 +87,19 @@ The skeleton is **built and wired** (no recognizer models yet):
|
||||
(env `VISION_*`), `schemas.py` (the `/analyze` contract incl. a not-yet-populated `vehicle` field
|
||||
for Job 2), `recognizer.py` (a `Recognizer` **Protocol** + `StubRecognizer` and `FastAlprRecognizer`
|
||||
— the [[device-adapter-pattern]] applied to the model).
|
||||
- **CI runs WITHOUT the extra** (`uv sync --frozen` in ci.yml and build-images.yml): a test that
|
||||
imports numpy/cv2 at module level breaks collection there even though it passes in a local venv
|
||||
that has `alpr`. Rule (2026-09-07, after three red runs): pure post-processing tests get numpy
|
||||
from the **dev group**; anything needing OpenCV uses `pytest.importorskip("cv2")`; the service
|
||||
itself imports both lazily inside functions.
|
||||
- **The same pattern, second package (2026-09-07):** `apps/trainer` (`@parking/trainer`,
|
||||
[[bodytype-classifier-training]]) — light core (numpy, opencv-headless, onnxruntime) + a
|
||||
`train` extra (CPU-only torch/torchvision/onnx/onnxscript from PyTorch's wheel index via
|
||||
`tool.uv.index`); CI syncs without it, torch tests `importorskip("torch")`, the module that
|
||||
imports torch is imported lazily by the `train` command only. Its own image
|
||||
(`parking-trainer`, context `apps/trainer`, uv base image, bakes `--extra train` + the
|
||||
ImageNet backbone weights) is built by build-images.yml beside the other three; both Python
|
||||
contexts now carry a `.dockerignore` (venv/caches/weights out). Workspace count 7→8.
|
||||
- **Light-core, heavy-optional:** core deps boot in **stub mode** (no model download) so `uv sync` +
|
||||
tests work offline; the real stack is the `alpr` extra (`uv sync --extra alpr` →
|
||||
fast-alpr + onnxruntime). `VISION_RECOGNIZER=fast_alpr` switches it on.
|
||||
|
||||
@@ -251,6 +251,10 @@ service's `/health` each tick and shows a **"Vision" chip** in the booth footer
|
||||
|
||||
## Vehicle body type (advisory) — the vehicle stage, phase A (2026-09-06)
|
||||
|
||||
> Phase B (the classifier that knows SUV from sedan), its training loop and the hardware it runs on
|
||||
> are on [[bodytype-classifier-training]] — built 2026-09-07, see §Phase B below; no model is
|
||||
> pinned yet (the stage is off until the first published version).
|
||||
|
||||
`/analyze` populates `vehicle.body_type` + `vehicle.confidence` from the shared vocabulary
|
||||
(car, sedan, hatchback, suv, minivan, pickup, van, truck, bus, motorcycle). Node records it beside
|
||||
the plate and the Car Wash desk pre-selects the category the site maps it to; the operator
|
||||
@@ -290,3 +294,31 @@ read. Composed `model_version` reads `<plate>+yolox:yolox_s.onnx@640`.
|
||||
operator's picks (untrusted — [[threat-model]]) but a trusted reviewer's, gathered through the
|
||||
[[vision-review-outbox]]. Expect 85–95 % on frontal gate views once tuned — enough to flag,
|
||||
never to bill, which is why the flag records and the site threshold exists.
|
||||
|
||||
### Phase B — the body-type classifier stage (built 2026-09-07)
|
||||
|
||||
`vehicle.py` gained a second stage: `BodyTypeClassifier` loads `bodytype.onnx` + its
|
||||
`bodytype.json` sidecar (produced by `apps/trainer`, [[bodytype-classifier-training]] §The
|
||||
contract) and `RefinedVehicleDetector` composes it over the YOLOX detector — the detector still
|
||||
finds and picks the vehicle, the classifier answers on its crop. `crop_vehicle` mirrors the
|
||||
outbox's `makeReviewCrop` (box + the sidecar's margin, plate strip Gaussian-blurred) so the
|
||||
booth sees what the model was trained on; resize is OpenCV `INTER_AREA` at the sidecar's
|
||||
`input_size`, raw RGB 0–255 in, normalisation inside the graph.
|
||||
|
||||
- **Rule:** the classifier runs only when the detector said `car` **or** a class the classifier
|
||||
trained on; a truck/bus/motorcycle it never saw is left alone. Below
|
||||
`VISION_VEHICLE_CLASSIFIER_MIN_CONFIDENCE` (0.6) the detector's class stands. When the stage
|
||||
ran, `vehicle.detector_class` carries the coarse class (Node ignores it today; the collector
|
||||
could show it). `model_version` reads `<plate>+yolox:…+bodytype:<version>`.
|
||||
- **Config:** `VISION_VEHICLE_CLASSIFIER_PATH` (the image sets `/app/models/bodytype.onnx`) and
|
||||
the min-confidence. **Path set but no file = the normal state before the first model** —
|
||||
phase A only, one log line, *no* `/health.detail` error. A file that fails to load IS an error
|
||||
in `detail` (`classifier: …`), and a classifier that throws per frame is caught, noted, and the
|
||||
detector's answer returned — the plate read is never at risk.
|
||||
- **Bake:** `apps/vision/models/bodytype.version` (tracked; empty) pins the published version the
|
||||
Dockerfile fetches from the Gitea generic package registry (BuildKit secret `bodytype_auth`);
|
||||
a pin that cannot be fetched fails the build, an empty pin passes with phase B off.
|
||||
- **Tests** (`tests/test_vehicle.py`): crop margin/clamp/blur, the refine rule (car → suv when
|
||||
confident; unsure → detector's class; unknown bus untouched; a classifier that knows trucks may
|
||||
override a truck), a throwing classifier survives and is reported, missing files → not ready,
|
||||
and the factory skips a missing model without an error.
|
||||
|
||||
@@ -133,6 +133,7 @@ Counts: 4 sources · 19 entities · 47 concepts · 8 decision records.
|
||||
- [[dingtian-vs-mqtt]] — transport choice: direct HTTP/UDP now, MQTT parked until multi-lane scale.
|
||||
- [[session-model]] — business layer start: session = projection; transient-first; pay-on-foot. New event types.
|
||||
- [[vision-service]] — build a host-side ANPR + vehicle-verification service; replaces edge-LPR; scoped AGPL exception.
|
||||
- [[bodytype-classifier-training]] — phase B (SUV vs sedan): `apps/trainer` trains on the collector host (time split, floor, features/finetune modes, feature cache) → report → publish to the Gitea generic package → `models/bodytype.version` pin bakes it into the vision image → TAG bump; sidecar = the preprocessing contract; CPU-only torch on the Xeon E3-1225 v5, Quadro FX 3800 unusable, cloud GPU rejected. Built 2026-09-07; first run waits for labels.
|
||||
- [[vision-service-packaging]] — the vision service lives in this monorepo (apps/vision/), separate process, wired into Turbo via a package.json shim; uv-managed Python.
|
||||
- [[event-streams-split]] — split the signed business ledger (ledger_events) from unsigned device telemetry (device_events).
|
||||
- [[desktop-shell-tauri]] — ✅ Tauri v2 chosen over Electron for the desktop kiosk shell; thin wrapper, server keeps all logic. Best case Ubuntu 26.04 LTS (resolves WebKitGTK); worst case Windows+WSL → kiosk browser, no native shell. Auto-updater mirrors signed releases to public `mca/public_releases` (source repo is private — field appliances have no Gitea creds).
|
||||
|
||||
+78
@@ -3121,3 +3121,81 @@ the overlay address; commented `trainer` profile seam for the GPU), a third buil
|
||||
build-images.yml, and a `wash-collector` stack on `art-docker-station` in komodo/resources.toml
|
||||
(secret refs to fill). Booth payload now carries `operatorCategory.classes`. Tests: app.test.ts.
|
||||
Updated [[vision-review-outbox]], [[fleet-deployment-komodo]].
|
||||
|
||||
## [2026-09-07] ingest | Entry-stream sampling for the review outbox
|
||||
User asked about feeding internet pictures through the collector; assessment: licensed only,
|
||||
separate folder, warm-up weight, never the evaluation set — and the stronger accelerator is the
|
||||
ENTRY stream (every car, the gate view, zero domain shift). Built: `deviceEvents.emitVehicleRead`
|
||||
from snapshot.ts (core announces; the module listens), `ReviewOutbox.sampleEntry()` (one in N,
|
||||
in-process) + `enqueueEntry()` (crop + camera class, no order/operator/category), env
|
||||
`CARWASH_REVIEW_ENTRY_SAMPLE` (compose + resources template + .env.example), packages carry
|
||||
`kind`; the collector stores kind, the review screen shows entry samples as such, export has a
|
||||
kind column, operator agreement is wash-only. Setup line shows "1 in N entries sampled". Also:
|
||||
Setup → Car wash is a two-column grid (the master-data card was squeezed at max-w-2xl). Tests
|
||||
on both sides. Updated [[vision-review-outbox]].
|
||||
|
||||
## [2026-09-07] decide | Phase B training path + hardware — recorded, not built
|
||||
User asked "now what about the training" and then "let's talk hardware". Recorded on the new
|
||||
[[bodytype-classifier-training]]: the five-step loop (train on the collector host → evaluate with a
|
||||
floor → publish weights to the registry → bake into the vision image → TAG bump; a booth gets a
|
||||
model the way it gets code, never a runtime fetch); ~200 reviewed crops per class before the first
|
||||
run; the Quadro FX 3800 is unusable (cc 1.3), the HD P530 irrelevant, the Xeon E3-1225 v5 is enough
|
||||
(feature-extraction head in minutes, full fine-tune ~1 h); trainer image = CPU-only torch, the
|
||||
compose seam drops the GPU reservation; cloud GPU rejected (crops stay on premises). Linked from
|
||||
[[opencv-anpr-service]], [[vision-review-outbox]], index. User: "No build just yet."
|
||||
|
||||
## [2026-09-07] decision | Desktop v0.2.0 — a minor bump, not a patch
|
||||
User: "Do you think we are ready for version 0.2.0? The actual version is 0.1.7." Yes: v0.1.x
|
||||
were all shell fixes; the bundled SPA now carries the module registry, Car Wash + per-till
|
||||
shifts, roles jobs, the vision category and the review outbox (31 commits since v0.1.7).
|
||||
`tauri.conf.json` set to 0.2.0, annotated tag `v0.2.0` created locally; the release gate in
|
||||
apps/desktop/README.md (real bundle, LIVE, a mutation, a frontend log row) stands between the
|
||||
tag and its push. Recorded on [[desktop-shell-tauri]].
|
||||
|
||||
## [2026-09-07] build | Training from the collector UI — the trainer becomes a job service
|
||||
User: the compose-profile trainer is "not so smart" (where is the compose file on a periphery
|
||||
host? why not a button on the collector UI?). Built: `parking-trainer serve` — a stdlib job API
|
||||
(`/health`, `/readiness`, `/versions`, `/versions/<v>/report`, `/jobs`), one job at a time, each
|
||||
job the CLI as a subprocess with state + log persisted under `/out/jobs/`; the collector gained
|
||||
`COLLECTOR_TRAINER_URL` + `/api/training/*` (reviewer-gated proxy, fixed paths, whitelisted
|
||||
knobs, trainer status codes passed through, 503 unconfigured / 502 unreachable) and a Training
|
||||
section on `/review` (readiness table, Train with mode/backbone/floor, live log, versions with
|
||||
Report / Evaluate / Publish, the pin reminder). Compose: `trainer` is a service now
|
||||
(`restart: unless-stopped`, `serve`, read-only data, own `trainer-out` volume, not published);
|
||||
the Docker socket route was rejected (root on the host for a service booths upload to). Tests:
|
||||
trainer 14, collector 7. Pages: [[bodytype-classifier-training]] (loop, running it, operating
|
||||
notes superseded), [[vision-review-outbox]], [[fleet-deployment-komodo]].
|
||||
|
||||
## [2026-09-07] ingest | Trainer deployed as a profile; operating notes
|
||||
User pushed `stage-f7a262a`, bumped the `wash-collector` TAG, redeployed, and asked why only one
|
||||
service runs on art-docker-station. Expected: the trainer is a compose profile, never started or
|
||||
pulled by a deploy; run by hand, exits. Recorded on [[bodytype-classifier-training]] §Operating
|
||||
notes (why the TAG bump still mattered, park-2 needs no bump until a pin, the registry login for
|
||||
the first pull, the order of commands) and [[fleet-deployment-komodo]].
|
||||
|
||||
## [2026-09-07] build | Phase B trainer + the classifier stage on the booth
|
||||
User: "Shall we go and build the trainer for the Xeon?" Built `apps/trainer` (`parking-trainer`:
|
||||
`inspect` / `train` / `evaluate` / `publish`; reads the collector volume read-only, time split,
|
||||
thin classes dropped, damped class weights, `features` mode with an on-disk feature cache and
|
||||
`finetune` mode with light augmentation, CPU-only torch from PyTorch's wheel index, ONNX export
|
||||
checked against the torch model, **no model file below the floor** — exit 3 with the report; exit
|
||||
2 = not enough labels), its image + a `.dockerignore`, and the `trainer` compose profile on the
|
||||
collector stack (CPU, read-only data volume, `TRAINER_OUT`). Vision side: `BodyTypeClassifier` +
|
||||
`RefinedVehicleDetector` (car or a known class only; min-confidence; `detector_class`; missing
|
||||
file = off without an error, broken file = health detail), `models/bodytype.version` pin fetched
|
||||
at build from the Gitea generic package (BuildKit secret; a pin that cannot be fetched fails the
|
||||
build). The sidecar is the preprocessing contract; verified a trainer model gives identical
|
||||
probabilities inside the vision service. CI: trainer synced without the `train` extra, torch tests
|
||||
skip. Tests: trainer 10 (6 in CI mode), vision 18. Pages: [[bodytype-classifier-training]]
|
||||
rewritten as built, [[opencv-anpr-service]] §Phase B, [[vision-review-outbox]],
|
||||
[[vision-service-packaging]], [[fleet-deployment-komodo]], index.
|
||||
|
||||
## [2026-09-07] ingest | Collector live on park-2; secrets shape, DNS vs bind, token format, CI rule
|
||||
Deployed: collector on art-docker-station + park-2 at stage-dbbb051, every entry sampled; review
|
||||
screen filling. Recorded on [[vision-review-outbox]]: one secret per booth referenced by both stacks
|
||||
(the combined-list secret was replaced the same day), URL by Netbird DNS name vs bind by IP, token =
|
||||
`openssl rand -hex 32` (opaque, ≥16, no separators, never shared), the operator hash needs no
|
||||
variable, deploy the collector before a booth that sends a new package kind, the export's formula
|
||||
neutralisation (security review finding), and why every entry is sent. On
|
||||
[[vision-service-packaging]]: CI syncs without the alpr extra — numpy in the dev group, cv2 tests
|
||||
importorskip (three red runs on 2026-09-07).
|
||||
|
||||
Reference in New Issue
Block a user