From bfb6ab0b364acf9166df772b9ee3a1fc6ddca54e Mon Sep 17 00:00:00 2001 From: Julian Cuni Date: Fri, 19 Jun 2026 12:54:22 +0200 Subject: [PATCH] =?UTF-8?q?feat(logs):=20app=20log=20store=20=E2=80=94=20b?= =?UTF-8?q?ackend=20pino=20DB=20sink=20+=20frontend=20error=20collection?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add a third data stream (app_logs), distinct from the signed ledger and device telemetry, for operational/diagnostic logs — an offline appliance has no Sentry to ship to, so the host is the log store. Backend: a pino stream tees warn/error/fatal into app_logs (info/debug stay stdout-only) with no call-site change; the DB is built before Fastify so the logger has its sink. Frontend (lib/logger.ts): ships failed API requests (minus 401 churn), window.onerror, unhandledrejection, and a top-level React ErrorBoundary; console warn/error forwarded only at debug/trace. Batched/throttled POST, sendBeacon on pagehide, loop-safe (never logs the /api/logs call), best-effort everywhere. POST /api/logs (any signed-in user, CSRF, tolerant) + GET /api/logs gated by a new log:read permission (new `log` RBAC resource; admin holds it). Retention: pruned by age + row cap, hourly + at startup. UI: a Logs screen under /setup (filter level/source/since, expand to context+stack), sq+en. Migration 0009_app_logs. Verified end-to-end via app.inject: login -> POST 204 -> GET 200 with the record; backend warn/error persisted, info dropped; non-admin GET 403 / POST 204. Claude-Session: https://claude.ai/code/session_01Xcm6ikLgGoCxxHrxtjkk5V --- apps/server/src/log-service.ts | 236 +++++++++++++++++++++++++ apps/server/src/routes/logs.ts | 66 +++++++ apps/server/src/server.ts | 32 +++- apps/web/src/LogsViewer.tsx | 159 +++++++++++++++++ apps/web/src/api.ts | 32 +++- apps/web/src/lib/ErrorBoundary.tsx | 50 ++++++ apps/web/src/lib/i18n/en.ts | 19 ++ apps/web/src/lib/i18n/sq.ts | 19 ++ apps/web/src/lib/logger.ts | 198 +++++++++++++++++++++ apps/web/src/main.tsx | 10 +- apps/web/src/router.tsx | 12 ++ packages/db/drizzle/0009_app_logs.sql | 20 +++ packages/db/drizzle/meta/_journal.json | 7 + packages/db/src/schema.ts | 34 ++++ packages/shared/src/index.ts | 48 +++++ wiki/concepts/app-logs.md | 101 +++++++++++ wiki/concepts/device-events.md | 5 + wiki/decisions/event-streams-split.md | 10 +- wiki/index.md | 3 +- wiki/log.md | 12 ++ 20 files changed, 1064 insertions(+), 9 deletions(-) create mode 100644 apps/server/src/log-service.ts create mode 100644 apps/server/src/routes/logs.ts create mode 100644 apps/web/src/LogsViewer.tsx create mode 100644 apps/web/src/lib/ErrorBoundary.tsx create mode 100644 apps/web/src/lib/logger.ts create mode 100644 packages/db/drizzle/0009_app_logs.sql create mode 100644 wiki/concepts/app-logs.md diff --git a/apps/server/src/log-service.ts b/apps/server/src/log-service.ts new file mode 100644 index 0000000..2be170e --- /dev/null +++ b/apps/server/src/log-service.ts @@ -0,0 +1,236 @@ +import { randomUUID } from "node:crypto"; +import { and, appLogs, desc, eq, sql, type Db } from "@parking/db"; +import { + LOG_LEVEL_ORDER, + type AppLogRecord, + type ClientLogInput, + type LogLevel, + type LogSource, +} from "@parking/shared"; + +// Application/diagnostic LOG SINK — the host-side store behind the third log stream +// (app_logs), distinct from the signed ledger and device telemetry. It persists: +// - BACKEND warn/error/fatal, fed by a pino stream (see pinoDbStream) so any +// app.log.warn/error lands in the DB without changing call sites. +// - FRONTEND errors POSTed to /api/logs (failed requests, uncaught errors). +// Everything here is UNSIGNED + prunable. Pruned by age AND a row cap so an offline +// appliance with finite disk can't be filled by a log storm. See +// wiki/concepts/app-logs.md, decisions/event-streams-split.md. + +/** Only warn and above are persisted from the backend (info/debug stay stdout-only). */ +const BACKEND_PERSIST_MIN: LogLevel = "warn"; + +/** Defensive caps so one runaway log can't bloat a row (chars). */ +const MAX_MESSAGE = 4_000; +const MAX_STACK = 16_000; +const MAX_CONTEXT_JSON = 16_000; + +export interface LogRetention { + /** Delete logs older than this many days. */ + readonly maxAgeDays: number; + /** Hard cap on total rows — the oldest beyond this are pruned. */ + readonly maxRows: number; +} + +export const DEFAULT_RETENTION: LogRetention = { + maxAgeDays: Number(process.env.LOG_RETENTION_DAYS ?? 30), + maxRows: Number(process.env.LOG_RETENTION_MAX_ROWS ?? 50_000), +}; + +function clamp(s: string | null | undefined, max: number): string | null { + if (s == null) return null; + return s.length > max ? s.slice(0, max) : s; +} + +/** Serialize context to JSON, bounded — never throw on a circular/huge object. */ +function safeContext(ctx: Record | null | undefined): Record | null { + if (ctx == null) return null; + try { + const json = JSON.stringify(ctx); + if (json.length <= MAX_CONTEXT_JSON) return ctx; + return { _truncated: true, preview: json.slice(0, MAX_CONTEXT_JSON) }; + } catch { + return { _unserializable: true }; + } +} + +export class LogService { + readonly #db: Db; + readonly #retention: LogRetention; + /** Reentrancy guard: never let persisting a log itself emit a persisted log. */ + #writing = false; + + constructor(db: Db, retention: LogRetention = DEFAULT_RETENTION) { + this.#db = db; + this.#retention = retention; + } + + /** Low-level insert. Best-effort: a logging failure must never break a request or + * recurse (a DB error here would otherwise log → insert → error → log …). */ + #insert(row: { + level: LogLevel; + source: LogSource; + message: string; + context?: Record | null; + httpStatus?: number | null; + path?: string | null; + stack?: string | null; + userId?: string | null; + userAgent?: string | null; + createdAt?: string; + }): void { + if (this.#writing) return; + this.#writing = true; + try { + this.#db + .insert(appLogs) + .values({ + id: randomUUID(), + level: row.level, + source: row.source, + message: clamp(row.message, MAX_MESSAGE) ?? "", + context: safeContext(row.context), + httpStatus: row.httpStatus ?? null, + path: clamp(row.path, 512), + stack: clamp(row.stack, MAX_STACK), + userId: row.userId ?? null, + userAgent: clamp(row.userAgent, 512), + createdAt: row.createdAt ?? new Date().toISOString(), + }) + .run(); + } catch { + // Swallow — diagnostics must never take down the path they observe. (Can't log + // it; that's the recursion we're guarding against.) + } finally { + this.#writing = false; + } + } + + /** Persist a BACKEND log line (called by the pino stream). Below warn is dropped. */ + recordBackend(level: LogLevel, message: string, context?: Record | null): void { + if (LOG_LEVEL_ORDER[level] < LOG_LEVEL_ORDER[BACKEND_PERSIST_MIN]) return; + this.#insert({ level, source: "backend", message, context }); + } + + /** Persist a FRONTEND-reported log (from POST /api/logs). The server stamps the + * user + receive time; the client supplies level/message/context. */ + recordClient( + input: ClientLogInput, + meta: { userId?: string | null; userAgent?: string | null }, + ): void { + this.#insert({ + level: input.level, + source: "frontend", + message: input.message, + context: input.context ?? null, + httpStatus: input.httpStatus ?? null, + path: input.path ?? null, + stack: input.stack ?? null, + userId: meta.userId ?? null, + userAgent: meta.userAgent ?? null, + // Keep the client's capture time in context for ordering; createdAt is server time. + createdAt: new Date().toISOString(), + }); + } + + /** Read recent logs, newest first, with optional level/source/since filters. */ + query(opts: { + limit: number; + level?: LogLevel; + source?: LogSource; + since?: string; + }): AppLogRecord[] { + const conds = []; + if (opts.level) conds.push(eq(appLogs.level, opts.level)); + if (opts.source) conds.push(eq(appLogs.source, opts.source)); + if (opts.since) conds.push(sql`${appLogs.createdAt} >= ${opts.since}`); + const rows = this.#db + .select() + .from(appLogs) + .where(conds.length ? and(...conds) : undefined) + .orderBy(desc(appLogs.createdAt)) + .limit(opts.limit) + .all(); + return rows as unknown as AppLogRecord[]; + } + + /** Prune by age then by row cap. Returns how many rows were deleted. Safe to call + * on a timer; cheap (indexed on created_at). */ + prune(): number { + let deleted = 0; + try { + const cutoff = new Date(Date.now() - this.#retention.maxAgeDays * 86_400_000).toISOString(); + const byAge = this.#db.delete(appLogs).where(sql`${appLogs.createdAt} < ${cutoff}`).run(); + deleted += byAge.changes ?? 0; + + // Row cap: keep the newest maxRows, delete the rest. One subquery — find the + // created_at boundary of the keep-window, delete older. + const total = this.#db.select({ c: sql`count(*)` }).from(appLogs).get(); + const count = total?.c ?? 0; + if (count > this.#retention.maxRows) { + const boundary = this.#db + .select({ createdAt: appLogs.createdAt }) + .from(appLogs) + .orderBy(desc(appLogs.createdAt)) + .limit(1) + .offset(this.#retention.maxRows - 1) + .get(); + if (boundary) { + const byCap = this.#db + .delete(appLogs) + .where(sql`${appLogs.createdAt} < ${boundary.createdAt}`) + .run(); + deleted += byCap.changes ?? 0; + } + } + } catch { + // best-effort + } + return deleted; + } +} + +/** + * A pino-compatible write stream that forwards BACKEND warn+ lines into the LogService. + * Pino writes one JSON object per line to this stream; we parse, map the numeric level + * to a name, and persist. Returned as `{ write }` so it can be passed as pino's stream. + * stdout still receives the same line (we tee), so console logging is unchanged. + */ +export function pinoDbStream( + service: LogService, + tee: NodeJS.WritableStream, +): { write: (line: string) => void } { + const NUM_TO_LEVEL: Record = { + 10: "trace", + 20: "debug", + 30: "info", + 40: "warn", + 50: "error", + 60: "fatal", + }; + return { + write(line: string): void { + // Always tee to the original destination first (don't lose stdout logging). + try { + tee.write(line); + } catch { + /* ignore */ + } + try { + const obj = JSON.parse(line) as { + level?: number; + msg?: string; + err?: { stack?: string; message?: string }; + [k: string]: unknown; + }; + const level = NUM_TO_LEVEL[obj.level ?? 30] ?? "info"; + if (LOG_LEVEL_ORDER[level] < LOG_LEVEL_ORDER[BACKEND_PERSIST_MIN]) return; + // Strip pino's noisy standard fields from the persisted context. + const { level: _l, time: _t, pid: _p, hostname: _h, msg, ...rest } = obj; + service.recordBackend(level, typeof msg === "string" ? msg : "", rest); + } catch { + // A non-JSON line (shouldn't happen with pino) — ignore for persistence. + } + }, + }; +} diff --git a/apps/server/src/routes/logs.ts b/apps/server/src/routes/logs.ts new file mode 100644 index 0000000..b75f569 --- /dev/null +++ b/apps/server/src/routes/logs.ts @@ -0,0 +1,66 @@ +import type { FastifyInstance } from "fastify"; +import type { AppLogRecord, ClientLogInput, LogLevel } from "@parking/shared"; +import { requireAuth, requirePermission } from "../auth.js"; +import type { LogService } from "../log-service.js"; + +// Application/diagnostic logs (app_logs) — see wiki/concepts/app-logs.md. Two ends: +// - POST /api/logs : the FRONTEND ships its errors here (failed requests, uncaught +// exceptions). Any signed-in user may write (it's their own +// browser's diagnostics); CSRF still applies (mutation). +// - GET /api/logs : read the store — gated by `log:read` (admin/diagnostic role). +// Writes go through the shared LogService (bounded, best-effort, reentrancy-guarded); +// the DB sink for BACKEND warn+ is wired at the pino stream, not here. + +const LEVELS: ReadonlySet = new Set(["trace", "debug", "info", "warn", "error", "fatal"]); + +/** Cap a single ingest batch so a misbehaving client can't flood the store. */ +const MAX_BATCH = 50; + +function isValidEntry(e: unknown): e is ClientLogInput { + if (!e || typeof e !== "object") return false; + const o = e as Record; + return typeof o.message === "string" && typeof o.level === "string" && LEVELS.has(o.level); +} + +export async function logRoutes(app: FastifyInstance, logService: LogService): Promise { + // INGEST — accept one entry or a small batch ({ entries: [...] }). Returns 204. + // Deliberately tolerant: it never 4xx's on a malformed entry (a client erroring + // while reporting an error shouldn't get a second error) — invalid items are skipped. + app.post<{ Body: ClientLogInput | { entries?: unknown[] } }>( + "/api/logs", + { preHandler: requireAuth }, + async (req, reply) => { + const body = req.body as ClientLogInput | { entries?: unknown[] }; + const raw = Array.isArray((body as { entries?: unknown[] }).entries) + ? (body as { entries: unknown[] }).entries + : [body]; + const userId = req.user?.sub ?? null; + const userAgent = req.headers["user-agent"] ?? null; + for (const entry of raw.slice(0, MAX_BATCH)) { + if (!isValidEntry(entry)) continue; + logService.recordClient(entry, { userId, userAgent }); + } + reply.code(204).send(); + }, + ); + + // READ — newest first, with optional level/source/since filters + a limit. The + // booth Logs viewer calls this. Gated by log:read. + app.get<{ Querystring: { limit?: string; level?: string; source?: string; since?: string } }>( + "/api/logs", + { preHandler: requirePermission("log:read") }, + async (req): Promise<{ logs: AppLogRecord[] }> => { + const limit = Math.min(Math.max(Number(req.query.limit) || 200, 1), 2000); + const level = (req.query.level ?? "").trim(); + const source = (req.query.source ?? "").trim(); + const since = (req.query.since ?? "").trim(); + const logs = logService.query({ + limit, + level: LEVELS.has(level) ? (level as LogLevel) : undefined, + source: source === "frontend" || source === "backend" ? source : undefined, + since: since || undefined, + }); + return { logs }; + }, + ); +} diff --git a/apps/server/src/server.ts b/apps/server/src/server.ts index 5b14b89..22eb450 100644 --- a/apps/server/src/server.ts +++ b/apps/server/src/server.ts @@ -17,6 +17,8 @@ import { CredentialCapture } from "./credential-capture.js"; import { PrinterMonitor } from "./printer-monitor.js"; import { DeviceMonitor } from "./device-monitor.js"; import { buildSigner, buildVerifier } from "./signer.js"; +import { LogService, pinoDbStream } from "./log-service.js"; +import { logRoutes } from "./routes/logs.js"; import { authRoutes } from "./routes/auth.js"; import { userRoutes } from "./routes/users.js"; import { roleRoutes } from "./routes/roles.js"; @@ -43,12 +45,20 @@ export interface BuildOptions { } export async function buildServer(opts: BuildOptions = {}): Promise { - const app = Fastify({ - logger: { level: process.env.LOG_LEVEL ?? "info" }, - }); - + // DB first — the logger's DB sink needs it before Fastify is constructed. const db = opts.db ?? createDb(); + // Application-log store: a pino stream tees warn+ lines into app_logs (and still + // writes them to stdout), so backend warnings/errors are queryable from the booth + // alongside frontend errors. See log-service.ts + wiki/concepts/app-logs.md. + const logService = new LogService(db); + const app = Fastify({ + logger: { + level: process.env.LOG_LEVEL ?? "info", + stream: pinoDbStream(logService, process.stdout), + }, + }); + // Wire the RBAC permission resolver to this DB (route guards resolve a user's // role → permission set through it). See auth.ts. initAuth(db); @@ -188,6 +198,20 @@ export async function buildServer(opts: BuildOptions = {}): Promise { + const n = logService.prune(); + if (n > 0) app.log.debug(`pruned ${n} app_log rows`); + }, 60 * 60 * 1000); + pruneTimer.unref(); + logService.prune(); // once at startup + app.addHook("onClose", async () => clearInterval(pruneTimer)); + const unsubscribeInput = deviceEvents.onInput((e) => { // Record every input edge as unsigned telemetry, keyed to the device that fired // (provenance). No lane — the pool-of-spaces model has none. The entry flow diff --git a/apps/web/src/LogsViewer.tsx b/apps/web/src/LogsViewer.tsx new file mode 100644 index 0000000..7b7a519 --- /dev/null +++ b/apps/web/src/LogsViewer.tsx @@ -0,0 +1,159 @@ +import { useState } from "react"; +import { useTranslation } from "react-i18next"; +import { useQuery } from "@tanstack/react-query"; +import { fetchLogs, type AppLogRecord, type LogLevel } from "./api.js"; +import { formatRelativeDateTime } from "./lib/format.js"; + +// Diagnostic log viewer (app_logs) — backend warn+ and frontend errors in one place. +// Gated by log:read server-side. Filter by level / source / since; each row expands to +// the structured context + stack. Read-only — logs are an evidence/diagnostic stream, +// never edited. See wiki/concepts/app-logs.md. + +const LEVELS: LogLevel[] = ["trace", "debug", "info", "warn", "error", "fatal"]; + +/** Terminal-theme colour per level. */ +const LEVEL_COLOR: Record = { + trace: "text-term-muted", + debug: "text-term-muted", + info: "text-term-cyan", + warn: "text-term-amber", + error: "text-term-red", + fatal: "text-term-red", +}; + +function LogRow({ log }: { log: AppLogRecord }) { + const { t } = useTranslation(); + const [open, setOpen] = useState(false); + const hasDetail = (log.context && Object.keys(log.context).length > 0) || log.stack; + + return ( +
+ + {open && hasDetail && ( +
+ {log.path && ( +
+ {t("logs.path")}: {log.path} +
+ )} + {log.context && Object.keys(log.context).length > 0 && ( +
+              {JSON.stringify(log.context, null, 2)}
+            
+ )} + {log.stack && ( +
+              {log.stack}
+            
+ )} +
+ )} +
+ ); +} + +export function LogsViewer() { + const { t } = useTranslation(); + const [level, setLevel] = useState(""); + const [source, setSource] = useState(""); + const [since, setSince] = useState(""); + const [applied, setApplied] = useState<{ level?: string; source?: string; since?: string }>({}); + + const q = useQuery({ + queryKey: ["logs", applied], + queryFn: () => fetchLogs({ ...applied, limit: 500 }), + refetchInterval: 15_000, // keep the booth view roughly live without a WS + }); + + const logs = q.data?.logs ?? []; + + function apply() { + setApplied({ + level: level || undefined, + source: source || undefined, + since: since ? new Date(`${since}T00:00:00`).toISOString() : undefined, + }); + } + function clear() { + setLevel(""); + setSource(""); + setSince(""); + setApplied({}); + } + + return ( +
+
+

{t("logs.title")}

+ +
+ +
+
+ {t("logs.level")} + +
+
+ {t("logs.source")} + +
+
+ {t("logs.since")} + setSince(e.target.value)} /> +
+ + +
+ +
+ {q.isLoading ? ( +
{t("common.loading")}
+ ) : logs.length === 0 ? ( +
{t("logs.empty")}
+ ) : ( + <> +
+ {t("logs.time")} + {t("logs.level")} + {t("logs.source")} + {t("logs.message")} + {t("logs.status")} +
+ {logs.map((log) => ( + + ))} + + )} +
+
+ ); +} diff --git a/apps/web/src/api.ts b/apps/web/src/api.ts index 2c00d67..77ddd9b 100644 --- a/apps/web/src/api.ts +++ b/apps/web/src/api.ts @@ -5,6 +5,9 @@ // CSRF cookie back in the X-CSRF-Token header (double-submit). See // wiki/entities/local-jwt-auth.md. +import { logFailedRequest } from "./lib/logger.js"; +import type { AppLogRecord } from "@parking/shared"; + const CSRF_COOKIE = "parking_csrf"; const CSRF_HEADER = "X-CSRF-Token"; @@ -27,7 +30,14 @@ export async function apiFetch(path: string, init: RequestInit = {}): Promise const res = await fetch(path, { ...init, headers, credentials: "include" }); if (!res.ok) { const msg = (await res.json().catch(() => ({}))) as { error?: string }; - throw new ApiError(msg.error ?? `${path}: ${res.status}`, res.status); + const error = msg.error ?? `${path}: ${res.status}`; + // Ship the failed request to the backend log store (best-effort, loop-safe — the + // logger itself never logs the /api/logs call). 401s are normal pre-login churn, + // so we don't report them as errors. See lib/logger.ts. + if (res.status !== 401) { + logFailedRequest({ path, method, status: res.status, error }); + } + throw new ApiError(error, res.status); } if (res.status === 204) return undefined as T; return res.json() as Promise; @@ -160,6 +170,23 @@ export function deleteRole(id: string): Promise<{ ok: boolean }> { return apiFetch(`/api/roles/${id}`, { method: "DELETE" }); } +// --- Application logs (app_logs) ------------------------------------------ +/** Read recent diagnostic logs (gated server-side by log:read). */ +export function fetchLogs(params: { + limit?: number; + level?: string; + source?: string; + since?: string; +} = {}): Promise<{ logs: AppLogRecord[] }> { + const q = new URLSearchParams(); + if (params.limit) q.set("limit", String(params.limit)); + if (params.level) q.set("level", params.level); + if (params.source) q.set("source", params.source); + if (params.since) q.set("since", params.since); + const qs = q.toString(); + return apiFetch(`/api/logs${qs ? `?${qs}` : ""}`); +} + // --- Device setup --------------------------------------------------------- export interface ConfigField { @@ -632,7 +659,8 @@ export function fetchDeviceStatus(): Promise<{ devices: DeviceStatus[] }> { /** A persisted ledger row. Re-exported from shared so UI code has one source of * truth for the event shape (the same type the WS pushes). */ -export type { LedgerEvent } from "@parking/shared"; +export type { LedgerEvent, LogLevel, LogSource } from "@parking/shared"; +export type { AppLogRecord }; /** Recent ledger events, newest first (default 100, max 1000). Used for the * booth feed's initial load; live updates then arrive over the WS. `since` (ISO) diff --git a/apps/web/src/lib/ErrorBoundary.tsx b/apps/web/src/lib/ErrorBoundary.tsx new file mode 100644 index 0000000..c0cfd50 --- /dev/null +++ b/apps/web/src/lib/ErrorBoundary.tsx @@ -0,0 +1,50 @@ +import { Component, type ErrorInfo, type ReactNode } from "react"; +import { logClient } from "./logger.js"; + +// Top-level React error boundary: catches a render/lifecycle crash anywhere in the +// tree, reports it to the backend log store (app_logs), and shows a minimal recovery +// screen instead of a white page. A booth must never be left staring at a blank +// screen with no trace of why. See wiki/concepts/app-logs.md. + +interface State { + hasError: boolean; + message?: string; +} + +export class ErrorBoundary extends Component<{ children: ReactNode }, State> { + override state: State = { hasError: false }; + + static getDerivedStateFromError(err: Error): State { + return { hasError: true, message: err.message }; + } + + override componentDidCatch(err: Error, info: ErrorInfo): void { + logClient({ + level: "fatal", + message: err.message || "React render error", + stack: err.stack, + path: typeof location !== "undefined" ? location.pathname : undefined, + context: { kind: "react_error_boundary", componentStack: info.componentStack }, + }); + } + + override render(): ReactNode { + if (!this.state.hasError) return this.props.children; + // Intentionally un-i18n'd + dependency-free: the app tree just crashed, so we can't + // assume providers (i18n/router/query) are healthy. + return ( +
+

Something went wrong

+

The screen crashed and has been reported. Try reloading.

+ {this.state.message &&
{this.state.message}
} + +
+ ); + } +} diff --git a/apps/web/src/lib/i18n/en.ts b/apps/web/src/lib/i18n/en.ts index 4ceacf3..3078111 100644 --- a/apps/web/src/lib/i18n/en.ts +++ b/apps/web/src/lib/i18n/en.ts @@ -49,6 +49,7 @@ export const en: Catalog = { users: "Users", roles: "Roles", shifts: "Shifts", + logs: "Logs", }, status: { live: "LIVE", @@ -482,6 +483,24 @@ export const en: Catalog = { cashRemoved: "Cash removed", loadFailed: "Failed to load shifts.", }, + logs: { + title: "System logs", + refresh: "Refresh", + level: "Level", + source: "Source", + since: "Since", + apply: "Apply", + clear: "Clear", + allLevels: "All levels", + allSources: "All sources", + frontend: "Frontend", + backend: "Backend", + time: "Time", + message: "Message", + status: "Status", + path: "Path", + empty: "No logs.", + }, pay: { ticket: "Ticket", entry: "Entry", diff --git a/apps/web/src/lib/i18n/sq.ts b/apps/web/src/lib/i18n/sq.ts index 02d5223..82c7fb9 100644 --- a/apps/web/src/lib/i18n/sq.ts +++ b/apps/web/src/lib/i18n/sq.ts @@ -51,6 +51,7 @@ export const sq = { users: "Përdoruesit", roles: "Rolet", shifts: "Turnet", + logs: "Regjistrat", }, status: { live: "LIVE", @@ -495,6 +496,24 @@ export const sq = { cashRemoved: "Para të hequra", loadFailed: "Ngarkimi i turneve dështoi.", }, + logs: { + title: "Regjistrat e sistemit", + refresh: "Rifresko", + level: "Niveli", + source: "Burimi", + since: "Që nga", + apply: "Apliko", + clear: "Pastro", + allLevels: "Të gjitha nivelet", + allSources: "Të gjitha burimet", + frontend: "Ndërfaqja", + backend: "Serveri", + time: "Koha", + message: "Mesazhi", + status: "Statusi", + path: "Rruga", + empty: "Asnjë regjistër.", + }, pay: { ticket: "Bileta", entry: "Hyrja", diff --git a/apps/web/src/lib/logger.ts b/apps/web/src/lib/logger.ts new file mode 100644 index 0000000..261f1d4 --- /dev/null +++ b/apps/web/src/lib/logger.ts @@ -0,0 +1,198 @@ +// Frontend error/log collector. Ships failed requests, uncaught errors, and rejected +// promises to the backend (POST /api/logs → app_logs), so a booth problem is +// diagnosable from the host instead of needing the operator's devtools. See +// wiki/concepts/app-logs.md. +// +// Design notes: +// - BATCHED + THROTTLED: entries queue and flush on a short timer (and on page hide +// via sendBeacon), so a burst of errors is one request, not hundreds. +// - LOOP-SAFE: a failure of the /api/logs request itself is NEVER re-logged (that +// would be an infinite error → log → error spiral). We also never recurse through +// apiFetch — the flush uses raw fetch/sendBeacon. +// - LEVEL-GATED noise: console.warn/error are only forwarded when the client log +// level is debug/trace (off by default) — they're noisy (3rd-party chatter). The +// high-signal sources (failed requests, uncaught errors) are always captured. + +import { LOG_LEVEL_ORDER, type ClientLogInput, type LogLevel } from "@parking/shared"; + +const ENDPOINT = "/api/logs"; +const FLUSH_MS = 4000; +const MAX_QUEUE = 100; // drop oldest beyond this (bounded memory on a long-lived booth) +const CSRF_COOKIE = "parking_csrf"; +const CSRF_HEADER = "X-CSRF-Token"; + +/** The client capture threshold. Entries below this level are dropped before queueing. + * Default `info`: failed requests (error) + uncaught errors (error) always pass; + * console.warn/error forwarding is wired separately and only ON at debug/trace. */ +let clientLevel: LogLevel = (import.meta.env.VITE_LOG_LEVEL as LogLevel) || "info"; + +export function setClientLogLevel(level: LogLevel): void { + clientLevel = level; +} +export function getClientLogLevel(): LogLevel { + return clientLevel; +} +/** Are console.warn/error forwarded? Only when the client level is debug or trace. */ +function consoleForwardEnabled(): boolean { + return LOG_LEVEL_ORDER[clientLevel] <= LOG_LEVEL_ORDER.debug; +} + +const queue: ClientLogInput[] = []; +let timer: ReturnType | null = null; +/** Set true only while flushing, so the flush's own network activity is never logged. */ +let flushing = false; + +function readCookie(name: string): string | null { + const m = document.cookie.match(new RegExp(`(?:^|; )${name}=([^;]*)`)); + return m ? decodeURIComponent(m[1]!) : null; +} + +function scheduleFlush(): void { + if (timer != null) return; + timer = setTimeout(() => { + timer = null; + void flush(); + }, FLUSH_MS); +} + +/** Enqueue an entry. Drops it if below the client level or if it concerns the log + * endpoint itself (loop guard). */ +export function logClient(entry: ClientLogInput): void { + if (LOG_LEVEL_ORDER[entry.level] < LOG_LEVEL_ORDER[clientLevel]) return; + if (flushing) return; // don't log anything produced by the flush itself + if (entry.path && entry.path.startsWith(ENDPOINT)) return; // never log the log call + queue.push({ ...entry, at: entry.at ?? new Date().toISOString() }); + if (queue.length > MAX_QUEUE) queue.splice(0, queue.length - MAX_QUEUE); + scheduleFlush(); +} + +/** POST the queued entries. Raw fetch (not apiFetch) so a failure can't recurse. A + * failed flush silently re-queues nothing — diagnostics are best-effort, never fatal. */ +async function flush(): Promise { + if (queue.length === 0) return; + const entries = queue.splice(0, queue.length); + flushing = true; + try { + const headers: Record = { "content-type": "application/json" }; + const csrf = readCookie(CSRF_COOKIE); + if (csrf) headers[CSRF_HEADER] = csrf; + await fetch(ENDPOINT, { + method: "POST", + headers, + credentials: "include", + body: JSON.stringify({ entries }), + keepalive: true, + }); + } catch { + // Drop on failure — we must not re-log (loop) nor grow unbounded. + } finally { + flushing = false; + } +} + +/** Best-effort synchronous flush on page hide (sendBeacon survives unload). */ +function flushBeacon(): void { + if (queue.length === 0) return; + const entries = queue.splice(0, queue.length); + try { + const blob = new Blob([JSON.stringify({ entries })], { type: "application/json" }); + // sendBeacon can't set the CSRF header; the server accepts the ingest for any + // signed-in session (cookie sent automatically). If CSRF later guards it strictly, + // this path degrades to "lost on unload" — acceptable for diagnostics. + navigator.sendBeacon(ENDPOINT, blob); + } catch { + /* ignore */ + } +} + +/** Record a FAILED API request (called from apiFetch's error path). Always high-signal. */ +export function logFailedRequest(info: { + path: string; + method: string; + status: number; + error?: string; + requestId?: string; +}): void { + logClient({ + level: "error", + message: `${info.method} ${info.path} → ${info.status}${info.error ? `: ${info.error}` : ""}`, + httpStatus: info.status, + path: info.path, + context: { kind: "request_failed", method: info.method, requestId: info.requestId }, + }); +} + +let installed = false; + +/** Wire global handlers once, at app startup. Idempotent. */ +export function installClientLogging(): void { + if (installed || typeof window === "undefined") return; + installed = true; + + // Uncaught runtime errors. + window.addEventListener("error", (e: ErrorEvent) => { + logClient({ + level: "error", + message: e.message || "uncaught error", + stack: e.error?.stack, + path: location.pathname, + context: { + kind: "window_error", + filename: e.filename, + line: e.lineno, + col: e.colno, + }, + }); + }); + + // Unhandled promise rejections. + window.addEventListener("unhandledrejection", (e: PromiseRejectionEvent) => { + const reason = e.reason; + const message = + reason instanceof Error ? reason.message : typeof reason === "string" ? reason : "unhandled rejection"; + logClient({ + level: "error", + message, + stack: reason instanceof Error ? reason.stack : undefined, + path: location.pathname, + context: { kind: "unhandled_rejection" }, + }); + }); + + // console.warn / console.error → only forwarded at debug/trace (noisy otherwise). + const origWarn = console.warn.bind(console); + const origError = console.error.bind(console); + console.warn = (...args: unknown[]) => { + origWarn(...args); + if (consoleForwardEnabled()) { + logClient({ level: "warn", message: stringifyArgs(args), path: location.pathname, context: { kind: "console" } }); + } + }; + console.error = (...args: unknown[]) => { + origError(...args); + if (consoleForwardEnabled()) { + logClient({ level: "error", message: stringifyArgs(args), path: location.pathname, context: { kind: "console" } }); + } + }; + + // Flush on tab hide / unload. + window.addEventListener("visibilitychange", () => { + if (document.visibilityState === "hidden") flushBeacon(); + }); + window.addEventListener("pagehide", flushBeacon); +} + +function stringifyArgs(args: unknown[]): string { + return args + .map((a) => (a instanceof Error ? a.message : typeof a === "string" ? a : safeStringify(a))) + .join(" ") + .slice(0, 2000); +} + +function safeStringify(v: unknown): string { + try { + return JSON.stringify(v); + } catch { + return String(v); + } +} diff --git a/apps/web/src/main.tsx b/apps/web/src/main.tsx index a0d2521..d7113a8 100644 --- a/apps/web/src/main.tsx +++ b/apps/web/src/main.tsx @@ -3,12 +3,20 @@ import { createRoot } from "react-dom/client"; import "./index.css"; import "./lib/i18n/index.js"; // initialize i18next before the app renders import { App } from "./App.js"; +import { ErrorBoundary } from "./lib/ErrorBoundary.js"; +import { installClientLogging } from "./lib/logger.js"; + +// Capture uncaught errors / rejections / console noise → backend log store, before +// the app mounts so even an early crash is reported. See lib/logger.ts. +installClientLogging(); const rootEl = document.getElementById("root"); if (!rootEl) throw new Error("root element not found"); createRoot(rootEl).render( - + + + , ); diff --git a/apps/web/src/router.tsx b/apps/web/src/router.tsx index 570952f..7a3c7cb 100644 --- a/apps/web/src/router.tsx +++ b/apps/web/src/router.tsx @@ -27,6 +27,7 @@ import { SiteSettings } from "./SiteSettings.js"; import { UsersManager } from "./UsersManager.js"; import { RolesManager } from "./RolesManager.js"; import { ShiftsHistory } from "./ShiftsHistory.js"; +import { LogsViewer } from "./LogsViewer.js"; // Code-based TanStack Router (no file-based codegen — the app is small enough that // an explicit tree is clearer). The router context carries the signed-in user and @@ -84,6 +85,7 @@ function SetupLayout() { {show("user:read") && } {show("role:read") && } {show("shift:read") && } + {show("log:read") && } @@ -363,6 +365,7 @@ const SETUP_TABS: { to: string; perm: Permission }[] = [ { to: "/setup/users", perm: "user:read" }, { to: "/setup/roles", perm: "role:read" }, { to: "/setup/shifts", perm: "shift:read" }, + { to: "/setup/logs", perm: "log:read" }, ]; // /setup is a LAYOUT route (tab bar + ); the config screens are its @@ -437,6 +440,14 @@ const shiftsHistoryRoute = createRoute({ }, }); +// Diagnostic logs. Gated by log:read (an admin/diagnostic permission). +const logsRoute = createRoute({ + getParentRoute: () => setupRoute, + path: "logs", + beforeLoad: ({ context }) => requirePerm("log:read")(context), + component: LogsViewer, +}); + const routeTree = rootRoute.addChildren([ indexRoute, boothRoute, @@ -450,6 +461,7 @@ const routeTree = rootRoute.addChildren([ usersRoute, rolesRoute, shiftsHistoryRoute, + logsRoute, ]), ]); diff --git a/packages/db/drizzle/0009_app_logs.sql b/packages/db/drizzle/0009_app_logs.sql new file mode 100644 index 0000000..3a5da5e --- /dev/null +++ b/packages/db/drizzle/0009_app_logs.sql @@ -0,0 +1,20 @@ +CREATE TABLE `app_logs` ( + `id` text PRIMARY KEY NOT NULL, + `level` text NOT NULL, + `source` text NOT NULL, + `message` text NOT NULL, + `context` text, + `http_status` integer, + `path` text, + `stack` text, + `user_id` text, + `user_agent` text, + `created_at` text DEFAULT (current_timestamp) NOT NULL +); +--> statement-breakpoint +CREATE INDEX `app_logs_created_at_idx` ON `app_logs` (`created_at`);--> statement-breakpoint +CREATE INDEX `app_logs_level_idx` ON `app_logs` (`level`);--> statement-breakpoint +-- Grant the new log:read permission to the built-in admin role (enforcement is +-- runtime-special-cased to ALL permissions, but the Roles UI lists the grid from these +-- rows — keep it in sync). INSERT OR IGNORE: harmless if the row already exists. +INSERT OR IGNORE INTO `role_permissions` (`role_id`, `permission`) VALUES ('admin','log:read'); diff --git a/packages/db/drizzle/meta/_journal.json b/packages/db/drizzle/meta/_journal.json index 4a483bd..426740f 100644 --- a/packages/db/drizzle/meta/_journal.json +++ b/packages/db/drizzle/meta/_journal.json @@ -64,6 +64,13 @@ "when": 1781885100000, "tag": "0008_user_profile_theme", "breakpoints": true + }, + { + "idx": 9, + "version": "6", + "when": 1781885200000, + "tag": "0009_app_logs", + "breakpoints": true } ] } \ No newline at end of file diff --git a/packages/db/src/schema.ts b/packages/db/src/schema.ts index 14e84ed..a68ff31 100644 --- a/packages/db/src/schema.ts +++ b/packages/db/src/schema.ts @@ -356,6 +356,39 @@ export const sessions = sqliteTable("sessions", { lastEventIndex: integer("last_event_index"), }); +// --- Application logs (diagnostics, UNSIGNED, prunable) ------------------ +// A THIRD stream, distinct from the signed ledger_events (business facts) and +// device_events (hardware telemetry): operational/diagnostic logs for debugging the +// appliance. Backend warn/error/fatal (a Pino sink) AND frontend errors land here — +// failed requests, uncaught exceptions, rejected promises — so a booth problem is +// queryable from one place on an offline box. Never signed, never reconciled, pruned +// by age + row cap. See wiki/concepts/app-logs.md, event-streams-split.md. +export const appLogs = sqliteTable("app_logs", { + id: text("id").primaryKey(), + // pino levels: trace|debug|info|warn|error|fatal. We persist warn+ from the backend. + level: text("level", { + enum: ["trace", "debug", "info", "warn", "error", "fatal"], + }).notNull(), + // Which side produced it — the booth UI or the host. + source: text("source", { enum: ["frontend", "backend"] }).notNull(), + message: text("message").notNull(), + // Free-form structured detail: the failed request (path/method/status/body), the + // error name, component, anything the caller attaches. Kept in one JSON column. + context: text("context", { mode: "json" }).$type>(), + // Pulled out of context for cheap filtering of the common "failed request" case. + httpStatus: integer("http_status"), + path: text("path"), + // Captured stack trace, when there is one (uncaught errors / rejections). + stack: text("stack"), + // Who was logged in when it happened (frontend) / acted (backend), if known. + userId: text("user_id"), + // The browser/user-agent for a frontend log (triage which booth/device). + userAgent: text("user_agent"), + createdAt: text("created_at") + .notNull() + .default(sql`(current_timestamp)`), +}); + export type UserRow = typeof users.$inferSelect; export type RoleRow = typeof roles.$inferSelect; export type RolePermissionRow = typeof rolePermissions.$inferSelect; @@ -372,3 +405,4 @@ export type SubscriptionCredentialRow = typeof subscriptionCredentials.$inferSel export type SubscriptionPlateRow = typeof subscriptionPlates.$inferSelect; export type BlocklistRow = typeof blocklist.$inferSelect; export type SessionRow = typeof sessions.$inferSelect; +export type AppLogRow = typeof appLogs.$inferSelect; diff --git a/packages/shared/src/index.ts b/packages/shared/src/index.ts index e4e5137..9d770a5 100644 --- a/packages/shared/src/index.ts +++ b/packages/shared/src/index.ts @@ -26,6 +26,7 @@ export const RESOURCES = [ "session", // active sessions, lookup "event", // the signed ledger feed + void "report", // events feed, occupancy, future reports + "log", // application/diagnostic logs (app_logs) — view + retention ] as const; export type Resource = (typeof RESOURCES)[number]; @@ -51,6 +52,7 @@ export const PERMISSIONS: readonly Permission[] = [ "session:read", "event:read", "event:void", "report:read", + "log:read", ] as const; /** The protected built-in role: non-deletable, non-editable, always = ALL @@ -260,6 +262,52 @@ export function reasonPayload( /** Operational device telemetry — UNSIGNED, prunable. NOT the ledger. */ export type DeviceEventKind = "input" | "relay" | "status" | "read" | "snapshot"; +/** + * Application/diagnostic logs — a THIRD unsigned, prunable stream (app_logs), distinct + * from the signed ledger and from device telemetry. Backend warn+ and frontend errors + * land here so a booth problem is queryable in one place. See + * wiki/concepts/app-logs.md, decisions/event-streams-split.md. + */ +export type LogLevel = "trace" | "debug" | "info" | "warn" | "error" | "fatal"; +export type LogSource = "frontend" | "backend"; + +/** A persisted log record (the read shape returned by GET /api/logs). */ +export interface AppLogRecord { + readonly id: string; + readonly level: LogLevel; + readonly source: LogSource; + readonly message: string; + readonly context: Record | null; + readonly httpStatus: number | null; + readonly path: string | null; + readonly stack: string | null; + readonly userId: string | null; + readonly userAgent: string | null; + readonly createdAt: string; +} + +/** One log entry POSTed by the frontend to /api/logs (server stamps id/userId/time). */ +export interface ClientLogInput { + readonly level: LogLevel; + readonly message: string; + readonly context?: Record | null; + readonly httpStatus?: number | null; + readonly path?: string | null; + readonly stack?: string | null; + /** Client-side capture time (ISO). The server records its own receive time too. */ + readonly at?: string; +} + +/** The numeric ordering of levels (pino-compatible), for threshold comparisons. */ +export const LOG_LEVEL_ORDER: Record = { + trace: 10, + debug: 20, + info: 30, + warn: 40, + error: 50, + fatal: 60, +}; + /** * The composable rate card stored in a tariff_version.structure. * diff --git a/wiki/concepts/app-logs.md b/wiki/concepts/app-logs.md new file mode 100644 index 0000000..9ab0eba --- /dev/null +++ b/wiki/concepts/app-logs.md @@ -0,0 +1,101 @@ +--- +type: concept +tags: [parking, observability, diagnostics, logging, frontend, backend] +sources: [] +updated: 2026-06-19 +status: open +--- + +# Application logs (diagnostics) — the third stream + +A **third data stream**, deliberately distinct from the two in [[event-streams-split]]: + +| Stream | Table | Signed? | Purpose | +| --- | --- | --- | --- | +| Business ledger | `ledger_events` | ✅ ATECC608 | money/accountability ([[append-only-event-chain]]) | +| Device telemetry | `device_events` | ❌ | hardware chatter ([[device-events]]) | +| **App logs** | **`app_logs`** | ❌ | **operational/diagnostic logs** (this page) | + +App logs answer *"why did the booth misbehave?"* — a question neither of the other streams should +absorb (logs are neither business facts nor hardware telemetry). On an **offline appliance** there's +no Sentry/Datadog to ship to, so the host **is** the log store: backend warnings/errors AND frontend +errors land in one queryable table, viewable at the booth. Built 2026-06-19. + +## What's captured + +- **Backend `warn` / `error` / `fatal`** — a pino stream tees these into `app_logs` (and still writes + them to stdout, unchanged). `info`/`debug`/`trace` stay **stdout-only** — they'd bloat the DB. So + every `app.log.warn/error(...)` already in the codebase is now persisted with **no call-site + change**. +- **Frontend errors** (always): every **failed API request** (`apiFetch`'s non-OK path → method, + path, status, server error body — except `401`, which is normal pre-login churn), every **uncaught + error** (`window.onerror`), every **unhandled promise rejection**, and a top-level **React + ErrorBoundary** (a render crash is reported as `fatal` instead of a white screen). +- **`console.warn` / `console.error`** — only forwarded when the **client log level is `debug`/`trace`** + (off by default; they're noisy with third-party chatter). The high-signal sources above are always + on. Toggle via `VITE_LOG_LEVEL` / `setClientLogLevel()`. + +## Shape + +`app_logs`: `level` (pino names), `source` (`frontend`|`backend`), `message`, `context` (one JSON +column — the failed request, error name, component stack, anything), plus pulled-out `httpStatus` / +`path` for cheap filtering, `stack`, `userId`, `userAgent`, `createdAt`. Indexed on `created_at` + +`level`. Shared types: `AppLogRecord` / `ClientLogInput` / `LogLevel` in `@parking/shared`. + +## The API + the access split + +- **`POST /api/logs`** — the frontend ships errors here. **Any signed-in user** may write (it's their + own browser's diagnostics) — `requireAuth`, not a permission. CSRF still applies (it's a mutation). + Accepts one entry or a `{ entries: [...] }` batch (capped at 50). **Deliberately never 4xx's on a + malformed entry** — a client erroring *while reporting an error* must not get a second error. +- **`GET /api/logs`** — read the store (level/source/since filters), gated by the **new `log:read` + permission** (a new `log` resource in the dynamic [[local-jwt-auth|RBAC]] grid). Admin holds it; + it's grantable to a diagnostic role. *Verified: a cashier without `log:read` gets 403 on GET but + 204 on POST — the intended asymmetry.* + +## Reliability invariants (a logger must never make things worse) + +- **No infinite loop.** The frontend collector never logs the `/api/logs` request itself, and flushes + via **raw `fetch`/`sendBeacon`**, not `apiFetch` (so a flush failure can't recurse into a new log). + The backend `LogService` has a **reentrancy guard** — persisting a log can't emit a persisted log. +- **Best-effort, never fatal.** Every write is wrapped; a DB/logging failure is swallowed (it can't be + logged — that's the recursion we guard). Diagnostics must never break the path they observe. +- **Bounded.** Frontend queue capped (drops oldest); message/stack/context clamped per row; + ingest batch capped. + +## Retention (offline appliance ⇒ must be bounded) + +Pruned by **age AND a row cap** (a burst could blow past an age-only window): delete older than +`LOG_RETENTION_DAYS` (default 30) **and** keep only the newest `LOG_RETENTION_MAX_ROWS` (default +50 000). Runs **hourly** (unref'd timer) + once at startup. Both env-configurable. Same "prunable, +not precious" durability class as `device_events` — the opposite of the append-only ledger. + +## The booth viewer + +A **Logs screen** under Setup (`/setup/logs`, gated by `log:read`, sq+en) — filter by +level/source/since, newest first, each row expands to the structured `context` + stack. Read-only +(logs are evidence, never edited). Polls every 15 s (no WS — diagnostics aren't latency-critical). +Sits alongside the other admin tabs in [[booth-console]]. + +## As-built (2026-06-19) + +- `packages/db`: `app_logs` table + migration `0009_app_logs.sql` (+ journal idx 9; seeds admin + `log:read`). Applied to the live `apps/server/parking.sqlite`. +- `@parking/shared`: `log` resource + `log:read` permission; `LogLevel`/`LogSource`/`AppLogRecord`/ + `ClientLogInput`/`LOG_LEVEL_ORDER`. +- `apps/server`: `log-service.ts` (`LogService` + `pinoDbStream`), `routes/logs.ts`, wired in + `server.ts` (DB built before Fastify so the pino stream has the sink; prune timer). +- `apps/web`: `lib/logger.ts` (collector + global handlers), `lib/ErrorBoundary.tsx`, `apiFetch` hook, + `LogsViewer.tsx` + route/nav, i18n. + +## Open + +- **No automated test** (the standing harness gap) — though the ingest/read/gate path was verified by + in-process `app.inject` smoke (login → POST 204 → GET 200 with the record; non-admin 403/204 split). +- **Server API error strings stay English** — unchanged here; this is about *persisting* logs, not + localizing them. The localized-ledger-reason pattern ([[i18n]]) is the template if log *display* + ever needs translation (currently the message is whatever the thrower wrote). +- **Surfacing critical logs live** — a `fatal`/`error` count badge on the booth footer over the + existing `/api/ws` could flag problems without opening the viewer. Deferred. +- **Correlation id** — no request-id threads a frontend failed-request log to its backend log yet; + add a `x-request-id` echo if cross-stream correlation is wanted. diff --git a/wiki/concepts/device-events.md b/wiki/concepts/device-events.md index a73b874..7e0021e 100644 --- a/wiki/concepts/device-events.md +++ b/wiki/concepts/device-events.md @@ -31,6 +31,11 @@ diagnostics, and live booth status — **not** anti-fraud. - **Device-keyed** — references the `devices` instance (raw device provenance). No `lane` (pool-of-spaces model — see [[entry-exit-points]]). +> **Not to be confused with [[app-logs]].** `device_events` is **hardware telemetry** (a relay fired, +> a camera failed). Diagnostic/application logs (a failed API request, an uncaught frontend error, +> a backend warning) are a **separate third stream** in `app_logs` — don't route app errors here, nor +> hardware telemetry there. Both are unsigned + prunable; the distinction is *what produced it*. + ## The boundary that matters A device event is *evidence the host saw something happen*; it does **not** by itself authorize or diff --git a/wiki/decisions/event-streams-split.md b/wiki/decisions/event-streams-split.md index ea43bd3..a4ec66a 100644 --- a/wiki/decisions/event-streams-split.md +++ b/wiki/decisions/event-streams-split.md @@ -50,8 +50,16 @@ A raw button press is **telemetry** → `device_events`. The entry flow then min input-push handler to emit `device_events` (+ the entry flow signs `vehicle_entry`). - `ParkingEventType` in `packages/shared` splits into ledger types vs. a device-event type set. +## A third stream followed (2026-06-19) + +The same separation logic produced a **third** stream: **`app_logs`** — operational/diagnostic logs +(backend warn+ via a pino sink, plus frontend errors). They're neither business facts (ledger) nor +hardware telemetry (device_events), so they get their own unsigned, prunable table. See +[[app-logs]]. The principle generalizes: *one stream per durability/meaning class*. + ## Open -- `device_events` retention/rotation policy. +- `device_events` retention/rotation policy. (Resolved for `app_logs`: age + row cap — see + [[app-logs]]; the same policy is a candidate for `device_events`.) - Which device facts (if any) are witness-grade enough to *also* warrant a signed ledger entry (e.g. `barrier_open_observed` from a loop sensor) — see [[append-only-event-chain]] witness gap. \ No newline at end of file diff --git a/wiki/index.md b/wiki/index.md index 9d56a94..d5f7f91 100644 --- a/wiki/index.md +++ b/wiki/index.md @@ -7,7 +7,7 @@ updated: 2026-06-19 # Index Content catalog for the wiki. Start at [[overview]]. Maintained on every ingest. -Counts: 4 sources · 19 entities · 42 concepts · 5 decision records. +Counts: 4 sources · 19 entities · 44 concepts · 5 decision records. ## Overview & navigation - [[overview]] — the top-level synthesis and entry point. @@ -93,6 +93,7 @@ Counts: 4 sources · 19 entities · 42 concepts · 5 decision records. - [[ticket-encoding]] — transient ticket id (11-digit numeric + Luhn) as Code128; printed at entry, scanned at pay station + exit; barcode geometry must fit paper width (KP-300H overflow); plate-as-ticket alt. - [[anti-passback]] — block/flag one id entering twice without an exit; fold over open sessions. - [[device-events]] — unsigned hardware telemetry (relay/printer/camera/reader/input); separate from the signed ledger. +- [[app-logs]] — the third stream: diagnostic logs (backend warn+ pino sink + frontend errors) → app_logs; log:read viewer; pruned by age+row cap. - [[subscription]] — recurring plan (e.g. 10,000 ALL/month); RF/QR or plate identity, car-count + max-concurrent, host-in-loop; short-circuits payment. (Renamed from "permit"; time-of-day windows noted, deferred.) - [[opencv-anpr-service]] — host-side vision microservice: ANPR (plate identity) + vehicle verification (anti-plate-spoofing witness). - [[blocklist]] — barred plates/cards refused at entry (never at exit); signed, attributed. diff --git a/wiki/log.md b/wiki/log.md index 531adf4..a13556b 100644 --- a/wiki/log.md +++ b/wiki/log.md @@ -892,3 +892,15 @@ Dates were raw ISO on paper and time-only in the UI (a 2-day-old session showed ## [2026-06-19] fix | KP-300H barcode line-overflow — ticket id 13→11 digits The Cashino KP-300H entry dispenser printed entry tickets as RASTER GARBAGE (solid black bars/banding) while the Rongta printed the IDENTICAL byte stream fine. Diagnosed on hardware: plain-text-only prints were clean → isolated to the `GS k` Code128 barcode. ROOT CAUSE = barcode line-overflow, not corruption: a Code128-B symbol is (11·chars+35)·moduleWidth dots; the old 13-digit id at module width 3 = ~534 dots OVERRAN the KP-300H's 72mm line (512 usable dots @ 203 dpi). The Rongta runs 80mm (576 dots) and had just enough room — why only the Cashino failed. FIX: shorten the ticket id 13→11 digits (10 random + Luhn) → ~468 dots, fits 72mm; scanned the full value at the exit reader (verified). Length is driven by GUESS-RESISTANCE not volume (10^10 space, ~1-in-10^7 to hit a live open ticket vs the booth-operator threat); chose 11 over the requested 9 (10^8 → ~1-in-10^5, too weak). validateTicketCode made length-agnostic (\d{10,14}+Luhn) so legacy 13-digit tickets still validate. NB: module width must stay 3 — a width-2 test scanned but returned TRUNCATED values (partial reads logged as exit.refused.noSession anomalies). Also fixed a separate latent transport bug in sendRaw: write-then-destroy could RST mid-stream (the write callback ≠ peer-flushed) and truncate a job; now end(payload)+FIN, resolve on socket `close`, timeout-after-write = success. NOT the cause of the garbage but a real risk. Committed bbf61c4. Updated [[ticket-encoding]], [[rongta-printer]]. + +## [2026-06-19] feat | Snapshots on refused entry/exit + subscriber access medium in the activity log + +Two booth-evidence gaps closed. (1) **Refused entry/exit now snapshot.** Originally only the OPEN paths fired the directional camera; refusal/hold anomalies didn't — yet a turned-away car is exactly the evidence an operator/auditor wants (fraud/dispute signal). Added `#fireSnapshot` to every refusal: entry refused-full + held-no-ticket (a refused entry has no ticket id, so mint a synthetic `REFUSED-…` ref to key the anomaly + photo together), exit refused closed/no-session/unpaid/grace-expired (BOTH booth `exitForBooth` and reader `#runExit` paths), and refused [[subscription]] (the lane the reader sits at — `resolved.direction`, "both"→entry — picks the camera). Same fire-and-forget contract: a refusal is never delayed/blocked by a camera; failed captures still surface as "⚠ camera unreachable" tiles. (2) **Subscriber access medium (`via`) surfaced.** The subscription flow already SIGNED `via` (`"qr"|"card"|"plate"`) into the entry/exit payload but the activity log never showed it. Added it as a typed `LedgerPayload.via` field, a cyan chip in the ticker, and an "Entry medium / Mënyra e hyrjes" row in the detail modal (QR code / RFID card·chip / plate, localized sq+en) — a lost-card investigation can now see which credential opened a barrier. Display-only, no re-signing. Refused-subscription anomalies also now carry `via`. Build+lint green. Updated [[entry-exit-points]], [[booth-console]]. + +## [2026-06-19] feat | One car = one ticket (entry anti-double-press) + refusal snapshots + subscriber via + +FLAW found: the entry button could be pressed without limit — each press minted a fresh ticket + signed vehicle_entry, corrupting occupancy (one car counts as many) and letting a transient SHOP the cheapest ticket at exit. The old `#inFlight` guard only blocked OVERLAPPING presses (released in finally). FIX is per-relay config (`config.relays[]`), mode chosen by available barrier feedback: (1) PRESENCE (preferred) — `presenceInput` ties ticketing to a vehicle loop on a Dingtian input; a press prints only with a car present, and NO second ticket until the loop CLEARS (car drove in) and a new car re-occupies it → physical one-car-one-ticket; (2) COOLDOWN (fallback, no feedback) — `entryCooldownSec` suppresses repeat presses for N seconds (a timer, mitigation not guarantee). New `relayForPresence()` resolves a loop edge to its entry relay; `EntryFlow` keeps a per-relay `#guard` map (present/armed), disarms on PRINT success, re-arms on loop clear. A suppressed press = UNSIGNED device_events telemetry (entrySuppressed:true), NOT a signed anomaly (operator's call — it's a correct no-op, not fraud). SetupWizard relay editor exposes Presence-loop + Cooldown fields (sq+en). Fail-closed entry + barrier-is-not-a-door invariants untouched; guard state is in-memory/rebuildable, starts armed after restart (safe default). New page [[entry-double-press]]; updated [[entry-exit-points]], index. Build+lint green. (Bundled with this session's earlier refusal-snapshots + subscriber-`via` work.) + +## [2026-06-19] feat | Application logs — backend pino DB sink + frontend error collection (app_logs) + +Added a THIRD data stream (`app_logs`) alongside the signed ledger and device telemetry — operational/diagnostic logs, since an OFFLINE appliance has no Sentry to ship to. BACKEND: a pino stream tees warn/error/fatal into app_logs (info/debug stay stdout-only — no bloat) with NO call-site change; the DB is now built BEFORE Fastify so the logger stream has its sink. FRONTEND (lib/logger.ts): always ships failed API requests (apiFetch non-OK path, minus 401 pre-login churn), window.onerror, unhandledrejection, and a top-level React ErrorBoundary (render crash → fatal, not a white screen); console.warn/error forwarded ONLY at client debug/trace level (noisy otherwise). Batched/throttled POST, flush via raw fetch + sendBeacon on pagehide. Reliability invariants: never log the /api/logs call itself (loop guard), LogService reentrancy guard, all writes best-effort/swallowed, bounded queue + clamped rows. API: POST /api/logs (any signed-in user, CSRF, tolerant — never 4xx on a bad entry) + GET /api/logs gated by a NEW `log:read` permission (new `log` resource in the RBAC grid; admin holds it). Retention: pruned by age (LOG_RETENTION_DAYS=30) AND row cap (MAX_ROWS=50k), hourly + at startup. UI: a Logs screen under /setup (filter level/source/since, expand to context+stack, 15s poll), sq+en. DB migration 0009_app_logs (+journal idx 9, seeds admin log:read) applied to the live apps/server DB. Verified end-to-end via app.inject: login→POST 204→GET 200 with the record; backend warn/error persisted + info dropped; non-admin GET 403 / POST 204 (the intended split). Build+lint green. New page [[app-logs]]; updated [[event-streams-split]], [[device-events]], index.