Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 343961902afdb2f8ba1ef71292a541e724387632
parent 502a6b6d54b9cbb05eed849a25c208cbfd495e02
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue, 18 Aug 2026 11:10:46 -0400

Fixture seeds and e2e/faces.spec.ts

The fixture's cover corners named v1...v11, which have no media at all -- fine
while the only question asked of a corner was whether accepting it would clash,
useless the moment a page wants to draw the frame it was cut from. They are
DERIVED from the symlinked media/ now, as suspect-sources.json already is, with
the sharing pattern untouched: alpha-b stays disjoint from the accepted
alpha-c, alpha-d still clashes on its first corner. face-verdicts.json is
seeded empty like every other pile. The detector comes in by reference too --
models/ beside the fixture's copy of facecrop.py, and the facedet venv where
SONG_SCRATCH looks for it -- because without them the detect route answers 503
and the test that checks the two crop implementations agree has nothing to
compare.

Twelve tests, shape and mechanism only, every count derived at runtime (which
matters more than usual here: deck.spec accepts a cover mid-suite, and the
accepted set is what the queue is ordered by).

The one that earns its keep is autoCrop() vs facecrop.py. The detect route now
returns BOTH boxes -- python's own and the TypeScript re-implementation's -- so
the comparison happens on every real detection rather than once in a test, the
deck can say loudly when they disagree, and the spec can check them without
spawning python. All 11 fixture corners agree exactly.

Two bugs the suite caught, both fixed here:

  * the opening crop was applied by an effect, so a recorded crop was absent
    from the server-rendered html and appeared a beat later. It is computed for
    the first render now -- the deck is linked to with ?key= from the bench and
    from notes, and a reader arriving that way is a real pre-hydration reader.
  * the frame and the detection were each fetched twice on arrival, because
    dev-mode double-invoke runs every effect twice and a guard held in state
    has not updated by the second run. Both are keyed ref caches now, so a
    corner costs one frame and one detection ever -- including on the way back
    through the queue.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

Diffstat:
Mumtool/components/FaceDeck.tsx | 71+++++++++++++++++++++++++++++++++++++++++++++++++++++------------------
Aumtool/e2e/faces.spec.ts | 448+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/e2e/fixtures/make-fixture.mjs | 59++++++++++++++++++++++++++++++++++++++++++++++++++++-------
Mumtool/lib/faces.ts | 34++++++++++++++++++++++++++++------
4 files changed, 581 insertions(+), 31 deletions(-)

diff --git a/umtool/components/FaceDeck.tsx b/umtool/components/FaceDeck.tsx @@ -67,6 +67,9 @@ export type FaceEntryView = { type Detection = { face: FaceBox | null; auto: Box | null; + /** facecrop.py's own box, so the deck can prove it is drawing the right one. */ + pyAuto: Box | null; + agrees: boolean; shove: number; frame: { w: number; h: number }; error?: string; @@ -93,15 +96,23 @@ export default function FaceDeck({ const [faces, setFaces] = useState(initial); const [spent, setSpent] = useState(rejected); + // The opening crop is computed for the FIRST RENDER, not applied by an effect + // afterwards. Server-rendering it is what makes a recorded crop present in the + // HTML rather than appearing a beat later: an effect-applied box is invisible + // to anything reading the page before hydration -- which is a real reader + // here, since the deck is linked to with ?key= from a note and from the bench. + const opening = queue[startAt]; + const first = opening ? (initial[opening.key] ?? null) : null; + // Detections are held per key for the life of the page. A second pass over a // corner already looked at must not cost another second of python. const [dets, setDets] = useState<Record<string, Detection>>({}); const [loading, setLoading] = useState(false); - const [crop, setCrop] = useState<Box | null>(null); - const [pending, setPending] = useState<FaceVerdict | null>(null); - const [code, setCode] = useState<number | null>(null); - const [text, setText] = useState(""); + const [crop, setCrop] = useState<Box | null>(first?.crop ?? opening?.crop ?? null); + const [pending, setPending] = useState<FaceVerdict | null>(first?.verdict ?? null); + const [code, setCode] = useState<number | null>(first?.code ?? null); + const [text, setText] = useState(first?.text ?? ""); const [guess, setGuess] = useState<Box | null>(null); const [busy, setBusy] = useState(false); const [error, setError] = useState<string | null>(null); @@ -109,6 +120,13 @@ export default function FaceDeck({ const canvas = useRef<HTMLCanvasElement | null>(null); const preview = useRef<HTMLCanvasElement | null>(null); const image = useRef<HTMLImageElement | null>(null); + // Both of these are keyed CACHES rather than plain flags, and they are refs + // rather than state for one reason: React's dev-mode double-invoke runs every + // effect twice, and a guard held in state has not updated by the second run. + // A corner already fetched must not be fetched again -- not on a re-mount, not + // on the way back through the queue, and above all not twice on arrival. + const images = useRef<Map<string, HTMLImageElement>>(new Map()); + const asked = useRef<Set<string>>(new Set()); const drag = useRef<{ mode: "move" | "resize"; ax: number; ay: number; ox: number; oy: number } | null>(null); const entry = queue[idx] ?? null; @@ -126,21 +144,25 @@ export default function FaceDeck({ const [ready, setReady] = useState(false); useEffect(() => { if (!frameSrc) return; - setReady(false); - const img = new Image(); - img.onload = () => { + let img = images.current.get(frameSrc); + if (!img) { + img = new Image(); + images.current.set(frameSrc, img); + img.src = frameSrc; // the ONE request for this corner's frame + } + if (img.complete && img.naturalWidth > 0) { image.current = img; setReady(true); + return; + } + setReady(false); + const shown = img; + const onLoad = () => { + image.current = shown; + setReady(true); }; - img.onerror = () => { - image.current = null; - setReady(false); - }; - img.src = frameSrc; - return () => { - img.onload = null; - img.onerror = null; - }; + shown.addEventListener("load", onLoad); + return () => shown.removeEventListener("load", onLoad); }, [frameSrc]); // ---- the detection -------------------------------------------------------- @@ -148,7 +170,8 @@ export default function FaceDeck({ // rows; detecting all of them to draw a list would be twenty seconds of work // for twenty-one pictures nobody is looking at yet. useEffect(() => { - if (!entry || dets[entry.key]) return; + if (!entry || dets[entry.key] || asked.current.has(entry.key)) return; + asked.current.add(entry.key); let live = true; setLoading(true); fetch(`/api/face/detect?video=${encodeURIComponent(entry.video)}&at=${entry.frameAt}`, { @@ -163,7 +186,11 @@ export default function FaceDeck({ if (live) { setDets((d) => ({ ...d, - [entry.key]: { face: null, auto: null, shove: 0, frame: { w: 0, h: 0 }, error: "the detector did not answer" }, + [entry.key]: { + face: null, auto: null, pyAuto: null, agrees: true, + shove: 0, frame: { w: 0, h: 0 }, + error: "the detector did not answer", + }, })); } }) @@ -638,6 +665,14 @@ export default function FaceDeck({ automatically, so crop it by hand </p> )} + {det?.face && !det.agrees && ( + /* The dashed box would not be what gets cut. Loud, because every + reading on this page assumes the two functions are one. */ + <p className="text-[12px] text-[var(--color-bad)]" data-disagree> + the drawn crop {boxAttr(det.auto)} is NOT what facecrop.py would cut ( + {boxAttr(det.pyAuto)}) — trust neither until they agree again + </p> + )} {det?.face && shove > SHOVE_FLOOR && ( <p className="text-[12px] text-[var(--color-dirty)]" data-shove-note> the automatic crop sat {Math.round(shove)} px off the face to stay inside the diff --git a/umtool/e2e/faces.spec.ts b/umtool/e2e/faces.spec.ts @@ -0,0 +1,448 @@ +import { test, expect, type APIRequestContext, type Page } from "@playwright/test"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import { autoCrop, scaleOf, CORNER_PX, SHOVE_FLOOR, type Box } from "../lib/face-types"; + +// The face judger, against the REAL episodes the fixture symlinks -- make-fixture +// derives its cover corners from media/ precisely so the frame and detect routes +// have something to serve. A synthetic 320x180 clip would exercise the plumbing +// and prove nothing about the thing this page exists for, which is that a webcam +// inset near a frame edge gets a crop shoved off the face. +// +// So no assertion here hardcodes a corner, a count or an episode id. Every one is +// a SHAPE, a MECHANISM or an INVARIANT, derived at runtime from the manifests the +// fixture actually wrote -- which matters more than usual here because deck.spec +// accepts a cover mid-suite, and the accepted set is what the queue is ordered by. + +const CODE = () => path.join(process.cwd(), ".e2e-song", "code"); +const readState = <T,>(name: string): T => + JSON.parse(readFileSync(path.join(CODE(), name), "utf8")) as T; + +type Corner = { video: string; srcStart?: number; frameAt?: number; crop?: Box }; +type Doc = { thumbs: Record<string, { corners?: Corner[] }> }; +type Entry = { + key: string; + video: string; + srcStart: number; + frameAt: number; + covers: string[]; + accepted: string[]; + onAccepted: boolean; + crop: Box | null; +}; +type Detection = { + face: (Box & { score: number; t: number }) | null; + auto: Box | null; + pyAuto: Box | null; + agrees: boolean; + shove: number; + frame: { w: number; h: number }; +}; +type FaceView = { queue: Entry[]; faces: Record<string, unknown>; rejected: string[] }; + +const view = async (request: APIRequestContext): Promise<FaceView> => { + const r = await request.get("/api/face"); + expect(r.ok(), `GET /api/face -> ${r.status()}`).toBe(true); + return (await r.json()) as FaceView; +}; + +const detect = async (request: APIRequestContext, e: Entry): Promise<Detection> => { + const r = await request.get( + `/api/face/detect?video=${encodeURIComponent(e.video)}&at=${e.frameAt}`, + ); + expect(r.ok(), `detect ${e.key} -> ${r.status()}`).toBe(true); + return (await r.json()) as Detection; +}; + +const attr = async (page: Page, sel: string, name: string): Promise<string> => + (await page.locator(sel).first().getAttribute(name)) ?? ""; + +const parseBox = (s: string): Box => { + const [x, y, w, h] = s.split(",").map(Number); + return { x, y, w, h }; +}; + +// --------------------------------------------------------------------------- +// 1. The route itself. +// --------------------------------------------------------------------------- + +// A STATIC segment inside app/browse/[song]/'s territory, the same trap +// /browse/find and /browse/decisions both document. When it regresses the page +// still renders perfectly on its own and only the URL stops working. +test("/browse/faces is not swallowed by the [song] route", async ({ page }) => { + const res = await page.goto("/browse/faces"); + expect(res?.status()).toBe(200); + await expect(page.getByRole("navigation", { name: "Breadcrumb" })).toContainText("faces"); + await expect(page.locator("[data-face-deck]")).toBeVisible(); +}); + +// The way in is the cover-art bench, not a tenth nav item -- AppNav is at nine +// and says a tenth wraps the header on a laptop. +test("the bench links to the judger and the nav does not", async ({ page }) => { + await page.goto("/browse/alpha"); + const link = page.locator("[data-judge-faces]").first(); + await expect(link).toBeVisible(); + const cover = (await link.getAttribute("data-judge-faces")) ?? ""; + await expect(link).toHaveAttribute("href", `/browse/faces?cover=${encodeURIComponent(cover)}`); + + // Nine items, and faces is not one of them. + const nav = page.getByRole("navigation").first(); + await expect(nav.getByRole("link", { name: "faces", exact: true })).toHaveCount(0); + + // And the link lands on a corner that cover actually uses. + await link.click(); + await expect(page.locator("[data-face-key]")).toBeVisible(); + const key = await attr(page, "[data-face-key]", "data-face-key"); + const manifest = readState<Doc>("thumb-manifest.json"); + const keys = (manifest.thumbs[cover]?.corners ?? []).map( + (c) => `${c.video}@${(c.srcStart ?? 0).toFixed(2)}`, + ); + expect(keys).toContain(key); +}); + +// --------------------------------------------------------------------------- +// 2. The queue is the corners on record, deduped, the shipped ones first. +// --------------------------------------------------------------------------- + +test("the queue is every corner deduped, accepted covers first", async ({ request }) => { + const manifest = readState<Doc>("thumb-manifest.json"); + const accepted = readState<Doc>("thumb-accepted.json"); + const acceptedNames = new Set(Object.keys(accepted.thumbs)); + + // Derived from the same files the app reads, never a hardcoded 22 -- or 11, + // which is what the fixture happens to hold today. + const entries: string[] = []; + const onAccepted = new Set<string>(); + for (const [name, e] of Object.entries({ ...manifest.thumbs, ...accepted.thumbs })) { + for (const c of e.corners ?? []) { + const k = `${c.video}@${(c.srcStart ?? 0).toFixed(2)}`; + entries.push(k); + if (acceptedNames.has(name)) onAccepted.add(k); + } + } + const distinct = new Set(entries); + expect(distinct.size).toBeGreaterThan(0); + // The fixture deliberately repeats corners across covers -- if it stopped, + // this test would pass while checking nothing. + expect(entries.length).toBeGreaterThan(distinct.size); + + const { queue } = await view(request); + expect(queue.length).toBe(distinct.size); + expect(new Set(queue.map((e) => e.key))).toEqual(distinct); + + // A corner on four covers is ONE entry that names all four. + const shared = queue.find((e) => e.covers.length > 1); + expect(shared, "no corner appears on more than one cover").toBeTruthy(); + + const flags = queue.map((e) => e.onAccepted); + expect(new Set(queue.filter((e) => e.onAccepted).map((e) => e.key))).toEqual(onAccepted); + // Every accepted-cover corner comes before every candidate-only one. + expect(flags.lastIndexOf(true)).toBeLessThan( + flags.indexOf(false) === -1 ? Number.MAX_SAFE_INTEGER : flags.indexOf(false), + ); +}); + +// --------------------------------------------------------------------------- +// 3. The two implementations of one function. +// --------------------------------------------------------------------------- + +// autoCrop() is a pure re-implementation of facecrop.py's crop maths, and the +// deck draws with it on every drag because a round trip per pixel would be +// absurd. A re-implementation is only worth having if it is provably the same +// function -- so the detect route returns BOTH boxes and this compares them for +// every corner in the queue, without the spec having to spawn python itself. +test("autoCrop() is the same function as facecrop.py's crop maths", async ({ request }) => { + test.slow(); + const { queue } = await view(request); + let compared = 0; + + for (const e of queue) { + const d = await detect(request, e); + if (!d.face) continue; + expect(d.pyAuto, `${e.key} detected a face but reported no crop`).toBeTruthy(); + expect(d.agrees, `${e.key}: route says the two disagree`).toBe(true); + // Recomputed here as well, so this is a THIRD evaluation of the same + // arithmetic rather than the route grading its own homework. + expect(autoCrop(d.face, d.frame.w, d.frame.h).box, `${e.key}`).toEqual(d.pyAuto); + compared += 1; + } + expect(compared, "no corner in the queue found a face at all").toBeGreaterThan(0); +}); + +// --------------------------------------------------------------------------- +// 4. The clamp is REPORTED, not hidden. +// --------------------------------------------------------------------------- + +// The invariant, on a face box built to sit against an edge. Pure, so it has a +// true answer whatever the corpus holds: a 200px face whose centre is 40px from +// the top cannot be centred in a 440px crop without leaving the frame, so the +// crop must slide and the slide must be reported. +test("autoCrop() reports the displacement it accepted", () => { + const face = { x: 600, y: -60, w: 200, h: 200 }; + const { box, shove } = autoCrop(face, 1280, 720); + expect(box.y).toBe(0); + expect(shove).toBeGreaterThan(SHOVE_FLOOR); + // And a face in open ground is not displaced at all. + const middle = autoCrop({ x: 540, y: 260, w: 200, h: 200 }, 1280, 720); + expect(middle.shove).toBeLessThanOrEqual(SHOVE_FLOOR); +}); + +// And the deck SAYS SO. This is the sentence that explains what went wrong on +// the Super Mario RPG corner; a number computed and not shown would leave the +// crop looking like an inexplicable choice rather than a measured mistake. +test("the deck states the shove on a corner whose crop was displaced", async ({ + page, + request, +}) => { + test.slow(); + const { queue } = await view(request); + let displaced: { key: string; shove: number } | null = null; + for (const e of queue) { + const d = await detect(request, e); + if (d.face && d.shove > SHOVE_FLOOR) { + displaced = { key: e.key, shove: d.shove }; + break; + } + } + expect(displaced, "no corner in the queue has a displaced crop").toBeTruthy(); + + await page.goto(`/browse/faces?key=${encodeURIComponent(displaced!.key)}`); + await expect(page.locator("[data-shove-note]")).toBeVisible({ timeout: 20_000 }); + await expect(page.locator("[data-shove-note]")).toContainText("off the face"); + await expect(page.locator("[data-shove-note]")).toContainText( + String(Math.round(displaced!.shove)), + ); + expect(Number(await attr(page, "[data-face-key]", "data-face-shove"))).toBeCloseTo( + displaced!.shove, + 0, + ); +}); + +// --------------------------------------------------------------------------- +// 5. The scale is a reading, and it is the real one. +// --------------------------------------------------------------------------- + +test("the scale on screen is 300 over the crop height", async ({ page, request }) => { + const { queue } = await view(request); + const target = queue[0]; + + // A crop with a known height, chosen so the answer is not 1.00 by accident. + const crop = { x: 100, y: 60, w: 450, h: 450 }; + const post = await request.post("/api/face", { + data: { key: target.key, verdict: "recrop", code: 1, crop }, + }); + expect(post.ok(), await post.text()).toBe(true); + + await page.goto(`/browse/faces?key=${encodeURIComponent(target.key)}`); + // Read straight off the SERVER-RENDERED html, with no wait: the reading is + // computed for the first render precisely so it does not appear a beat late. + const shown = await attr(page, "[data-face-key]", "data-face-scale"); + expect(shown).toBe((CORNER_PX / crop.h).toFixed(2)); + expect(shown).toBe(scaleOf(crop).toFixed(2)); + await expect(page.locator("[data-scale-note]")).toContainText(`${shown}×`); + await expect(page.locator("[data-scale-note]")).toContainText(`${crop.w}px`); +}); + +// --------------------------------------------------------------------------- +// 6. A crop round-trips in SOURCE pixels. +// --------------------------------------------------------------------------- + +// The units are the whole risk here. The canvas draws at the source's own +// resolution precisely so a dragged box needs no conversion, and a crop that +// came back scaled by the display width would be silently wrong rather than +// obviously broken -- it would still be a square on a face, just the wrong one. +test("a recorded crop survives a reload, in source pixels", async ({ page, request }) => { + const { queue } = await view(request); + const target = queue[queue.length - 1]; + const crop = { x: 812, y: 137, w: 331, h: 331 }; + + const post = await request.post("/api/face", { + data: { key: target.key, verdict: "recrop", code: 1, crop }, + }); + expect(post.ok(), await post.text()).toBe(true); + const saved = (await post.json()) as { judgement: { crop: Box; frameAt: number } }; + expect(saved.judgement.crop).toEqual(crop); + + await page.goto(`/browse/faces?key=${encodeURIComponent(target.key)}`); + expect(parseBox(await attr(page, "[data-face-key]", "data-crop"))).toEqual(crop); + + // And it is in the file make-thumb.mjs reads, not only in the response. + const file = readState<{ faces: Record<string, { crop: Box }> }>("face-verdicts.json"); + expect(file.faces[target.key].crop).toEqual(crop); + + // The frame the canvas draws is the frame the crop is expressed in. + const frame = await attr(page, "[data-face-key]", "data-frame"); + if (frame) { + const [w, h] = frame.split("x").map(Number); + expect(crop.x + crop.w).toBeLessThanOrEqual(w); + expect(crop.y + crop.h).toBeLessThanOrEqual(h); + } +}); + +// --------------------------------------------------------------------------- +// 7. Nothing spawns on load. +// --------------------------------------------------------------------------- + +// The detector is ~0.94s of python. Detecting the whole queue to draw a list of +// buttons would be twenty seconds of work for pictures nobody is looking at, and +// it would happen on every navigation. One candidate is focused, so exactly one +// frame and one detection. +test("only the focused candidate costs anything", async ({ page, request }) => { + const { queue } = await view(request); + expect(queue.length).toBeGreaterThan(2); + + const detects: string[] = []; + const framesHit: string[] = []; + page.on("request", (r) => { + const u = new URL(r.url()); + if (u.pathname === "/api/face/detect") detects.push(u.searchParams.get("video") ?? ""); + if (u.pathname === "/api/face/frame") framesHit.push(u.searchParams.get("video") ?? ""); + }); + + await page.goto("/browse/faces"); + await expect(page.locator("[data-face-canvas]")).toBeVisible(); + // The rows are all rendered -- this is not passing because the list is empty. + expect(await page.locator("[data-face-row]").count()).toBe(queue.length); + await expect + .poll(() => detects.length, { timeout: 20_000, message: "no detection ran at all" }) + .toBe(1); + expect(framesHit.length).toBe(1); + expect(detects[0]).toBe(queue[0].video); + + // Moving on costs exactly one more of each, and not one per row. + await page.locator("body").press("j"); + await expect + .poll(() => detects.length, { timeout: 20_000 }) + .toBe(2); + expect(framesHit.length).toBe(2); + expect(detects[1]).toBe(queue[1].video); + + // Going back costs NOTHING: the detection is held for the life of the page. + await page.locator("body").press("k"); + await page.waitForTimeout(1500); + expect(detects.length).toBe(2); +}); + +// --------------------------------------------------------------------------- +// 8. A rejection is spent, and make-thumb.mjs is what spends it. +// --------------------------------------------------------------------------- + +test("reject puts the video in the set make-thumb reads", async ({ request }) => { + const { queue, rejected: before } = await view(request); + // The LAST corner, so this never retires the episode an earlier test focused. + const target = queue[queue.length - 1]; + + const post = await request.post("/api/face", { + data: { key: target.key, verdict: "reject", code: 1 }, + }); + expect(post.ok(), await post.text()).toBe(true); + const body = (await post.json()) as { rejected: string[] }; + expect(body.rejected).toContain(target.video); + expect(body.rejected.length).toBe(before.length + (before.includes(target.video) ? 0 : 1)); + + const after = await view(request); + expect(after.rejected).toContain(target.video); + + // The FILE, read the way make-thumb.mjs reads it -- REJECTED is built from + // exactly this, so a verdict that never reached the file would be a verdict + // the next build silently ignores. + const file = readState<{ faces: Record<string, { verdict: string; video: string }> }>( + "face-verdicts.json", + ); + expect(file.faces[target.key].verdict).toBe("reject"); + expect(file.faces[target.key].video).toBe(target.video); + + // A reject carries no crop: there would be nothing to frame, and a stored one + // would be a framing make-thumb can never reach. + expect((file.faces[target.key] as { crop?: Box }).crop).toBeUndefined(); +}); + +// --------------------------------------------------------------------------- +// 9. What a proposal is not allowed to be. +// --------------------------------------------------------------------------- + +test("a crop outside the frame is refused and an unknown corner is a 404", async ({ + request, +}) => { + const { queue } = await view(request); + const target = queue[0]; + + const outside = await request.post("/api/face", { + data: { key: target.key, verdict: "recrop", code: 1, crop: { x: 1200, y: 600, w: 400, h: 400 } }, + }); + expect(outside.status()).toBe(400); + expect((await outside.json()).error).toContain("frame"); + + const zero = await request.post("/api/face", { + data: { key: target.key, verdict: "recrop", code: 1, crop: { x: 10, y: 10, w: 0, h: 0 } }, + }); + expect(zero.status()).toBe(400); + + const unknown = await request.post("/api/face", { + data: { key: "nosuchvideo@1.00", verdict: "clean" }, + }); + expect(unknown.status()).toBe(404); + + const nonsense = await request.post("/api/face", { data: { key: target.key, verdict: "maybe" } }); + expect(nonsense.status()).toBe(400); + + // Free text is REQUIRED on the other-reason code, or the reason says nothing. + const wordless = await request.post("/api/face", { + data: { key: target.key, verdict: "reject", code: 99 }, + }); + expect(wordless.status()).toBe(400); + + // The media routes guard their own inputs rather than trusting the page. + expect((await request.get("/api/face/frame?video=../etc/passwd&at=1")).status()).toBe(400); + expect((await request.get("/api/face/detect?video=nosuchvideo&at=1")).status()).toBe(404); + expect( + (await request.get(`/api/face/frame?video=${target.video}&at=-5`)).status(), + ).toBe(400); +}); + +// --------------------------------------------------------------------------- +// 10. The frame route serves the SOURCE's pixels. +// --------------------------------------------------------------------------- + +// Native resolution is what makes the canvas's coordinates source coordinates, +// which is what makes a dragged box a crop facecrop.py can cut without a ratio +// in the middle. If this ever started scaling by default, every crop the deck +// recorded would be wrong by that ratio and nothing on screen would say so. +test("the frame comes back at the source's own size unless a width is asked for", async ({ + request, +}) => { + const { queue } = await view(request); + const e = queue[0]; + const d = await detect(request, e); + + const native = await request.get( + `/api/face/frame?video=${encodeURIComponent(e.video)}&at=${e.frameAt}`, + ); + expect(native.ok()).toBe(true); + expect(native.headers()["content-type"]).toBe("image/jpeg"); + const bytes = await native.body(); + expect(jpegSize(bytes)).toEqual({ w: d.frame.w, h: d.frame.h }); + + const small = await request.get( + `/api/face/frame?video=${encodeURIComponent(e.video)}&at=${e.frameAt}&w=320`, + ); + expect(jpegSize(await small.body()).w).toBe(320); +}); + +/** The dimensions out of a JPEG's SOF marker -- no decoder, no dependency. */ +function jpegSize(buf: Buffer): { w: number; h: number } { + let i = 2; + while (i < buf.length) { + if (buf[i] !== 0xff) { + i += 1; + continue; + } + const marker = buf[i + 1]; + // SOF0..SOF15, skipping the four that are not frame headers. + if (marker >= 0xc0 && marker <= 0xcf && ![0xc4, 0xc8, 0xcc].includes(marker)) { + return { h: buf.readUInt16BE(i + 5), w: buf.readUInt16BE(i + 7) }; + } + i += 2 + buf.readUInt16BE(i + 2); + } + throw new Error("no SOF marker in that jpeg"); +} diff --git a/umtool/e2e/fixtures/make-fixture.mjs b/umtool/e2e/fixtures/make-fixture.mjs @@ -68,6 +68,11 @@ const seed = { // So do the shortlists. /browse/find writes here, and inheriting real ones // would mean a spec reordering a list somebody is actually cutting from. "shortlist.json": { version: 1, lists: {} }, + // And the face judgements. This is the ONE file /browse/faces writes, and + // make-thumb.mjs reads it to decide which source videos to skip -- so + // inheriting real ones would mean a spec quietly retiring an episode + // somebody's next cover was going to draw from. + "face-verdicts.json": { version: 1, faces: {} }, }; for (const [f, v] of Object.entries(seed)) { writeFileSync(path.join(dest, "code", f), JSON.stringify(v, null, 1)); @@ -101,6 +106,22 @@ for (const d of ["wav48", "asr", "media"]) { if (existsSync(src)) symlinkSync(src, path.join(dest, "data", d)); } +// The DETECTOR, also by reference, and it needs two things beside the code. +// +// facecrop.py resolves its model relative to its own directory (`../models`), +// and the fixture's copy of it lives in code/ -- so the model has to sit beside +// that copy or every detection fails with a missing onnx. lib/faces.ts looks +// for the venv at SONG_SCRATCH/facedet, and SONG_SCRATCH is the parent of +// SONG_DIR, which in the fixture is the fixture root. +// +// Symlinked rather than copied for the same reason wav48/ is: a venv is ~200 MB +// and read-only in practice. Without these the detect route answers 503 and the +// spec that checks autoCrop() agrees with facecrop.py has nothing to agree with. +const MODELS = path.resolve(path.dirname(new URL(import.meta.url).pathname), "..", "..", "models"); +if (existsSync(MODELS)) symlinkSync(MODELS, path.join(dest, "models")); +const FACEDET = path.join(path.dirname(SONG_DATA), "facedet"); +if (existsSync(FACEDET)) symlinkSync(FACEDET, path.join(dest, "facedet")); + // -- one FLAGGED SOURCE, derived from the symlinked asr/ ---------------------- // // suspect-sources.json was neither copied nor seeded before this, so the @@ -420,14 +441,32 @@ for (const [n, colour] of [["alpha-c.jpg", "green"], ["alpha-b.jpg", "blue"], [" "-frames:v", "1", path.join(reports, "thumbs", n), ]); } -const corners = (ids) => ids.map((v, i) => ({ video: v, srcStart: 10 + i, frameAt: 10 + i })); +// The corner videos are DERIVED from the symlinked media/, not invented. +// +// They used to be `v1`...`v11`, which have no media at all -- fine while the +// only question asked of a corner was whether accepting it would clash, and +// useless the moment /browse/faces wanted to draw the frame it was cut from and +// run a detector over it. So the eleven slots are filled with real episode ids, +// in sorted order so the fixture is the same on every run, and the SHARING +// PATTERN below is untouched: alpha-b is disjoint from the accepted alpha-c and +// may be accepted; alpha-d shares its first corner with alpha-c and must be +// refused BY NAME. When media/ is missing the synthetic names come back, so the +// clash assertions still hold on a machine with no corpus. +const mediaDir = path.join(dest, "data", "media"); +const episodes = existsSync(mediaDir) + ? readdirSync(mediaDir).filter((f) => f.endsWith(".mp4")).sort().map((f) => f.slice(0, -4)) + : []; +const slot = (n) => episodes[n - 1] ?? `v${n}`; +// A distinct second per corner, and well inside every episode -- these clips run +// for hours, so any of these lands on a real decodable frame. +const corners = (ns) => ns.map((n, i) => ({ video: slot(n), srcStart: 60 + 7 * i, frameAt: 60 + 7 * i })); const thumbManifest = { version: 1, thumbs: { - "alpha-a": { out: path.join(reports, "thumbs", "alpha-a-auto.jpg"), corners: corners(["v1", "v2", "v3", "v4"]) }, - "alpha-c": { out: path.join(reports, "thumbs", "alpha-c.jpg"), corners: corners(["v1", "v2", "v3", "v4"]) }, - "alpha-b": { out: path.join(reports, "thumbs", "alpha-b.jpg"), corners: corners(["v5", "v6", "v7", "v8"]) }, - "alpha-d": { out: path.join(reports, "thumbs", "alpha-d.jpg"), corners: corners(["v1", "v9", "v10", "v11"]) }, + "alpha-a": { out: path.join(reports, "thumbs", "alpha-a-auto.jpg"), corners: corners([1, 2, 3, 4]) }, + "alpha-c": { out: path.join(reports, "thumbs", "alpha-c.jpg"), corners: corners([1, 2, 3, 4]) }, + "alpha-b": { out: path.join(reports, "thumbs", "alpha-b.jpg"), corners: corners([5, 6, 7, 8]) }, + "alpha-d": { out: path.join(reports, "thumbs", "alpha-d.jpg"), corners: corners([1, 9, 10, 11]) }, }, used: [], }; @@ -435,7 +474,11 @@ writeFileSync(path.join(dest, "code", "thumb-manifest.json"), JSON.stringify(thu writeFileSync( path.join(dest, "code", "thumb-accepted.json"), JSON.stringify( - { version: 1, thumbs: { "alpha-c": thumbManifest.thumbs["alpha-c"] }, used: ["v1", "v2", "v3", "v4"] }, + { + version: 1, + thumbs: { "alpha-c": thumbManifest.thumbs["alpha-c"] }, + used: [1, 2, 3, 4].map(slot), + }, null, 1, ), @@ -548,7 +591,9 @@ if (planned) console.log(` planned clip (used in a build): ${planned}`); console.log(` videos/: alpha (4 cuts, 3 variants), beta (2 cuts), deck (1 cut, 2 variants)`); console.log(` deck: 1 spec error, 1 stale recipe, 1 unjudged variant, 1 judged one`); console.log(` loudness: deck wide.mp4 vs variants/wide-quiet.mp4, 10 dB apart`); -console.log(` thumbs: alpha-c accepted, alpha-b free, alpha-d clashes on v1`); +console.log(` thumbs: alpha-c accepted, alpha-b free, alpha-d clashes on ${slot(1)}`); +console.log(` faces: 8 distinct corners over 4 covers, from ${episodes.length ? "real episodes" : "synthetic ids"}`); +console.log(` detector: ${existsSync(path.join(dest, "facedet")) ? "facedet symlinked" : "NO facedet -- detect routes will 503"}`); console.log(` plans: alpha (4 notes), alpha-v2 (+bass), alpha-body.json (depth 3), overlays.json (not a plan)`); console.log(` trim set: mk-hooks (10s stem, 2 hooks)`); console.log(` flagged source: ${flagged ? flagged.video : "none — no asr/"}`); diff --git a/umtool/lib/faces.ts b/umtool/lib/faces.ts @@ -191,7 +191,12 @@ export const FACECROP_PY = () => path.join(SONG_CODE, "facecrop.py"); export type Detection = { face: FaceBox | null; + /** What autoCrop() says -- the function the deck draws with. */ auto: Box | null; + /** What facecrop.py itself said -- the function that CUTS. */ + pyAuto: Box | null; + /** Whether the two are the same box. If this is ever false the deck is lying. */ + agrees: boolean; shove: number; frame: { w: number; h: number }; }; @@ -241,16 +246,33 @@ export async function detectFace(video: string, at: number): Promise<Detection> const j = JSON.parse(stdout.trim()) as { face: FaceBox | null; + auto: Box | null; frame: { w: number; h: number }; }; - if (!j.face) return { face: null, auto: null, shove: 0, frame: j.frame }; + if (!j.face) { + return { face: null, auto: null, pyAuto: null, agrees: true, shove: 0, frame: j.frame }; + } - // The AUTO CROP is recomputed here from the face box rather than taken from - // python's own answer. Not distrust: it is the assertion that the pure - // re-implementation the deck draws with is the same function as the one that - // cuts, checked on every single detection instead of once in a test. + // BOTH ANSWERS ARE RETURNED, and this is the point of the route carrying + // `pyAuto` at all. + // + // autoCrop() is a pure re-implementation of facecrop.py's crop maths, and the + // deck draws with it on every drag because a round trip per pixel would be + // absurd. A re-implementation is only worth having if it is provably the same + // function, so python's own box comes back beside it -- checked on every + // single detection rather than once in a test, and available to a test that + // then does not have to spawn python to make the comparison. const { box, shove } = autoCrop(j.face, j.frame.w, j.frame.h); - return { face: j.face, auto: box, shove, frame: j.frame }; + const same = (a: Box | null, b: Box | null) => + !!a && !!b && a.x === b.x && a.y === b.y && a.w === b.w && a.h === b.h; + return { + face: j.face, + auto: box, + pyAuto: j.auto ?? null, + agrees: same(box, j.auto ?? null), + shove, + frame: j.frame, + }; } // ---------------------------------------------------------------------------