commit 502a6b6d54b9cbb05eed849a25c208cbfd495e02
parent 9fcb5b8dd6e43c0832e42cb64f6ee7cf362ac5db
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 18 Aug 2026 11:02:32 -0400
/browse/faces: judge the crop, not just the face
One candidate on screen with the crop as a draggable box, the way the um deck
puts one clip on screen with its window as draggable edges. That deck exists
because ten of 23 free-text notes said "cut off" -- clips rejected for their
WINDOW rather than their CONTENT. Here a corner gets rejected for its CROP
rather than its FACE, and the fix is the same one.
Three routes:
GET/POST /api/face the queue and the judgements
GET /api/face/frame ONE STILL, at native resolution. Native is the
point: the canvas's pixels are then SOURCE
pixels, so a dragged box is already in the units
facecrop.py cuts in with no ratio in between.
/api/clip/[key]/video could not be reused -- it
re-encodes a WINDOW scaled to 360px as mp4: the
wrong shape, no addressable pixels, ~50x the
bytes for a picture nobody will play.
GET /api/face/detect ~0.94s of python, so it is its own route. Nothing
detects until a candidate is focused; bundling it
into the frame route would make every arrow key
wait on YuNet for a cached picture.
The deck states three numbers rather than implying them. THE SHOVE -- how far
the automatic crop sat off the face to stay inside the frame, ~92px on the
SMRPG corner, and those 92px ARE the YouTube chrome. THE SCALE -- 300px of
output over the crop's height, in the meter colour and never a verdict hue,
because half the corners already upscale and trading softness for a tighter
crop is the human's call. THE SPEND -- a reject costs an episode the
unique-face rule never gives back.
`b` guesses the inset border by scanning the canvas's own ImageData outward
from the face box for a sustained straight gradient. OFFERED, NEVER APPLIED,
and labelled a guess: a variance spike found the inset on one of three test
frames and failed wherever the screen share had video playing. Applying a guess
automatically would be the clamp's mistake again with a nicer heuristic.
A static segment under app/browse/, so it wins over [song]. NOT a tenth nav
item -- AppNav is at nine and says a tenth wraps the header. It is reached from
the cover-art bench, which is where somebody is looking at four faces when they
notice one is a strip of stream UI.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
6 files changed, 1113 insertions(+), 1 deletion(-)
diff --git a/umtool/app/api/face/detect/route.ts b/umtool/app/api/face/detect/route.ts
@@ -0,0 +1,38 @@
+import { FaceError, detectFace } from "@/lib/faces";
+
+export const dynamic = "force-dynamic";
+
+// WHERE THE DETECTOR LOOKED.
+//
+// Its own route, and not folded into the frame route, because the two cost
+// wildly different amounts: a still is ~80ms of ffmpeg and this is ~0.94s of
+// python decoding five frames and running YuNet on each. Bundling them would
+// make every arrow key wait on the detector for a picture that is already
+// cached, and the deck would feel broken while doing exactly what it should.
+//
+// So the deck fetches the frame immediately and the detection separately, and
+// nothing is detected until a candidate is actually focused. Two of the corners
+// on record find NO FACE AT ALL at their recorded time, which is a real answer
+// this route has to be able to give: `face: null` with the frame size still
+// filled in, so the deck can draw the picture and let a person crop it by hand.
+export async function GET(request: Request) {
+ const url = new URL(request.url);
+ const video = url.searchParams.get("video") ?? "";
+ const at = Number(url.searchParams.get("at"));
+
+ if (!/^[A-Za-z0-9_-]+$/.test(video)) {
+ return Response.json({ error: "bad video" }, { status: 400 });
+ }
+ if (!Number.isFinite(at) || at < 0) {
+ return Response.json({ error: "bad time" }, { status: 400 });
+ }
+
+ try {
+ return Response.json(await detectFace(video, at), {
+ headers: { "cache-control": "no-store" },
+ });
+ } catch (e) {
+ if (e instanceof FaceError) return Response.json({ error: e.message }, { status: e.status });
+ return Response.json({ error: "could not run the detector" }, { status: 500 });
+ }
+}
diff --git a/umtool/app/api/face/frame/route.ts b/umtool/app/api/face/frame/route.ts
@@ -0,0 +1,78 @@
+import { createHash } from "node:crypto";
+import { existsSync } from "node:fs";
+import { mkdir, readFile, rename } from "node:fs/promises";
+import path from "node:path";
+import { execFile } from "node:child_process";
+import { promisify } from "node:util";
+import { sourceVideo } from "@/lib/clips";
+import { CACHE_DIR } from "@/lib/paths";
+
+const run = promisify(execFile);
+
+export const dynamic = "force-dynamic";
+
+// ONE STILL FRAME, whole.
+//
+// A new media route, where the phrase console reused /api/clip/[key]/video
+// verbatim -- so it is worth saying why this one could not.
+//
+// The deck drags a crop box OVER the picture, which means it needs the WHOLE
+// frame in source proportions and it needs to read the pixels back out of a
+// canvas to guess at the webcam inset's border. /api/clip/[key]/video answers a
+// different question: it re-encodes a WINDOW of video, scaled to 360px wide, as
+// an mp4 -- the wrong shape, no addressable pixels, and about fifty times the
+// bytes for a picture nobody is going to play.
+//
+// Cached on a sha1 of (video, at, w), exactly as that route caches its window.
+// Decoding the same frame on every arrow key would be about 80ms of ffmpeg per
+// keystroke for a picture that cannot have changed.
+export async function GET(request: Request) {
+ const url = new URL(request.url);
+ const video = url.searchParams.get("video") ?? "";
+ const at = Number(url.searchParams.get("at"));
+ // NO `w` MEANS NATIVE, and that is what the deck asks for.
+ //
+ // Serving the source's own pixels makes the canvas's coordinate system the
+ // SOURCE coordinate system: a crop box dragged on screen is already in the
+ // units facecrop.py cuts in, with no ratio in the middle to get backwards.
+ // Every episode is 1280x720, so this is ~120 KB and decodes instantly.
+ //
+ // `w` is still honoured, and capped rather than trusted -- a client asking for
+ // 40000 would ask ffmpeg to allocate accordingly.
+ const raw = url.searchParams.get("w");
+ const w = raw === null ? null : Math.max(160, Math.min(1920, Number(raw) || 960));
+
+ if (!/^[A-Za-z0-9_-]+$/.test(video)) return new Response("bad video", { status: 400 });
+ if (!Number.isFinite(at) || at < 0) return new Response("bad time", { status: 400 });
+ if (raw !== null && !Number.isFinite(Number(raw))) return new Response("bad width", { status: 400 });
+
+ const file = sourceVideo(video);
+ // media/ is bulk data and re-derivable, so a missing source is a normal state
+ // rather than an error -- the deck simply says it cannot draw this one.
+ if (!existsSync(file)) return new Response("no video", { status: 404 });
+
+ const stamp = createHash("sha1")
+ .update(`face1|${video}|${at.toFixed(3)}|${w ?? "native"}`)
+ .digest("hex")
+ .slice(0, 16);
+ const cached = path.join(CACHE_DIR, `${stamp}.jpg`);
+
+ if (!existsSync(cached)) {
+ await mkdir(CACHE_DIR, { recursive: true });
+ const tmp = `${cached}.tmp.jpg`;
+ // -ss BEFORE -i for the fast seek. When a width is asked for it is scaled to
+ // an even one with the aspect preserved; when it is not, the frame comes out
+ // at the source's own size and no mapping is needed at all.
+ await run("ffmpeg", [
+ "-nostdin", "-v", "error", "-y",
+ "-ss", at.toFixed(3), "-i", file,
+ "-frames:v", "1", ...(w === null ? [] : ["-vf", `scale=${w}:-2`]),
+ "-q:v", "3", tmp,
+ ]);
+ await rename(tmp, cached).catch(() => {});
+ }
+
+ return new Response(new Uint8Array(await readFile(cached)), {
+ headers: { "content-type": "image/jpeg", "cache-control": "no-store" },
+ });
+}
diff --git a/umtool/app/api/face/route.ts b/umtool/app/api/face/route.ts
@@ -0,0 +1,60 @@
+import {
+ FaceError,
+ faceQueue,
+ isFaceBody,
+ judgeFace,
+ readFaceVerdicts,
+ rejectedVideos,
+} from "@/lib/faces";
+
+export const dynamic = "force-dynamic";
+
+// The face judger's queue and its judgements.
+//
+// GET the 22 corners on record, deduped, accepted-cover ones first, plus
+// every judgement made so far and the videos a rebuild will now skip
+// POST one judgement
+//
+// The queue is READ FROM THE MANIFESTS on every request, like every other pile
+// in this tool. A cached copy would go stale the moment make-thumb.mjs ran,
+// which is the one command most likely to be running while this page is open.
+//
+// Nothing here writes thumb-manifest.json. lib/thumbs.ts sets out why at
+// length; the short version is that accept-thumb.mjs is its sole writer and a
+// second writer with different atomicity guarantees on the authority file is
+// not acceptable. The judgements go in a file this app owns outright, and
+// make-thumb.mjs reads it.
+
+const noStore = { "cache-control": "no-store" };
+
+export async function GET() {
+ const [queue, verdicts, rejected] = await Promise.all([
+ faceQueue(),
+ readFaceVerdicts(),
+ rejectedVideos(),
+ ]);
+ return Response.json({ queue, faces: verdicts.faces, rejected }, { headers: noStore });
+}
+
+export async function POST(request: Request) {
+ let body: unknown;
+ try {
+ body = await request.json();
+ } catch {
+ return Response.json({ error: "bad json" }, { status: 400 });
+ }
+ if (!isFaceBody(body)) {
+ return Response.json({ error: "not a face judgement" }, { status: 400 });
+ }
+
+ try {
+ const judgement = await judgeFace(body);
+ return Response.json(
+ { ok: true, judgement, rejected: await rejectedVideos() },
+ { headers: noStore },
+ );
+ } catch (e) {
+ if (e instanceof FaceError) return Response.json({ error: e.message }, { status: e.status });
+ return Response.json({ error: "could not write the judgement" }, { status: 500 });
+ }
+}
diff --git a/umtool/app/browse/faces/page.tsx b/umtool/app/browse/faces/page.tsx
@@ -0,0 +1,85 @@
+import BrowseHeader from "@/components/BrowseHeader";
+import FaceDeck from "@/components/FaceDeck";
+import { faceQueue, readFaceVerdicts, rejectedVideos } from "@/lib/faces";
+
+export const dynamic = "force-dynamic";
+
+// THE FACE JUDGER: every corner that has ever been cut into a cover, one at a
+// time, with the crop as something you can see and move.
+//
+// A STATIC segment under app/browse/, so it wins over app/browse/[song]/ --
+// which would otherwise catch /browse/faces, fail readSong() and 404. The same
+// trap /browse/decisions and /browse/find both document, and the suite asserts
+// it here too, because when this regresses the page still renders perfectly in
+// isolation and only the URL stops working.
+//
+// It is NOT a tenth nav item. AppNav is at nine and says why: a tenth wraps the
+// header on a laptop, and a nav that wraps stops reading as one row of places.
+// This is reached from the cover-art bench on a song page -- which is where
+// somebody is looking at the four faces when they notice one is wrong.
+//
+// The page reads the queue directly rather than fetching its own API, like
+// /browse/find does: a render that waited on a round trip to itself would be
+// serialising for nothing. /api/face exists so the suite and a script can ask
+// the same question of the same function.
+
+type Params = Record<string, string | string[] | undefined>;
+const one = (v: string | string[] | undefined) => (Array.isArray(v) ? v[0] : v) ?? null;
+
+export default async function FacesPage({ searchParams }: { searchParams: Promise<Params> }) {
+ const params = await searchParams;
+ const [queue, verdicts, rejected] = await Promise.all([
+ faceQueue(),
+ readFaceVerdicts(),
+ rejectedVideos(),
+ ]);
+
+ // Two ways in, because the two callers know different things. The bench knows
+ // a COVER and wants its four faces; a link written in a note knows the exact
+ // corner. Neither is made to learn the other's identifier.
+ const wantKey = one(params.key);
+ const wantCover = one(params.cover);
+ const start =
+ (wantKey && queue.find((e) => e.key === wantKey)?.key) ??
+ (wantCover && queue.find((e) => e.covers.includes(wantCover))?.key) ??
+ null;
+
+ const onAccepted = queue.filter((e) => e.onAccepted).length;
+ const unrecorded = queue.filter((e) => !e.crop).length;
+
+ return (
+ <div className="flex h-full flex-col">
+ <BrowseHeader
+ crumbs={[{ href: "/browse", label: "songs" }, { label: "faces" }]}
+ note={`${queue.length} corners · ${onAccepted} on accepted covers · ${Object.keys(verdicts.faces).length} judged`}
+ />
+
+ <main className="deck-main flex-1 space-y-3 p-4">
+ <p className="text-[12px] text-[var(--color-dim)]">
+ Every corner on record, deduped by episode and moment, with the ones that ship
+ first. The dashed amber box is what <span className="num">facecrop.py</span> would
+ cut; the solid one is yours. Judging writes{" "}
+ <span className="num">face-verdicts.json</span>, which{" "}
+ <span className="num">make-thumb.mjs</span> reads — this page never touches the
+ thumb manifests.
+ {unrecorded > 0 && (
+ <>
+ {" "}
+ <span className="text-[var(--color-dirty)]" data-unrecorded={unrecorded}>
+ {unrecorded} of them have no box recorded at all, so what they were cut from
+ is not known — only what they were cut near.
+ </span>
+ </>
+ )}
+ </p>
+
+ <FaceDeck
+ queue={queue}
+ initial={verdicts.faces}
+ rejected={rejected}
+ start={start}
+ />
+ </main>
+ </div>
+ );
+}
diff --git a/umtool/components/FaceDeck.tsx b/umtool/components/FaceDeck.tsx
@@ -0,0 +1,834 @@
+"use client";
+
+import { useCallback, useEffect, useMemo, useRef, useState } from "react";
+import {
+ autoCrop,
+ boxInFrame,
+ clampBox,
+ CORNER_PX,
+ FACE_REASONS,
+ OTHER_CODE,
+ reasonText,
+ scaleOf,
+ SHOVE_FLOOR,
+ type Box,
+ type FaceBox,
+ type FaceJudgement,
+ type FaceVerdict,
+} from "@/lib/face-types";
+
+// ---------------------------------------------------------------------------
+// THE FACE JUDGER.
+//
+// Several shipped cover corners carry stream UI rather than a face. The
+// bottom-left of the Super Mario RPG cover is the clearest: a strip of YouTube
+// chrome, a dark block, and the webcam inset's own red border, with Jeremy in
+// the right 60% of the picture.
+//
+// This is the um sorter's lesson on a different axis. That deck exists because
+// ten of 23 free-text notes said "cut off" and six were filed as `not an um` --
+// clips rejected for their WINDOW rather than for their CONTENT. Here a corner
+// gets rejected for its CROP rather than for its FACE, and the fix is the same:
+// make the crop a parameter you can see and move, instead of an output you can
+// only accept or discard.
+//
+// So the page is one candidate at a time, like the um deck, and the crop is
+// draggable. Three things are stated rather than implied, because each of them
+// is a number somebody would otherwise have to guess at:
+//
+// THE SHOVE how far the automatic crop sat off the face to stay inside the
+// frame. On the SMRPG corner that is ~92px, and those 92px ARE
+// the YouTube chrome. Unmeasured it is a mystery; measured it is
+// the sentence that explains what went wrong.
+// THE SCALE 300px of output over the crop's own height. Half the corners on
+// record already upscale, so tightening a crop to exclude UI
+// trades contamination for softness -- and that trade is the
+// human's to make, not this component's to make quietly.
+// THE SPEND a rejection costs a source episode out of a pool the
+// unique-face rule never gives back.
+//
+// The crop box the deck draws is computed by autoCrop(), a pure
+// re-implementation of facecrop.py's maths -- so what you see beside your own
+// framing is what the CLI would actually do, on every drag, with no round trip.
+// ---------------------------------------------------------------------------
+
+export type FaceEntryView = {
+ key: string;
+ video: string;
+ srcStart: number;
+ frameAt: number;
+ covers: string[];
+ accepted: string[];
+ onAccepted: boolean;
+ crop: Box | null;
+ link: string | null;
+};
+
+type Detection = {
+ face: FaceBox | null;
+ auto: Box | null;
+ shove: number;
+ frame: { w: number; h: number };
+ error?: string;
+};
+
+const HANDLE = 10; // display px within which a pointer grabs a corner
+const MIN_SIDE = 40; // source px; below this the corner is unusable anyway
+const fmt = (n: number) => n.toFixed(2);
+const boxAttr = (b: Box | null) => (b ? `${b.x},${b.y},${b.w},${b.h}` : "");
+
+export default function FaceDeck({
+ queue,
+ initial,
+ rejected,
+ start,
+}: {
+ queue: FaceEntryView[];
+ initial: Record<string, FaceJudgement>;
+ rejected: string[];
+ start: string | null;
+}) {
+ const startAt = Math.max(0, queue.findIndex((e) => e.key === start));
+ const [idx, setIdx] = useState(startAt);
+ const [faces, setFaces] = useState(initial);
+ const [spent, setSpent] = useState(rejected);
+
+ // Detections are held per key for the life of the page. A second pass over a
+ // corner already looked at must not cost another second of python.
+ const [dets, setDets] = useState<Record<string, Detection>>({});
+ const [loading, setLoading] = useState(false);
+
+ const [crop, setCrop] = useState<Box | null>(null);
+ const [pending, setPending] = useState<FaceVerdict | null>(null);
+ const [code, setCode] = useState<number | null>(null);
+ const [text, setText] = useState("");
+ const [guess, setGuess] = useState<Box | null>(null);
+ const [busy, setBusy] = useState(false);
+ const [error, setError] = useState<string | null>(null);
+
+ const canvas = useRef<HTMLCanvasElement | null>(null);
+ const preview = useRef<HTMLCanvasElement | null>(null);
+ const image = useRef<HTMLImageElement | null>(null);
+ const drag = useRef<{ mode: "move" | "resize"; ax: number; ay: number; ox: number; oy: number } | null>(null);
+
+ const entry = queue[idx] ?? null;
+ const det = entry ? dets[entry.key] : undefined;
+ const judged = entry ? faces[entry.key] : undefined;
+ const frame = det?.frame ?? null;
+
+ // ---- the frame ------------------------------------------------------------
+ // Requested at NATIVE resolution, so the canvas's pixels ARE source pixels and
+ // a dragged box needs no mapping to become a crop facecrop.py can cut.
+ const frameSrc = entry
+ ? `/api/face/frame?video=${encodeURIComponent(entry.video)}&at=${entry.frameAt}`
+ : null;
+
+ const [ready, setReady] = useState(false);
+ useEffect(() => {
+ if (!frameSrc) return;
+ setReady(false);
+ const img = new Image();
+ img.onload = () => {
+ image.current = img;
+ setReady(true);
+ };
+ img.onerror = () => {
+ image.current = null;
+ setReady(false);
+ };
+ img.src = frameSrc;
+ return () => {
+ img.onload = null;
+ img.onerror = null;
+ };
+ }, [frameSrc]);
+
+ // ---- the detection --------------------------------------------------------
+ // NOTHING SPAWNS ON LOAD. The detector is ~0.94s of python and the queue is 22
+ // rows; detecting all of them to draw a list would be twenty seconds of work
+ // for twenty-one pictures nobody is looking at yet.
+ useEffect(() => {
+ if (!entry || dets[entry.key]) return;
+ let live = true;
+ setLoading(true);
+ fetch(`/api/face/detect?video=${encodeURIComponent(entry.video)}&at=${entry.frameAt}`, {
+ cache: "no-store",
+ })
+ .then(async (r) => {
+ const j = (await r.json()) as Detection & { error?: string };
+ if (!live) return;
+ setDets((d) => ({ ...d, [entry.key]: j }));
+ })
+ .catch(() => {
+ if (live) {
+ setDets((d) => ({
+ ...d,
+ [entry.key]: { face: null, auto: null, shove: 0, frame: { w: 0, h: 0 }, error: "the detector did not answer" },
+ }));
+ }
+ })
+ .finally(() => live && setLoading(false));
+ return () => {
+ live = false;
+ };
+ }, [entry, dets]);
+
+ // ---- what the crop starts as ---------------------------------------------
+ // A judgement already made wins, then the box the manifest recorded, then the
+ // automatic one. That order is the point of the tool: a human framing is never
+ // re-derived out from under itself.
+ useEffect(() => {
+ if (!entry) return;
+ setPending(judged?.verdict ?? null);
+ setCode(judged?.code ?? null);
+ setText(judged?.text ?? "");
+ setGuess(null);
+ setError(null);
+ setCrop(judged?.crop ?? entry.crop ?? det?.auto ?? null);
+ // Deliberately keyed on the CORNER, not on the detection: re-running this
+ // when the detection lands would throw away a box already being dragged.
+ // eslint-disable-next-line react-hooks/exhaustive-deps
+ }, [entry?.key]);
+
+ // The one case where a late detection may set the box: there was no box at all.
+ useEffect(() => {
+ if (crop === null && det?.auto) setCrop(det.auto);
+ }, [crop, det]);
+
+ // ---- drawing --------------------------------------------------------------
+ const draw = useCallback(() => {
+ const c = canvas.current;
+ const img = image.current;
+ if (!c || !img) return;
+ c.width = img.naturalWidth;
+ c.height = img.naturalHeight;
+ const ctx = c.getContext("2d");
+ if (!ctx) return;
+ ctx.drawImage(img, 0, 0);
+
+ const stroke = (b: Box, colour: string, dash: number[], width: number) => {
+ ctx.save();
+ ctx.setLineDash(dash);
+ ctx.lineWidth = width;
+ ctx.strokeStyle = colour;
+ ctx.strokeRect(b.x + 0.5, b.y + 0.5, b.w, b.h);
+ ctx.restore();
+ };
+
+ if (det?.face) stroke(det.face, "rgba(132,150,168,0.75)", [4, 4], 2);
+ // WHAT THE CLI WOULD DO, always drawn, so the diff between the automatic
+ // crop and the chosen one is visible rather than remembered.
+ if (det?.auto) stroke(det.auto, "rgba(210,153,34,0.85)", [10, 6], 3);
+ if (guess) stroke(guess, "rgba(86,212,196,0.9)", [2, 6], 3);
+ if (crop) {
+ stroke(crop, "#58a6ff", [], 4);
+ ctx.save();
+ ctx.fillStyle = "#58a6ff";
+ for (const [hx, hy] of [
+ [crop.x, crop.y],
+ [crop.x + crop.w, crop.y],
+ [crop.x, crop.y + crop.h],
+ [crop.x + crop.w, crop.y + crop.h],
+ ]) {
+ ctx.fillRect(hx - 7, hy - 7, 14, 14);
+ }
+ ctx.restore();
+ }
+
+ const p = preview.current;
+ if (p && crop) {
+ p.width = CORNER_PX;
+ p.height = CORNER_PX;
+ const pctx = p.getContext("2d");
+ if (pctx) {
+ pctx.clearRect(0, 0, CORNER_PX, CORNER_PX);
+ // The SAME resample the corner gets: source box in, 300x300 out.
+ pctx.drawImage(img, crop.x, crop.y, crop.w, crop.h, 0, 0, CORNER_PX, CORNER_PX);
+ }
+ }
+ }, [crop, det, guess]);
+
+ useEffect(() => {
+ draw();
+ }, [draw, ready]);
+
+ // ---- dragging -------------------------------------------------------------
+ // Square-locked, because CORNER_W and CORNER_H are both 300 and a rectangle
+ // would be squashed rather than cropped on the way in.
+ const toSource = (e: React.PointerEvent<HTMLCanvasElement>) => {
+ const c = canvas.current;
+ if (!c) return { x: 0, y: 0, k: 1 };
+ const r = c.getBoundingClientRect();
+ // The canvas is drawn at SOURCE resolution and displayed at whatever the
+ // layout gives it, so this ratio is the only mapping in the component --
+ // and it maps display pixels to source pixels, never the other way.
+ const k = r.width > 0 ? c.width / r.width : 1;
+ return { x: (e.clientX - r.left) * k, y: (e.clientY - r.top) * k, k };
+ };
+
+ const onDown = (e: React.PointerEvent<HTMLCanvasElement>) => {
+ if (!crop || !frame) return;
+ const { x, y, k } = toSource(e);
+ const near = HANDLE * k;
+ const corners: [number, number, number, number][] = [
+ [crop.x, crop.y, crop.x + crop.w, crop.y + crop.h],
+ [crop.x + crop.w, crop.y, crop.x, crop.y + crop.h],
+ [crop.x, crop.y + crop.h, crop.x + crop.w, crop.y],
+ [crop.x + crop.w, crop.y + crop.h, crop.x, crop.y],
+ ];
+ for (const [cx, cy, ax, ay] of corners) {
+ if (Math.abs(x - cx) <= near && Math.abs(y - cy) <= near) {
+ drag.current = { mode: "resize", ax, ay, ox: 0, oy: 0 };
+ e.currentTarget.setPointerCapture(e.pointerId);
+ return;
+ }
+ }
+ if (x >= crop.x && x <= crop.x + crop.w && y >= crop.y && y <= crop.y + crop.h) {
+ drag.current = { mode: "move", ax: 0, ay: 0, ox: x - crop.x, oy: y - crop.y };
+ e.currentTarget.setPointerCapture(e.pointerId);
+ }
+ };
+
+ const onMove = (e: React.PointerEvent<HTMLCanvasElement>) => {
+ const d = drag.current;
+ if (!d || !crop || !frame) return;
+ const { x, y } = toSource(e);
+ if (d.mode === "move") {
+ setCrop(clampBox({ ...crop, x: x - d.ox, y: y - d.oy }, frame.w, frame.h));
+ return;
+ }
+ // The opposite corner stays put and the side follows the longer reach, so
+ // the box grows the way a square selection is expected to.
+ const side = Math.max(MIN_SIDE, Math.min(Math.abs(x - d.ax), Math.abs(y - d.ay)));
+ const nx = x < d.ax ? d.ax - side : d.ax;
+ const ny = y < d.ay ? d.ay - side : d.ay;
+ setCrop(clampBox({ x: nx, y: ny, w: side, h: side }, frame.w, frame.h));
+ };
+
+ const onUp = () => {
+ drag.current = null;
+ };
+
+ // ---- the border guess -----------------------------------------------------
+ // Client-side, on exactly the pixels on screen -- no route, no python. It
+ // scans outward from the face box for a sustained straight gradient, which is
+ // what the edge of a webcam inset looks like: a border, a bezel, a hard cut
+ // between a video feed and a browser window.
+ //
+ // OFFERED, NEVER APPLIED. A variance spike found the inset on one of three
+ // test frames and failed wherever the screen share had video playing, so this
+ // is a guess and it is labelled as one. Applying a guess automatically would
+ // be the clamp's mistake again, with a nicer heuristic.
+ const suggest = useCallback(() => {
+ const c = canvas.current;
+ const img = image.current;
+ if (!c || !img || !frame) return;
+ const ctx = c.getContext("2d");
+ if (!ctx) return;
+ const face = det?.face ?? (crop ? { x: crop.x, y: crop.y, w: crop.w, h: crop.h } : null);
+ if (!face) return;
+
+ const W = c.width;
+ const H = c.height;
+ const data = ctx.getImageData(0, 0, W, H).data;
+ const luma = (x: number, y: number) => {
+ const i = ((y | 0) * W + (x | 0)) * 4;
+ return 0.299 * data[i] + 0.587 * data[i + 1] + 0.114 * data[i + 2];
+ };
+
+ const fx0 = Math.max(0, face.x);
+ const fy0 = Math.max(0, face.y);
+ const fx1 = Math.min(W - 1, face.x + face.w);
+ const fy1 = Math.min(H - 1, face.y + face.h);
+ // How far out to look: an inset is a few face-widths at most, and searching
+ // the whole frame would find the browser chrome instead of the bezel.
+ const reach = Math.round(Math.max(face.w, face.h) * 2.2);
+
+ /** The strongest sustained vertical edge in [from,to), scanning `dir`. */
+ const vertical = (from: number, to: number, dir: 1 | -1) => {
+ let best = { x: dir === 1 ? to : from, score: 0 };
+ for (let x = from; dir === 1 ? x < to : x > to; x += dir) {
+ if (x < 1 || x >= W - 1) continue;
+ let s = 0;
+ let n = 0;
+ for (let y = fy0; y <= fy1; y += 2) {
+ s += Math.abs(luma(x + 1, y) - luma(x - 1, y));
+ n += 1;
+ }
+ const score = n ? s / n : 0;
+ if (score > best.score) best = { x, score };
+ }
+ return best;
+ };
+ const horizontal = (from: number, to: number, dir: 1 | -1) => {
+ let best = { y: dir === 1 ? to : from, score: 0 };
+ for (let y = from; dir === 1 ? y < to : y > to; y += dir) {
+ if (y < 1 || y >= H - 1) continue;
+ let s = 0;
+ let n = 0;
+ for (let x = fx0; x <= fx1; x += 2) {
+ s += Math.abs(luma(x, y + 1) - luma(x, y - 1));
+ n += 1;
+ }
+ const score = n ? s / n : 0;
+ if (score > best.score) best = { y, score };
+ }
+ return best;
+ };
+
+ // A flat wall reads as a weak edge everywhere; below this there is nothing
+ // to propose and saying so is better than proposing noise.
+ const FLOOR = 12;
+ const left = vertical(Math.max(1, fx0 - reach), fx0, 1);
+ const right = vertical(Math.min(W - 2, fx1 + reach), fx1, -1);
+ const top = horizontal(Math.max(1, fy0 - reach), fy0, 1);
+ const bottom = horizontal(Math.min(H - 2, fy1 + reach), fy1, -1);
+ if ([left.score, right.score, top.score, bottom.score].every((s) => s < FLOOR)) {
+ setError("no inset border found in these pixels — crop it by eye");
+ return;
+ }
+
+ const x0 = left.score >= FLOOR ? left.x : fx0;
+ const x1 = right.score >= FLOOR ? right.x : fx1;
+ const y0 = top.score >= FLOOR ? top.y : fy0;
+ const y1 = bottom.score >= FLOOR ? bottom.y : fy1;
+ // Squared off, because the corner is square. The inset's SHORTER side wins:
+ // a square that overflows the border would put back exactly the content the
+ // border was found to exclude.
+ const side = Math.max(MIN_SIDE, Math.min(x1 - x0, y1 - y0));
+ setGuess(clampBox({ x: x0, y: y0, w: side, h: side }, frame.w, frame.h));
+ setError(null);
+ }, [crop, det, frame]);
+
+ // ---- committing -----------------------------------------------------------
+ const commit = useCallback(async () => {
+ if (!entry || !pending) return;
+ setBusy(true);
+ setError(null);
+ try {
+ const res = await fetch("/api/face", {
+ method: "POST",
+ headers: { "content-type": "application/json" },
+ cache: "no-store",
+ body: JSON.stringify({
+ key: entry.key,
+ verdict: pending,
+ ...(code !== null ? { code } : {}),
+ ...(text ? { text } : {}),
+ // A REJECT carries no crop: there is nothing to frame, and storing one
+ // would leave a framing behind that make-thumb would never reach.
+ ...(pending !== "reject" && crop ? { crop } : {}),
+ ...(det?.auto ? { auto: det.auto } : {}),
+ }),
+ });
+ const j = (await res.json()) as { error?: string; judgement?: FaceJudgement; rejected?: string[] };
+ if (!res.ok || !j.judgement) throw new Error(j.error || `HTTP ${res.status}`);
+ setFaces((f) => ({ ...f, [entry.key]: j.judgement as FaceJudgement }));
+ if (j.rejected) setSpent(j.rejected);
+ // Straight to the next one still unjudged, the way the um deck moves on.
+ const next = queue.findIndex((e, i) => i > idx && !faces[e.key] && e.key !== entry.key);
+ setIdx(next >= 0 ? next : Math.min(queue.length - 1, idx + 1));
+ } catch (e) {
+ setError(e instanceof Error ? e.message : String(e));
+ } finally {
+ setBusy(false);
+ }
+ }, [entry, pending, code, text, crop, det, queue, idx, faces]);
+
+ const resetCrop = useCallback(() => {
+ setGuess(null);
+ if (det?.auto) setCrop(det.auto);
+ else if (entry?.crop) setCrop(entry.crop);
+ }, [det, entry]);
+
+ // ---- keyboard -------------------------------------------------------------
+ // Same shape as Deck.tsx and WordSelector.tsx: a pending verdict owns the
+ // digits, so 1/2/3 pick the verdict and then the digits pick its reason. Two
+ // meanings for one key is what the um deck already does, and the alternative
+ // is a second row of letters nobody would remember.
+ useEffect(() => {
+ const onKey = (e: KeyboardEvent) => {
+ const t = e.target as HTMLElement | null;
+ if (t && (t.tagName === "INPUT" || t.tagName === "TEXTAREA")) return;
+ const k = e.key;
+
+ if (k === "Enter") {
+ e.preventDefault();
+ if (pending) void commit();
+ return;
+ }
+ if (k === "Escape") {
+ e.preventDefault();
+ setPending(null);
+ setCode(null);
+ return;
+ }
+ if (k === "j") {
+ e.preventDefault();
+ setIdx((i) => Math.min(queue.length - 1, i + 1));
+ return;
+ }
+ if (k === "k") {
+ e.preventDefault();
+ setIdx((i) => Math.max(0, i - 1));
+ return;
+ }
+ if (k === "r") {
+ e.preventDefault();
+ resetCrop();
+ return;
+ }
+ if (k === "b") {
+ e.preventDefault();
+ suggest();
+ return;
+ }
+ if (/^[0-9]$/.test(k)) {
+ e.preventDefault();
+ // A verdict that wants a reason takes the digits next; `clean` never
+ // does, so 1/2/3 keep meaning the verdict while it is selected.
+ if (pending && pending !== "clean") {
+ const n = k === "9" ? OTHER_CODE : Number(k);
+ if (n === OTHER_CODE || (n >= 1 && n <= FACE_REASONS.length)) setCode(n);
+ return;
+ }
+ const v = ({ "1": "clean", "2": "recrop", "3": "reject" } as const)[k as "1" | "2" | "3"];
+ if (v) {
+ setPending(v);
+ setCode(null);
+ }
+ }
+ };
+ window.addEventListener("keydown", onKey);
+ return () => window.removeEventListener("keydown", onKey);
+ }, [pending, queue.length, commit, resetCrop, suggest]);
+
+ // ---- readings -------------------------------------------------------------
+ const scale = crop ? scaleOf(crop) : 0;
+ // Recomputed here from the face box rather than taken from the route, so what
+ // is on screen is the pure function the CLI mirrors -- the same reason the
+ // route recomputes it too.
+ const local = det?.face && frame ? autoCrop(det.face, frame.w, frame.h) : null;
+ const shove = local?.shove ?? det?.shove ?? 0;
+ const moved = useMemo(
+ () => (crop && det?.auto ? boxAttr(crop) !== boxAttr(det.auto) : false),
+ [crop, det],
+ );
+ const inFrame = crop && frame ? boxInFrame(crop, frame.w, frame.h) : true;
+
+ if (!entry) {
+ return (
+ <p className="text-[12px] text-[var(--color-dim)]" data-face-deck data-count="0">
+ No corners on record. make-thumb.mjs writes them; there is nothing to judge until it
+ has run.
+ </p>
+ );
+ }
+
+ const done = Object.keys(faces).length;
+
+ return (
+ <div
+ data-face-deck
+ data-count={queue.length}
+ data-judged={done}
+ data-rejected={spent.length}
+ className="flex flex-col gap-3"
+ >
+ {/* ---- the candidate ---------------------------------------------- */}
+ <section
+ data-face-key={entry.key}
+ data-face-scale={crop ? scale.toFixed(2) : ""}
+ data-face-shove={det ? shove.toFixed(1) : ""}
+ data-crop={boxAttr(crop)}
+ data-auto={boxAttr(det?.auto ?? null)}
+ data-verdict={judged?.verdict ?? ""}
+ data-pending={pending ?? ""}
+ data-frame={frame ? `${frame.w}x${frame.h}` : ""}
+ className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] p-3"
+ >
+ <div className="mb-2 flex flex-wrap items-baseline gap-2">
+ <span className="num text-[13px] text-[var(--color-text)]">{entry.video}</span>
+ <span className="num text-[12px] text-[var(--color-dim)]">@{fmt(entry.srcStart)}</span>
+ <span className="micro">
+ {idx + 1} of {queue.length}
+ </span>
+ {entry.onAccepted ? (
+ <span className="text-[11px] text-[var(--color-sel)]" data-on-accepted="1">
+ on the shipped {entry.accepted.join(" + ")} cover
+ </span>
+ ) : (
+ <span className="text-[11px] text-[var(--color-dim)]" data-on-accepted="0">
+ candidate only — {entry.covers.join(", ")}
+ </span>
+ )}
+ {entry.link && (
+ <a
+ href={entry.link}
+ target="_blank"
+ rel="noreferrer"
+ className="text-[11px] text-[var(--color-sel)] underline"
+ >
+ the moment
+ </a>
+ )}
+ {judged && (
+ <span className="text-[11px] text-[var(--color-meter)]">
+ judged {judged.verdict}
+ {judged.code ? ` — ${reasonText(judged.code, judged.text)}` : ""}
+ </span>
+ )}
+ </div>
+
+ <div className="flex flex-wrap items-start gap-4">
+ <div className="min-w-0 flex-1">
+ <canvas
+ ref={canvas}
+ data-face-canvas
+ onPointerDown={onDown}
+ onPointerMove={onMove}
+ onPointerUp={onUp}
+ onPointerCancel={onUp}
+ className="w-full max-w-[860px] touch-none rounded border border-[var(--color-line)] bg-[var(--color-panel-2)]"
+ />
+ {!ready && (
+ <p className="micro mt-1" data-frame-state="loading">
+ fetching the frame at {fmt(entry.frameAt)}s…
+ </p>
+ )}
+ </div>
+
+ <div className="w-[300px] shrink-0 space-y-2">
+ <canvas
+ ref={preview}
+ data-face-preview
+ width={CORNER_PX}
+ height={CORNER_PX}
+ className="h-[300px] w-[300px] rounded border border-[var(--color-sel)] bg-[var(--color-panel-2)]"
+ />
+ <p className="micro">the corner, at the size it is cut</p>
+
+ {/* THE READING, in the meter colour and never a verdict hue. It
+ says what the resample is; whether that is acceptable is a
+ judgement this component does not get to make. */}
+ {crop && (
+ <p className="text-[12px] text-[var(--color-meter)]" data-scale-note>
+ side {crop.w}px → {CORNER_PX}px = {scale.toFixed(2)}×
+ {scale > 1 ? " — soft" : ""}
+ </p>
+ )}
+ {!inFrame && (
+ <p className="text-[12px] text-[var(--color-bad)]">
+ that box is outside the frame
+ </p>
+ )}
+ </div>
+ </div>
+
+ {/* ---- what the automatic crop did ------------------------------- */}
+ <div className="mt-2 space-y-1">
+ {loading && !det && (
+ <p className="micro" data-detect-state="running">
+ running the detector…
+ </p>
+ )}
+ {det && !det.face && (
+ <p className="text-[12px] text-[var(--color-dirty)]" data-no-face>
+ no face found at {fmt(entry.frameAt)}s — this corner cannot be reproduced
+ automatically, so crop it by hand
+ </p>
+ )}
+ {det?.face && shove > SHOVE_FLOOR && (
+ <p className="text-[12px] text-[var(--color-dirty)]" data-shove-note>
+ the automatic crop sat {Math.round(shove)} px off the face to stay inside the
+ frame — that displacement is what fills the corner with stream UI
+ </p>
+ )}
+ {det?.face && shove <= SHOVE_FLOOR && (
+ <p className="micro" data-shove-note>
+ the automatic crop is centred on the face (
+ {shove.toFixed(1)} px off) — whatever is wrong here is inside the inset, not
+ at the frame edge
+ </p>
+ )}
+ {moved && (
+ <p className="micro" data-moved>
+ your crop differs from the automatic one — <kbd>r</kbd> puts it back
+ </p>
+ )}
+ {guess && (
+ <p className="text-[12px] text-[var(--color-meter)]" data-guess>
+ a guess at the inset border: {boxAttr(guess)} —{" "}
+ <button
+ type="button"
+ onClick={() => {
+ setCrop(guess);
+ setGuess(null);
+ }}
+ data-take-guess
+ className="text-[var(--color-sel)] underline"
+ >
+ take it
+ </button>{" "}
+ or ignore it. The scan finds a straight edge, not a webcam.
+ </p>
+ )}
+ </div>
+
+ {/* ---- the verdict ------------------------------------------------ */}
+ <div className="mt-3 flex flex-wrap items-center gap-1.5">
+ {(
+ [
+ ["clean", "1", "the crop is fine"],
+ ["recrop", "2", "the face is fine, the box is not"],
+ ["reject", "3", "spend the episode, use another"],
+ ] as const
+ ).map(([v, key, why]) => (
+ <button
+ key={v}
+ type="button"
+ title={why}
+ onClick={() => {
+ setPending(v);
+ setCode(null);
+ }}
+ data-verdict-btn={v}
+ aria-pressed={pending === v}
+ className={`rounded border px-2 py-0.5 font-mono text-[12px] ${
+ pending === v
+ ? "border-[var(--color-sel)] text-[var(--color-sel)]"
+ : "border-[var(--color-line)] text-[var(--color-dim)] hover:text-[var(--color-text)]"
+ }`}
+ >
+ {v} <span className="micro">{key}</span>
+ </button>
+ ))}
+
+ <button
+ type="button"
+ onClick={resetCrop}
+ data-reset-crop
+ className="ml-2 rounded border border-[var(--color-line)] px-2 py-0.5 font-mono text-[12px] text-[var(--color-dim)] hover:text-[var(--color-text)]"
+ >
+ reset to auto <span className="micro">r</span>
+ </button>
+ <button
+ type="button"
+ onClick={suggest}
+ data-suggest
+ className="rounded border border-[var(--color-line)] px-2 py-0.5 font-mono text-[12px] text-[var(--color-dim)] hover:text-[var(--color-text)]"
+ >
+ suggest the border <span className="micro">b</span>
+ </button>
+ <button
+ type="button"
+ disabled={!pending || busy}
+ onClick={() => void commit()}
+ data-commit
+ className="rounded border border-[var(--color-sel)] px-2 py-0.5 font-mono text-[12px] text-[var(--color-sel)] disabled:opacity-40"
+ >
+ {busy ? "saving…" : "commit"} <span className="micro">↵</span>
+ </button>
+ </div>
+
+ {pending && pending !== "clean" && (
+ <div className="mt-2 flex flex-wrap items-center gap-1.5" data-reasons>
+ <span className="micro">why</span>
+ {FACE_REASONS.map((r, i) => (
+ <button
+ key={r}
+ type="button"
+ onClick={() => setCode(i + 1)}
+ data-reason={i + 1}
+ aria-pressed={code === i + 1}
+ className={`rounded border px-1.5 py-0.5 font-mono text-[11px] ${
+ code === i + 1
+ ? "border-[var(--color-sel)] text-[var(--color-sel)]"
+ : "border-[var(--color-line)] text-[var(--color-dim)] hover:text-[var(--color-text)]"
+ }`}
+ >
+ {i + 1} {r}
+ </button>
+ ))}
+ <button
+ type="button"
+ onClick={() => setCode(OTHER_CODE)}
+ data-reason={OTHER_CODE}
+ aria-pressed={code === OTHER_CODE}
+ className={`rounded border px-1.5 py-0.5 font-mono text-[11px] ${
+ code === OTHER_CODE
+ ? "border-[var(--color-sel)] text-[var(--color-sel)]"
+ : "border-[var(--color-line)] text-[var(--color-dim)] hover:text-[var(--color-text)]"
+ }`}
+ >
+ 9 other
+ </button>
+ {code === OTHER_CODE && (
+ <input
+ value={text}
+ onChange={(e) => setText(e.target.value)}
+ onKeyDown={(e) => e.stopPropagation()}
+ placeholder="in your own words"
+ aria-label="reason"
+ data-reason-text
+ className="w-72 rounded border border-[var(--color-line)] bg-[var(--color-panel-2)] px-2 py-0.5 font-mono text-[12px] text-[var(--color-text)] outline-none focus:border-[var(--color-sel)]"
+ />
+ )}
+ </div>
+ )}
+
+ {pending === "reject" && (
+ <p className="mt-2 text-[12px] text-[var(--color-dirty)]" data-spend-warning>
+ rejecting spends {entry.video} — the unique-face rule never gives an episode
+ back, and {spent.length} {spent.length === 1 ? "is" : "are"} spent this way
+ already
+ </p>
+ )}
+
+ {error && <p className="mt-2 text-[11px] text-[var(--color-bad)]">{error}</p>}
+
+ <p className="micro mt-2">
+ <kbd>j</kbd>/<kbd>k</kbd> move · <kbd>1</kbd>/<kbd>2</kbd>/<kbd>3</kbd> verdict ·
+ digits pick a reason once a verdict is chosen · <kbd>r</kbd> reset · <kbd>b</kbd>{" "}
+ suggest · <kbd>↵</kbd> commit
+ </p>
+ </section>
+
+ {/* ---- the queue --------------------------------------------------- */}
+ <section className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] p-2">
+ <p className="micro mb-1">
+ {queue.length} corners on record · {done} judged · {spent.length} episodes spent
+ </p>
+ <ul className="flex flex-wrap gap-1">
+ {queue.map((e, i) => {
+ const j = faces[e.key];
+ return (
+ <li key={e.key}>
+ <button
+ type="button"
+ onClick={() => setIdx(i)}
+ data-face-row={e.key}
+ data-row-accepted={e.onAccepted ? "1" : "0"}
+ data-row-verdict={j?.verdict ?? ""}
+ aria-current={i === idx ? "true" : undefined}
+ className={`rounded border px-1.5 py-0.5 font-mono text-[11px] ${
+ i === idx
+ ? "border-[var(--color-sel)] text-[var(--color-sel)]"
+ : j
+ ? "border-[var(--color-line)] text-[var(--color-meter)]"
+ : e.onAccepted
+ ? "border-[var(--color-line)] text-[var(--color-text)]"
+ : "border-[var(--color-line)] text-[var(--color-dim)]"
+ }`}
+ >
+ {e.video}
+ {j ? ` ${j.verdict === "reject" ? "✕" : "✓"}` : ""}
+ </button>
+ </li>
+ );
+ })}
+ </ul>
+ </section>
+ </div>
+ );
+}
diff --git a/umtool/components/ThumbBench.tsx b/umtool/components/ThumbBench.tsx
@@ -1,6 +1,7 @@
"use client";
import { useState } from "react";
+import Link from "next/link";
import NoteField from "@/components/NoteField";
import type { NoteMap } from "@/lib/notes";
import type { ThumbView } from "@/lib/thumbs";
@@ -126,7 +127,7 @@ export default function ThumbBench({
{/* Which four episodes this cover spends. The clip: note target
is the one the arrangement already uses, so a note written
here shows up beside the same clip everywhere else. */}
- <ul className="flex flex-wrap gap-x-3 gap-y-0.5">
+ <ul className="flex flex-wrap items-baseline gap-x-3 gap-y-0.5">
{c.corners.map((v) => (
<li key={v.clipId} className="num text-[11px]">
{v.link ? (
@@ -144,6 +145,22 @@ export default function ThumbBench({
<span className="ml-1 text-[var(--color-dim)]">@{v.srcStart.toFixed(1)}</span>
</li>
))}
+ {/* THE WAY IN to the face judger. It is here rather than in
+ the nav -- AppNav is at nine items and a tenth wraps the
+ header -- and here is also where the question comes up: a
+ person looking at these four faces is the person who
+ notices one of them is a strip of YouTube chrome. */}
+ {c.corners.length > 0 && (
+ <li>
+ <Link
+ href={`/browse/faces?cover=${encodeURIComponent(c.name)}`}
+ data-judge-faces={c.name}
+ className="text-[11px] text-[var(--color-sel)] hover:underline"
+ >
+ judge {c.corners.length === 1 ? "this face" : `these ${c.corners.length} faces`} →
+ </Link>
+ </li>
+ )}
</ul>
{c.clash ? (