"use client"; import { useCallback, useEffect, useMemo, useRef, useState } from "react"; import AppNav from "./AppNav"; import PitchRail, { type CorpusPitch } from "./PitchRail"; import Waveform, { type Peaks } from "./Waveform"; import WordSelector, { type Vocab } from "./WordSelector"; import VerdictRow, { choicesFor, type Choice, type Reasons } from "./VerdictRow"; import { type Clip, type ClipPage, type Mode } from "@/lib/types"; const VIEW_PAD = 1.0; // seconds each side of the selection, to start with const WIDEN = 1.0; // seconds a `[` / `]` or an edge-drag adds /** Is an ASR token a filler sound rather than a real word? */ const isFillerToken = (w: string) => /^(?:u+[mh]+|e+r+m?|h+m+|m+h*m+)$/i.test(w.replace(/[^a-z]/gi, "")); /** * The word label this window ALREADY implies, read off the ASR. * * Everything needed was on the card and nothing used it: `tBHm09OttbQ@94.17` * has token "um", `wordIn` "and", and the ASR word "and" (94.08-94.32) sitting * inside the default window (94.10-94.723). "and um" was derivable without * anyone typing it, and the word selector still asked. * * Ordering is by MIDPOINT against the candidate's midpoint, not by whether the * word ends before the um starts. ASR boundaries overlap constantly -- "and" * above ends 0.15s INSIDE the detected um -- so an ends-before test would call * it neither leading nor trailing and give up on exactly the clips this is for. * * Recomputed from the live selection, so dragging a handle over a neighbour * changes the proposal, which is the honest behaviour: the label describes the * window, and the window is what is being edited. */ export function inferWord( detail: { token: string; candStart: number; candEnd: number; words: { start: number; end: number; w: string }[] } | null, sel: { from: number; to: number }, ): string { if (!detail) return ""; const inside = detail.words .filter((w) => w.end > sel.from + 0.03 && w.start < sel.to - 0.03) .map((w) => ({ ...w, t: w.w.toLowerCase().replace(/[^a-z'-]/g, "") })) .filter((w) => w.t) .sort((a, b) => a.start - b.start); // Does the window still cover the um this card is about? A SECOND capture in // the same window usually does not -- it is a different sound entirely, and // its label is simply what the ASR says is there ("you know"), with no filler // to compose around. const overlapsCandidate = sel.to > detail.candStart + 0.02 && sel.from < detail.candEnd - 0.02; const filler = (detail.token || "um").toLowerCase().replace(/[^a-z-]/g, "") || "um"; const candMid = (detail.candStart + detail.candEnd) / 2; const asrHasFiller = inside.some((w) => isFillerToken(w.t)); const parts: string[] = []; let placed = false; for (const w of inside) { // Insert the um at its own position in the running order, so a word AFTER // it reads "um and" and one before reads "and um" without a second rule. if (overlapsCandidate && !asrHasFiller && !placed && (w.start + w.end) / 2 > candMid) { parts.push(filler); placed = true; } parts.push(w.t); } if (overlapsCandidate && !asrHasFiller && !placed) parts.push(filler); if (!parts.length) return ""; // A label names a sound; past three words the window has swallowed speech and // the answer is to fix the window, not to name it. if (parts.length > 3) return ""; // Nothing to say when all the window holds is the um the card already knows. if (parts.length === 1 && isFillerToken(parts[0])) return ""; return parts.join(" "); } type Props = { page: ClipPage; vocab: Vocab; reasons: Reasons }; // Mirrors lib/usage.ts's Use. `build` is the unique root-relative id and is // what distinct builds are counted by; `label` is what a person reads, because // once videos//plan/ is indexed the id is a path. type Use = { build: string; label: string; song: string | null; voice: string; at: number }; const useName = (u: Use) => (u.song ? `${u.song}/${u.label}` : u.label); type Evidence = { flatness: number | null; flatMax: number; hissy: boolean; wordsInside: string[]; gate: number; source: { clips: number; f0s: number[]; lo: number; hi: number; spread: number; wide: boolean; suspect: boolean; }; }; type Counts = ClipPage["counts"]; export default function Deck({ page, vocab: initialVocab, reasons }: Props) { const { mode, clips, total, orderNote } = page; const [idx, setIdx] = useState(0); const [detail, setDetail] = useState(null); const [view, setView] = useState({ from: 0, to: 1 }); const [sel, setSel] = useState({ from: 0, to: 1 }); const [peaks, setPeaks] = useState(null); const [pending, setPending] = useState(null); const [freeText, setFreeText] = useState(""); const [word, setWord] = useState(""); const [wordCustom, setWordCustom] = useState(false); // The label was proposed from the ASR rather than chosen. Tracked so a human // choice is never overwritten by a later re-inference, and so the card can // say where the word came from. const [wordInferred, setWordInferred] = useState(false); const [wordTouched, setWordTouched] = useState(false); const [wordOpen, setWordOpen] = useState(false); const [vocab, setVocab] = useState(initialVocab); const [counts, setCounts] = useState(page.counts); const [judged, setJudged] = useState>({}); const [note, setNote] = useState(""); const [busy, setBusy] = useState(false); const [playhead, setPlayhead] = useState(null); const [silFrac, setSilFrac] = useState(0.06); const [autoState, setAutoState] = useState<{ auto: boolean; off: boolean; dl: number; dr: number }>( { auto: false, off: false, dl: 0, dr: 0 }, ); const [dragged, setDragged] = useState(false); // ---- more than one sound in the same window ------------------------------- // // A window often holds two usable sounds ("um ... uh", a restarted um). The // deck advances on every verdict, so taking the second one meant finding the // clip again by key. `moreHere` says "keep this window after I submit": the // card stays put, what was just taken is drawn on the waveform so it cannot // be taken twice, and the selection moves past it ready for the next. // // It CLEARS on every submission by design. Staying is the exception, so the // exception is what you re-assert; a sticky version would silently hold the // deck still after you had stopped thinking about it. const [moreHere, setMoreHere] = useState(false); const [captured, setCaptured] = useState<{ from: number; to: number; verdict: number }[]>([]); // media/ is bulk data and not every source has it, so a missing video is a // normal state rather than an error -- hide the player instead of showing a // broken one. Reset per clip, since the next source may well have it. const [videoGone, setVideoGone] = useState(false); // ---- measurement, as opposed to judgement -------------------------------- // The corpus distribution is fetched once and never changes underneath a // sitting; the per-clip evidence is measured on the window CURRENTLY on // screen, so dragging a handle changes the answer. const [corpus, setCorpus] = useState(null); const [evidence, setEvidence] = useState(null); // Where this clip is already sounding, and a drop held back for confirmation. // Dropping a clip that is placed in a rendered plan means the next arrange // cannot pick it, so a video that exists stops being reproducible from the // palette that made it. That is a judgement with consequences, and nothing // used to say so. const [uses, setUses] = useState([]); const [confirmDrop, setConfirmDrop] = useState<{ verdict: number; code: number; text?: string } | null>( null, ); const clip = clips[idx]; // ---- audio --------------------------------------------------------------- const ctxRef = useRef(null); const bufRef = useRef(null); const bufWindow = useRef<{ from: number; to: number } | null>(null); const srcRef = useRef(null); const rafRef = useRef(null); const reqRef = useRef(0); // ---- the next clip, fetched before it is asked for ----------------------- // // Every clip costs three round trips before it can be judged: the detail // (which opens a wav window and runs clipWindow), the audio slice, and the // peaks. At judging speed that gap lands between every verdict, and a pause // you did not ask for reads as the tool thinking rather than as loading. // // It is all prefetchable, because a clip's DEFAULT window is knowable without // anyone looking at it -- so the next card's audio can be decoded and waiting // while this one is still on screen. What is cached is exactly what the load // path produces, so a hit is a straight assignment and there is no second // code path that could disagree with the first. type Prefetched = { detail: Clip; view: { from: number; to: number }; buf: AudioBuffer | null; bufWindow: { from: number; to: number } | null; peaks: Peaks | null; }; const preRef = useRef>(new Map()); const preInFlight = useRef>(new Set()); // ---- the window you left a clip on --------------------------------------- // // Going back and returning used to hand back the SERVER's default window, so // any adjustment made before moving on was lost the moment you looked at the // clip again -- and the natural reason to go back is to reconsider the very // edit that was discarded. // // Held for the session only, and only in memory: this is where you left the // card, not a decision. A decision is a verdict, and those are written. `R` // still returns the card to the server's default, which is the other half of // the same behaviour -- remembering is worthless if you cannot get back to // the untouched window. const editsRef = useRef>( new Map(), ); // What the view effect has already fetched, so a cache hit does not // immediately re-request the very bytes it was given. const fetchedRef = useRef(""); const viewTag = (key: string, v: { from: number; to: number }) => `${key}|${v.from.toFixed(3)}|${v.to.toFixed(3)}`; const ctx = () => { if (!ctxRef.current) ctxRef.current = new AudioContext(); return ctxRef.current; }; // The picture follows the same clock as the sound. It is MUTED and never the // audio source: the audio route levels every window against a fixed reference // region so the volume cannot jump mid-judgement, and playing the video's own // track would throw that away. const vidRef = useRef(null); const stop = useCallback(() => { if (srcRef.current) { try { srcRef.current.stop(); } catch { /* already stopped */ } srcRef.current = null; } if (vidRef.current) vidRef.current.pause(); if (rafRef.current) cancelAnimationFrame(rafRef.current); rafRef.current = null; setPlayhead(null); }, []); const play = useCallback( (from: number, to: number) => { const buf = bufRef.current; const w = bufWindow.current; if (!buf || !w) return; stop(); const a = Math.max(0, from - w.from); const dur = Math.max(0.01, Math.min(buf.duration - a, to - from)); const src = ctx().createBufferSource(); src.buffer = buf; src.connect(ctx().destination); // Sample-exact: an explicit offset and duration, not a seek-and-stop. The // old preview seeked an mp3 and stopped on an animation frame, so tails // sounded included that were never selected -- and clips went into the // palette tens of ms short. src.start(0, a, dur); srcRef.current = src; // Same window as the audio, so the offset into the clip is the same // number -- no second mapping to drift. const vid = vidRef.current; if (vid && vid.readyState >= 1) { try { vid.currentTime = a; void vid.play().catch(() => {}); } catch { /* not seekable yet */ } } const t0 = ctx().currentTime; const tick = () => { const el = ctx().currentTime - t0; if (el > dur) { if (vidRef.current) vidRef.current.pause(); setPlayhead(null); rafRef.current = null; return; } setPlayhead(from + el); rafRef.current = requestAnimationFrame(tick); }; rafRef.current = requestAnimationFrame(tick); }, [stop], ); // ---- loading a clip ------------------------------------------------------ useEffect(() => { if (!clip) return; let cancelled = false; setDetail(null); setPeaks(null); setPending(null); setFreeText(""); setWordOpen(false); setAutoState({ auto: false, off: false, dl: 0, dr: 0 }); setDragged(false); setVideoGone(false); // Both are per-WINDOW, and this is the only place the window really changes. setCaptured([]); setMoreHere(false); setWordInferred(false); setWordTouched(false); stop(); // DROP THE PREVIOUS CLIP'S AUDIO. Keeping it is what made the handles // vanish after every verdict: the auto-expand effect fires as soon as // `detail` arrives, and it only checks that SOME buffer exists. With the // old buffer still in hand it walked the previous clip's envelope and wrote // a selection in the previous clip's absolute seconds -- hundreds of // seconds away from the new view, so both handles were drawn off-canvas. // It stuck because nothing recomputes the selection afterwards, which is // why only a reload brought them back. bufRef.current = null; bufWindow.current = null; // Already waiting? Then there is no load at all -- state goes straight in, // in the same order the network path sets it. const hit = preRef.current.get(clip.key); if (hit) { preRef.current.delete(clip.key); setDetail(hit.detail); setWord(hit.detail.word ?? ""); setWordCustom(false); const kept = editsRef.current.get(clip.key); setSel(kept ? kept.sel : { from: hit.detail.selStart, to: hit.detail.selEnd }); setView(kept ? kept.view : hit.view); setPeaks(hit.peaks); bufRef.current = hit.buf; bufWindow.current = hit.bufWindow; // Claim the fetch that would otherwise re-request these exact bytes -- // but only when the view is the one the prefetch actually fetched. A // remembered view is a different window and must be fetched. if (hit.buf && hit.peaks && !kept) fetchedRef.current = viewTag(clip.key, hit.view); return; } (async () => { const r = await fetch(`/api/clip/${encodeURIComponent(clip.key)}`, { cache: "no-store" }); if (!r.ok || cancelled) return; const d = (await r.json()) as Clip; if (cancelled) return; setDetail(d); setWord(d.word ?? ""); setWordCustom(false); const kept = editsRef.current.get(clip.key); const dur = d.sourceDuration || d.selEnd + VIEW_PAD; setSel(kept ? kept.sel : { from: d.selStart, to: d.selEnd }); setView( kept ? kept.view : { from: Math.max(0, d.selStart - VIEW_PAD), to: Math.min(dur || d.selEnd + VIEW_PAD, d.selEnd + VIEW_PAD), }, ); })(); return () => { cancelled = true; }; }, [clip, stop]); // ---- filling the queue ahead --------------------------------------------- const prefetchClip = useCallback(async (key: string) => { if (!key || preRef.current.has(key) || preInFlight.current.has(key)) return; preInFlight.current.add(key); try { const r = await fetch(`/api/clip/${encodeURIComponent(key)}`, { cache: "no-store" }); if (!r.ok) return; const d = (await r.json()) as Clip; const dur = d.sourceDuration || d.selEnd + VIEW_PAD; const view = { from: Math.max(0, d.selStart - VIEW_PAD), to: Math.min(dur || d.selEnd + VIEW_PAD, d.selEnd + VIEW_PAD), }; const entry: Prefetched = { detail: d, view, buf: null, bufWindow: null, peaks: null }; preRef.current.set(key, entry); // Oldest out first. Three is enough to stay ahead of judging without // holding a pile of decoded audio -- each entry is a few seconds of PCM. while (preRef.current.size > 3) { const oldest = preRef.current.keys().next().value; if (!oldest || oldest === key) break; preRef.current.delete(oldest); } if (d.sourceMissing) return; const qs = `from=${view.from.toFixed(3)}&to=${view.to.toFixed(3)}`; const [aRes, pRes] = await Promise.all([ fetch(`/api/clip/${encodeURIComponent(key)}/audio?${qs}`, { cache: "no-store" }), fetch(`/api/clip/${encodeURIComponent(key)}/peaks?${qs}`, { cache: "no-store" }), ]); if (pRes.ok) entry.peaks = (await pRes.json()) as Peaks; if (aRes.ok) { const bytes = await aRes.arrayBuffer(); entry.buf = await ctx().decodeAudioData(bytes.slice(0)); const hdr = aRes.headers.get("x-window")?.split(",").map(Number); entry.bufWindow = hdr && hdr.length === 2 && Number.isFinite(hdr[0]) ? { from: hdr[0], to: hdr[1] } : view; } } catch { // A prefetch is an optimisation and nothing more: on any failure the // normal load path runs when the clip is actually reached. preRef.current.delete(key); } finally { preInFlight.current.delete(key); } }, []); // Start once THIS clip is playable, so the prefetch never competes with the // audio someone is waiting on. useEffect(() => { if (!detail || !peaks) return; const t = setTimeout(() => { for (const c of [clips[idx + 1], clips[idx + 2]]) if (c) void prefetchClip(c.key); }, 120); return () => clearTimeout(t); }, [detail, peaks, idx, clips, prefetchClip]); // Where this clip already sounds. Per clip, not per window: the question is // about the clip that exists in finished plans, not about the edit on screen. useEffect(() => { if (!clip) { setUses([]); return; } let cancelled = false; setUses([]); setConfirmDrop(null); void (async () => { try { const r = await fetch(`/api/usage?key=${encodeURIComponent(clip.key)}`, { cache: "no-store" }); if (!r.ok || cancelled) return; const j = (await r.json()) as { uses: Use[] }; setUses(j.uses ?? []); } catch { /* no plans readable: behave exactly as "used nowhere" */ } })(); return () => { cancelled = true; }; }, [clip]); // Remember where this card was left. Written on every settled change, so // stepping back lands on the window you were looking at rather than on the // one the server proposed. useEffect(() => { if (!detail) return; editsRef.current.set(detail.key, { sel, view }); }, [detail, sel, view]); // Propose the label the window implies, unless a human has already said. useEffect(() => { if (!detail) return; if (wordTouched) return; // The clip's recorded label describes the FIRST take from this window. A // second capture is a different sound -- often not an um at all, just what // the ASR says is there ("you know") -- so once something has been taken, // the proposal comes from the window rather than from the clip's history. if (detail.word && captured.length === 0) return; const proposed = inferWord(detail, sel); setWord(proposed); setWordCustom(false); setWordInferred(!!proposed); }, [detail, sel, wordTouched, captured.length]); // The corpus's pitch distribution, once per sitting. useEffect(() => { void (async () => { try { const r = await fetch("/api/corpus", { cache: "no-store" }); if (r.ok) setCorpus((await r.json()) as CorpusPitch); } catch { /* the rail simply does not draw */ } })(); }, []); // Evidence for the window ON SCREEN. Debounced, because it re-measures on // every handle drag and each measurement is an FFT over the selection. useEffect(() => { if (!detail || detail.sourceMissing) { setEvidence(null); return; } let cancelled = false; const t = setTimeout(() => { void (async () => { try { const qs = `from=${sel.from.toFixed(3)}&to=${sel.to.toFixed(3)}`; const r = await fetch(`/api/clip/${encodeURIComponent(detail.key)}/evidence?${qs}`, { cache: "no-store", }); if (!r.ok || cancelled) return; setEvidence((await r.json()) as Evidence); } catch { /* evidence is additive: its absence must never block a verdict */ } })(); }, 220); return () => { cancelled = true; clearTimeout(t); }; }, [detail, sel]); // ---- fetching the view (audio + peaks) ----------------------------------- useEffect(() => { if (!detail || detail.sourceMissing) return; // A prefetched clip arrives with its audio and peaks already in hand; // without this the effect would immediately re-fetch and re-decode the // identical window, which is the very cost the prefetch just paid. if (fetchedRef.current === viewTag(detail.key, view)) return; const my = ++reqRef.current; const key = encodeURIComponent(detail.key); const qs = `from=${view.from.toFixed(3)}&to=${view.to.toFixed(3)}`; let cancelled = false; (async () => { const [aRes, pRes] = await Promise.all([ fetch(`/api/clip/${key}/audio?${qs}`, { cache: "no-store" }), fetch(`/api/clip/${key}/peaks?${qs}`, { cache: "no-store" }), ]); if (cancelled || my !== reqRef.current) return; if (pRes.ok) setPeaks((await pRes.json()) as Peaks); if (aRes.ok) { const bytes = await aRes.arrayBuffer(); if (cancelled || my !== reqRef.current) return; const decoded = await ctx().decodeAudioData(bytes.slice(0)); if (cancelled || my !== reqRef.current) return; bufRef.current = decoded; // The server clamps to the episode, so trust what it says it sent. const hdr = aRes.headers.get("x-window")?.split(",").map(Number); bufWindow.current = hdr && hdr.length === 2 && Number.isFinite(hdr[0]) ? { from: hdr[0], to: hdr[1] } : { from: view.from, to: view.to }; // Only once BOTH have landed, so a failed half re-requests next time // rather than being remembered as satisfied. if (pRes.ok) fetchedRef.current = viewTag(detail.key, view); } })(); return () => { cancelled = true; }; }, [detail, view]); // Auto-expand once the audio for a fresh, untouched clip has arrived. const autoTried = useRef(""); useEffect(() => { if (!detail || !bufRef.current) return; if (autoTried.current === detail.key) return; if (dragged || autoState.off || judged[detail.key]) return; autoTried.current = detail.key; const g = proposeExpand(); if (g && (g.dl || g.dr)) { setSel({ from: g.from, to: g.to }); setAutoState({ auto: true, off: false, dl: g.dl, dr: g.dr }); } // eslint-disable-next-line react-hooks/exhaustive-deps }, [detail, peaks]); // ---- auto-expand --------------------------------------------------------- // Ported from the static page, including the two bugs worth not // reintroducing: `lastLoud` starts UNSET (seeded with the starting frame, a // walk that found nothing still reported the edge, so every apply pushed the // window out one hop), and a frame only counts once TWO in a row clear the // threshold (a lone 5ms blip could otherwise drag an edge into silence). const proposeExpand = useCallback(() => { const buf = bufRef.current; const w = bufWindow.current; if (!buf || !w || !detail) return null; // The buffer must actually COVER this selection. Clearing it per clip is // the primary fix; this is the one that makes the failure impossible rather // than merely unlikely, because every path that leaves a buffer from one // clip in front of another clip's window ends here. Without it the walk // below runs on unrelated samples and returns times from the wrong part of // the episode, which is not a worse selection -- it is a selection with no // relationship to what is on screen. if (sel.from < w.from - 1e-6 || sel.to > w.to + 1e-6) return null; const ch = buf.getChannelData(0); const sr = buf.sampleRate; const HOPS = 0.005; const HOP = Math.max(1, Math.round(HOPS * sr)); const env: number[] = []; for (let k = 0; k + HOP <= ch.length; k += HOP) { let acc = 0; for (let j = 0; j < HOP; j += 1) acc += ch[k + j] * ch[k + j]; env.push(Math.sqrt(acc / HOP)); } if (!env.length) return null; const fr = (t: number) => Math.max(0, Math.min(env.length - 1, Math.round((t - w.from) / HOPS))); // Threshold from the SELECTION's own peak, not the clip's: a quiet um next // to a loud word would otherwise be measured against the word and never // expand. let selPk = 0; for (let k = fr(sel.from); k <= fr(sel.to); k += 1) selPk = Math.max(selPk, env[k]); const sorted = env.slice().sort((p, q) => p - q); const floor = sorted[Math.floor(sorted.length * 0.15)] || 0; const th = Math.max(selPk * silFrac, floor * 1.8); // A dip must LAST to count as silence, or the gap between two syllables ends // the expansion one syllable in. const HOLD = Math.round(0.03 / HOPS); const walk = (from: number, dir: number) => { let quiet = 0; let loud = 0; let lastLoud = -1; for (let k = from; k >= 0 && k < env.length; k += dir) { if (env[k] >= th) { quiet = 0; loud += 1; if (loud >= 2) lastLoud = k; } else { loud = 0; if (++quiet >= HOLD) break; } } return lastLoud; }; const la = walk(fr(sel.from), -1); const lb = walk(fr(sel.to), +1); let A = la < 0 ? sel.from : Math.min(sel.from, w.from + la * HOPS); let B = lb < 0 ? sel.to : Math.max(sel.to, w.from + (lb + 1) * HOPS); // Never expand INTO a neighbouring word. Stop at the word boundary PLUS // clipwindow.mjs's own guards, not at the bare timestamp: clamping to the // bare boundary let the walk cross the 40ms guard into the tail of the // previous word, which is what produced a rash of clips that sounded like // "and uh". The guards are asymmetric because parakeet timestamps a word's // start late. const GUARD_PREV = 0.04; const GUARD_NEXT = 0.07; for (const wd of detail.words) { if (wd.end <= sel.from + 0.02 && wd.end + GUARD_PREV > A) A = wd.end + GUARD_PREV; if (wd.start >= sel.to - 0.02 && wd.start - GUARD_NEXT < B) B = wd.start - GUARD_NEXT; } A = Math.max(w.from, Math.min(A, sel.from)); B = Math.min(w.to, Math.max(B, sel.to)); return { from: A, to: B, dl: Math.round((sel.from - A) * 1000), dr: Math.round((B - sel.to) * 1000), }; }, [detail, sel, silFrac]); const applyExpand = useCallback( (withPlay = true) => { const g = proposeExpand(); if (!g || (!g.dl && !g.dr)) return false; setSel({ from: g.from, to: g.to }); setAutoState({ auto: true, off: false, dl: g.dl, dr: g.dr }); if (withPlay) play(g.from, g.to); return true; }, [proposeExpand, play], ); // Back to the SERVER's window, discarding both the auto-expansion and // anything remembered from a previous visit. The other half of remembering: // a kept selection is only safe to keep if there is one key that always // returns the card to untouched. const undoExpand = useCallback(() => { if (!detail) return; const dur = detail.sourceDuration || detail.defaultEnd + VIEW_PAD; const freshView = { from: Math.max(0, detail.defaultStart - VIEW_PAD), to: Math.min(dur || detail.defaultEnd + VIEW_PAD, detail.defaultEnd + VIEW_PAD), }; editsRef.current.delete(detail.key); setSel({ from: detail.defaultStart, to: detail.defaultEnd }); setView(freshView); setAutoState({ auto: false, off: true, dl: 0, dr: 0 }); setDragged(false); setWordTouched(false); play(detail.defaultStart, detail.defaultEnd); }, [detail, play]); // Un-expand ONE side. The auto-expansion is two independent decisions -- it // reports +dl and +dr separately -- but undoing it was all-or-nothing, so a // clip whose start was right and whose end swallowed a breath had to be reset // whole and re-dragged by hand. The default edge is always inside the current // one, so putting an edge back can never cross the other. const resetSide = useCallback( (side: "start" | "end") => { if (!detail) return; const next = side === "start" ? { from: detail.defaultStart, to: sel.to } : { from: sel.from, to: detail.defaultEnd }; if (next.to - next.from < 0.05) return; setSel(next); setAutoState((a) => ({ ...a, auto: false, off: side === "start" ? !a.dr : !a.dl, dl: side === "start" ? 0 : a.dl, dr: side === "end" ? 0 : a.dr, })); play(next.from, next.to); }, [detail, sel, play], ); // Hear the whole VIEW, not the selection. Judging an edge means hearing what // is on the other side of it, and until now the only way to do that was to // drag a handle out, listen, and drag it back -- which edits the thing you // were trying to check. const playView = useCallback(() => { play(view.from, view.to); }, [play, view]); // ---- widening ------------------------------------------------------------ const widen = useCallback( (side: "start" | "end") => { if (!detail) return; const dur = detail.sourceDuration || view.to + WIDEN; setView((v) => ({ from: side === "start" ? Math.max(0, v.from - WIDEN) : v.from, to: side === "end" ? Math.min(dur, v.to + WIDEN) : v.to, })); }, [detail, view.to], ); // ---- committing ---------------------------------------------------------- const commit = useCallback( async (verdict: number, code: number, text?: string) => { if (!detail || busy) return; // A DROP of a clip that is sounding in a finished build is held for // confirmation. Only a drop: accepting or parking it changes nothing // about a plan that already places it, and interrupting those would make // the confirmation noise rather than a warning. if (verdict === 0 && uses.length && !confirmDrop) { stop(); setConfirmDrop({ verdict, code, text }); return; } setConfirmDrop(null); setBusy(true); stop(); try { const res = await fetch("/api/verdict", { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify({ key: detail.key, mode, verdict, code, text, word: word || undefined, custom: wordCustom || undefined, window: { from: +sel.from.toFixed(4), to: +sel.to.toFixed(4) }, defaultWindow: { from: detail.defaultStart, to: detail.defaultEnd }, needsContext: detail.sourceMissing ? "both" : null, }), }); const out = await res.json(); if (!res.ok || out.error) { setNote(`write failed: ${out.error ?? res.status}`); return; } setCounts(out.counts); setJudged((j) => ({ ...j, [detail.key]: labelFor(mode, verdict) })); setNote( `${labelFor(mode, verdict)} · ${out.finalKey}${out.moved ? " (re-trimmed, new key)" : ""}`, ); setPending(null); setFreeText(""); if (moreHere) { // Stay on this window. What was just taken is remembered so it can be // drawn and not taken twice, and the selection steps past it at the // same length -- a starting point to drag from, not a guess at where // the next sound is. const taken = { from: +sel.from.toFixed(4), to: +sel.to.toFixed(4), verdict }; setCaptured((c) => [...c, taken]); const len = Math.max(0.14, taken.to - taken.from); const nextFrom = Math.min(view.to - len - 0.02, taken.to + 0.05); if (nextFrom > taken.to) setSel({ from: nextFrom, to: nextFrom + len }); setAutoState({ auto: false, off: true, dl: 0, dr: 0 }); setDragged(true); // never auto-expand a hand-placed follow-up // Let the NEXT span propose its own label. Without this the word just // committed would ride along onto a different sound. setWordTouched(false); setWord(""); setWordInferred(false); // The checkbox is per-submission: re-assert it if there is another. setMoreHere(false); setNote( `${labelFor(mode, verdict)} · ${out.finalKey} — staying on this window (${ captured.length + 1 } taken)`, ); return; } setIdx((i) => Math.min(clips.length - 1, i + 1)); } catch (err) { setNote(`write failed: ${String(err)}`); } finally { setBusy(false); } }, [detail, busy, mode, word, wordCustom, sel, view.to, clips.length, stop, moreHere, captured.length, uses.length, confirmDrop], ); /** * Take back the last sound captured from this window. * * A capture is already WRITTEN -- it went through the same verdict machinery * as any other -- so this cannot just drop the marker and pretend. It undoes * the verdict on the server (restoring the exact pile membership the journal * recorded), then puts the selection back where that capture was so it can be * redone. */ const undoCapture = useCallback(async () => { if (!captured.length || busy) return false; setBusy(true); stop(); try { const res = await fetch("/api/verdict", { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify({ undo: true }), }); const out = await res.json(); if (!res.ok || out.error) { setNote(`undo failed: ${out.error ?? res.status}`); return false; } const last = captured[captured.length - 1]; setCaptured((c) => c.slice(0, -1)); setCounts(out.counts); setSel({ from: last.from, to: last.to }); setWordTouched(false); setNote(`undone · ${out.undone.finalKey} — back to that window`); return true; } catch (err) { setNote(`undo failed: ${String(err)}`); return false; } finally { setBusy(false); } }, [captured, busy, stop]); const choose = useCallback( (c: Choice) => { if (c.reasons) setPending(c); else void commit(c.verdict, 0); }, [commit], ); // ---- keyboard ------------------------------------------------------------ useEffect(() => { const onKey = (e: KeyboardEvent) => { const target = e.target as HTMLElement | null; if (target && (target.tagName === "INPUT" || target.tagName === "TEXTAREA")) return; if (wordOpen) return; // Same guard as the buttons: a verdict before the window has arrived would // be dropped silently, and a keyboard-driven pass is where that would go // unnoticed longest. if (!detail) return; const k = e.key; // A held drop owns the keyboard: the arrows underneath it would otherwise // commit a different verdict on a card that is asking a question. if (confirmDrop) { if (k === "Escape" || k === "Backspace") { e.preventDefault(); // Backing out abandons the DROP, not just the confirmation. Clearing // only the confirm would return to the reason list -- still mid-drop, // one keystroke from doing the thing that was just declined. setConfirmDrop(null); setPending(null); } return; } if (pending) { if (k === "Escape" || k === "Backspace") { e.preventDefault(); setPending(null); return; } if (/^[0-9]$/.test(k)) { e.preventDefault(); void commit(pending.verdict, Number(k)); return; } return; } const list = choicesFor(mode); // Bind by the choice's OWN key, never by its position in the list. The // row is ordered to read left-to-right as the arrows do, and when that // order changed, position-indexing silently swapped → and ← -- so the // key that means "good" committed "not an um". One property now drives // both what is drawn and what the key does, so they cannot disagree. const byArrow = (arrow: string) => list.find((c) => c.key === arrow); const arrow = { ArrowRight: "→", ArrowDown: "↓", ArrowUp: "↑", ArrowLeft: "←" }[k]; if (arrow) { const c = byArrow(arrow); if (c) { e.preventDefault(); choose(c); } } else if (k === "7" || k === "8") { // The side, chosen at the moment of judgement rather than inferred // afterwards -- in sort this parks to unclean already tagged. if (mode !== "keeps") { e.preventDefault(); void commit(2, Number(k)); } } else if (k === " ") { e.preventDefault(); play(sel.from, sel.to); } else if (k === "w" || k === "W") { e.preventDefault(); setWordOpen(true); } else if (k === "e") { e.preventDefault(); const g = proposeExpand(); if (g) play(g.from, g.to); } else if (k === "E") { e.preventDefault(); applyExpand(true); } else if (k === "m" || k === "M") { e.preventDefault(); setMoreHere((v) => !v); } else if (k === "r" || k === "R") { e.preventDefault(); undoExpand(); } else if (k === ",") { e.preventDefault(); resetSide("start"); } else if (k === ".") { e.preventDefault(); resetSide("end"); } else if (k === "a" || k === "A") { // `a`, as the static page had it -- the muscle memory is older than // this tool and there is no reason to make it relearn a key. e.preventDefault(); playView(); } else if (k === "[") { e.preventDefault(); widen("start"); } else if (k === "]") { e.preventDefault(); widen("end"); } else if (k === "-" || k === "_") { e.preventDefault(); setSilFrac((s) => Math.max(0.01, +(s - (s > 0.1 ? 0.02 : 0.01)).toFixed(3))); } else if (k === "=" || k === "+") { e.preventDefault(); setSilFrac((s) => Math.min(0.4, +(s + (s >= 0.1 ? 0.02 : 0.01)).toFixed(3))); } else if (k === "Backspace") { e.preventDefault(); // Take back the captures from THIS window first, one press each, and // only leave the card once there are none. Stepping away while a // half-finished multi-capture is on screen is almost never what was // meant, and the captures are the thing most recently done. if (captured.length) void undoCapture(); else setIdx((i) => Math.max(0, i - 1)); } else if (k === "ArrowRight" && e.shiftKey) { e.preventDefault(); setIdx((i) => Math.min(clips.length - 1, i + 1)); } }; window.addEventListener("keydown", onKey); return () => window.removeEventListener("keydown", onKey); }, [ mode, pending, wordOpen, sel, detail, choose, commit, play, proposeExpand, confirmDrop, captured.length, undoCapture, applyExpand, undoExpand, resetSide, playView, widen, clips.length, ]); const buildNames = useMemo(() => [...new Set(uses.map((u) => u.build))], [uses]); const remaining = clips.length - idx; const viewSpan = view.to - view.from; const selMs = Math.round((sel.to - sel.from) * 1000); const canWidenLeft = view.from > 0.001; const canWidenRight = !detail?.sourceDuration || view.to < detail.sourceDuration - 0.001; const header = useMemo( () => (
{clips.length} on this page of {total} · {orderNote}
palette {counts.accepted} unclean {counts.unclean} dirty {counts.impure} rejected {counts.rejected}
), [mode, clips.length, total, orderNote, counts], ); if (!clips.length) { return (
{header}

Nothing left in the {mode} pile.

{mode === "salvage" ? "Park some clips as unclean in sort, and they show up here." : mode === "verify" ? "Run find-risky.mjs to flag accepted clips worth a second listen." : "Try another mode."}

); } return (
{header} {/* overflow:auto + align-items:start, so a tall card SCROLLS rather than being centred and clipped at both ends. */}
{/* The rail. A permanent instrument, and the one thing that answers "is this pitch normal for this corpus?" -- a question no number on a card can answer on its own. Hidden below lg, where it rotates flat above the card instead. */}
{ setWord(w); setWordCustom(custom); setWordInferred(false); setWordTouched(true); if (custom) { setVocab((v) => ({ ...v, custom: [...v.custom, w] })); void fetch("/api/vocab", { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify({ word: w }), }); } }} />
{clip?.key}
{clip?.title}
{clip?.date} · {clip?.f0 || "?"} Hz · {remaining} left
{detail?.sourceMissing ? (
wav48/{clip?.video}.wav is missing — this is the only case that genuinely needs more context. The edge will be recorded for a later fetch pass rather than the window being silently truncated.
) : null} {/* The face, for the question the waveform cannot answer: whether this is Jer or the guest sitting next to him. Muted and driven by the same play() as the audio. */} {detail && !detail.sourceMissing && !videoGone ? (
) : null} { setSel(s); setDragged(true); setAutoState((a) => ({ ...a, auto: false })); if (!dragging) play(s.from, s.to); }} onReachEdge={widen} />
selection {selMs} ms ·{" "} {sel.from.toFixed(2)}–{sel.to.toFixed(2)}s view {viewSpan.toFixed(1)}s of{" "} {detail?.sourceDuration ? `${detail.sourceDuration.toFixed(0)}s` : "?"} {captured.length ? ( {captured.length} taken from this window ) : null} un-expand silence {(silFrac * 100).toFixed(0)}% {autoState.auto ? ( auto-expanded +{autoState.dl}/+{autoState.dr} ms · R undoes ) : autoState.off ? ( auto-expand undone ) : null} {detail?.stripMs ? ( stripped {detail.stripMs} ms — {detail.stripWhy} ) : null} {detail?.wordIn ? ( word inside: {detail.wordIn} ) : null} {detail?.risk ? ( flagged: {detail.risk} ) : null} {/* MEASUREMENT, in the measurement colour. None of these is a verdict, and none of them wears a verdict's hue unless it is reporting a threshold that has actually been crossed. */} {evidence?.flatness != null ? ( noise {evidence.flatness.toFixed(3)} {evidence.hissy ? " — hissy" : ""} ) : null} {evidence?.wordsInside.length ? ( words in window: “{evidence.wordsInside.join(" ")}” ) : null} {evidence?.source.clips ? ( source {Math.round(evidence.source.lo)}–{Math.round(evidence.source.hi)} Hz ·{" "} {evidence.source.spread.toFixed(1)} st over {evidence.source.clips} {evidence.source.wide ? " — two voices?" : ""} ) : null} {wordInferred && word ? ( word from ASR: “{word}” ) : null} {uses.length ? ( `${useName(u)} · ${u.voice} @ ${u.at.toFixed(2)}s`).join("\n")} className="num rounded border border-[var(--color-good)] px-1.5 text-[var(--color-good)]" > used in {buildNames.length} {buildNames.length === 1 ? "build" : "builds"} {uses.length > buildNames.length ? ` · ${uses.length} placements` : ""} ) : null} {evidence?.source.suspect ? ( flagged source ) : null}
{confirmDrop ? (
This clip is sounding in {buildNames.length} finished{" "} {buildNames.length === 1 ? "build" : "builds"}.
{uses .slice(0, 6) .map((u) => `${useName(u)} · ${u.voice} @ ${u.at.toFixed(2)}s`) .join(" ")} {uses.length > 6 ? ` +${uses.length - 6} more` : ""}

Dropping it removes it from the palette, so the next arrange cannot pick it and those renders stop being reproducible from the palette that made them. The rendered files are not touched.

) : null}
pending && void commit(pending.verdict, code, text)} onBack={() => setPending(null)} freeText={freeText} onFreeText={setFreeText} ready={!!detail} />
space play · A whole window · W word ·{" "} E hear expand · shift+E apply ·{" "} M more in this window · R undo both ·{" "} ,/. un-expand one side · [/] widen ·{" "} −/= silence · ⌫ undo a capture, then back {mode !== "keeps" ? ( <> {" "} · 7 “and um” · 8 “um and” ) : null} {busy ? "writing…" : note}
); } function labelFor(mode: Mode, verdict: number): string { if (mode === "keeps") { return { 0: "not a keep", 1: "leading", 2: "trailing", 3: "both sides" }[verdict] ?? "?"; } const c = choicesFor(mode).find((x) => x.verdict === verdict); return c?.label ?? String(verdict); }