commit 839ec457c059cc6b20891b240f076ede3569372d
parent d853116d6040a12650b3d4ad6ef7c711e151925c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 19 Sep 2026 03:28:13 -0400
clip bench: read the peeked cues as a transcript, not as a rail
The rail is a timeline. It squeezes every cue into a column a few characters
wide, which is right for dragging an edge against and useless for reading --
and reading is the entire reason for peeking past the cache. So the same cues
also run under it as a transcript, in the layout the export site already uses
(common/components/TranscriptModal.tsx): a monospace m:ss column, the text
beside it, the active line lit. Copied rather than imported -- that component
comes with PlayerProvider, radix, tanstack-virtual and lucide, none of which
umtool has -- but the shape is deliberately the same, because somebody reading
a transcript in this repo should not have to learn a second one.
What the pane adds over the modal is what this bench knows: the selection is
lit, the playing line is marked and followed (only while playing -- scrolling
the words out from under somebody reading ahead would be worse than not
following at all), lines inside another clip's window carry that clip's id,
and the cache edge is one labelled rule instead of a shade of grey to squint
at. Clicking a line inside the cache plays from it; clicking one outside
fetches to it, one side only.
The player and the pane share the leftover height 3:2 rather than either
taking fixed pixels: a first pass at a fixed 144px pane left a 133px video on
a 768-tall screen, which is not a player.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
3 files changed, 168 insertions(+), 2 deletions(-)
diff --git a/umtool/components/projects/ClipBench.tsx b/umtool/components/projects/ClipBench.tsx
@@ -311,6 +311,7 @@ export default function ClipBench({ data }: { data: ClipBenchData }) {
// `x` answers "no" by putting the cursor in the note, which is the answer.
const correctionBox = useRef<HTMLTextAreaElement | null>(null);
const segVideo = useRef<HTMLVideoElement | null>(null);
+ const transcriptBox = useRef<HTMLDivElement | null>(null);
// One auto-audition per clip, keyed by id: `canplay` fires again after a
// seek, and a clip that replays itself every time you drag is unusable.
const autoPlayed = useRef<string | null>(null);
@@ -338,6 +339,17 @@ export default function ClipBench({ data }: { data: ClipBenchData }) {
};
}, [cached, data.project, clip.id]);
+ // Follow the playhead in the reading pane, and ONLY while something is
+ // playing: scrolling the words out from under somebody who is reading ahead
+ // would be worse than not following at all.
+ useEffect(() => {
+ if (playhead == null) return;
+ const box = transcriptBox.current;
+ if (!box) return;
+ const row = box.querySelector(`[data-tcue="${cues.find((c) => playhead >= c.start && playhead < c.end)?.start ?? -1}"]`);
+ if (row) row.scrollIntoView({ block: "nearest" });
+ }, [playhead, cues]);
+
// ---- reading ahead --------------------------------------------------------
//
// The rail used to stop where the CACHE stops, so "should I fetch more" could
@@ -1084,12 +1096,16 @@ export default function ClipBench({ data }: { data: ClipBenchData }) {
src={`/api/report/raw?project=${encodeURIComponent(data.project)}&clip=${encodeURIComponent(clip.id)}&file=${encodeURIComponent(cached.name)}`}
// object-contain, because the height is now whatever is left
// rather than whatever 16:9 asks for.
- className="aspect-video w-full rounded border border-[var(--color-line)] bg-black object-contain lg:aspect-auto lg:min-h-0 lg:flex-1"
+ // The picture and the words SHARE what is left, 3:2, rather
+ // than either taking a fixed number of pixels: a laptop and a
+ // 1080p screen have very different amounts to give away, and a
+ // 130px player is not a player.
+ className="aspect-video w-full rounded border border-[var(--color-line)] bg-black object-contain lg:aspect-auto lg:min-h-0 lg:flex-[3_1_0%]"
preload="metadata"
controls
/>
) : (
- <div className="flex aspect-video w-full items-center justify-center rounded border border-dashed border-[var(--color-line)] text-[12px] text-[var(--color-dim)] lg:aspect-auto lg:min-h-0 lg:flex-1">
+ <div className="flex aspect-video w-full items-center justify-center rounded border border-dashed border-[var(--color-line)] text-[12px] text-[var(--color-dim)] lg:aspect-auto lg:min-h-0 lg:flex-[3_1_0%]">
nothing fetched for this clip yet
</div>
)}
@@ -1437,6 +1453,101 @@ export default function ClipBench({ data }: { data: ClipBenchData }) {
);
})}
</div>
+ {/* ---- THE TRANSCRIPT, as something you can actually read ----
+ The rail is a timeline: it squeezes every cue into a column a few
+ characters wide, which is right for dragging an edge against and
+ useless for reading. The words are why anybody peeks ahead, so
+ they also run here in reading order -- the export site's
+ transcript viewer's layout (a monospace m:ss column, the text
+ beside it, the active line lit), because somebody reading a
+ transcript in this repo should not have to learn a second one. */}
+ <div
+ data-transcript=""
+ ref={transcriptBox}
+ className="h-48 min-h-0 overflow-y-auto rounded border border-[var(--color-line)] bg-[var(--color-panel-2)] lg:h-auto lg:flex-[2_1_0%]"
+ >
+ <ul>
+ {railCues.map((c, i) => {
+ const outside = c.start < view.from - 0.02 || c.end > view.to + 0.02;
+ const prev = railCues[i - 1];
+ const prevOutside = prev
+ ? prev.start < view.from - 0.02 || prev.end > view.to + 0.02
+ : outside;
+ // The cache edge, drawn once, where the material stops being
+ // on disk. Reading past it is free; playing past it is not.
+ const rule = i > 0 && prevOutside !== outside;
+ const inSel = c.end > sel.from && c.start < sel.to;
+ const playing = playhead != null && playhead >= c.start && playhead < c.end;
+ const owner = siblingAt(c);
+ return (
+ <li key={`t-${c.start}-${c.end}`}>
+ {rule && (
+ <div
+ data-cache-rule={outside ? "end" : "start"}
+ className="flex items-center gap-2 border-t border-dashed border-[var(--color-meter)] px-2 py-0.5 text-[10px] uppercase tracking-wider text-[var(--color-meter)]"
+ >
+ {outside ? "not fetched — reading only" : "cached from here"}
+ </div>
+ )}
+ <button
+ type="button"
+ data-tcue={c.start}
+ data-peek={outside ? "1" : "0"}
+ data-in-sibling={owner ? owner.id : ""}
+ title={
+ outside
+ ? `not fetched — click to fetch to ${padLabelForCue(c)}`
+ : "click to play from here"
+ }
+ onClick={() => {
+ if (outside) {
+ if (!atMaxPad) void fetchMore(padsForCue(c));
+ return;
+ }
+ play(c.start, Math.min(c.end + 4, view.to));
+ }}
+ className={`flex w-full gap-2 px-2 py-1 text-left hover:bg-[color-mix(in_srgb,var(--color-sel)_10%,transparent)] ${
+ inSel ? "bg-[color-mix(in_srgb,var(--color-sel)_14%,transparent)]" : ""
+ } ${playing ? "border-l-2 border-[var(--color-meter)]" : "border-l-2 border-transparent"}`}
+ >
+ <span
+ className={`num w-10 shrink-0 font-mono text-[10px] ${
+ playing ? "text-[var(--color-meter)]" : "text-[var(--color-dim)]"
+ }`}
+ >
+ {clock(c.start)}
+ </span>
+ <span
+ className={`min-w-0 text-[12px] leading-snug ${
+ outside
+ ? "text-[var(--color-dim)] opacity-70"
+ : playing
+ ? "text-[var(--color-text)]"
+ : "text-[var(--color-text)]"
+ }`}
+ >
+ {owner && (
+ <span
+ className="mr-1.5 rounded border border-[var(--color-sel)] px-1 text-[10px] text-[var(--color-sel)]"
+ title={`already in the cut as ${owner.id} (${owner.where})`}
+ >
+ {owner.id}
+ </span>
+ )}
+ {c.endsSentence && (
+ <span className="mr-1 text-[var(--color-meter)]" aria-hidden>
+ ¶
+ </span>
+ )}
+ {c.text}
+ </span>
+ </button>
+ </li>
+ );
+ })}
+ </ul>
+ </div>
+
<div className="micro flex flex-wrap items-center gap-2">
<span>
what is being said — a marked cue closes a sentence; click one to snap an edge
diff --git a/umtool/docs/clip-bench.md b/umtool/docs/clip-bench.md
@@ -90,6 +90,17 @@ the one it replaces. A side that cannot grow says which reason it is — *"the
recording starts here"* or the pad cap — rather than offering a press that
cannot help. The message after a fetch names the side that moved.
+**The words run as words, under the rail.** The rail is a timeline — it
+squeezes each cue into a column a few characters wide, which is right for
+dragging an edge against and useless for reading, and reading is why anybody
+peeks ahead. So the same cues also run in a transcript pane below it, in the
+export site's own layout (`common/components/TranscriptModal.tsx`): a monospace
+`m:ss` column, the text beside it, the selection lit, the playing line marked
+and followed, sibling clips' lines badged with their id, and a single rule
+where the cache ends rather than a shade of grey to squint at. Clicking a line
+inside the cache plays from it; clicking one outside fetches to it. The pane
+and the player share the leftover height 2:3, so neither is a strip.
+
**Read ahead before you pay for it.** The cue rail runs ±60 s past the cached
file (one request to `/api/report/cues`, text from the archive and cheap beside
media). Everything outside the cache is dimmed, both cache edges are marked, and
diff --git a/umtool/e2e/clip-bench.spec.ts b/umtool/e2e/clip-bench.spec.ts
@@ -805,6 +805,50 @@ test("the rail reads past the cache, dimmed, and a dimmed cue fetches to itself"
// clip.
// ---------------------------------------------------------------------------
+test("the peeked cues are readable, in order, with the cache edge marked", async ({ page }) => {
+ await page.goto(bench("c01"));
+ const rows = page.locator("[data-tcue]");
+ await expect(rows.first()).toBeVisible();
+
+ // Reading order, which is the whole point: the rail squeezes the same cues
+ // into columns a few characters wide, and that is unreadable by design --
+ // it is a timeline.
+ const starts = await rows.evaluateAll((els) =>
+ els.map((e) => Number(e.getAttribute("data-tcue"))),
+ );
+ expect(starts.length).toBeGreaterThan(0);
+ expect(starts).toEqual([...starts].sort((a, b) => a - b));
+
+ // Both sides of the cache are here, and the boundary between them is drawn
+ // once rather than left to be inferred from a shade of grey.
+ const cached = await page.locator("[data-tcue][data-peek='0']").count();
+ const peeked = await page.locator("[data-tcue][data-peek='1']").count();
+ expect(cached).toBeGreaterThan(0);
+ expect(peeked).toBeGreaterThan(0);
+ await expect(page.locator("[data-cache-rule=end]")).toContainText("not fetched");
+});
+
+test("clicking a peeked line fetches only that side", async ({ page, request }) => {
+ const clip = (await (
+ await request.get(`/api/report/clip?project=${encodeURIComponent(PROJECT)}&clip=c01`)
+ ).json()) as { clip: { start: number }; windows: { from: number }[] };
+ const reachBefore = Number((clip.clip.start - clip.windows[0].from).toFixed(2));
+
+ await page.goto(bench("c01"));
+ let body: { padBefore?: number; padAfter?: number } = {};
+ await page.route("**/api/report/fetch", async (route) => {
+ if (route.request().method() !== "POST") return route.continue();
+ body = route.request().postDataJSON() as { padBefore?: number; padAfter?: number };
+ await route.fulfill({ status: 409, json: { error: "intercepted" } });
+ });
+
+ await page.locator("[data-tcue][data-peek='1']").first().click();
+ await expect.poll(() => body.padAfter).toBeGreaterThan(0);
+ // The lead-in is already on disk; paying for it again to hear the sentence
+ // that follows is the waste one-sided fetching exists to stop.
+ expect(body.padBefore).toBe(reachBefore);
+});
+
test("a clip lists the other clips cut from its recording", async ({ page, request }) => {
const r = await request.get(
`/api/report/clip?project=${encodeURIComponent(PROJECT)}&clip=c01`,