// The report-video kind's own reading: the manifest, what is wrong with it, and // what state the build is in. // // Plain ESM, no TypeScript and no app imports, because `umtool check` has to run // this from a terminal with no server. It is also the only place that knows the // manifest's shape -- the walk knows a marker file, the page knows a summary, // and neither parses JSON. import { readdir, readFile, stat } from "node:fs/promises"; import path from "node:path"; import { DEFAULT_VARIANT, cachedWindowsFor } from "umtool-report-to-video/build-video"; import { SHADOW_CHANNELS as SOURCES_SHADOW_CHANNELS, channelsDirFor as sourcesChannelsDirFor, corpusWindowsOf, videoDirOf, } from "umtool-report-to-video/sources"; import { rawCacheOf } from "../report/raw-cache.mjs"; import { deliverablesState, finishCommand, outDirState, leftoversOf } from "../report/storage.mjs"; import { CHANNELS_DIR, inside } from "../paths.mjs"; import { channelName, cleanTitle } from "umtool-report-to-video/attribution"; import { teaserTitle } from "umtool-report-to-video/deck"; /** * Where a build's per-entry segments live. * * They moved under `out//` when the pipeline learned to cut two * versions of the same manifest. `out/segments` is still checked, because every * other report on disk was built before that and its files are still there — * and a poster tile silently going blank is exactly the kind of regression * nothing would have reported. */ const segmentDirs = (outDir) => [ path.join(outDir, DEFAULT_VARIANT, "segments"), path.join(outDir, "segments"), ]; import { adjudicationGaps, ledgerTotals, unadjudicatedOf } from "umtool-report-to-video/ledger-totals"; import { widen } from "umtool-report-to-video/resolve-windows"; // widen() and the cache's window naming are IMPORTED, never reimplemented. The // bench's "extend to sentence end" has to be the same function the CLI runs, or // the UI and `resolve-windows --write` will disagree about where a clip ends -- // and the CLI is the one that wins, silently, on the next build. export { widen }; export const MANIFEST_NAME = "video.manifest.json"; // ONE definition: lib/paths.mjs's CHANNELS_DIR (the env var, else // TRANSCRIPTS_DIR/channels, else the checkout's own transcripts/channels). The // bench's read root and this reader cannot disagree about which corpus it is. export const GLOBAL_CHANNELS_DIR = () => CHANNELS_DIR; /** The conventional name make-shadow-channels.sh builds. */ export const SHADOW_CHANNELS = SOURCES_SHADOW_CHANNELS; // --------------------------------------------------------------------------- // Which channels directory THIS project reads. // // Not always the global one, and assuming it was produced three confident, // wrong "the build dies here" findings on the first run against real data. // quartering-flagging-takedowns cites three deleted YouTube uploads and cuts // them from live Rumble mirrors the corpus has no entry for; rather than write // into transcripts/channels/ (which would perturb the export build's change // detection) it builds a SHADOW tree that symlinks every real channel and adds // just those three. Its documented build command sets CHANNELS_DIR to it. // // So the project gets to say. `provenance.channelsDir` is the explicit form and // is resolved relative to the project; `.shadow-channels/` is the convention, // detected because it already exists and nothing had to change on disk for it // to work. Neither is a guess: both are things the project itself wrote down. // --------------------------------------------------------------------------- // // The rule itself is report-to-video/sources.mjs's, so the render looks for // local media in the tree this reads. export function channelsDirFor(dir, manifest, { shadowExists = false } = {}) { return sourcesChannelsDirFor(dir, manifest, { shadowExists, fallback: GLOBAL_CHANNELS_DIR() }); } export const hasShadowChannels = (dir) => stat(path.join(dir, SHADOW_CHANNELS)).then((s) => s.isDirectory(), () => false); /** The same regex resolve-windows.mjs uses. Imported there, duplicated nowhere. */ export const ENDS_SENTENCE = /[.!?]["'”’)\]]*\s*$/; export const IS_FILLER = /^\s*(\[[^\]]*\]|>>|♪|—|-)*\s*$/; // A QR built from any of these resolves to nothing on somebody else's phone. // Both of the real defects this catches had already shipped: one manifest has no // siteOrigin at all (19 codes reading `undefined/?v=…`) and one has localhost. const DEAD_ORIGIN = /^https?:\/\/(localhost|127\.|0\.0\.0\.0|\[::1\]|10\.|192\.168\.|172\.(1[6-9]|2\d|3[01])\.)/i; export const isDeadOrigin = (o) => !o || typeof o !== "string" || DEAD_ORIGIN.test(o); const stat0 = (p) => stat(p).then((s) => s, () => null); export const manifestPath = (dir) => path.join(dir, MANIFEST_NAME); export async function readManifest(dir) { try { return JSON.parse(await readFile(manifestPath(dir), "utf8")); } catch { return null; } } /** Clips only, in timeline order. Array order IS the cut; nothing sorts. */ export const clipsOf = (m) => (m?.timeline ?? []).filter((e) => e?.type === "clip"); export const cardsOf = (m) => (m?.timeline ?? []).filter((e) => e?.type === "card"); export const channelFor = (m, e) => e.channel ?? m?.provenance?.channelSlug ?? null; /** * Where a clip's QR points. THE SAME RULE build-video.mjs's qrForEntry() uses: * an explicit per-clip `citeUrl` wins (the Rumble mirror case -- a local slug * is not the id the site serves), else the viewer moment URL is derived. */ export const derivedCiteUrl = (m, e) => `${m?.provenance?.siteOrigin ?? ""}/?v=${encodeURIComponent(`${channelFor(m, e)}/${e.video}`)}&t=${Math.floor(e.start ?? 0)}`; export const citeUrlFor = (m, e) => e.citeUrl ?? derivedCiteUrl(m, e); /** * The clips whose `correction` says the REPORT got something wrong. * * Not a render input and not a decision: a correction is a message to the next * report pass -- wrong speaker, wrong addressee, wrong date -- written while * somebody was watching the clip and could see it. The page and `umtool * corrections` read it from here so a list that is copied into a prompt and one * that is read on screen cannot drift apart. */ export function correctionsOf(m) { return clipsOf(m) .filter((e) => String(e.correction ?? "").trim()) .map((e) => ({ id: e.id, channel: channelFor(m, e), video: e.video, at: e.cite ?? e.start ?? 0, href: citeUrlFor(m, e), text: String(e.correction).trim(), // What the walk said about the clip this text is attached to. A note on a // CONFIRMED clip is not a defect in the report, and a reader who cannot // tell the two apart will go and "fix" a clip nobody complained about. verdict: clipVerdict(e), })); } /** * Has anybody looked at this clip, and did they agree with the description? * * "confirmed" -> the description is accurate. A `correction` here is a NOTE: * why a good clip's window moved, a caveat for the writers. * "incorrect" -> it is not, and the `correction` says how. The writer refuses * one without the other. * absent -> nobody has been here yet. * * LEGACY: a clip carrying a `correction` and no `verdict` was written by the * bench's `x` before `incorrect` existed -- two of them are in a live manifest * -- and it means exactly what `incorrect` means. Read that way rather than * migrated: a rewrite of somebody's manifest to teach this file a value it can * already infer is a worse trade than one branch. */ export const clipVerdict = (e) => { const v = e?.verdict; if (v === "confirmed" || v === "incorrect") return v; return String(e?.correction ?? "").trim() ? "incorrect" : "unreviewed"; }; /** Does this clip carry a written note, whatever the verdict says? */ export const clipHasNote = (e) => !!String(e?.correction ?? "").trim(); /** * The walk's coverage: how much of the cut has been looked at, and what is left. * * One definition, read by the bench header, the project page and `umtool * corrections` -- the same reason correctionsOf() is one definition. */ export function reviewOf(m) { const rows = clipsOf(m).map((e) => ({ id: e.id, verdict: clipVerdict(e), note: clipHasNote(e), })); const of = (v) => rows.filter((r) => r.verdict === v); return { total: rows.length, clips: rows, confirmed: of("confirmed").length, /** Confirmed AND annotated: a note on a good clip is not a defect. */ confirmedWithNote: of("confirmed").filter((r) => r.note).length, incorrect: of("incorrect").length, unreviewed: of("unreviewed").length, unreviewedIds: of("unreviewed").map((r) => r.id), /** Confirmed or incorrect: somebody has been here and said something. */ reviewed: rows.length - of("unreviewed").length, }; } /** * What the WALK will actually visit, and how much of it can be visited today. * * The walk is somebody at a desk answering one question per clip, so it goes to * the clips that still need an answer AND can be watched end to end right now: * * needing -> nobody has judged it (`clipVerdict` is "unreviewed"). The same * rule reviewOf() counts by, so "12 of 19 reviewed" and "ready 3 * of 7" cannot disagree about the same clip. * ready -> needing AND `fetched`, which readClipDetail computes from the * clips-raw cache: a file holding the clip's own window. * * Walking onto a clip with nothing to play is a dead end -- there is nothing to * judge and the only move is to press `n` again -- and walking back onto one * already answered is the same round trip a walk exists to remove. `readyIds` * is in TIMELINE ORDER, because the cut's order is the order you watch it in. * * @param {{id: string, kind?: string, fetched?: boolean}[]} entries readClipDetail's entries */ /** * The clips the walk is WAITING ON: needing judgement, with nothing cached that * holds them end to end. Exactly the gap between `ready` and `needing` above, * as a list — so "ready 3 of 7" and the button that closes it cannot disagree * about which four clips those are. * * In timeline order, because that is the order somebody would watch them in * and therefore the order they are worth having. * * @param {{id: string, kind?: string, fetched?: boolean}[]} entries */ export function unfetchedNeedingJudgement(entries) { return (entries ?? []) .filter((e) => (e.kind ?? e.type) === "clip") .filter((e) => clipVerdict(e) === "unreviewed" && !e.fetched) .map((e) => e.id); } export function walkReadiness(entries) { const clips = (entries ?? []).filter((e) => (e.kind ?? e.type) === "clip"); const needing = clips.filter((e) => clipVerdict(e) === "unreviewed"); const ready = needing.filter((e) => !!e.fetched); return { needing: needing.length, ready: ready.length, needingIds: needing.map((e) => e.id), readyIds: ready.map((e) => e.id), }; } /** * The OTHER clips in the cut that come from this same recording. * * "Does this clip need more context, or is the context already coming up as * another clip?" is a question the manifest can answer and nothing in the * bench could: a widened window that runs into the next clip from the same * stream is a duplicate, and twenty seconds of missing context that the cut * never picks up is a hole. Both look identical from inside one clip. * * Matched on the VIDEO, and on the channel when the entries carry one -- two * recordings of the same event on different channels are different clocks, and * saying "c07 covers this" about a different upload would be worse than saying * nothing. Position is the distance in the CUT (array order is the cut), and * the gap is measured against this clip's own edges, which is what somebody * dragging one is actually asking about. * * @param {any} m manifest * @param {string} clipId */ export function siblingsOf(m, clipId) { const clips = clipsOf(m); const i = clips.findIndex((e) => e.id === clipId); if (i < 0) return []; const me = clips[i]; const where = (steps) => { if (steps === 1) return "next in cut"; if (steps === -1) return "previous in cut"; return steps > 0 ? `${steps} later` : `${-steps} earlier`; }; return clips .map((e, j) => ({ e, j })) .filter( ({ e, j }) => j !== i && e.video === me.video && (e.channel ?? null) === (me.channel ?? null), ) .map(({ e, j }) => { const after = e.start >= me.end; const before = e.end <= me.start; const quote = String(e.quote ?? "").trim(); return { id: e.id, start: e.start, end: e.end, cite: e.cite ?? null, quote: quote.length > 80 ? `${quote.slice(0, 80)}…` : quote, note: e.note ?? null, /** Distance in the cut: +1 is the next clip, -2 is two clips back. */ steps: j - i, where: where(j - i), /** Where it sits on the SOURCE clock relative to this clip's window. */ side: after ? "after" : before ? "before" : "overlap", /** Seconds of source between the two windows; null when they overlap. */ gap: after ? Number((e.start - me.end).toFixed(2)) : before ? Number((me.start - e.end).toFixed(2)) : null, }; }) .sort((a, b) => a.start - b.start); } /** The recorded preflight, and when it ran. Never a live probe. */ export async function readAvailability(dir) { const file = path.join(dir, "out", "availability.json"); const [st, text] = await Promise.all([stat0(file), readFile(file, "utf8").catch(() => null)]); if (!st || text === null) return null; try { const doc = JSON.parse(text); return { ...doc, checkedAtMs: Date.parse(doc.checkedAt ?? "") || st.mtimeMs, file }; } catch { return null; } } /** * The windows OUTSIDE the project that also hold this manifest's videos, as * `[{ video, windows }]` for rawCacheOf's `extraWindows`. * * Listed by the render's own module (report-to-video/sources.mjs), so the * bench's "fetched" and the build's "local" are one set: the editor's window * cache in channels//data//clips/ -- managed, polite, provenanced, * and REUSABLE, which is the whole point: a window one report paid for is a * window the next one does not -- and a saved whole source, reached through * the store's pointer (`full: true` puts the recording there, not in clips/) * or still in the video dir. A dangling link -- a drive not mounted -- is no * window, never an error. One entry per distinct (channel, video) the * timeline cites. * * A WHOLE CONTAINER'S SPAN COMES FROM THE CUES, not from an ffprobe as the * render's does: the cue doc is the archive's own record of how long the * recording is, it is memoised against mtime (see below), and it is asked for * only when a container is actually there -- which keeps the index load as * cheap as it was. No duration is no window rather than a guess. */ export const clipWindowDirs = async (m, channelsDir) => { const root = channelsDir ?? GLOBAL_CHANNELS_DIR(); const seen = new Set(); const out = []; for (const e of clipsOf(m)) { const chan = channelFor(m, e); if (!chan || !e.video) continue; const key = `${chan}/${e.video}`; if (seen.has(key)) continue; seen.add(key); const cueFile = path.join(videoDirOf(root, chan, e.video), "transcript.cues.json"); const probe = async () => { const duration = Number((await readCues(cueFile))?.duration); return Number.isFinite(duration) && duration > 0 ? { duration } : null; }; const windows = await corpusWindowsOf({ video: e.video, slug: chan, channelsDir: root, probe }); if (windows.length) out.push({ video: e.video, windows }); } return out; }; export const cuePathFor = (m, e, channelsDir) => { const chan = channelFor(m, e); if (!chan) return null; return path.join(channelsDir ?? GLOBAL_CHANNELS_DIR(), chan, "data", e.video, "transcript.cues.json"); }; // --------------------------------------------------------------------------- // Cue files are the expensive input: a nine-hour stream's cues run to megabytes, // and six reports citing thirteen videos each would be a hundred-odd megabytes of // JSON on every index load. So the DERIVED answers are memoised against the // file's own mtime and size -- a cue file is an archive artefact and does not // change under us, so a hit is permanent in practice. // --------------------------------------------------------------------------- const cueMemo = new Map(); export async function readCues(file) { const st = await stat0(file); if (!st) return null; const key = `${file}|${Math.round(st.mtimeMs)}|${st.size}`; const hit = cueMemo.get(key); if (hit) return hit; let doc; try { doc = JSON.parse(await readFile(file, "utf8")); } catch { return null; } const cues = doc.cues ?? []; const value = { cues, title: doc.title, uploadDate: doc.uploadDate, webpageUrl: doc.webpageUrl, duration: doc.duration, // The uploader's DISPLAY name, which heads the burned-in header. Without it // here the bench previewed the raw slug while the renderer drew "Destiny" -- // a preview promising a line the renderer would not draw, which is the one // thing attribution.mjs exists to prevent. channel: doc.channel, // What fraction of cues close a sentence. Below ~10% the upload's ASR // carries no punctuation worth the name, and sentence-widening cannot help // -- which is a thing to SAY, not a thing to let somebody rediscover. punctuationRate: cues.length === 0 ? 0 : cues.filter((c) => ENDS_SENTENCE.test(c.text ?? "")).length / cues.length, }; cueMemo.set(key, value); return value; } /** The cue whose span contains t, else the nearest on the right. */ export function cueAt(cues, t, which = "start") { const EPS = 0.02; if (!cues.length) return null; if (which === "end") { const j = cues.findIndex((c) => c.end >= t - EPS); return cues[j < 0 ? cues.length - 1 : j]; } const i = cues.findIndex((c) => c.end > t); return cues[i < 0 ? cues.length - 1 : i]; } // --------------------------------------------------------------------------- // The build's own state, read from the output directory. // // Deliberately NOT a readdir of out/ -- it holds 43 to 63 files per project and // the index would pay for all of them. One stat for the deliverable, one readdir // of clips-raw (a few dozen names) to tell "windows written" from "clips // fetched", and nothing else. // --------------------------------------------------------------------------- export async function buildStateOf(dir, manifest) { const outDir = path.join(dir, "out"); const slug = manifest?.slug ?? path.basename(dir); const finalPath = path.join(outDir, `${slug}.mp4`); // The `full` cut's file, when the manifest has one. One extra stat; it is a // deliverable too and the dashboard's byte count would lie without it. const fullPath = path.join(outDir, `${slug}-full.mp4`); const [fin, man, full] = await Promise.all([stat0(finalPath), stat0(manifestPath(dir)), stat0(fullPath)]); let rawCount = 0; let segCount = 0; let firstSeg = null; let firstRaw = null; let segRel = path.posix.join("out", "segments"); if (!fin) { const dirs = segmentDirs(outDir); const [raws, ...segLists] = await Promise.all([ readdir(path.join(outDir, "clips-raw")).catch(() => []), ...dirs.map((d) => readdir(d).catch(() => [])), ]); const which = segLists.findIndex((l) => l.length); const segs = which < 0 ? [] : segLists[which]; segRel = which === 0 ? path.posix.join("out", DEFAULT_VARIANT, "segments") : path.posix.join("out", "segments"); const rawMp4 = raws.filter((n) => n.endsWith(".mp4")).sort(); const segMp4 = segs.filter((n) => n.endsWith(".mp4")).sort(); rawCount = rawMp4.length; segCount = segMp4.length; firstSeg = segMp4[0] ?? null; firstRaw = rawMp4[0] ?? null; } const stale = !!(fin && man && fin.mtimeMs < man.mtimeMs); return { outDir, slug, finalPath, built: !!fin, stale, finalSize: fin?.size ?? 0, finalMtimeMs: fin?.mtimeMs ?? 0, fullPath, fullSize: full?.size ?? 0, fullMtimeMs: full?.mtimeMs ?? 0, manifestMtimeMs: man?.mtimeMs ?? 0, rawCount, segCount, firstSeg, firstRaw, segRel, }; } /** The signature the summary and the decisions are cached against. */ export async function reportSignature(dir) { const [man, out, avail, rev] = await Promise.all([ stat0(manifestPath(dir)), stat0(path.join(dir, "out")), // Rewritten IN PLACE by every preflight, which leaves out/'s own mtime // where it was -- so it has to be signed on its own or a re-check would // never reach the index. stat0(path.join(dir, "out", "availability.json")), stat0(path.join(dir, "revisions")), ]); return [ Math.round(man?.mtimeMs ?? 0), man?.size ?? 0, Math.round(out?.mtimeMs ?? 0), Math.round(avail?.mtimeMs ?? 0), Math.round(rev?.mtimeMs ?? 0), ].join(":"); } const fmtDur = (s) => { const t = Math.round(s); const m = Math.floor(t / 60); return m >= 60 ? `${Math.floor(m / 60)}h${String(m % 60).padStart(2, "0")}m` : `${m}m${String(t % 60).padStart(2, "0")}s`; }; // --------------------------------------------------------------------------- // Sources: one row per distinct (channel, video) a report draws on. // // This is where the long-form problem lives. Eight to eighteen sources per // video, across channels and platforms; a transcript that stops before the clip // it is cited for; a Rumble id that is not the id the site serves; sources that // go dead after the manifest is written. Every column here is READ from // something already on disk -- the cue file, out/availability.json, the // manifest -- and nothing is probed. // // Cue coverage is the one that had no home at all: gout's c03 (861–897 s) sat // on a transcript whose last cue ended at 880 s, and the only record of it was // a paragraph in provenance.transcriptGapNote. // --------------------------------------------------------------------------- /** * @returns {Promise<{ rows: Array<{ * key: string, channel: string|null, video: string, clips: string[], * cues: "present"|"missing"|"no-punctuation", cueFile: string|null, * cuesEnd: number|null, latestClipEnd: number, coverageGap: null | { clip: string, needs: number, cuesEnd: number }, * availability: null | { state: string, ok: boolean, checkedAtMs: number, title: string|null, url: string|null, error: string|null }, * cite: { derived: string, override: string|null, differs: boolean }, * title: string|null, * }>, checkedAtMs: number|null, channelsDir: string }>} */ export async function sourcesOf(dir, { manifest = null } = {}) { const m = manifest ?? (await readManifest(dir)); if (!m) return { rows: [], checkedAtMs: null, channelsDir: GLOBAL_CHANNELS_DIR() }; const shadowExists = await hasShadowChannels(dir); const channelsDir = channelsDirFor(dir, m, { shadowExists }); const avail = await readAvailability(dir); const availBy = new Map((avail?.sources ?? []).map((s) => [s.key, s])); const byKey = new Map(); for (const e of clipsOf(m)) { const chan = channelFor(m, e); const key = `${chan}/${e.video}`; if (!byKey.has(key)) byKey.set(key, { chan, video: e.video, clips: [] }); byKey.get(key).clips.push(e); } const rows = []; for (const [key, { chan, video, clips }] of byKey) { const cueFile = cuePathFor(m, clips[0], channelsDir); const doc = cueFile ? await readCues(cueFile) : null; const cuesEnd = doc?.cues?.length ? Number(doc.cues[doc.cues.length - 1].end) : null; const latest = clips.reduce((best, e) => (Number(e.end) > (best?.end ?? -1) ? { clip: e.id, end: Number(e.end) } : best), null); // A gap is the cue file ENDING before a clip does: the tail of the quote // has no words behind it, so the widener cannot see it and a build cuts // footage the transcript never described. Half a second of slack, because // a cue's end and a window's end are both rounded. const coverageGap = doc && cuesEnd != null && latest && latest.end > cuesEnd + 0.5 ? { clip: latest.clip, needs: latest.end, cuesEnd } : null; const a = availBy.get(key) ?? null; // The QR/cite target. "Differs" is what marks the Rumble case: every clip // in one real manifest carries a citeUrl, and only the ones that point // somewhere other than the derived moment are overrides worth flagging. const first = clips[0]; const derived = derivedCiteUrl(m, first); const override = first.citeUrl ?? null; rows.push({ key, channel: chan, video, clips: clips.map((e) => e.id), cues: !doc ? "missing" : doc.punctuationRate < 0.1 ? "no-punctuation" : "present", cueFile, cuesEnd, latestClipEnd: latest?.end ?? 0, coverageGap, availability: a ? { state: a.state ?? (a.ok ? "ok" : "unknown"), ok: !!a.ok, checkedAtMs: Date.parse(a.checkedAt ?? "") || avail.checkedAtMs, title: a.title ?? null, url: a.url ?? null, error: a.error ?? null, } : null, cite: { derived, override, differs: !!override && override !== derived }, title: doc?.title ?? a?.title ?? null, }); } return { rows, checkedAtMs: avail?.checkedAtMs ?? null, channelsDir, shadowExists }; } /** * The card. Cheap by construction: the manifest, one stat, one readdir. * * Runtime is SUMMED FROM THE WINDOWS, not probed. It is the right number anyway * -- it is what the cut will be if it is built -- and a probe per project would * put a second in front of the index. */ export async function summariseReport(ctx) { const { dir } = ctx; const m = await readManifest(dir); const clips = clipsOf(m); const cards = cardsOf(m); // Anything that is neither. The vocabulary is open, so a card that says // "19 clips · 10 cards" about a 31-entry timeline is lying by omission. const others = (m?.timeline ?? []).filter((e) => e?.type !== "clip" && e?.type !== "card"); const build = await buildStateOf(dir, m); const runtime = clips.reduce((n, e) => n + Math.max(0, (e.end ?? 0) - (e.start ?? 0)), 0); const sources = new Set(clips.map((e) => `${channelFor(m, e)}/${e.video}`)).size; const locked = clips.filter((e) => e.lock).length; // The source facts the dashboard's Sources panel and /browse read. The cue // reads this costs are memoised per file for the life of the process, and // the record is cached in the index against the signature -- so the first // load of a project pays for its cue files once, and no page does again. const src = m ? await sourcesOf(dir, { manifest: m }) : { rows: [], checkedAtMs: null }; const sourcesDead = src.rows.filter((r) => r.availability && !r.availability.ok && r.availability.state !== "no-cues").length; const sourcesMissing = src.rows.filter((r) => r.cues === "missing").length; const cueGaps = src.rows.filter((r) => r.coverageGap).length; const state = !m ? "draft" : build.stale ? "stale" : build.built ? "built" : build.segCount > 0 || build.rawCount > 0 ? "fetched" : clips.length ? "windows" : "draft"; const facts = []; if (clips.length) facts.push(`${clips.length} clip${clips.length === 1 ? "" : "s"}`); if (cards.length) facts.push(`${cards.length} card${cards.length === 1 ? "" : "s"}`); if (others.length) { const kinds = [...new Set(others.map((e) => e.type ?? "entry"))].sort(); facts.push(`${others.length} ${kinds.join("/")}`); } if (runtime > 0) facts.push(fmtDur(runtime)); if (sources) facts.push(`${sources} source${sources === 1 ? "" : "s"}`); if (locked) facts.push(`${locked} locked`); if (build.built) facts.push(`${(build.finalSize / 1e6).toFixed(0)} MB`); const flags = []; if (isDeadOrigin(m?.provenance?.siteOrigin)) flags.push("dead QR origin"); if (!m?.provenance?.channelSlug) flags.push("no channelSlug"); if (build.stale) flags.push("older than its manifest"); if (sourcesDead) flags.push(`${sourcesDead} dead source${sourcesDead === 1 ? "" : "s"}`); if (cueGaps) flags.push(`${cueGaps} cue gap${cueGaps === 1 ? "" : "s"}`); return { title: m?.title ?? ctx.name, subtitle: m?.subtitle ?? m?.generatedOn ?? null, state, newestMtimeMs: Math.max(build.manifestMtimeMs, build.finalMtimeMs, build.fullMtimeMs), facts, flags, // A card should look like the video as soon as anything of it exists. A // built SEGMENT already carries the chrome, so it is the better mid-build // poster than a raw clip; a raw clip is better than a blank tile. posterRel: build.built ? path.posix.join("out", `${build.slug}.mp4`) : build.firstSeg ? path.posix.join(build.segRel, build.firstSeg) : build.firstRaw ? path.posix.join("out", "clips-raw", build.firstRaw) : null, haystack: [ ctx.id, m?.title, m?.subtitle, m?.slug, m?.provenance?.channel, m?.provenance?.channelSlug, ...clips.map((e) => e.video), ] .filter(Boolean) .join(" ") .toLowerCase(), attrs: { clips: String(clips.length), cards: String(cards.length), other: String(others.length), sources: String(sources), locked: String(locked), // Omitted when never checked, so "never" is the attribute not being // there rather than a zero somebody has to know the meaning of. ...(src.checkedAtMs ? { "sources-checked-at": String(Math.round(src.checkedAtMs)) } : {}), ...(sourcesDead ? { "sources-dead": String(sourcesDead) } : {}), ...(sourcesMissing ? { "sources-missing": String(sourcesMissing) } : {}), ...(cueGaps ? { "cue-gaps": String(cueGaps) } : {}), ...(build.built ? { built: "1", "final-bytes": String(build.finalSize + build.fullSize) } : {}), ...(isDeadOrigin(m?.provenance?.siteOrigin) ? { "dead-origin": "1" } : {}), }, // Kept for the project page and the decisions pass, so neither re-reads. manifest: m, build, sources: src, }; } // --------------------------------------------------------------------------- // Decisions. // // Severity is EARNED. `blocking` means something downstream would LIE or die if // you acted on it: a QR that resolves nowhere, a clip whose cue file is absent // (the build dies there), a source that is gone. Everything else is `open` at // most, and a manifest with no build yet is `info` -- that is a normal state to // be in, not a decision anybody is waiting on. // --------------------------------------------------------------------------- export const REPORT_DECISION_KINDS = [ "manifest-invalid", "clip-no-cues", "clip-unfetchable", "clip-mid-sentence", "window-overlap", "no-punctuation", // The cue file ENDS before a clip does. `open`, not blocking: the build does // not die (it cuts from the audio), but the tail of that clip has no words // behind it, so nothing can say whether it ends on a sentence -- and the // fix (a recovered transcript, or a shorter window) is editorial. The // reducer reads cue files already; it does not measure media. "clip-cue-gap", // A `cutStart`/`cutEnd` that is not inside its clip's own window. BLOCKING: // the build would cut somewhere nobody reviewed, or -- with the pair half // written -- silently ignore it and render the whole extent instead. Both // are the render disagreeing with the manifest about what the clip IS. "clip-cut-outside", // A ledger entry nobody has ruled on. BLOCKING, which is earned here: both // the stated and the implied total lie if you act on an unadjudicated ledger, // and they lie quietly, in a chart, with his name on it. "claim-unadjudicated", // A fired coherence predicate. `open`, not blocking, because the decision is // EDITORIAL rather than mechanical: is the flag right, and does it belong on // screen? Neither answer stops a build. "claim-incoherent", "stale-build", "unbuilt", // `out/`, `clips/` or a `share-*/` is a link whose target is not there: the // media drive is not mounted (release 17). BLOCKING -- a build, a cut or a // batch refuses at that directory, and a reader sees "nothing built/cut". "storage-unreachable", // Where the deliverables are disagrees with `storage.deliverables`, or a // move was cut. `open`: nothing is lost, and one command settles it. "storage-mismatch", ]; export async function reportDecisions(ctx, summary) { const { id, dir } = ctx; const s = summary ?? (await summariseReport(ctx)); const m = s.manifest; const href = `/browse/${id}`; const clipHref = (cid) => `/browse/${id}/clip/${cid}`; const at = s.newestMtimeMs; const out = []; if (!m) return out; const add = (kind, target, why, severity, extra = {}) => out.push({ kind, project: id, target, why, href, severity, at, ...extra }); const shadowExists = await hasShadowChannels(dir); const channelsDir = channelsDirFor(dir, m, { shadowExists }); // A project that ships a builder for its shadow tree but has not run it is a // DIFFERENT problem from one whose sources are gone, and the fix is one // command rather than an editorial decision. const shadowBuilder = !shadowExists && (await stat(path.join(dir, "make-shadow-channels.sh")).then(() => true, () => false)); // --- the manifest itself ------------------------------------------------- const origin = m.provenance?.siteOrigin; if (isDeadOrigin(origin)) { add( "manifest-invalid", "provenance.siteOrigin", origin ? `\`${origin}\` — every QR in this cut resolves to nothing on anyone else's phone` : "missing — every QR in this cut encodes `undefined/?v=…`", "blocking", ); } if (!m.provenance?.channelSlug) { add( "manifest-invalid", "provenance.channelSlug", "missing — a clip with no `channel` of its own has no cue file to find", "blocking", ); } const clips = clipsOf(m); const seen = new Map(); for (const e of m.timeline ?? []) { if (!e?.id) continue; seen.set(e.id, (seen.get(e.id) ?? 0) + 1); } for (const [eid, n] of seen) { if (n > 1) { add( "manifest-invalid", eid, `${n} entries share the id \`${eid}\` — segments overwrite each other`, "blocking", ); } } const nodeCount = (m.timelineNodes ?? []).length; if (nodeCount > 0) { for (const e of clips) { if (typeof e.section === "number" && (e.section < 0 || e.section >= nodeCount)) { add( "manifest-invalid", e.id, `section ${e.section} but only ${nodeCount} timeline node(s) — the footer marker would run off the track`, "blocking", { href: clipHref(e.id) }, ); } } } // --- availability, from the recorded check (never a live one) ------------ // The reducer must not shell out: an inbox that runs yt-dlp once per source is // an inbox that takes a minute to open. So it reads what the preflight wrote, // and says when nobody has run one. const avail = await readFile(path.join(dir, "out", "availability.json"), "utf8").then( (t) => JSON.parse(t), () => null, ); if (avail) { for (const src of avail.sources ?? []) { if (src.ok) continue; // SEVERITY FOLLOWS THE CLIPS, not the source. // // The probe covers ledger sources as well as clip sources now, and most // dead ledger sources are cited by no clip at all — the cut quotes them on // a card precisely BECAUSE they are gone. Calling those blocking made the // inbox say "0 clip(s) cite it; the build dies here", which refutes itself // in its own sentence, and put six permanent red rows in front of a // manifest that builds cleanly. const cited = src.clips?.length ?? 0; const claimed = src.claims?.length ?? 0; add( src.state === "no-cues" ? "clip-no-cues" : "clip-unfetchable", src.key, cited ? `${src.state} — ${cited} clip(s) cite it; the build dies here` : `${src.state} — no clip cites it${claimed ? `, ${claimed} ledger claim(s) do` : ""}. ` + "A cut that quotes it on a card is unaffected; one that adds a clip from it will not build", cited ? "blocking" : "info", ); } } // --- per clip, against the cues ----------------------------------------- const byVideo = new Map(); for (const e of clips) { // A clip with its own media (`src`, `cues`) has no corpus cue file: the // build checks its paths and its window itself. if (e.src != null) continue; const key = `${channelFor(m, e)}/${e.video}`; if (!byVideo.has(key)) byVideo.set(key, []); byVideo.get(key).push(e); } const unpunctuated = []; for (const [key, list] of byVideo) { const file = cuePathFor(m, list[0], channelsDir); const doc = file ? await readCues(file) : null; if (!doc) { // Reported once per source, not once per clip. Almost always the Rumble // two-ids trap: the manifest names the site/MCP id while the cue file // lives under the URL slug. if (!avail?.sources?.some((s) => s.key === key && !s.ok)) { add( "clip-no-cues", key, shadowBuilder ? "no transcript.cues.json — but this project ships make-shadow-channels.sh, " + `which is what puts it there. Run it before building ${list.map((e) => e.id).join(", ")}` : `no transcript.cues.json — the build dies at ${list.map((e) => e.id).join(", ")}. ` + "On Rumble, check `video` is the URL slug and not the site id", "blocking", ); } continue; } if (doc.punctuationRate < 0.1) unpunctuated.push(key); for (const e of list) { // The standing rule, mechanised. One 14-clip cut shipped with 8 clips // ending mid-thought. `lockEnd` is the ACKNOWLEDGEMENT -- setting it is // the author saying "I meant to cut here" -- so it clears this. if (!e.lockEnd && !e.lock && doc.punctuationRate >= 0.1) { const c = cueAt(doc.cues, e.end, "end"); if (c && !ENDS_SENTENCE.test(c.text ?? "") && !IS_FILLER.test(c.text ?? "")) { add( "clip-mid-sentence", e.id, `ends mid-sentence: “…${String(c.text ?? "").trim().slice(-48)}”`, "open", { href: clipHref(e.id) }, ); } } } // Two clips from one source that overlap play as the same footage twice. // resolve-windows de-overlaps them -- unless the earlier one has lockEnd, // where it warns and refuses. That refusal is the decision. const sorted = [...list].sort((a, b) => a.start - b.start); for (let i = 0; i < sorted.length - 1; i += 1) { const a = sorted[i]; const b = sorted[i + 1]; if (a.end <= b.start) continue; add( "window-overlap", a.id, `overlaps ${b.id} by ${(a.end - b.start).toFixed(1)}s` + (a.lockEnd ? " and has lockEnd, so nothing will trim it" : ""), a.lockEnd ? "open" : "info", { href: clipHref(a.id) }, ); } } // THE CUT MUST BE INSIDE THE EXTENT. `start`/`end` is the reviewed extent and // `cutStart`/`cutEnd` is what actually plays; a cut outside it renders // seconds nobody looked at, and half a pair renders the whole extent while // the manifest reads as though it were trimmed. for (const e of clips) { const hasStart = Number.isFinite(Number(e.cutStart)); const hasEnd = Number.isFinite(Number(e.cutEnd)); if (!hasStart && !hasEnd) continue; if (hasStart !== hasEnd) { add( "clip-cut-outside", e.id, `has ${hasStart ? "cutStart" : "cutEnd"} and not the other — the build honours the pair or neither`, "blocking", { href: clipHref(e.id) }, ); continue; } if (Number(e.cutEnd) - Number(e.cutStart) < 0.5) { add("clip-cut-outside", e.id, `cut ${e.cutStart}–${e.cutEnd} is under half a second`, "blocking", { href: clipHref(e.id) }); continue; } if (Number(e.cutStart) < Number(e.start) - 0.02 || Number(e.cutEnd) > Number(e.end) + 0.02) { add( "clip-cut-outside", e.id, `cut ${e.cutStart}–${e.cutEnd} lies outside its window ${e.start}–${e.end} — the build would cut seconds nobody reviewed`, "blocking", { href: clipHref(e.id) }, ); } } // A clip with no numeric window cannot be built, widened or benched. The // scaffold's `--seed chapters` writes one when the last chapter has no // duration to end at, and says so; this is where it stays visible. for (const e of clips) { if (Number.isFinite(Number(e.start)) && Number.isFinite(Number(e.end))) continue; add("manifest-invalid", e.id, `no ${Number.isFinite(Number(e.start)) ? "end" : "start"} — a clip needs both edges in source seconds`, "blocking", { href: clipHref(e.id) }); } // The cue file ends before the clip does. One row per SOURCE, naming the // clip that reaches furthest past it. for (const r of (s.sources?.rows ?? [])) { if (!r.coverageGap) continue; add( "clip-cue-gap", r.coverageGap.clip, `cues end at ${r.coverageGap.cuesEnd.toFixed(0)} s · ${r.coverageGap.clip} needs ${r.coverageGap.needs.toFixed(0)} s — ` + `the transcript of ${r.key} stops short; recover it (local whisper) or shorten the window`, "open", { href: clipHref(r.coverageGap.clip) }, ); } // One row, not one per source. Six real projects produce forty-odd of these // between them, and an inbox that says the same true thing forty times is an // inbox whose blocking rows have scrolled off the top. if (unpunctuated.length) { add( "no-punctuation", unpunctuated.length === 1 ? unpunctuated[0] : `${unpunctuated.length} sources`, `unpunctuated ASR in ${unpunctuated.slice(0, 3).join(", ")}` + `${unpunctuated.length > 3 ? ` and ${unpunctuated.length - 3} more` : ""}` + " — widening cannot help there; set those edges by ear and lock them", "info", ); } // --- the ledger ---------------------------------------------------------- // The same rule the availability read follows: this reducer NEVER measures. // ledgerTotals() is pure arithmetic over JSON already parsed into memory, so // the inbox opens in the time it did before -- audio is fetched only when a // claim page is opened, one claim at a time. const ledger = Array.isArray(m.ledger) ? m.ledger : []; if (ledger.length) { const claimHref = (cid) => `/browse/${id}/claim/${cid}`; // One row per entry, not one collapsed row. Forty unpunctuated sources are // the same true thing said forty times; forty unadjudicated claims are // forty DIFFERENT decisions, and the inbox being empty is the sign-off. for (const e of unadjudicatedOf(ledger)) { const gaps = adjudicationGaps(e); add( "claim-unadjudicated", e.id, `${e.date ?? "undated"} · ${e.display ?? e.value ?? "—"} — no ruling on ` + `${gaps.join(", ")}. Settle it against ±90s of context, not the quote`, "blocking", { href: claimHref(e.id) }, ); } // Coherence over whatever HAS been ruled on, so these appear as the work // proceeds rather than all at once at the end. let totals = null; try { totals = ledgerTotals(ledger, { strict: false }); } catch { /* a malformed ledger is already reported as unadjudicated rows */ } for (const step of totals?.steps ?? []) { for (const f of step.flags) { add( "claim-incoherent", step.id, `${f.rule.replace(/_/g, " ")} — ${f.text}`, "open", { href: claimHref(step.id) }, ); } } } // --- where the bulk lives (release 17) ----------------------------------- // `out/` follows UMTOOL_MEDIA_DIR; the deliverables follow // `storage.deliverables`. A link whose drive is not there reads as "never // built" and "nothing cut" everywhere else, which is the sentence this // exists to replace. const outState = await outDirState(dir); if (outState.state === "dangling") { add("storage-unreachable", "out", `a link to ${outState.target}, which is not there — is the media drive mounted? Nothing can be built until it is`, "blocking"); } const outLeft = await leftoversOf(dir, "out"); if (outLeft.length) { add( "storage-mismatch", "out", `a move of out/ was cut (left: ${outLeft.map((p) => path.basename(p)).join(", ")}) — \`${finishCommand("out", outLeft.some((p) => p.endsWith(".incoming"))).replace("", id)}\` finishes it`, "open", ); } const store = await deliverablesState(dir); if (!store.mode) add("manifest-invalid", "storage.deliverables", `${store.error} — a cut or a batch refuses until it is`, "blocking"); for (const d of store.dirs) { if (d.state === "dangling") { add("storage-unreachable", d.name, `a link to ${d.target}, which is not there — is the media drive mounted? A cut or a batch refuses until it is`, "blocking"); } if (d.leftovers.length) { add( "storage-mismatch", d.name, `a move of ${d.name}/ was cut (left: ${d.leftovers.join(", ")}) — \`${finishCommand(d.name, d.leftovers.some((l) => l.endsWith(".incoming"))).replace("", id)}\` finishes it`, "open", ); } else if (store.mode === "media" && d.state === "dir") { add("storage-mismatch", d.name, `a directory in the project, while storage.deliverables is media — \`umtool storage deliverables ${id} --to media\` moves it`, "open"); } else if ( store.mode === "local" && d.state === "link" && // Absent means local, so a link INTO the media root under no key at all // is the same disagreement (review N1). A hand-made link elsewhere under // no key is nobody's decision to second-guess; under an explicit // "local", any link is. (store.value !== undefined || (store.mediaRoot && d.target && inside(store.mediaRoot, d.target))) ) { add( "storage-mismatch", d.name, `a link to ${d.target}, while storage.deliverables is ${store.value === undefined ? "unset (local)" : "local"} — \`umtool storage deliverables ${id} --to local\` brings it back, \`--to media\` records it`, "open", ); } } if (store.mode === "media" && !store.tiered) { add( "storage-mismatch", "storage.deliverables", "media, and UMTOOL_MEDIA_DIR is not set here — a new clips/ or batch would be refused. Set it where umtool runs", "open", ); } // --- the build ----------------------------------------------------------- if (s.build.stale) { add( "stale-build", `out/${s.build.slug}.mp4`, "older than the manifest that describes it — the file is a lead, not a fact", "open", ); } else if (!s.build.built) { add("unbuilt", `out/${s.build.slug}.mp4`, "never built", "info"); } return out; } // --------------------------------------------------------------------------- // Per-clip detail: what the project page lists and the clip bench edits. // // This is the expensive read -- one cue file per distinct source -- so it is // NOT what the index calls. summariseReport() is. // --------------------------------------------------------------------------- export async function readClipDetail(dir, { manifest = null } = {}) { const m = manifest ?? (await readManifest(dir)); if (!m) return null; const shadowExists = await hasShadowChannels(dir); const channelsDir = channelsDirFor(dir, m, { shadowExists }); const build = await buildStateOf(dir, m); // ONE listing of out/clips-raw for the whole project. This used to be two // readdirs PER CLIP -- the padded lookup and the full list -- so a forty-clip // cut paid eighty directory reads to draw one page. const raw = await rawCacheOf(dir, { extraWindows: await clipWindowDirs(m, channelsDir) }); const segDirs = segmentDirs(path.join(dir, "out")); const segLists = await Promise.all(segDirs.map((d) => readdir(d).catch(() => []))); const segWhich = segLists.findIndex((l) => l.length); const segNames = new Set(segWhich < 0 ? [] : segLists[segWhich]); const segRel = segWhich === 0 ? path.posix.join("out", DEFAULT_VARIANT, "segments") : path.posix.join("out", "segments"); const cueCache = new Map(); const entries = []; for (const e of m.timeline ?? []) { // A CLIP is `type === "clip"`. Everything else is a non-clip entry with no // window and no source. // // Not `!== "card"`: the timeline's vocabulary is OPEN. quartering-employee- // count carries `scroll` and `chart` entries beside its cards, and treating // anything-that-is-not-a-card as a clip sent `undefined` into path.join() // and 500'd the whole project page. Every other real manifest is clips only, // which is exactly why this survived testing. if (e.type !== "clip") { // A teaser has no heading or title of its own: its row names its lines. entries.push({ ...e, kind: e.type ?? "entry", ...(e.type === "teaser" ? { label: teaserTitle(e) } : {}) }); continue; } const chan = channelFor(m, e); const key = `${chan}/${e.video}`; if (!cueCache.has(key)) { const file = cuePathFor(m, e, channelsDir); cueCache.set(key, file ? await readCues(file) : null); } const doc = cueCache.get(key); // The pad the build would use, so "is this clip cached" answers the same // question the build will ask. const pad = m.render?.fetchPad ?? 3.0; const from = Math.max(0, e.start - pad); const to = e.end + pad; const cached = raw.containing(e.video, from, to); const allWindows = raw.windows(e.video); // The bench wants the WIDEST containing file (room to drag); the build wants // the tightest (least to decode). They are different questions. const widest = allWindows .filter((w) => w.from <= e.start && w.to >= e.end) .sort((a, b) => b.to - b.from - (a.to - a.from))[0] ?? null; const endCue = doc ? cueAt(doc.cues, e.end, "end") : null; // NULL means "cannot be known", and that is a third answer worth having. // // In an upload whose ASR emitted no terminators, every clip "ends // mid-sentence" and the fact says nothing about the cut. Reporting it as // FALSE put four ferret-rescue clips under a warning the decisions inbox // (which has always had this gate) correctly stayed silent about -- the page // and the inbox disagreeing about the same clip. The honest answer is that // the source cannot support the question. const noPunctuation = !!doc && doc.punctuationRate < 0.1; const endsSentence = !endCue || noPunctuation ? null : ENDS_SENTENCE.test(endCue.text ?? ""); // What resolve-windows WOULD do, computed in-process because widen() is pure // once the cues are read. It is the difference between "run the widener and // see" and knowing before you touch anything. let proposed = null; if (doc?.cues?.length && !e.lock) { const w = widen(doc.cues, e.start, e.end); if (e.lockStart) w.start = e.start; if (e.lockEnd) w.end = e.end; const moved = Math.abs(w.start - e.start) > 0.05 || Math.abs(w.end - e.end) > 0.05; proposed = moved ? { start: Number(w.start.toFixed(2)), end: Number(w.end.toFixed(2)) } : null; } entries.push({ ...e, kind: "clip", channel: chan, cueFile: cuePathFor(m, e, channelsDir), hasCues: !!doc, duration: doc?.duration ?? null, sourceTitle: doc?.title ?? null, // What the burned-in header would say if this clip carried no overrides: // the title with the emoji and !commands stripped, and the ARCHIVED // copy's upload date. The bench shows both beside the fields that replace // them, so "2019" next to a 2016 stream is visible rather than inferred. sourceTitleClean: doc?.title ? cleanTitle(doc.title) : null, uploadDate: doc?.uploadDate ?? null, // WHO the header will name, resolved here rather than in the component: // the answer walks the clip's override, the record, the sweep's own // display name and the slug, and a preview that re-implemented that walk // is exactly the drift `attribution.mjs` exists to prevent. sourceChannel: channelName(e, doc ?? {}, m.provenance ?? {}), punctuationRate: doc?.punctuationRate ?? null, endCueText: endCue?.text ?? null, endsSentence, noPunctuation, proposed, // `path` rides along because a window is no longer always under // out/clips-raw: the editor fetches into the corpus, and a caller that // rebuilds the project-local path from the name names a file that is not // there. cached: cached ? { name: cached.name, path: cached.path, from: cached.from, to: cached.to } : null, // `cached` is the BUILD's question (is the padded window on disk); this is // the PLAYER's (can this clip be watched end to end right now), and it is // what the walk skips on. Same predicate, different window. fetched: raw.isFetched(e), widest: widest ? { name: widest.name, path: widest.path, from: widest.from, to: widest.to } : null, segment: segNames.has(`${e.id}.mp4`) ? path.posix.join(segRel, `${e.id}.mp4`) : null, wantFrom: from, wantTo: to, }); } return { manifest: m, build, channelsDir, shadowExists, entries }; } /** The cues a clip bench draws, trimmed to a window. Absolute source seconds. */ export async function cuesInWindow(dir, clipId, from, to) { const m = await readManifest(dir); const e = clipsOf(m).find((x) => x.id === clipId); if (!e) return null; const shadowExists = await hasShadowChannels(dir); const file = cuePathFor(m, e, channelsDirFor(dir, m, { shadowExists })); const doc = file ? await readCues(file) : null; if (!doc) return null; return { duration: doc.duration ?? null, punctuationRate: doc.punctuationRate, cues: doc.cues .filter((c) => c.end >= from && c.start <= to) .map((c) => ({ start: c.start, end: c.end, text: c.text, endsSentence: ENDS_SENTENCE.test(c.text ?? "") && !IS_FILLER.test(c.text ?? ""), })), }; } // --------------------------------------------------------------------------- // One CLAIM, ready to adjudicate. // // The claim bench asks a different question from the clip bench. A clip asks // "where exactly does this cut?", so it wants sample-accurate edges. A claim // asks "what did he mean by that?", so it wants ROOM -- the standing rule is // that a first-person quote is routinely the host reading someone else's words // or being sarcastic, and neither is visible inside the quote itself. Hence a // default of +/-90s, and hence context is returned as cues rather than as a // waveform: the words on either side are what settle a scope. // --------------------------------------------------------------------------- export const CLAIM_CONTEXT_PAD = 90; export async function readClaimDetail(dir, claimId, { manifest = null, pad = CLAIM_CONTEXT_PAD } = {}) { const m = manifest ?? (await readManifest(dir)); if (!m) return null; const claim = (m.ledger ?? []).find((e) => e.id === claimId); if (!claim) return null; const shadowExists = await hasShadowChannels(dir); const channelsDir = channelsDirFor(dir, m, { shadowExists }); const file = claim.video ? cuePathFor(m, claim, channelsDir) : null; const doc = file ? await readCues(file) : null; const at = Number.isFinite(Number(claim.cite)) ? Number(claim.cite) : null; const from = at == null ? 0 : Math.max(0, at - pad); const to = at == null ? 0 : at + pad; // Which cached raw windows already cover this moment. A claim that lands // inside one a clip fetched earlier is playable with no download at all, // which is most of them once a build has run. const rawDir = path.join(dir, "out", "clips-raw"); const windows = claim.video ? (await cachedWindowsFor(rawDir, claim.video)).filter( (w) => at != null && w.from <= at && w.to >= at, ) : []; windows.sort((a, b) => b.to - b.from - (a.to - a.from)); return { claim, at, view: { from, to }, pad, // Every cue in the window, with the cited one marked. The mark is what // makes "he is quoting a tweet here" visible: the quote sits in a paragraph // rather than alone. cues: (doc?.cues ?? []) .filter((c) => c.end >= from && c.start <= to) .map((c) => ({ start: c.start, end: c.end, text: c.text, cited: at != null && c.start <= at && c.end >= at, })), source: doc ? { title: doc.title ?? null, uploadDate: doc.uploadDate ?? null, duration: doc.duration ?? null, webpageUrl: doc.webpageUrl ?? null } : null, noCues: !doc, windows: windows.map((w) => ({ name: w.name, from: w.from, to: w.to })), gaps: adjudicationGaps(claim), }; }