Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a3f629619c71a4f35e9c84e89b2a66e1e41c9ec5
parent 94e9c6d78429a46f19b69d9362a35245b060bcb8
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 03:13:25 -0400

Merge report-r2-media (evidence media for report sites: one source lookup in common, the 720p evidence cutter, cited-only post captures, archilyzer reports prepare + job + ops route)

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
MPUBLISH.md | 12++++++++++++
Mcommon/bin/archilyzer.ts | 11+++++++++++
Acommon/bin/reports-prepare.ts | 30++++++++++++++++++++++++++++++
Mcommon/jobs/jobKinds.test.ts | 3+++
Mcommon/jobs/jobKinds.ts | 17+++++++++++++++++
Acommon/lib/evidenceClip-server.test.ts | 240+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/evidenceClip-server.ts | 421+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/evidenceClip.mjs | 400+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/package.json | 1+
Acommon/publish/citedPostCaptures.test.ts | 132+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/publish/citedPostCaptures.ts | 149+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/publish/reportMedia.test.ts | 229+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/publish/reportMedia.ts | 370+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/postCapture.test.ts | 23++++++++++++++++-------
Meditor/CHANGELOG.md | 1+
Aeditor/app/api/ops/reports-prepare/route.test.ts | 49+++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/app/api/ops/reports-prepare/route.ts | 23+++++++++++++++++++++++
Meditor/app/jobs/jobReplayRegistry.ts | 8++++++++
Aeditor/app/sites/lib/reportsPrepareAction.ts | 51+++++++++++++++++++++++++++++++++++++++++++++++++++
Mscripts/archilyzer-ops.mjs | 8++++++++
Mscripts/archilyzer-ops.test.mjs | 10++++++++++
Mumtool/report-to-video/build-video.mjs | 7++++---
Mumtool/report-to-video/sources.mjs | 296+++++++++++++------------------------------------------------------------------
23 files changed, 2233 insertions(+), 258 deletions(-)

diff --git a/PUBLISH.md b/PUBLISH.md @@ -53,6 +53,7 @@ entry points in `common/publish/build.ts`. | The hub | /sites → Hub → **Build hub** / **Deploy hub** | `build-hub`, `deploy-hub` | `build hub`, `deploy hub [--preview <branch>]` | | The homepage | /sites → Homepage → **Build homepage** (tick *Deploy after build*) / **Deploy homepage**, with an optional preview branch | `build-homepage` (`{"deploy":true}` to deploy after), `deploy-homepage` (`{"preview":"<branch>"}`) | `build homepage [--no-source]`, `deploy homepage [--preview <branch>]` | | The source mirror alone | — (every homepage build runs it) | — | `source publish [--force] [--check] [--keep-scratch]`, `source audit [<git dir>]` | +| A site's report evidence media | — (a Reports tab is to come) | `reports-prepare` (`{"siteId"}`) | `reports prepare <id>` | `pnpm archilyzer <command>` is the short form of `pnpm --filter yt-dlp-transcript-common exec tsx bin/archilyzer.ts <command>`; @@ -61,6 +62,17 @@ this machine has what a build needs. From `export/`, `pnpm run build` is `archilyzer build site` and `pnpm run deploy` is `archilyzer deploy site`; both take the site from `SITE_ID` when no id is given. +A site with `reports` (site.json) has one more step, before its build and on the +host: **reports prepare** cuts the evidence clip of every span its published reports +cite — from the editor's clip windows, the saved-video store or a record's audio, +fitted inside 1280×720 (H.264 crf 23, AAC; an audio span is an `.m4a`) — and copies +the screenshot and media of every cited post, only those, into +`.export-index/sites/<id>/report-media/`, with a manifest (`index.json`) of each +moment's file, size, hash and duration. Nothing is fetched: a citation whose media +is not on disk, a clip over 24 MiB or an invalid report is listed and fails the run +(exit 1, or a failed job) — fetch the window or persist the video, capture the +post, and run it again; what is already cut is reused. + Deploy-only ships whatever is in `export/out`, which the basic build composes one site at a time into a single shared directory — so it **refuses, before starting a job, if `export/out` holds a build of another site** (or no build at all), naming the diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -151,6 +151,17 @@ export const COMMANDS: Command[] = [ }, }, { + path: ["reports", "prepare"], + usage: + "<id> cut every clip and copy every post capture the site's published reports cite into its report-media cache, before its build (exit 1 when a citation lacks media; default id: SITE_ID)", + maxPositionals: 1, + run: async ({ positionals, env }) => { + const siteId = siteIdFrom(positionals, env, "reports prepare"); + if (!siteId) return 2; + return (await import("./reports-prepare")).main({ siteId, signal: interrupted() }); + }, + }, + { path: ["source", "publish"], usage: "[--force] [--check] [--keep-scratch] the scrubbed git mirror, raw tree, history pages (stagit, when installed) and tarball into homepage/public, behind the denied-literal gate (--check: audit and count, write nothing)", diff --git a/common/bin/reports-prepare.ts b/common/bin/reports-prepare.ts @@ -0,0 +1,30 @@ +// `archilyzer reports prepare <siteId>` — cut and copy the evidence media of a +// site's published reports into `.export-index/sites/<siteId>/report-media/`, +// before the site's build. The work is publish/reportMedia.ts's, the same the +// editor's `reports-prepare` job runs; this file prints its log and its +// problems. +// +// Exit 0 when every cited moment has its media; 1 when any problem is listed +// (the manifest is written either way, problems included); 2 for a site that +// does not exist. + +import { formatReportMediaProblems, prepareReportMedia } from "../publish/reportMedia"; + +type Out = { log: (s: string) => void; error: (s: string) => void }; + +export async function main( + opts: { siteId: string; signal?: AbortSignal }, + out: Out = console, +): Promise<number> { + let index; + try { + index = await prepareReportMedia({ siteId: opts.siteId, signal: opts.signal, onLog: out.log }); + } catch (err) { + out.error(`reports prepare: ${(err as Error).message}`); + return opts.signal?.aborted ? 1 : 2; + } + if (index.problems.length === 0) return 0; + out.error(`reports prepare ${opts.siteId}: ${index.problems.length} problem(s):`); + for (const line of formatReportMediaProblems(index.problems)) out.error(` ${line}`); + return 1; +} diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts @@ -84,6 +84,8 @@ const ADDED_KINDS: Record<string, { label: string; drainable: boolean }> = { }, // A list of videos persisted one at a time: a drain stops between them. "persist-videos": { label: "Persist videos", drainable: true }, + // A report site's evidence media: one pass, cancelled rather than drained. + "reports-prepare": { label: "Prepare report media", drainable: false }, }; test("added kinds carry their pinned label and drainability", () => { @@ -176,6 +178,7 @@ const STILL_MEDIA = [ "clean-audio-transcribed", "clean-extra-audio-formats", "remove-wrong-format-audio", + "reports-prepare", ]; test("release 17: the text kinds flipped from needsMedia to needsText", () => { diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -630,6 +630,23 @@ const JOB_KINDS: Record<string, JobKindMeta> = { replayable: false, queueKeyStrategy: "custom", }, + // A REPORT SITE'S EVIDENCE MEDIA (publish/reportMedia.ts): every clip its + // published reports cite, cut from the media on disk, and every cited post + // capture copied, into the site's report-media cache before its build. + // `needsMedia`: it opens saved containers and audio, which may be on another + // drive. It spans channels and runs with no `channelSlug`, so the guard in + // runManagedFunction does not ask; the step itself reports a span on an + // unreachable channel as unreachable rather than missing. Its own queue + // (`reports-prepare`): one at a time, behind nothing — not the download + // queues, not the build queue. Replayable: the spec is the site. + "reports-prepare": { + kind: "reports-prepare", + label: "Prepare report media", + drainable: false, + replayable: true, + queueKeyStrategy: "custom", + needsMedia: true, + }, // THE PER-VIDEO WRITERS THAT WERE NOT IN THIS TABLE (release 16 slice RM). // Each runs with a channelSlug and writes under `data/<id>/` — a single // video's transcription (the video page's two Transcribe buttons, and its diff --git a/common/lib/evidenceClip-server.test.ts b/common/lib/evidenceClip-server.test.ts @@ -0,0 +1,240 @@ +// The evidence cutter: where a cited span's media is found, the span it cuts, +// the cut itself and the cache in front of it. +// +// Real ffmpeg over tiny lavfi containers (a few frames at 160×90, a second of +// sine), so a run costs little memory and a second or two. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/evidenceClip-server.test.ts + +import { after, before, test } from "node:test"; +import assert from "node:assert/strict"; +import { execFileSync } from "node:child_process"; +import { copyFile, mkdir, mkdtemp, readdir, rm, symlink, utimes, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { + evidenceCutArgs, + evidenceHash, + evidenceSpan, + prepareEvidenceClip, + probeMedia, + resolveEvidenceSource, + widerPad, +} from "./evidenceClip-server"; + +const SLUG = "demo-channel"; +const ID = "abc123"; + +let ROOT = ""; +// A 6 s 160×90 container with sound, a 1 s 1440×810 one, and 6 s of sound. +let SMALL = ""; +let WIDE = ""; +let SOUND = ""; + +function ff(args: string[]): void { + execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", ...args]); +} + +before(async () => { + ROOT = await mkdtemp(path.join(tmpdir(), "evidence-clip-")); + const fx = path.join(ROOT, "fixtures"); + await mkdir(fx, { recursive: true }); + SMALL = path.join(fx, "small.mp4"); + WIDE = path.join(fx, "wide.mp4"); + SOUND = path.join(fx, "sound.m4a"); + ff([ + "-f", "lavfi", "-i", "testsrc=size=160x90:rate=10:duration=6", + "-f", "lavfi", "-i", "sine=frequency=440:duration=6", + "-c:v", "libx264", "-preset", "ultrafast", "-pix_fmt", "yuv420p", "-c:a", "aac", "-shortest", SMALL, + ]); + ff([ + "-f", "lavfi", "-i", "testsrc=size=1440x810:rate=5:duration=1", + "-c:v", "libx264", "-preset", "ultrafast", "-pix_fmt", "yuv420p", WIDE, + ]); + ff(["-f", "lavfi", "-i", "sine=frequency=220:duration=6", "-c:a", "aac", SOUND]); +}); + +after(() => rm(ROOT, { recursive: true, force: true })); + +// A fresh channels tree per test: `<root>/channels/<slug>/data/<id>/`. +async function corpus(): Promise<{ channelsDir: string; videoDir: string; cacheDir: string; root: string }> { + const root = await mkdtemp(path.join(ROOT, "t-")); + const channelsDir = path.join(root, "channels"); + const videoDir = path.join(channelsDir, SLUG, "data", ID); + await mkdir(videoDir, { recursive: true }); + return { channelsDir, videoDir, cacheDir: path.join(root, "report-media"), root }; +} + +test("evidenceSpan: the moment's rounded seconds, widened by the pad, clamped at 0", () => { + assert.deepEqual(evidenceSpan({ start: 12.345, end: 20 }), { from: 12.35, to: 20 }); + assert.deepEqual(evidenceSpan({ start: 12, end: 20, pad: { before: 5, after: 2.5 } }), { from: 7, to: 22.5 }); + assert.deepEqual(evidenceSpan({ start: 3, end: 9, pad: { before: 5 } }), { from: 0, to: 9 }); + // A negative pad is no pad. + assert.deepEqual(evidenceSpan({ start: 3, end: 9, pad: { before: -2, after: -1 } }), { from: 3, to: 9 }); +}); + +test("widerPad: two citations of one moment get the wider context on each side", () => { + assert.equal(widerPad(undefined, undefined), undefined); + assert.deepEqual(widerPad({ before: 2 }, undefined), { before: 2 }); + assert.deepEqual(widerPad({ before: 2, after: 1 }, { before: 1, after: 4 }), { before: 2, after: 4 }); +}); + +test("tiers: a clip window, else the saved container, else nothing; the window wins when both hold the span", async () => { + const { channelsDir, videoDir, root } = await corpus(); + const span = { from: 11, to: 13 }; + const ask = (audio = false) => resolveEvidenceSource({ channelsDir, slug: SLUG, id: ID, span, audio }); + + assert.equal(await ask(), null); + + // The saved-video store's pointer: the container is [0, its duration]. + const store = path.join(root, "saved-videos", SLUG, ID); + await mkdir(store, { recursive: true }); + await copyFile(SMALL, path.join(store, "source-media.mp4")); + await writeFile(path.join(videoDir, "saved-video.json"), JSON.stringify({ dir: store, file: "source-media.mp4" })); + // 6 s does not reach 11–13. + assert.equal(await ask(), null); + const early = await resolveEvidenceSource({ channelsDir, slug: SLUG, id: ID, span: { from: 1, to: 3 }, audio: false }); + assert.equal(early?.kind, "saved-video"); + assert.equal(early?.windowStart, 0); + + // A window named for the seconds it holds. + await mkdir(path.join(videoDir, "clips"), { recursive: true }); + await copyFile(SMALL, path.join(videoDir, "clips", "10.00-16.00.mp4")); + const win = await ask(); + assert.equal(win?.kind, "corpus-window"); + assert.equal(win?.windowStart, 10); + // Both hold 1–3 once a window does: the window wins. + await copyFile(SMALL, path.join(videoDir, "clips", "0.00-6.00.mp4")); + const both = await resolveEvidenceSource({ channelsDir, slug: SLUG, id: ID, span: { from: 1, to: 3 }, audio: false }); + assert.equal(both?.kind, "corpus-window"); +}); + +test("tiers: the sound is asked only when allowed, and never beats a picture", async () => { + const { channelsDir, videoDir } = await corpus(); + const span = { from: 1, to: 3 }; + await copyFile(SOUND, path.join(videoDir, "audio.m4a")); + assert.equal(await resolveEvidenceSource({ channelsDir, slug: SLUG, id: ID, span, audio: false }), null); + assert.equal((await resolveEvidenceSource({ channelsDir, slug: SLUG, id: ID, span, audio: true }))?.kind, "audio"); + await mkdir(path.join(videoDir, "clips"), { recursive: true }); + await copyFile(SMALL, path.join(videoDir, "clips", "0.00-6.00.mp4")); + assert.equal( + (await resolveEvidenceSource({ channelsDir, slug: SLUG, id: ID, span, audio: true }))?.kind, + "corpus-window", + ); +}); + +test("tiers: a dangling link (an unmounted drive) is not there — never a throw — and the next tier answers", async () => { + const { channelsDir, videoDir, root } = await corpus(); + const span = { from: 1, to: 3 }; + await mkdir(path.join(videoDir, "clips"), { recursive: true }); + await symlink(path.join(root, "gone", "0.00-6.00.mp4"), path.join(videoDir, "clips", "0.00-6.00.mp4")); + // The media tier's relative link into media/, itself pointing nowhere. + await symlink("../../media/abc123/source-media.mp4", path.join(videoDir, "source-media.mp4")); + assert.equal(await resolveEvidenceSource({ channelsDir, slug: SLUG, id: ID, span, audio: false }), null); + // A live relative link into media/ is followed. + const media = path.join(channelsDir, SLUG, "media", ID); + await mkdir(media, { recursive: true }); + await copyFile(SMALL, path.join(media, "source-media.mp4")); + const hit = await resolveEvidenceSource({ channelsDir, slug: SLUG, id: ID, span, audio: false }); + assert.equal(hit?.kind, "saved-video"); + assert.equal(hit?.name, "source-media.mp4"); +}); + +test("the cut: the span exactly, a small source keeps its size, metadata dropped", async () => { + const { channelsDir, videoDir, cacheDir } = await corpus(); + await mkdir(path.join(videoDir, "clips"), { recursive: true }); + await copyFile(SMALL, path.join(videoDir, "clips", "10.00-16.00.mp4")); + const r = await prepareEvidenceClip({ + channelsDir, slug: SLUG, id: ID, kind: "video", span: { from: 11, to: 13.5 }, cacheDir, + }); + assert.ok(r.ok, JSON.stringify(r)); + assert.equal(r.cached, false); + assert.equal(r.media.kind, "video"); + assert.match(r.media.file, /^[0-9a-f]{32}\.mp4$/); + assert.equal(r.media.width, 160); + assert.equal(r.media.height, 90); + assert.ok(Math.abs((r.media.durationSec ?? 0) - 2.5) < 0.15, `duration ${r.media.durationSec}`); + assert.match(r.media.sha256, /^[0-9a-f]{64}$/); + const probe = await probeMedia(path.join(cacheDir, r.media.file)); + assert.equal(probe?.hasAudio, true); + // The clip and its sidecar, and no temp file left behind. + assert.deepEqual((await readdir(cacheDir)).sort(), [r.media.file, r.media.file.replace(/\.mp4$/, ".json")].sort()); +}); + +test("the cut: a larger source is fitted inside 1280×720, sides even", async () => { + const { channelsDir, videoDir, cacheDir } = await corpus(); + await mkdir(path.join(videoDir, "clips"), { recursive: true }); + await copyFile(WIDE, path.join(videoDir, "clips", "0.00-1.00.mp4")); + const r = await prepareEvidenceClip({ channelsDir, slug: SLUG, id: ID, kind: "video", span: { from: 0, to: 1 }, cacheDir }); + assert.ok(r.ok, JSON.stringify(r)); + assert.equal(r.media.width, 1280); + assert.equal(r.media.height, 720); + const args = evidenceCutArgs("in.mp4", 1, 2, "video", "out"); + assert.ok(args.includes("libx264") && args.includes("23") && args.includes("+faststart")); + assert.match(args[args.indexOf("-vf") + 1], /min\(1280,iw\).*min\(720,ih\).*force_divisible_by=2/); +}); + +test("audio: an audio citation is an .m4a with no picture; a video citation of a sound-only record too", async () => { + const { channelsDir, videoDir, cacheDir } = await corpus(); + await copyFile(SOUND, path.join(videoDir, "audio.m4a")); + const span = { from: 1, to: 3 }; + const a = await prepareEvidenceClip({ channelsDir, slug: SLUG, id: ID, kind: "audio", span, cacheDir }); + assert.ok(a.ok, JSON.stringify(a)); + assert.equal(a.media.kind, "audio"); + assert.match(a.media.file, /\.m4a$/); + assert.equal(a.media.width, null); + assert.equal((await probeMedia(path.join(cacheDir, a.media.file)))?.hasVideo, false); + + // A video citation does not reach the sound unless the record has no picture. + const v = await prepareEvidenceClip({ channelsDir, slug: SLUG, id: ID, kind: "video", span, cacheDir }); + assert.equal(v.ok, false); + assert.equal(!v.ok && v.reason, "missing"); + const p = await prepareEvidenceClip({ channelsDir, slug: SLUG, id: ID, kind: "video", span, cacheDir, audioOnlyRecord: true }); + assert.ok(p.ok, JSON.stringify(p)); + assert.equal(p.media.kind, "audio"); + // The same source, span and kind: the same file. + assert.equal(p.media.file, a.media.file); +}); + +test("cache: a hit by source identity + span; a changed source or span is a new cut", async () => { + const { channelsDir, videoDir, cacheDir } = await corpus(); + await mkdir(path.join(videoDir, "clips"), { recursive: true }); + const win = path.join(videoDir, "clips", "0.00-6.00.mp4"); + await copyFile(SMALL, win); + const span = { from: 1, to: 2 }; + const first = await prepareEvidenceClip({ channelsDir, slug: SLUG, id: ID, kind: "video", span, cacheDir }); + const again = await prepareEvidenceClip({ channelsDir, slug: SLUG, id: ID, kind: "video", span, cacheDir }); + assert.ok(first.ok && again.ok); + assert.equal(again.cached, true); + assert.deepEqual(again.media, first.media); + + const other = await prepareEvidenceClip({ channelsDir, slug: SLUG, id: ID, kind: "video", span: { from: 1, to: 2.5 }, cacheDir }); + assert.ok(other.ok); + assert.equal(other.cached, false); + assert.notEqual(other.media.file, first.media.file); + + // A re-fetched window (a new mtime) is a new identity. + await utimes(win, new Date(), new Date(Date.now() + 60_000)); + const refetched = await prepareEvidenceClip({ channelsDir, slug: SLUG, id: ID, kind: "video", span, cacheDir }); + assert.ok(refetched.ok); + assert.equal(refetched.cached, false); + assert.notEqual(refetched.media.file, first.media.file); + + const id = { path: "/x/a.mp4", bytes: 10, mtimeMs: 5 }; + assert.equal(evidenceHash(id, span, "video"), evidenceHash({ ...id }, { ...span }, "video")); + assert.notEqual(evidenceHash(id, span, "video"), evidenceHash(id, span, "audio")); + assert.notEqual(evidenceHash(id, span, "video"), evidenceHash({ ...id, bytes: 11 }, span, "video")); +}); + +test("size: a clip over the limit is refused, with its size, and nothing is kept", async () => { + const { channelsDir, videoDir, cacheDir } = await corpus(); + await mkdir(path.join(videoDir, "clips"), { recursive: true }); + await copyFile(SMALL, path.join(videoDir, "clips", "0.00-6.00.mp4")); + const r = await prepareEvidenceClip({ + channelsDir, slug: SLUG, id: ID, kind: "video", span: { from: 0, to: 4 }, cacheDir, maxBytes: 1000, + }); + assert.equal(r.ok, false); + assert.equal(!r.ok && r.reason, "too-big"); + assert.ok(!r.ok && (r.bytes ?? 0) > 1000); + assert.deepEqual(await readdir(cacheDir), []); +}); diff --git a/common/lib/evidenceClip-server.ts b/common/lib/evidenceClip-server.ts @@ -0,0 +1,421 @@ +// AN EVIDENCE CLIP: the self-hosted cut of one cited span that a report site's +// moment page plays — the span, a little context either side, and nothing +// else (plans/report-sites.md, "Evidence media"). +// +// Clips are cut ON THE HOST, before a site's build, because the build (a +// docker container with read-only mounts and no ffmpeg) only copies them. So +// this module is the whole of "cut": find the span's media on disk, cut it +// once, keep it in a cache keyed by what it was cut from. +// +// WHERE THE MEDIA COMES FROM is ./evidenceClip.mjs — the corpus tiers umtool's +// report-to-video build asks too (clip window, saved-video store, audio), so +// the two cannot disagree about what is local. A tier hit through a dangling +// link (an unmounted drive) is "not here", never a throw. Nothing here fetches: +// a span with no media on disk is MISSING, and the editor's fetch-window (or +// persist) is what fills it. +// +// THE CUT is accurate — the input is seeked and re-encoded, so the first frame +// is the asked-for second, never the keyframe before it: +// video fitted inside 1280×720 (never upscaled; a smaller source keeps its +// size; both sides even), H.264 crf 23 + AAC 128k, faststart. +// audio an `.m4a`: AAC 128k, faststart, no picture. Chosen over an mp4 +// with a still poster frame because the moment page draws the poster +// itself (the record's thumbnail): a still encoded into every audio +// clip would cost bytes for a picture the page already has. An +// `audio` citation is always cut this way, and so is a `video` +// citation whose only media on disk has no picture stream. +// Container metadata is dropped (`-map_metadata -1`): a clip carries its +// seconds, not its source file's tags. +// +// THE CACHE is `<dir>/<hash>.<ext>` + `<hash>.json`, the hash a function of the +// source file's identity (path, size, mtime), the padded span, the output kind +// and the profile. Same source, same span → the same file, never re-cut; a +// re-fetched window or a re-persisted container is a new identity and a new +// cut. The sidecar holds what the manifest needs (bytes, sha256, geometry, +// duration), so a hit costs a stat and a small read, not a probe and a hash. +// +// THE SIZE LIMIT: a static host refuses a file over 25 MiB, so a clip over +// EVIDENCE_MAX_BYTES (24 MiB) is refused — removed, and reported as too big +// with its size. A 120 s span at 720p is well under it; one that is not is a +// span to shorten. +// +// SERVER-ONLY (node:fs, ffmpeg). + +import { createHash } from "node:crypto"; +import { createReadStream } from "node:fs"; +import { mkdir, readFile, rename, rm, stat } from "node:fs/promises"; +import path from "node:path"; +import { execa } from "execa"; +import { tmpPathFor, writeJsonAtomic } from "./jsonFile-server"; +import { roundMomentSeconds } from "./citations/moments"; +import type { CitationPad } from "./citations/schema"; +import { + AUDIO_ONLY_PLATFORMS, + cutArgs, + ffprobeSource, + resolveCorpusSource, +} from "./evidenceClip.mjs"; + +// Bumped whenever the arguments below change what a clip looks like: it is in +// every hash, so a new profile re-cuts rather than serving an old file. +export const EVIDENCE_PROFILE = "evidence-v1"; + +export const EVIDENCE_MAX_WIDTH = 1280; +export const EVIDENCE_MAX_HEIGHT = 720; +export const EVIDENCE_CRF = 23; +export const EVIDENCE_AUDIO_BITRATE = "128k"; + +// 24 MiB: a Pages file limit is 25 MiB, and a clip is published as one file. +export const EVIDENCE_MAX_BYTES = 24 * 1024 * 1024; + +// A cut of a cached window is seconds of work; a whole saved container on a +// platter seeks once. Generous, so only a wedged ffmpeg reaches it. +const CUT_TIMEOUT_MS = 10 * 60_000; + +export type EvidenceKind = "video" | "audio"; + +export const EVIDENCE_EXT: Record<EvidenceKind, string> = { video: ".mp4", audio: ".m4a" }; + +// The seconds a clip covers, in the record's clock. +export type EvidenceSpan = { from: number; to: number }; + +// The span a citation's clip covers: its moment's (rounded) start and end, +// widened by its pad, the start clamped at 0. Rounded to the millisecond so a +// float's last digits never change a hash. +export function evidenceSpan(c: { start: number; end: number; pad?: CitationPad }): EvidenceSpan { + const start = roundMomentSeconds(c.start); + const end = roundMomentSeconds(c.end); + const before = Math.max(0, c.pad?.before ?? 0); + const after = Math.max(0, c.pad?.after ?? 0); + return { + from: Number(Math.max(0, start - before).toFixed(3)), + to: Number((end + after).toFixed(3)), + }; +} + +// Two citations of one moment share its page and so its clip: the clip gets +// the WIDER pad on each side, so neither citation loses the context it asked for. +export function widerPad(a: CitationPad | undefined, b: CitationPad | undefined): CitationPad | undefined { + if (!a) return b; + if (!b) return a; + return { + before: Math.max(a.before ?? 0, b.before ?? 0), + after: Math.max(a.after ?? 0, b.after ?? 0), + }; +} + +export type EvidenceSourceKind = "corpus-window" | "saved-video" | "audio"; + +export type EvidenceSource = { + kind: EvidenceSourceKind; + // The file as the tier names it (a `data/<id>/…` path may be a link). + path: string; + name: string; + // The record seconds the file holds. + windowStart: number; + windowEnd: number; +}; + +// The corpus file holding [from, to] of a record, or null. `audio` admits the +// sound-only tier: always for an audio citation, and for a video citation of a +// record whose platform has no picture. +export async function resolveEvidenceSource(opts: { + channelsDir: string; + slug: string; + id: string; + span: EvidenceSpan; + audio: boolean; + ffprobeBin?: string; +}): Promise<EvidenceSource | null> { + const probe = (file: string) => ffprobeSource(file, opts.ffprobeBin); + const hit = await resolveCorpusSource( + { video: opts.id, slug: opts.slug, from: opts.span.from, to: opts.span.to }, + { channelsDir: opts.channelsDir, probe, audio: opts.audio }, + ); + if (!hit) return null; + return { + kind: hit.kind as EvidenceSourceKind, + path: hit.path, + name: hit.name, + windowStart: hit.windowStart, + windowEnd: hit.windowEnd, + }; +} + +// Whether a channel's records have no picture (a feed of episodes). +export function isAudioOnlyPlatform(platform: string | null | undefined): boolean { + return AUDIO_ONLY_PLATFORMS.has(String(platform ?? "").toLowerCase()); +} + +export type SourceIdentity = { path: string; bytes: number; mtimeMs: number }; + +// The source file's identity, through its links; null when it is not there. +export async function sourceIdentity(file: string): Promise<SourceIdentity | null> { + try { + const st = await stat(file); + if (!st.isFile()) return null; + return { path: file, bytes: st.size, mtimeMs: Math.round(st.mtimeMs) }; + } catch { + return null; + } +} + +// The cache key: source identity + span + output kind + profile. +export function evidenceHash(source: SourceIdentity, span: EvidenceSpan, kind: EvidenceKind): string { + const key = JSON.stringify([ + EVIDENCE_PROFILE, + kind, + source.path, + source.bytes, + source.mtimeMs, + span.from.toFixed(3), + span.to.toFixed(3), + ]); + return createHash("sha256").update(key).digest("hex").slice(0, 32); +} + +// The video filter: fit inside the box, never upscale, even sides. +const FIT_FILTER = + `scale='min(${EVIDENCE_MAX_WIDTH},iw)':'min(${EVIDENCE_MAX_HEIGHT},ih)'` + + ":force_original_aspect_ratio=decrease:force_divisible_by=2,format=yuv420p"; + +// The ffmpeg arguments that cut [a, b] seconds INTO `file` as `kind`, writing +// `out` (whose name need not carry an extension: the muxer is named). +export function evidenceCutArgs(file: string, a: number, b: number, kind: EvidenceKind, out: string): string[] { + const common = ["-map_metadata", "-1", "-sn", "-dn"]; + const audio = ["-c:a", "aac", "-b:a", EVIDENCE_AUDIO_BITRATE, "-ac", "2"]; + const tail = ["-movflags", "+faststart", "-f", "mp4", out]; + const head = ["-nostdin", "-v", "error", "-y", ...cutArgs(file, a, b)]; + if (kind === "audio") { + return [...head, "-map", "0:a:0", "-vn", ...audio, ...common, ...tail]; + } + return [ + ...head, + "-map", "0:v:0", + "-map", "0:a:0?", + "-vf", FIT_FILTER, + "-c:v", "libx264", "-preset", "medium", "-profile:v", "high", "-crf", String(EVIDENCE_CRF), + ...audio, + ...common, + ...tail, + ]; +} + +export type MediaProbe = { + durationSec: number | null; + width: number | null; + height: number | null; + hasVideo: boolean; + hasAudio: boolean; +}; + +// One ffprobe: the container's duration and its first picture's size, and +// which kinds of stream it has. Null when ffprobe cannot read it. An attached +// cover picture (`disposition.attached_pic`, an mp3's album art) is not a +// picture stream. +export async function probeMedia(file: string, ffprobeBin = "ffprobe"): Promise<MediaProbe | null> { + const r = await execa( + ffprobeBin, + [ + "-v", "error", + "-show_entries", "stream=codec_type,width,height:stream_disposition=attached_pic:format=duration", + "-of", "json", + file, + ], + { reject: false }, + ); + if (r.exitCode !== 0) return null; + let doc: { + streams?: { codec_type?: string; width?: number; height?: number; disposition?: { attached_pic?: number } }[]; + format?: { duration?: string }; + }; + try { + doc = JSON.parse(String(r.stdout)); + } catch { + return null; + } + const streams = doc.streams ?? []; + const video = streams.find((s) => s.codec_type === "video" && !s.disposition?.attached_pic); + const duration = Number(doc.format?.duration); + return { + durationSec: Number.isFinite(duration) && duration > 0 ? Number(duration.toFixed(3)) : null, + width: Number.isInteger(video?.width) ? (video!.width as number) : null, + height: Number.isInteger(video?.height) ? (video!.height as number) : null, + hasVideo: video !== undefined, + hasAudio: streams.some((s) => s.codec_type === "audio"), + }; +} + +export async function sha256File(file: string): Promise<string> { + const hash = createHash("sha256"); + for await (const chunk of createReadStream(file)) hash.update(chunk as Buffer); + return hash.digest("hex"); +} + +// One prepared clip, as the manifest lists it. `file` is relative to the cache +// directory. +export type EvidenceMedia = { + kind: EvidenceKind; + file: string; + bytes: number; + sha256: string; + width: number | null; + height: number | null; + durationSec: number | null; +}; + +type Sidecar = EvidenceMedia & { + profile: string; + span: EvidenceSpan; + source: { kind: EvidenceSourceKind; name: string }; +}; + +export type EvidenceClipResult = + | { ok: true; media: EvidenceMedia; cached: boolean; source: EvidenceSource } + | { + ok: false; + reason: "missing" | "no-audio" | "too-big" | "cut-failed"; + message: string; + bytes?: number; + }; + +async function readSidecar(file: string): Promise<Sidecar | null> { + try { + const raw = JSON.parse(await readFile(file, "utf8")) as Partial<Sidecar>; + if (raw.profile !== EVIDENCE_PROFILE || typeof raw.file !== "string") return null; + if (typeof raw.bytes !== "number" || typeof raw.sha256 !== "string") return null; + return raw as Sidecar; + } catch { + return null; + } +} + +export function sidecarPathFor(cacheDir: string, hash: string): string { + return path.join(cacheDir, `${hash}.json`); +} + +// Find, cut and cache the clip of one span of one record. +// +// `kind` is what the citation asks for; the clip is cut as audio when the +// citation is audio or when the source on disk has no picture. The result +// names the cache file (relative to `cacheDir`) — or why there is none: no +// media on disk (`missing`), a source with no sound to cut for an audio clip +// (`no-audio`), a clip over the limit (`too-big`, with its size), or an +// ffmpeg failure (`cut-failed`). +export async function prepareEvidenceClip(opts: { + channelsDir: string; + slug: string; + id: string; + kind: EvidenceKind; + span: EvidenceSpan; + cacheDir: string; + // The record's platform has no picture: a video citation may be served by + // its sound. + audioOnlyRecord?: boolean; + ffmpegBin?: string; + ffprobeBin?: string; + maxBytes?: number; + signal?: AbortSignal; +}): Promise<EvidenceClipResult> { + const ffmpegBin = opts.ffmpegBin ?? "ffmpeg"; + const ffprobeBin = opts.ffprobeBin ?? "ffprobe"; + const maxBytes = opts.maxBytes ?? EVIDENCE_MAX_BYTES; + const { span } = opts; + + const source = await resolveEvidenceSource({ + channelsDir: opts.channelsDir, + slug: opts.slug, + id: opts.id, + span, + audio: opts.kind === "audio" || opts.audioOnlyRecord === true, + ffprobeBin, + }); + const identity = source ? await sourceIdentity(source.path) : null; + if (!source || !identity) { + return { + ok: false, + reason: "missing", + message: `no media on disk holds ${span.from.toFixed(2)}–${span.to.toFixed(2)} s (no clip window, saved video${opts.kind === "audio" || opts.audioOnlyRecord ? " or audio" : ""} covers it)`, + }; + } + + // The output kind decides the hash, and a video citation over a sound-only + // source is an audio clip — so the source is probed first. An audio-tier + // source is sound by definition and is not probed. + let outKind: EvidenceKind = opts.kind; + if (outKind === "video") { + if (source.kind === "audio") outKind = "audio"; + else { + const p = await probeMedia(source.path, ffprobeBin); + if (p && !p.hasVideo) outKind = "audio"; + } + } + + const hash = evidenceHash(identity, span, outKind); + const name = `${hash}${EVIDENCE_EXT[outKind]}`; + const out = path.join(opts.cacheDir, name); + const sidecarFile = sidecarPathFor(opts.cacheDir, hash); + + // A hit: the sidecar describes a file of exactly its size. + const cached = await readSidecar(sidecarFile); + if (cached && cached.file === name) { + const st = await stat(out).catch(() => null); + if (st?.isFile() && st.size === cached.bytes) { + const { kind, file, bytes, sha256, width, height, durationSec } = cached; + return { ok: true, cached: true, source, media: { kind, file, bytes, sha256, width, height, durationSec } }; + } + } + + const a = Math.max(0, span.from - source.windowStart); + const b = Math.min(span.to, source.windowEnd) - source.windowStart; + await mkdir(opts.cacheDir, { recursive: true }); + const tmp = tmpPathFor(out); + try { + const r = await execa(ffmpegBin, evidenceCutArgs(source.path, a, b, outKind, tmp), { + reject: false, + timeout: CUT_TIMEOUT_MS, + cancelSignal: opts.signal, + }); + if (r.exitCode !== 0) { + const err = String(r.stderr ?? "").trim().split("\n").slice(-3).join(" | "); + if (outKind === "audio" && /matches no streams|does not contain any stream/i.test(err)) { + return { ok: false, reason: "no-audio", message: `${source.name} has no sound to cut` }; + } + return { + ok: false, + reason: "cut-failed", + message: `ffmpeg failed cutting ${source.name}${err ? `: ${err}` : ` (exit ${r.exitCode ?? "?"})`}`, + }; + } + const bytes = (await stat(tmp)).size; + if (bytes > maxBytes) { + return { + ok: false, + reason: "too-big", + bytes, + message: `the clip is ${(bytes / 1024 / 1024).toFixed(1)} MiB, over the ${(maxBytes / 1024 / 1024).toFixed(0)} MiB limit — shorten the span or its pad`, + }; + } + const probe = await probeMedia(tmp, ffprobeBin); + const media: EvidenceMedia = { + kind: outKind, + file: name, + bytes, + sha256: await sha256File(tmp), + width: outKind === "video" ? (probe?.width ?? null) : null, + height: outKind === "video" ? (probe?.height ?? null) : null, + durationSec: probe?.durationSec ?? null, + }; + await rename(tmp, out); + const sidecar: Sidecar = { + ...media, + profile: EVIDENCE_PROFILE, + span, + source: { kind: source.kind, name: source.name }, + }; + await writeJsonAtomic(sidecarFile, sidecar); + return { ok: true, cached: false, source, media }; + } finally { + await rm(tmp, { force: true }).catch(() => {}); + } +} diff --git a/common/lib/evidenceClip.mjs b/common/lib/evidenceClip.mjs @@ -0,0 +1,400 @@ +// evidenceClip.mjs — where a cited span's media already is in the corpus, and +// the cut that takes the span out of it. +// +// Two consumers ask the same question — "which file on this disk holds these +// seconds of this record?" — and must get the same answer: +// - umtool's report-to-video build (report-to-video/sources.mjs), which asks +// its own raw cache first and then these tiers; +// - the report-site prepare step (lib/evidenceClip-server.ts), which cuts a +// self-hosted evidence clip of every cited span. +// So the corpus half of the lookup lives HERE, once, and sources.mjs imports +// and re-exports it. Plain ESM with no app imports: umtool's scripts run under +// bare node, which cannot load a `.ts` file. +// +// THE CORPUS TIERS, in the order they are consulted: +// +// 1. corpus-window the editor's window cache, +// `channels/<slug>/data/<id>/clips/<from>-<to>.<ext>` +// (lib/clipWindow.ts) +// 2. saved-video the whole source container: the saved-video store's +// pointer (`data/<id>/saved-video.json` -> `<dir>/<file>`), +// else a `data/<id>/source-media.<ext>` not yet moved there. +// It is the window [0, its probed duration]. +// 3. audio `data/<id>/audio.<ext>`: the recording's SOUND alone, the +// window [0, its probed duration]. Consulted only when the +// caller allows it (`audio: true` on the tier): being last, +// it never beats a picture already on disk. +// +// A window qualifies only if it holds the REQUESTED span whole — the span plus +// its pad — to WIN_EPS. Within a tier the tightest wins; across tiers the order +// above wins. +// +// A file in `data/<id>/` may be a RELATIVE symlink into `media/`, which may be +// an absolute symlink to another drive. Every candidate is `stat`ed through its +// links before it is returned, and a dangling one — an unmounted drive — is +// "not here", never an error: the lookup falls through to the next tier. +// +// It never computes a path from the cwd: every root arrives as an argument +// (umtool's Next build bundles this file, and a cwd join makes its tracer walk +// the corpus). +import { execFile } from "node:child_process"; +import { readdir, readFile, stat } from "node:fs/promises"; +import path from "node:path"; +import { promisify } from "node:util"; + +const execFileP = promisify(execFile); + +/** + * Platforms whose records have no picture: a feed of episodes. A span of one + * is served by its sound. + */ +export const AUDIO_ONLY_PLATFORMS = new Set(["podcast", "feed", "rss"]); + +// A window read back from a 2 dp name can sit a hair outside the request that +// produced it; the same tolerance lib/clipWindow.ts and resolve-windows.mjs +// use, for the same reason. +export const WIN_EPS = 0.02; + +export const WINDOW_RE = /^(\d+(?:\.\d+)?)-(\d+(?:\.\d+)?)$/; + +/** + * A directory listing, or [] for a directory that is not there. + * @param {string} dir + * @returns {Promise<string[]>} + */ +export async function listNames(dir) { + try { + return await readdir(dir); + } catch { + return []; + } +} + +/** + * @typedef {{ name: string, path: string, from: number, to: number, height?: number }} SourceWindow + * @typedef {{ kind: string, path: string, name: string, windowStart: number, windowEnd: number, height?: number }} LocalSource + * @typedef {(file: string) => Promise<{ duration: number, height?: number } | null>} SourceProbe + * @typedef {{ video: string, slug?: string | null, channelsDir?: string | null, rawDir?: string | null, probe?: SourceProbe }} TierContext + * @typedef {{ kind: string, windows: (ctx: TierContext) => Promise<SourceWindow[]>, audio?: boolean }} SourceTier + */ + +/** + * Does this window hold [from, to] whole, to the tolerance? + * @param {{ from: number, to: number }} w + * @param {number} from + * @param {number} to + */ +export function windowContains(w, from, to) { + return !(w.from > from + WIN_EPS || w.to < to - WIN_EPS); +} + +/** + * The tightest of `windows` containing [from, to], or null. + * @template {{ from: number, to: number }} W + * @param {W[]} windows + * @param {number} from + * @param {number} to + * @returns {W | null} + */ +export function tightestContaining(windows, from, to) { + /** @type {W | null} */ + let best = null; + for (const w of windows) { + if (!windowContains(w, from, to)) continue; + if (!best || w.to - w.from < best.to - best.from) best = w; + } + return best; +} + +/** + * A video's directory in a channels tree: `<channelsDir>/<slug>/data/<id>`. + * @param {string} channelsDir + * @param {string} slug + * @param {string} video + */ +export function videoDirOf(channelsDir, slug, video) { + return path.join(/* turbopackIgnore: true */ channelsDir, slug, "data", video); +} + +/** + * Where the editor's fetch-window puts a video's windows (lib/clipWindow.ts). + * @param {string} videoDir + */ +export function corpusClipsDir(videoDir) { + return path.join(/* turbopackIgnore: true */ videoDir, "clips"); +} + +// The extensions a corpus window may wear -- lib/clipWindow.ts's +// CLIP_WINDOW_EXTS. The editor writes `.mp4`; the other two are what an older +// fetch left. Never `.json` (the sidecar) and never `.part.mp4` (in flight: +// its stem is not `a-b`, so the pattern refuses it). +export const CORPUS_WINDOW_EXTS = [".mp4", ".mkv", ".webm"]; + +/** + * The windows in a corpus clips dir: un-prefixed, because the directory is + * already per video -- `<from>-<to>.<ext>`. + * @param {string[]} names + * @param {string} dir + * @returns {SourceWindow[]} + */ +export function windowsFromBareNames(names, dir) { + /** @type {SourceWindow[]} */ + const out = []; + for (const name of names) { + const ext = CORPUS_WINDOW_EXTS.find((x) => name.endsWith(x)); + if (!ext) continue; + const m = WINDOW_RE.exec(name.slice(0, -ext.length)); + if (!m) continue; + const from = Number(m[1]); + const to = Number(m[2]); + if (!(to > from)) continue; + out.push({ name, path: path.join(/* turbopackIgnore: true */ dir, name), from, to }); + } + return out; +} + +export const SAVED_VIDEO_POINTER = "saved-video.json"; + +// lib/mediaFiles.ts's anchored `source-media.<ext>`, over its +// VIDEO_CONTAINER_EXTS: a picture is the point here, so an audio-only +// container is not a source for this tier. +const SOURCE_MEDIA_RE = /^source-media\.(?:mp4|webm|mkv|mov|m4v|ogv|avi)$/i; + +/** + * The saved-video store's pointer for a video, read the way lib/savedVideo.ts + * parses it (`dir` and `file`, both non-empty), or null. Re-implemented rather + * than imported: this runs under bare node. `height` is the format the persist + * recorded taking, when it recorded one. + * @param {string} videoDir + * @returns {Promise<{ dir: string, file: string, path: string, height?: number } | null>} + */ +export async function readSavedVideoPointer(videoDir) { + let raw; + try { + raw = JSON.parse(await readFile(path.join(/* turbopackIgnore: true */ videoDir, SAVED_VIDEO_POINTER), "utf8")); + } catch { + return null; + } + if (!raw || typeof raw !== "object") return null; + if (typeof raw.dir !== "string" || raw.dir === "") return null; + if (typeof raw.file !== "string" || raw.file === "") return null; + const height = Number(raw.format?.height); + return { + dir: raw.dir, + file: raw.file, + path: path.join(/* turbopackIgnore: true */ raw.dir, raw.file), + ...(Number.isInteger(height) && height > 0 ? { height } : {}), + }; +} + +/** + * The whole-source containers a video has: the store's (through the pointer) + * first, then any `source-media.<ext>` still in the video dir. Not yet checked + * for existence -- that is `present`'s job, once, for every tier. + * @param {string} videoDir + * @returns {Promise<{ name: string, path: string, height?: number }[]>} + */ +export async function wholeContainersOf(videoDir) { + /** @type {{ name: string, path: string, height?: number }[]} */ + const out = []; + const pointer = await readSavedVideoPointer(videoDir); + if (pointer) out.push({ name: pointer.file, path: pointer.path, height: pointer.height }); + const local = (await listNames(videoDir)).filter((n) => SOURCE_MEDIA_RE.test(n)).sort(); + for (const name of local) { + const p = path.join(/* turbopackIgnore: true */ videoDir, name); + if (!out.some((c) => c.path === p)) out.push({ name, path: p }); + } + return out; +} + +// The sound files a video dir may hold: lib/mediaFiles.ts's anchored +// `audio.<ext>` over AUDIO_EXTS, plus the `.webm`/`.mp4` an extract-to-mp3 that +// failed leaves behind (sound only). Read in AUDIO_PREFERENCE's order -- the +// file the transcribe path reads -- then by name. +const AUDIO_FILE_RE = /^audio\.(?:mp3|m4a|aac|ogg|oga|opus|wav|flac|webm|mp4)$/i; +export const AUDIO_PREFERENCE = ["audio.mp3", "audio.m4a", "audio.opus"]; + +/** + * A video dir's audio files, best first. Not yet checked for existence. + * @param {string} videoDir + * @returns {Promise<{ name: string, path: string }[]>} + */ +export async function audioFilesOf(videoDir) { + const names = (await listNames(videoDir)).filter((n) => AUDIO_FILE_RE.test(n)); + const rank = (/** @type {string} */ n) => { + const i = AUDIO_PREFERENCE.indexOf(n); + return i < 0 ? AUDIO_PREFERENCE.length : i; + }; + names.sort((a, b) => rank(a) - rank(b) || a.localeCompare(b)); + return names.map((name) => ({ name, path: path.join(/* turbopackIgnore: true */ videoDir, name) })); +} + +/** + * Is there a readable file at `p`, through every link on the way? A dangling + * link, a missing file and an unanswering drive are all "no", never a throw. + * @param {string} p + */ +export async function present(p) { + try { + return (await stat(p)).isFile(); + } catch { + return false; + } +} + +/** + * The default probe: one ffprobe for the container's duration and its first + * video stream's height. Null when ffprobe cannot read it -- which makes the + * container "not a source", not a failure. + * @param {string} file + * @param {string} [bin] the ffprobe binary (default: FFPROBE_BIN, else `ffprobe`) + * @returns {Promise<{ duration: number, height?: number } | null>} + */ +export async function ffprobeSource(file, bin = process.env.FFPROBE_BIN ?? "ffprobe") { + try { + const { stdout } = await execFileP(bin, [ + "-v", "error", "-select_streams", "v:0", + "-show_entries", "stream=height:format=duration", + "-of", "json", file, + ]); + const doc = JSON.parse(stdout); + const duration = Number(doc?.format?.duration); + if (!Number.isFinite(duration) || duration <= 0) return null; + const height = Number(doc?.streams?.[0]?.height); + return { duration, ...(Number.isInteger(height) && height > 0 ? { height } : {}) }; + } catch { + return null; + } +} + +// ---- the tiers --------------------------------------------------------------- +// Each lists a video's candidate windows `{name, path, from, to, height?}` for +// one context `{video, slug, channelsDir, probe}`. A tier that cannot apply (no +// channelsDir or slug) lists nothing. + +/** @param {TierContext} ctx */ +async function corpusWindows({ video, slug, channelsDir }) { + if (!channelsDir || !slug) return []; + const dir = corpusClipsDir(videoDirOf(channelsDir, slug, video)); + return windowsFromBareNames(await listNames(dir), dir); +} + +/** @param {TierContext} ctx */ +async function savedVideoWindows({ video, slug, channelsDir, probe = ffprobeSource }) { + if (!channelsDir || !slug) return []; + /** @type {SourceWindow[]} */ + const out = []; + for (const c of await wholeContainersOf(videoDirOf(channelsDir, slug, video))) { + // Existence BEFORE the probe: an unmounted drive must not cost a timeout. + if (!(await present(c.path))) continue; + const info = await probe(c.path); + const duration = Number(info?.duration); + // No duration is no entry rather than a guess: claiming a span a file may + // not cover is the one failure worse than a miss. + if (!Number.isFinite(duration) || duration <= 0) continue; + const height = c.height ?? info?.height; + out.push({ name: c.name, path: c.path, from: 0, to: duration, ...(height ? { height } : {}) }); + } + return out; +} + +// The ONE best audio file, not every one: they are all the same recording, and +// probing three formats of it would buy nothing. +/** @param {TierContext} ctx */ +async function audioWindows({ video, slug, channelsDir, probe = ffprobeSource }) { + if (!channelsDir || !slug) return []; + for (const f of await audioFilesOf(videoDirOf(channelsDir, slug, video))) { + if (!(await present(f.path))) continue; + const duration = Number((await probe(f.path))?.duration); + if (!Number.isFinite(duration) || duration <= 0) continue; + return [{ name: f.name, path: f.path, from: 0, to: duration }]; + } + return []; +} + +/** + * The corpus tiers, in order. `audio: true` marks a tier consulted only when + * the caller allows it. + * @type {SourceTier[]} + */ +export const CORPUS_TIERS = [ + { kind: "corpus-window", windows: corpusWindows }, + { kind: "saved-video", windows: savedVideoWindows }, + { kind: "audio", windows: audioWindows, audio: true }, +]; + +/** + * @param {string} kind + * @param {SourceWindow} w + * @returns {LocalSource} + */ +export const asSource = (kind, w) => ({ + kind, + path: w.path, + name: w.name, + windowStart: w.from, + windowEnd: w.to, + ...(w.height ? { height: w.height } : {}), +}); + +/** + * The first source among `tiers` holding [from, to] whole and present on disk. + * A tier is only listed when every tier before it missed, so the saved-video + * probe (an ffprobe, perhaps on a platter) is paid for only by a span nothing + * nearer could serve. + * + * @param {SourceTier[]} tiers + * @param {TierContext} ctx + * @param {number} from + * @param {number} to + * @param {{ allow?: (tier: SourceTier) => boolean, preferName?: (tier: SourceTier) => string | null }} [opts] + * `allow` skips a tier (default: every tier but an audio one); + * `preferName` names, per tier, a file that wins over the tightest. + * @returns {Promise<LocalSource | null>} + */ +export async function resolveFromTiers(tiers, ctx, from, to, opts = {}) { + const { allow = (tier) => !tier.audio, preferName = () => null } = opts; + for (const tier of tiers) { + if (!allow(tier)) continue; + const windows = (await tier.windows(ctx)).filter((w) => windowContains(w, from, to)); + const exactName = preferName(tier); + windows.sort((a, b) => + Number(b.name === exactName) - Number(a.name === exactName) || (a.to - a.from) - (b.to - b.from)); + for (const w of windows) { + if (await present(w.path)) return asSource(tier.kind, w); + } + } + return null; +} + +/** + * The corpus source for one span of one record, or null when nothing on disk + * holds it. + * + * @param {{ video: string, slug: string, from: number, to: number }} want + * `from`/`to` are the PADDED span. + * @param {{ channelsDir: string, probe?: SourceProbe, audio?: boolean }} config + * `audio` admits the audio tier (default false). + * @returns {Promise<LocalSource | null>} + */ +export async function resolveCorpusSource(want, config) { + const { video, slug, from, to } = want; + const { channelsDir, probe, audio = false } = config; + if (!video || !slug || !Number.isFinite(from) || !Number.isFinite(to)) return null; + return resolveFromTiers(CORPUS_TIERS, { video, slug, channelsDir, probe }, from, to, { + allow: (tier) => (tier.audio ? audio : true), + }); +} + +/** + * The input half of a cut of [a, b] seconds of `raw`: seek on the input, so a + * re-encode starts on exactly the asked-for frame. report-to-video's segment + * pass and the evidence cutter both use it, so the seconds a video renders and + * the seconds a moment page plays are the same arithmetic. + * @param {string} raw + * @param {number} a + * @param {number} b + */ +export function cutArgs(raw, a, b) { + return ["-ss", a.toFixed(3), "-to", b.toFixed(3), "-i", raw]; +} diff --git a/common/package.json b/common/package.json @@ -32,6 +32,7 @@ "./components/virtualizer": "./components/virtualizer.ts", "./components/*": "./components/*.tsx", "./lib/detectPlatform.mjs": "./lib/detectPlatform.mjs", + "./lib/evidenceClip.mjs": "./lib/evidenceClip.mjs", "./lib/ports.mjs": "./lib/ports.mjs", "./lib/report/verdicts.mjs": "./lib/report/verdicts.mjs", "./lib/toolProbe.mjs": "./lib/toolProbe.mjs", diff --git a/common/publish/citedPostCaptures.test.ts b/common/publish/citedPostCaptures.test.ts @@ -0,0 +1,132 @@ +// The one publish module that reads post captures copies the CITED ones and +// nothing else — the other half of the guard in the post capture tests. +// (The capture directory is reached through the module's own helper, so this +// file does not name it and needs no exception of its own.) +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test publish/citedPostCaptures.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { createHash } from "node:crypto"; +import { mkdir, mkdtemp, readdir, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { citedCaptureSourceDir, copyCitedPostCaptures, pngSize } from "./citedPostCaptures"; + +const ROOT = await mkdtemp(path.join(tmpdir(), "cited-captures-")); +after(() => rm(ROOT, { recursive: true, force: true })); + +// A PNG header for a w×h image (the bytes after IHDR's size do not matter here). +function png(w: number, h: number): Buffer { + const b = Buffer.alloc(33); + b.writeUInt32BE(0x89504e47, 0); + b.writeUInt32BE(0x0d0a1a0a, 4); + b.writeUInt32BE(13, 8); + b.write("IHDR", 12, "latin1"); + b.writeUInt32BE(w, 16); + b.writeUInt32BE(h, 20); + return b; +} + +async function capture(channelsDir: string, channel: string, id: string, files: Record<string, string | Buffer>) { + const dir = citedCaptureSourceDir(channelsDir, channel, id); + await mkdir(dir, { recursive: true }); + for (const [name, body] of Object.entries(files)) await writeFile(path.join(dir, name), body); +} + +async function tree(dir: string): Promise<string[]> { + const out: string[] = []; + const walk = async (d: string, rel: string) => { + for (const e of await readdir(d, { withFileTypes: true }).catch(() => [])) { + const r = rel ? `${rel}/${e.name}` : e.name; + if (e.isDirectory()) await walk(path.join(d, e.name), r); + else out.push(r); + } + }; + await walk(dir, ""); + return out.sort(); +} + +test("copies only the cited posts' screenshot and media — never the record, the article or an uncited post", async () => { + const channelsDir = path.join(ROOT, "a", "channels"); + const destRoot = path.join(ROOT, "a", "report-media"); + await capture(channelsDir, "demo-x", "111", { + "shot.png": png(600, 400), + "capture.json": "{}", + "111_1.jpg": "jpeg one", + "article.md": "an article", + "article-img-1.jpg": "inline", + "111_2.mp4.part": "partial", + }); + await capture(channelsDir, "demo-x", "222", { "shot.png": png(10, 10), "222_1.jpg": "not cited" }); + await capture(channelsDir, "other-x", "333", { "shot.png": png(10, 10) }); + + const r = await copyCitedPostCaptures({ channelsDir, destRoot, posts: [{ channel: "demo-x", id: "111" }] }); + assert.deepEqual(r.missing, []); + assert.deepEqual(await tree(destRoot), ["posts/demo-x/111/111_1.jpg", "posts/demo-x/111/shot.png"]); + assert.equal(r.copied.length, 1); + const c = r.copied[0]; + assert.equal(c.shot.file, "posts/demo-x/111/shot.png"); + assert.equal(c.shot.width, 600); + assert.equal(c.shot.height, 400); + assert.deepEqual(c.media, [ + { + file: "posts/demo-x/111/111_1.jpg", + bytes: "jpeg one".length, + sha256: createHash("sha256").update("jpeg one").digest("hex"), + }, + ]); +}); + +test("a later run with a different citation set leaves exactly that set", async () => { + const channelsDir = path.join(ROOT, "b", "channels"); + const destRoot = path.join(ROOT, "b", "report-media"); + await capture(channelsDir, "demo-x", "111", { "shot.png": png(1, 1), "111_1.jpg": "x" }); + await capture(channelsDir, "demo-x", "222", { "shot.png": png(1, 1) }); + await capture(channelsDir, "other-x", "333", { "shot.png": png(1, 1) }); + await copyCitedPostCaptures({ + channelsDir, + destRoot, + posts: [ + { channel: "demo-x", id: "111" }, + { channel: "other-x", id: "333" }, + ], + }); + // The capture lost its media since; 111 is no longer cited at all. + await rm(path.join(citedCaptureSourceDir(channelsDir, "demo-x", "111"), "111_1.jpg")); + await writeFile(path.join(destRoot, "posts", "stray.txt"), "left by hand"); + const r = await copyCitedPostCaptures({ channelsDir, destRoot, posts: [{ channel: "demo-x", id: "222" }] }); + assert.equal(r.copied.length, 1); + assert.deepEqual(await tree(destRoot), ["posts/demo-x/222/shot.png"]); +}); + +test("a cited post with no screenshot, or an id that is not one, is missing — and nothing is copied for it", async () => { + const channelsDir = path.join(ROOT, "c", "channels"); + const destRoot = path.join(ROOT, "c", "report-media"); + await capture(channelsDir, "demo-x", "111", { "111_1.jpg": "media only" }); + const r = await copyCitedPostCaptures({ + channelsDir, + destRoot, + posts: [ + { channel: "demo-x", id: "111" }, + { channel: "demo-x", id: "999" }, + { channel: "demo-x", id: "../escape" }, + ], + }); + assert.deepEqual(r.copied, []); + assert.deepEqual( + r.missing.map((m) => m.id), + ["111", "999", "../escape"], + ); + assert.match(r.missing[2].message, /is not a post id/); + assert.deepEqual(await tree(destRoot), []); +}); + +test("pngSize: the IHDR size, or null for anything else", async () => { + const f = path.join(ROOT, "x.png"); + await writeFile(f, png(1280, 720)); + assert.deepEqual(await pngSize(f), { width: 1280, height: 720 }); + await writeFile(f, "not a png at all, but long enough"); + assert.equal(await pngSize(f), null); + assert.equal(await pngSize(path.join(ROOT, "absent.png")), null); +}); diff --git a/common/publish/citedPostCaptures.ts b/common/publish/citedPostCaptures.ts @@ -0,0 +1,149 @@ +// THE ONE PUBLISH MODULE THAT READS A POST CAPTURE — and only the captures a +// report cites. +// +// A post capture (`channels/<slug>/posts-media/<id>/`: the post's screenshot, +// its attached media, its record) is editor-only material: nothing the export +// builds may carry it, and social/postCapture.test.ts holds every publish and +// export source to that by name. A report site is the one exception the plan +// makes (plans/report-sites.md, "Evidence media"): a cited post's moment page +// shows its screenshot and media. So the exception is THIS module, named in +// that guard, and it copies exactly the posts it is handed — the caller's +// cited list, already narrowed by the site's channel pool and the post +// visibility rule (lib/postsVisibility.ts) — and nothing else: +// +// - `shot.png` and the media files (`listCapturedMediaFiles`): never the +// record (`capture.json`), never the article half, never a leftover; +// - into `<destRoot>/posts/<channel>/<id>/`, each post's directory rebuilt +// from scratch, so a file the capture no longer has does not linger; +// - and every other directory under `<destRoot>/posts/` is REMOVED: what is +// there after a run is the cited set, whatever an earlier run cited. +// +// A post with no screenshot is missing (its moment page has nothing to show); +// the editor's Capture posts fills it. + +import { open, readdir, rm, stat } from "node:fs/promises"; +import path from "node:path"; +import { copyFileAtomic } from "../lib/jsonFile-server"; +import { + fileDigest, + listCapturedMediaFiles, + postCaptureDir, + postsMediaDir, + SHOT_FILENAME, +} from "../social/postCapture"; + +// The directory under the report-media cache that holds the copied captures. +export const REPORT_POSTS_DIRNAME = "posts"; + +export type CitedPost = { channel: string; id: string }; + +export type CopiedFile = { + // Relative to the report-media cache directory, `/`-separated. + file: string; + bytes: number; + sha256: string; +}; + +export type CopiedCapture = CitedPost & { + shot: CopiedFile & { width: number | null; height: number | null }; + media: CopiedFile[]; +}; + +export type MissingCapture = CitedPost & { message: string }; + +// A PNG's size from its header: the IHDR chunk is always first, its width and +// height big-endian at bytes 16 and 20. Null for anything that is not a PNG. +export async function pngSize(file: string): Promise<{ width: number; height: number } | null> { + const head = Buffer.alloc(24); + try { + const fh = await open(file, "r"); + try { + await fh.read(head, 0, 24, 0); + } finally { + await fh.close(); + } + } catch { + return null; + } + if (head.readUInt32BE(0) !== 0x89504e47 || head.toString("latin1", 12, 16) !== "IHDR") { + return null; + } + return { width: head.readUInt32BE(16), height: head.readUInt32BE(20) }; +} + +// Where a post's capture is: its channel's capture directory, the id checked +// (lib's postCaptureDir throws on one that is not a post id). +export function citedCaptureSourceDir(channelsDir: string, channel: string, id: string): string { + return postCaptureDir(postsMediaDir(path.join(channelsDir, channel)), id); +} + +const isFile = async (p: string) => (await stat(p).catch(() => null))?.isFile() === true; + +// Copy the cited posts' captures into `<destRoot>/posts/`, and remove every +// capture there that is not cited. `channelsDir` is the corpus's +// `channels/`. +export async function copyCitedPostCaptures(opts: { + channelsDir: string; + destRoot: string; + posts: readonly CitedPost[]; +}): Promise<{ copied: CopiedCapture[]; missing: MissingCapture[] }> { + const postsRoot = path.join(opts.destRoot, REPORT_POSTS_DIRNAME); + const copied: CopiedCapture[] = []; + const missing: MissingCapture[] = []; + const keep = new Set<string>(); + + for (const post of opts.posts) { + const key = `${post.channel}/${post.id}`; + if (keep.has(key)) continue; + keep.add(key); + let src: string; + try { + src = citedCaptureSourceDir(opts.channelsDir, post.channel, post.id); + } catch (e) { + missing.push({ ...post, message: (e as Error).message }); + continue; + } + const dest = path.join(postsRoot, post.channel, post.id); + await rm(dest, { recursive: true, force: true }); + if (!(await isFile(path.join(src, SHOT_FILENAME)))) { + missing.push({ ...post, message: "the post has no captured screenshot (capture it on the channel's Posts page)" }); + continue; + } + const copy = async (name: string): Promise<CopiedFile> => { + await copyFileAtomic(path.join(src, name), path.join(dest, name), { mkdir: true }); + const digest = await fileDigest(path.join(dest, name)); + return { file: path.posix.join(REPORT_POSTS_DIRNAME, post.channel, post.id, name), ...digest }; + }; + const shot = await copy(SHOT_FILENAME); + const size = await pngSize(path.join(dest, SHOT_FILENAME)); + const media: CopiedFile[] = []; + for (const name of await listCapturedMediaFiles(src)) { + // A file through its links; a directory or a dangling link is skipped. + if (await isFile(path.join(src, name))) media.push(await copy(name)); + } + copied.push({ + ...post, + shot: { ...shot, width: size?.width ?? null, height: size?.height ?? null }, + media, + }); + } + + // Everything else goes: an uncited post, an uncited channel. + for (const channel of await readdir(postsRoot).catch(() => [] as string[])) { + const channelDir = path.join(postsRoot, channel); + const ids = await readdir(channelDir).catch(() => null); + if (ids === null) { + await rm(channelDir, { recursive: true, force: true }); + continue; + } + let kept = 0; + for (const id of ids) { + if (keep.has(`${channel}/${id}`)) kept++; + else await rm(path.join(channelDir, id), { recursive: true, force: true }); + } + if (kept === 0) await rm(channelDir, { recursive: true, force: true }); + } + if (keep.size === 0) await rm(postsRoot, { recursive: true, force: true }); + + return { copied, missing }; +} diff --git a/common/publish/reportMedia.test.ts b/common/publish/reportMedia.test.ts @@ -0,0 +1,229 @@ +// `reports prepare` over a temp site: two reports' citations become clips and +// post captures in the site's report-media cache, with a manifest and every +// problem listed — through the real report checker, the real tier lookup and +// real ffmpeg over a few frames of lavfi. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test publish/reportMedia.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { execFileSync } from "node:child_process"; +import { existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +// Every path getPaths() can resolve to a place this file's code may write is +// pinned under ROOT before anything calls it. +const ROOT = mkdtempSync(path.join(tmpdir(), "reports-prepare-")); +Object.assign(process.env, { + TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), + SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), + SITES_DIR: path.join(ROOT, "transcripts", "sites"), + SETTINGS_FILE: path.join(ROOT, "settings.json"), + EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), + EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), + EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), + ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), +}); +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const { getPaths } = await import("../lib/paths"); +const { prepareReportMedia, readReportMediaIndex, reportMediaDir, citedMoments } = await import("./reportMedia"); +const { citedCaptureSourceDir } = await import("./citedPostCaptures"); +const { main: prepareMain } = await import("../bin/reports-prepare"); + +const paths = getPaths(); +const SITE = "demo-site"; +const CH = "demo-channel"; +const X = "demo-x"; + +const writeJson = (file: string, value: unknown) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value, null, 2)); +}; +const ff = (args: string[]) => execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", ...args]); +const videoDir = (slug: string, id: string) => path.join(paths.channelsDir, slug, "data", id); + +// A PNG header is all the copier reads of a screenshot. +function png(w: number, h: number): Buffer { + const b = Buffer.alloc(33); + b.writeUInt32BE(0x89504e47, 0); + b.writeUInt32BE(0x0d0a1a0a, 4); + b.writeUInt32BE(13, 8); + b.write("IHDR", 12, "latin1"); + b.writeUInt32BE(w, 16); + b.writeUInt32BE(h, 20); + return b; +} + +function report(id: string, citations: Record<string, unknown>) { + return { + format: "archilyzer-report", + version: 1, + id, + kind: "sweep", + title: `Report ${id}`, + citations, + sections: [ + { + id: "s1", + title: "One", + body: Object.keys(citations).map((c) => `[${c}](cite:${c})`).join(" "), + }, + ], + }; +} + +// The corpus: a video channel with a fetched window and a recording's sound, +// an X channel with two captured posts, and a channel no site has. +writeJson(path.join(paths.channelsDir, CH, "config.json"), { + handling: "transcribe", + name: "Demo", + url: "https://example.test/demo", +}); +writeJson(path.join(paths.channelsDir, X, "config.json"), { + handling: "transcribe", + name: "Demo (X)", + url: "https://x.com/demo", + sourceKind: "social", + platform: "twitter", +}); +mkdirSync(path.join(videoDir(CH, "abc123"), "clips"), { recursive: true }); +ff([ + "-f", "lavfi", "-i", "testsrc=size=160x90:rate=10:duration=6", + "-f", "lavfi", "-i", "sine=frequency=440:duration=6", + "-c:v", "libx264", "-preset", "ultrafast", "-pix_fmt", "yuv420p", "-c:a", "aac", "-shortest", + path.join(videoDir(CH, "abc123"), "clips", "0.00-6.00.mp4"), +]); +mkdirSync(videoDir(CH, "pod1"), { recursive: true }); +ff(["-f", "lavfi", "-i", "sine=frequency=220:duration=6", "-c:a", "aac", path.join(videoDir(CH, "pod1"), "audio.m4a")]); +for (const id of ["111", "222"]) { + const dir = citedCaptureSourceDir(paths.channelsDir, X, id); + mkdirSync(dir, { recursive: true }); + writeFileSync(path.join(dir, "shot.png"), png(600, 400)); + writeFileSync(path.join(dir, `${id}_1.jpg`), `media of ${id}`); + writeFileSync(path.join(dir, "capture.json"), "{}"); +} + +writeJson(path.join(paths.sitesDir, SITE, "site.json"), { + title: "Demo", + channels: [{ slug: CH }, { slug: X }], + publish: "cited", + reports: ["r1", "r2", "r-gone"], +}); +writeJson( + path.join(paths.sitesDir, SITE, "reports", "r1", "report.json"), + report("r1", { + c01: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { before: 0.5 }, quote: "one" }, + c02: { kind: "audio", channel: CH, id: "pod1", start: 1, end: 3, quote: "two" }, + p01: { kind: "post", channel: X, id: "111", quote: "three" }, + c03: { kind: "video", channel: CH, id: "missing1", start: 10, end: 12, quote: "four" }, + c04: { kind: "video", channel: "elsewhere", id: "xyz", start: 1, end: 2, quote: "five" }, + }), +); +// The same moment as r1's c01, with more context after it. +writeJson( + path.join(paths.sitesDir, SITE, "reports", "r2", "report.json"), + report("r2", { k1: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { after: 1 }, quote: "one" } }), +); +// A draft: in reports/, not in site.json. Never read. +writeJson( + path.join(paths.sitesDir, SITE, "reports", "draft", "report.json"), + report("draft", { d1: { kind: "post", channel: X, id: "222", quote: "uncited" } }), +); + +const PUBLIC = { social: { x: { visibility: "public" } } }; +const VIDEO_KEY = `${CH}/abc123/1.00-2.00`; +const AUDIO_KEY = `${CH}/pod1/1.00-3.00`; +const POST_KEY = `${X}/111`; + +test("citedMoments: one moment per span, cited by both reports, with the wider pad on each side", () => { + const moments = citedMoments([ + report("r1", { a: { kind: "video", channel: CH, id: "abc123", start: 1, end: 2, pad: { before: 0.5 }, quote: "q" } }), + report("r2", { b: { kind: "audio", channel: CH, id: "abc123", start: 1.001, end: 2, pad: { after: 1 }, quote: "q" } }), + ] as never); + assert.equal(moments.length, 1); + assert.equal(moments[0].key, VIDEO_KEY); + assert.equal(moments[0].kind, "video"); + assert.deepEqual(moments[0].pad, { before: 0.5, after: 1 }); + assert.deepEqual(moments[0].citedBy, ["r1#a", "r2#b"]); +}); + +test("prepare: clips, the cited capture, a manifest, and every problem", async () => { + const lines: string[] = []; + const index = await prepareReportMedia({ siteId: SITE, settings: PUBLIC, onLog: (l) => lines.push(l) }); + const dir = reportMediaDir(paths, SITE); + + assert.deepEqual(Object.keys(index.moments), [VIDEO_KEY, POST_KEY, AUDIO_KEY].sort()); + + const video = index.moments[VIDEO_KEY]; + assert.equal(video.kind, "video"); + assert.match(video.file, /^[0-9a-f]{32}\.mp4$/); + assert.equal(video.width, 160); + assert.equal(video.height, 90); + // 1–2 widened by r1's 0.5 before and r2's 1 after: 0.5–3. + assert.ok(Math.abs((video.durationSec ?? 0) - 2.5) < 0.15, `duration ${video.durationSec}`); + assert.equal(readFileSync(path.join(dir, video.file)).length, video.bytes); + + const audio = index.moments[AUDIO_KEY]; + assert.equal(audio.kind, "audio"); + assert.match(audio.file, /\.m4a$/); + assert.equal(audio.width, null); + + const post = index.moments[POST_KEY]; + assert.equal(post.kind, "post"); + assert.equal(post.file, `posts/${X}/111/shot.png`); + assert.equal(post.width, 600); + assert.ok(post.kind === "post" && post.media.map((m) => m.file).join() === `posts/${X}/111/111_1.jpg`); + // Only the cited post: not the draft's, not the record. + assert.deepEqual(readdirSync(path.join(dir, "posts", X)), ["111"]); + assert.deepEqual(readdirSync(path.join(dir, "posts", X, "111")).sort(), ["111_1.jpg", "shot.png"]); + + const byKind = (k: string) => index.problems.filter((p) => p.kind === k); + assert.equal(byKind("missing-report").length, 1); + assert.equal(byKind("missing-report")[0].report, "r-gone"); + assert.deepEqual(byKind("missing-media").map((p) => p.moment), [`${CH}/missing1/10.00-12.00`]); + assert.deepEqual(byKind("missing-media")[0].citations, ["r1#c03"]); + assert.deepEqual(byKind("not-in-site").map((p) => p.moment), ["elsewhere/xyz/1.00-2.00"]); + assert.equal(index.problems.length, 3); + + // The manifest on disk is what was returned. + assert.deepEqual(await readReportMediaIndex(paths, SITE), index); + assert.ok(lines.some((l) => l.includes(VIDEO_KEY) && l.includes("corpus-window"))); +}); + +test("a second run cuts nothing it already has; the CLI exits 1 and names each problem", async () => { + const first = await readReportMediaIndex(paths, SITE); + const out: string[] = []; + const err: string[] = []; + const code = await prepareMain({ siteId: SITE }, { log: (s) => out.push(s), error: (s) => err.push(s) }); + assert.equal(code, 1); + const second = await readReportMediaIndex(paths, SITE); + assert.equal(second?.moments[VIDEO_KEY].file, first?.moments[VIDEO_KEY].file); + assert.ok(out.some((l) => l.includes(VIDEO_KEY) && l.includes("cached"))); + assert.ok(err.some((l) => l.startsWith(" missing-media:") && l.includes("missing1") && l.includes("r1#c03"))); + assert.ok(err.some((l) => l.includes("missing-report"))); + assert.equal(await prepareMain({ siteId: "no-such-site" }, { log: () => {}, error: () => {} }), 2); +}); + +test("a post the site may not carry is a problem, and its capture leaves the cache", async () => { + const index = await prepareReportMedia({ siteId: SITE, settings: { social: { x: { visibility: "private" } } } }); + assert.equal(index.moments[POST_KEY], undefined); + assert.deepEqual( + index.problems.filter((p) => p.kind === "not-visible").map((p) => p.moment), + [POST_KEY], + ); + assert.equal(existsSync(path.join(reportMediaDir(paths, SITE), "posts", X)), false); +}); + +test("a clean site: no problems, and a clip no longer cited leaves the cache", async () => { + const site = path.join(paths.sitesDir, SITE, "site.json"); + writeJson(site, { title: "Demo", channels: [{ slug: CH }, { slug: X }], reports: ["r2"] }); + const index = await prepareReportMedia({ siteId: SITE, settings: PUBLIC }); + assert.deepEqual(index.problems, []); + assert.deepEqual(Object.keys(index.moments), [VIDEO_KEY]); + const files = readdirSync(reportMediaDir(paths, SITE)).sort(); + const clip = index.moments[VIDEO_KEY].file; + assert.deepEqual(files, [clip.replace(/\.mp4$/, ".json"), clip, "index.json"].sort()); + assert.equal(await prepareMain({ siteId: SITE }, { log: () => {}, error: () => {} }), 0); +}); diff --git a/common/publish/reportMedia.ts b/common/publish/reportMedia.ts @@ -0,0 +1,370 @@ +// PREPARE A SITE'S EVIDENCE MEDIA — every clip and post capture its published +// reports cite, cut and copied on the host, before the site's build +// (plans/report-sites.md, "Evidence media"). `archilyzer reports prepare +// <siteId>` and the editor's `reports-prepare` job both run `prepareReportMedia`. +// +// What it reads: the site's `reports` (site.json, in order), each +// `sites/<siteId>/reports/<reportId>/report.json`, parsed and validated by the +// report document's own checker (lib/report/validate.ts). What it does, per +// cited MOMENT (lib/citations/moments.ts — two citations of one span share one +// page and so one clip, cut with the wider of their pads): +// +// video / audio span lib/evidenceClip-server.ts finds the span's media on +// disk (clip window, saved container, audio) and cuts it +// into the cache, or answers why it cannot; +// post ./citedPostCaptures.ts copies the post's screenshot +// and media — only cited posts — into the cache. +// +// A citation must resolve against the site's own `channels` (the pool a cited +// site's citations may draw on), and a post must be one the site may carry +// (lib/postsVisibility.ts — a public site never shows a private platform's +// posts, cited or not). +// +// What it writes: `.export-index/sites/<siteId>/report-media/` — +// `<hash>.mp4` / `<hash>.m4a` (+ `<hash>.json`) the clips +// `posts/<channel>/<id>/…` the cited captures +// `index.json` the manifest: +// { format, version, siteId, preparedAt, +// moments: { <momentKey>: { kind, file, bytes, sha256, width, height, +// durationSec, media? } }, +// problems: [ { kind, message, report?, citations?, moment?, path? } ] } +// — and nothing else: a clip or capture no moment names is removed, so the +// cache holds exactly what the site cites. The build copies from here and +// never runs ffmpeg (it has none). +// +// A RUN WITH PROBLEMS STILL WRITES THE MANIFEST (the editor lists the problems +// from it) and FAILS: a citation without media, a clip over the size limit, an +// invalid or missing report. Nothing here fetches — the editor's fetch-window +// and persist (for spans) and Capture posts (for posts) fill what is missing. +// +// AN UNMOUNTED DRIVE IS NOT A MISSING CLIP. A span whose media is not found on +// a channel whose media tier is not reachable (lib/channelMedia.ts) is +// reported as unreachable, with the reason, rather than as missing. + +import { readdir, rm } from "node:fs/promises"; +import path from "node:path"; +import { readJsonFile, writeJsonAtomic } from "../lib/jsonFile-server"; +import { getPaths, type Paths } from "../lib/paths"; +import { getSettings } from "../lib/settings"; +import { getSite, listSiteIds, siteChannelSlugs, siteDir, siteIndexDir, type Site } from "../lib/site"; +import { postsVisibleTo } from "../lib/postsVisibility"; +import { inspectChannelMedia } from "../lib/channelMedia"; +import type { ChannelConfig } from "../lib/channelConfig"; +import { readChannelConfig } from "../controller/channels"; +import { momentKey, momentOf, momentProblem, type Moment } from "../lib/citations/moments"; +import type { CitationPad } from "../lib/citations/schema"; +import { parseReport } from "../lib/report/validate"; +import type { Report } from "../lib/report/schema"; +import { + evidenceSpan, + isAudioOnlyPlatform, + prepareEvidenceClip, + widerPad, + type EvidenceKind, + type EvidenceMedia, +} from "../lib/evidenceClip-server"; +import { copyCitedPostCaptures, REPORT_POSTS_DIRNAME, type CopiedFile } from "./citedPostCaptures"; + +export const REPORT_MEDIA_FORMAT = "archilyzer-report-media"; +export const REPORT_MEDIA_VERSION = 1; +export const REPORT_MEDIA_DIRNAME = "report-media"; +export const REPORT_MEDIA_INDEX_FILENAME = "index.json"; + +// `.export-index/sites/<siteId>/report-media/`: the prepared media, not served. +export function reportMediaDir(paths: Paths, siteId: string): string { + return path.join(siteIndexDir(paths, siteId), REPORT_MEDIA_DIRNAME); +} + +export function reportMediaIndexFile(paths: Paths, siteId: string): string { + return path.join(reportMediaDir(paths, siteId), REPORT_MEDIA_INDEX_FILENAME); +} + +// `sites/<siteId>/reports/<reportId>/`: a report's directory. +export function siteReportDir(paths: Paths, siteId: string, reportId: string): string { + return path.join(siteDir(paths, siteId), "reports", reportId); +} + +export function siteReportFile(paths: Paths, siteId: string, reportId: string): string { + return path.join(siteReportDir(paths, siteId, reportId), "report.json"); +} + +export type ReportMediaEntry = + | EvidenceMedia + | { + kind: "post"; + // The screenshot. + file: string; + bytes: number; + sha256: string; + width: number | null; + height: number | null; + durationSec: null; + // The post's attached media, copied beside it. + media: CopiedFile[]; + }; + +export type ReportMediaProblemKind = + | "missing-report" + | "invalid-report" + | "not-in-site" + | "not-visible" + | "missing-media" + | "unreachable" + | "too-big" + | "cut-failed"; + +export type ReportMediaProblem = { + kind: ReportMediaProblemKind; + message: string; + report?: string; + // `<reportId>#<citationId>` for every citation of the moment. + citations?: string[]; + moment?: string; + // A JSON path in the report (an invalid report's problems). + path?: string; +}; + +export type ReportMediaIndex = { + format: typeof REPORT_MEDIA_FORMAT; + version: typeof REPORT_MEDIA_VERSION; + siteId: string; + preparedAt: string; + moments: Record<string, ReportMediaEntry>; + problems: ReportMediaProblem[]; +}; + +// One cited moment and everything that cites it. +type CitedMoment = { + key: string; + moment: Moment; + // A span cited as video anywhere is a video clip; else audio. + kind: EvidenceKind | "post"; + pad?: CitationPad; + citedBy: string[]; +}; + +// The published reports of a site, read and validated. A report that does not +// parse contributes its problems and no moments; one that parses with value +// problems contributes both — its media is still prepared, so fixing the +// document does not wait on a re-cut. +export async function loadSiteReports( + paths: Paths, + site: Site, +): Promise<{ reports: Report[]; problems: ReportMediaProblem[] }> { + const reports: Report[] = []; + const problems: ReportMediaProblem[] = []; + for (const id of site.reports ?? []) { + const file = siteReportFile(paths, site.siteId, id); + const read = await readJsonFile(file); + if (!read.ok) { + problems.push({ + kind: read.reason === "absent" ? "missing-report" : "invalid-report", + report: id, + message: + read.reason === "absent" + ? `the site lists report "${id}", but sites/${site.siteId}/reports/${id}/report.json does not exist` + : `sites/${site.siteId}/reports/${id}/report.json is not readable JSON`, + }); + continue; + } + const parsed = parseReport(read.value, { id }); + for (const p of parsed.problems) { + problems.push({ kind: "invalid-report", report: id, path: p.path, message: p.message }); + } + if (parsed.ok) reports.push(parsed.value); + } + return { reports, problems }; +} + +// Every moment the reports cite, keyed, in first-cited order. +export function citedMoments(reports: readonly Report[]): CitedMoment[] { + const byKey = new Map<string, CitedMoment>(); + for (const report of reports) { + for (const [cid, c] of Object.entries(report.citations ?? {})) { + const moment = momentOf(c); + // A kind without a page, or one validation already names as unsafe. + if (!moment || momentProblem(moment)) continue; + const key = momentKey(moment); + const kind: CitedMoment["kind"] = c.kind === "post" ? "post" : c.kind === "audio" ? "audio" : "video"; + const pad = c.kind === "video" || c.kind === "audio" ? c.pad : undefined; + const seen = byKey.get(key); + if (!seen) { + byKey.set(key, { key, moment, kind, pad, citedBy: [`${report.id}#${cid}`] }); + continue; + } + seen.citedBy.push(`${report.id}#${cid}`); + seen.pad = widerPad(seen.pad, pad); + if (kind === "video") seen.kind = "video"; + } + } + return [...byKey.values()]; +} + +export type PrepareReportMediaOptions = { + siteId: string; + paths?: Paths; + onLog?: (line: string) => void; + signal?: AbortSignal; + // `social.x.visibility` and friends; default the live settings. + settings?: { social?: { x?: { visibility?: unknown } } }; + now?: () => Date; +}; + +const mib = (n: number) => `${(n / 1024 / 1024).toFixed(1)} MiB`; + +export async function prepareReportMedia(opts: PrepareReportMediaOptions): Promise<ReportMediaIndex> { + const paths = opts.paths ?? getPaths(); + const log = opts.onLog ?? (() => {}); + const { siteId } = opts; + if (!listSiteIds(paths).includes(siteId)) { + throw new Error(`no site "${siteId}" (no sites/${siteId}/site.json)`); + } + const site = getSite(siteId, paths); + const settings = opts.settings ?? getSettings(); + const cacheDir = reportMediaDir(paths, siteId); + + const { reports, problems } = await loadSiteReports(paths, site); + log(`${siteId}: ${site.reports?.length ?? 0} published report(s), ${reports.length} readable.`); + const moments = citedMoments(reports); + log(`${moments.length} cited moment(s) with media.`); + + const pool = siteChannelSlugs(site); + const configs = new Map<string, ChannelConfig | null>(); + const configOf = async (slug: string) => { + if (!configs.has(slug)) configs.set(slug, await readChannelConfig(paths, slug)); + return configs.get(slug) ?? null; + }; + const entries: Record<string, ReportMediaEntry> = {}; + const fail = (m: CitedMoment, kind: ReportMediaProblemKind, message: string) => { + problems.push({ kind, moment: m.key, citations: m.citedBy, message }); + log(` ✗ ${m.key}: ${message}`); + }; + + const posts: CitedMoment[] = []; + for (const m of moments) { + if (opts.signal?.aborted) throw new Error("prepare cancelled"); + const slug = m.moment.channel; + if (!pool.has(slug)) { + fail(m, "not-in-site", `channel "${slug}" is not one of this site's channels`); + continue; + } + const config = await configOf(slug); + if (m.kind === "post") { + if (!postsVisibleTo(site, config, settings)) { + fail(m, "not-visible", `this site may not carry posts of "${slug}" (the post visibility rule)`); + continue; + } + posts.push(m); + continue; + } + if (m.moment.kind !== "span") continue; + const span = evidenceSpan({ start: m.moment.start, end: m.moment.end, pad: m.pad }); + const r = await prepareEvidenceClip({ + channelsDir: paths.channelsDir, + slug, + id: m.moment.id, + kind: m.kind, + span, + cacheDir, + audioOnlyRecord: isAudioOnlyPlatform(config?.platform), + ffmpegBin: paths.ffmpegBin, + ffprobeBin: paths.ffprobeBin, + signal: opts.signal, + }); + if (r.ok) { + entries[m.key] = r.media; + log( + ` ${r.cached ? "=" : "+"} ${m.key} ← ${r.source.kind} ${r.source.name}: ` + + `${r.media.file} (${mib(r.media.bytes)}${r.cached ? ", cached" : ""})`, + ); + continue; + } + if (r.reason === "missing") { + const where = await inspectChannelMedia(paths, slug, config, { fresh: true }).catch(() => null); + if (where && where.status !== "ok" && where.status !== "in-place") { + fail(m, "unreachable", `${r.message}; the channel's media is ${where.status}${where.detail ? ` (${where.detail})` : ""}`); + continue; + } + fail(m, "missing-media", `${r.message} — fetch the window or persist the video in the editor`); + continue; + } + fail(m, r.reason === "too-big" ? "too-big" : r.reason === "no-audio" ? "missing-media" : "cut-failed", r.message); + } + + if (opts.signal?.aborted) throw new Error("prepare cancelled"); + const copiedPosts = await copyCitedPostCaptures({ + channelsDir: paths.channelsDir, + destRoot: cacheDir, + posts: posts.map((m) => ({ channel: m.moment.channel, id: m.moment.id })), + }); + const postMoment = new Map(posts.map((m) => [m.key, m])); + for (const c of copiedPosts.copied) { + const key = `${c.channel}/${c.id}`; + entries[key] = { kind: "post", ...c.shot, durationSec: null, media: c.media }; + log(` + ${key}: screenshot${c.media.length ? ` and ${c.media.length} media file(s)` : ""}`); + } + for (const miss of copiedPosts.missing) { + const m = postMoment.get(`${miss.channel}/${miss.id}`); + if (m) fail(m, "missing-media", miss.message); + } + + // The cache holds what the manifest names: a clip no moment names goes. + await pruneClips(cacheDir, entries); + + const index: ReportMediaIndex = { + format: REPORT_MEDIA_FORMAT, + version: REPORT_MEDIA_VERSION, + siteId, + preparedAt: (opts.now?.() ?? new Date()).toISOString(), + moments: Object.fromEntries(Object.keys(entries).sort().map((k) => [k, entries[k]])), + problems, + }; + await writeJsonAtomic(reportMediaIndexFile(paths, siteId), index, { mkdir: true }); + const bytes = Object.values(entries).reduce( + (n, e) => n + e.bytes + (e.kind === "post" ? e.media.reduce((s, f) => s + f.bytes, 0) : 0), + 0, + ); + log( + `${Object.keys(entries).length} of ${moments.length} moment(s) prepared (${mib(bytes)}); ` + + `${problems.length} problem(s).`, + ); + return index; +} + +const CLIP_FILE_RE = /^[0-9a-f]{32}\.(?:mp4|m4a|json)$/; + +async function pruneClips(cacheDir: string, entries: Record<string, ReportMediaEntry>): Promise<void> { + const keep = new Set<string>([REPORT_MEDIA_INDEX_FILENAME, REPORT_POSTS_DIRNAME]); + for (const e of Object.values(entries)) { + if (e.kind === "post") continue; + keep.add(e.file); + keep.add(e.file.replace(/\.[^.]+$/, ".json")); + } + for (const name of await readdir(cacheDir).catch(() => [] as string[])) { + if (keep.has(name)) continue; + // Only what this module writes: a clip, its sidecar, a temp file of either. + if (CLIP_FILE_RE.test(name) || /^[0-9a-f]{32}\.(?:mp4|m4a|json)\.tmp-/.test(name)) { + await rm(path.join(cacheDir, name), { force: true }); + } + } +} + +// The problems as lines, for a log or a terminal. +export function formatReportMediaProblems(problems: readonly ReportMediaProblem[]): string[] { + return problems.map((p) => { + const where = p.moment + ? `${p.moment}${p.citations?.length ? ` (cited by ${p.citations.join(", ")})` : ""}` + : `${p.report ?? "?"}${p.path ? ` at ${p.path}` : ""}`; + return `${p.kind}: ${where}: ${p.message}`; + }); +} + +// Read a prepared manifest, or null when there is none (never prepared, or +// unreadable). +export async function readReportMediaIndex(paths: Paths, siteId: string): Promise<ReportMediaIndex | null> { + const read = await readJsonFile(reportMediaIndexFile(paths, siteId)); + if (!read.ok) return null; + const v = read.value as Partial<ReportMediaIndex> | null; + if (!v || v.format !== REPORT_MEDIA_FORMAT || v.version !== REPORT_MEDIA_VERSION) return null; + return v as ReportMediaIndex; +} diff --git a/common/social/postCapture.test.ts b/common/social/postCapture.test.ts @@ -167,12 +167,16 @@ test("article files are named by the layout, images numbered from 1", () => { assert.ok(!isArticleFile("articles.txt")); }); -// THE EXPORT NEVER PUBLISHES A CAPTURE. The export serves the index's JSON -// pages and copies trees out of export/public — it never reads a channel -// directory — so nothing it builds can carry posts-media/. Held here the cheap -// way: no module of the export build, the publish layer or the export app -// names the directory. -test("no export or publish source names posts-media/", async () => { +// THE EXPORT NEVER PUBLISHES A CAPTURE — but the one a report cites. The +// export serves the index's JSON pages and copies trees out of export/public — +// it never reads a channel directory — so nothing it builds can carry +// posts-media/. Held here the cheap way: no module of the export build, the +// publish layer or the export app names the directory, except EXACTLY the one +// that copies a report site's CITED captures (publish/citedPostCaptures.ts, +// whose own test proves it copies nothing else). +const CITED_CAPTURE_COPIER = path.join("common", "publish", "citedPostCaptures.ts"); + +test("no export or publish source names posts-media/ but the cited-capture copier", async () => { const HERE = path.dirname(fileURLToPath(import.meta.url)); const repo = path.resolve(HERE, "..", ".."); const roots = [ @@ -182,6 +186,7 @@ test("no export or publish source names posts-media/", async () => { path.join(repo, "export", "lib"), ]; const offenders: string[] = []; + const allowed: string[] = []; const walk = async (dir: string): Promise<void> => { let entries; try { @@ -196,11 +201,15 @@ test("no export or publish source names posts-media/", async () => { else if (/\.(ts|tsx|mjs|js)$/.test(e.name)) { const text = await readFile(p, "utf8"); if (text.includes(POSTS_MEDIA_DIRNAME) || text.includes("POSTS_MEDIA_DIRNAME") || text.includes("social/postCapture")) { - offenders.push(path.relative(repo, p)); + const rel = path.relative(repo, p); + if (rel === CITED_CAPTURE_COPIER) allowed.push(rel); + else offenders.push(rel); } } } }; for (const r of roots) await walk(r); assert.deepEqual(offenders, []); + // The exception exists and is the only one: if it moves, this moves with it. + assert.deepEqual(allowed, [CITED_CAPTURE_COPIER]); }); diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -3,6 +3,7 @@ ## [Unreleased] - **A site can say what it publishes, and which reports.** `site.json` takes `publish` — `"full"`, the searchable corpus every site has been (the default, never written), or `"cited"`, only the site's reports and the moments they cite — and `reports`, the ordered ids of its published reports (each a slug; invalid and repeated ids are dropped). The site form has a Publish control and lists the site's reports read-only; saving the form keeps the stored list. A cited site still builds as a full one until the reports pipeline applies the scope. SITE.md documents both keys. - **A report and its citations now have one written format, checked before anything is built from them.** A cited report is a `report.json` (`archilyzer-report`, version 1): a summary, then sections of claims, each claim with an optional verdict, the reviewed document's own sentence, findings in markdown and the citations it rests on. A citation is one of five kinds — a span of a video, a span of an audio record, a post, a sentence of a source document, or a web page — with a verbatim quote, and is cited from any markdown in the report as `[label](cite:<id>)`. The checker lists every problem at once with where it is: a citation, a source or a `cite:` link that names nothing, a span that ends before it starts or runs past 120 seconds with its context, a still that points outside the report's folder, an id used twice. Each cited span and post has one page address, `/m/<channel>/<id>/<start>-<end>/` or `/m/<channel>/<id>/`. Nothing builds or shows reports yet. The fact-check verdicts (Corroborated, Partly true, Contradicted, Not found, Untestable) and their colours are now kept in one place, which the report video's stamps and tally read too. `REPORT.md` and `CITATIONS.md` list every key. +- **A site's reports can have their evidence media prepared: every cited span cut to a clip, every cited post's capture copied.** `archilyzer reports prepare <site>`, the `reports-prepare` job (`POST /api/ops/reports-prepare`, `pnpm ops reports-prepare`) reads the site's published reports, checks them, and for each cited moment cuts the span (with its context) out of the media already on disk — a fetched clip window, the saved video, or the recording's audio — fitted inside 1280×720 with H.264 and AAC, or as an `.m4a` for an audio span; and copies the screenshot and attached media of each cited post, and of no other post, beside them. Everything lands in the site's build staging (`.export-index/sites/<site>/report-media/`) with an `index.json` naming each moment's file, size, checksum, size in pixels and duration. A clip is cut once and reused while its source file and span are unchanged; a clip or capture no longer cited is removed. Nothing is fetched: a citation whose media is not on disk, a post without a screenshot, a clip over 24 MiB, a citation of a channel outside the site, a post the site may not show, or an invalid or missing report is listed with the citations it affects, and the run fails (exit 1, or a failed job) — a span on a drive that is not mounted is reported as such rather than as missing. The lookup of a span's media on disk is now shared with report-to-video, which finds the same files it did. - **A long report video no longer runs out of memory while its clips are crossfaded.** `build-video.mjs` used to join every segment of a cut in one ffmpeg command, which grows with the number of segments: a cut of a few hundred clips could use more memory than the machine had and be stopped. Past 24 segments the build now crossfades them in batches of consecutive segments, each into a file under `out/<variant>/xfade-batches/`, then crossfades those files together with the same transition and lays the on-screen deck, the rail and the dips over them. Every transition, the deck's schedule and the chapters land on the same frames as before, at the cost of one more video encode on such a cut. The batch size is `render.xfadeBatch` in the manifest or `REPORT_VIDEO_XFADE_BATCH` in the environment (which wins); `0` never batches. A batch file is reused while its segments, their holds and moves and the encode settings are unchanged, so a `--chrome-only` run that moves no footage redoes only the last pass. - **Auto-download no longer tries a video the metadata scan already found members-only or private.** The scan records why it could not read a video, but only a failed download used to take a video out of the auto-download queue, so each members-only video the scan had found was still downloaded once: four yt-dlp requests, two with browser cookies, about 30 seconds each. A members-only or private answer from the scan now keeps the video out of the queue and counts it under the channel's members-only or private exclusions, and it stays in **Needs cookies** for a manual cookie run. A scan that reads the video later lifts this. A video the scan saw only as "Video unavailable" is still tried, since YouTube gives that answer when it is throttling too. - **X post fetches stop on Drain, and an account with no posts is not searched.** Draining a `fetch-posts` job used to do nothing until gallery-dl finished its whole run. Now the timeline fetch stops at the next page boundary (at once when gallery-dl is between pages or waiting out a rate limit), the older-posts walk stops its current window's search at once and never starts the 45–120 second pause between windows, and both keep their resume point: the job ends done, not failed, and the log says "Drained; the next run resumes …". An older-posts walk is refused when nothing is archived and the last timeline fetch finished having read no posts, since it would only repeat empty searches; `"force": true` (`--force` on `archilyzer posts fetch`) walks anyway. A walk with nothing archived that finds nothing ends after two empty three-month windows instead of four, and records why; a walk that has posts keeps the year-of-empty-windows rule. Capture-posts already stopped between posts on Drain. diff --git a/editor/app/api/ops/reports-prepare/route.test.ts b/editor/app/api/ops/reports-prepare/route.test.ts @@ -0,0 +1,49 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, rm } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +// Run with: +// pnpm -C editor exec tsx --test "app/api/ops/reports-prepare/route.test.ts" +// +// The body's shape and the refusals that come before any job, answered from an +// empty temp corpus: no job is queued and no media is touched. + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "reports-prepare-route-")); +// Set before the route (and getPaths, which caches) is first imported. +process.env.WORKER_TOKEN = "test-token"; +process.env.TRANSCRIPTS_DIR = ROOT; +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); +const { POST } = await import("./route"); +test.after(() => rm(ROOT, { recursive: true, force: true })); + +async function post(body: Record<string, unknown>): Promise<{ status: number; error: string }> { + const res = await POST( + new Request("http://localhost/api/ops/reports-prepare", { + method: "POST", + headers: { + authorization: "Bearer test-token", + "content-type": "application/json", + }, + body: JSON.stringify(body), + }), + ); + return { status: res.status, error: ((await res.json()) as { error?: string }).error ?? "" }; +} + +test("siteId is required, a site id, and the only key", async () => { + assert.match((await post({})).error, /"siteId" is required/); + const bad = await post({ siteId: "../x" }); + assert.equal(bad.status, 400); + assert.match(bad.error, /not a valid site id/); + const unknown = await post({ siteId: "demo-site", reportId: "r1" }); + assert.equal(unknown.status, 400); + assert.match(unknown.error, /unknown key\(s\): reportId — this route accepts siteId/); +}); + +test("a site that does not exist is refused before any job", async () => { + const r = await post({ siteId: "demo-site" }); + assert.equal(r.status, 400); + assert.match(r.error, /No site "demo-site"/); +}); diff --git a/editor/app/api/ops/reports-prepare/route.ts b/editor/app/api/ops/reports-prepare/route.ts @@ -0,0 +1,23 @@ +import { reportsPrepareAction } from "../../../sites/lib/reportsPrepareAction"; +import { jobResponse, OpsInputError, ops, reqString } from "../_lib"; +import { isValidSiteId } from "yt-dlp-transcript-common/lib/site"; + +export const dynamic = "force-dynamic"; + +// POST { siteId: string } -> { ok: true, jobId } +// +// An adapter: one call to the action the site's Reports tab posts. Cuts every +// clip and copies every post capture the site's published reports cite into +// its report-media cache, as one `reports-prepare` job on its own queue. The +// job fails, naming each citation, when any lacks its media. +export async function POST(request: Request) { + return ops(request, ["siteId"], async (body) => { + const siteId = reqString(body, "siteId"); + if (!isValidSiteId(siteId)) { + throw new OpsInputError( + `"${siteId}" is not a valid site id (lowercase letters, digits and "-"; must start with a letter or digit)`, + ); + } + return jobResponse(await reportsPrepareAction(siteId)); + }); +} diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -54,6 +54,7 @@ import { redownloadShortAudioBucketAction, } from "../channels/[slug]/incompleteTranscriptActions"; import { replayFetchWindowAction } from "../channels/[slug]/videos/[id]/videoActions"; +import { reportsPrepareAction } from "../sites/lib/reportsPrepareAction"; export type ReplayHandler = (spec: JobSpec) => Promise<StreamActionResult>; @@ -89,6 +90,13 @@ function params(spec: JobSpec): { } export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { + // A report site's evidence media. The spec's slug is the SITE (the job has + // no channel); a re-run re-reads the site's reports and cuts only what its + // cache does not already hold. + "reports-prepare": (spec) => { + const { p } = params(spec); + return reportsPrepareAction(str(p.siteId) ?? spec.slug); + }, // A clip window sourced for another tool. Replay RE-DERIVES from disk like // every bucket job does: if the window (or a wider one covering it) has // arrived since, the retry says so neutrally rather than paying twice. diff --git a/editor/app/sites/lib/reportsPrepareAction.ts b/editor/app/sites/lib/reportsPrepareAction.ts @@ -0,0 +1,51 @@ +"use server"; + +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { isValidSiteId, listSiteIds } from "yt-dlp-transcript-common/lib/site"; +import { + runManagedFunction, + type StreamActionResult, +} from "yt-dlp-transcript-common/jobs/streamCommand"; +import { + formatReportMediaProblems, + prepareReportMedia, +} from "yt-dlp-transcript-common/publish/reportMedia"; + +// Its own queue: one prepare at a time, waiting on no download and no build. +const REPORTS_PREPARE_QUEUE = "reports-prepare"; + +// PREPARE A REPORT SITE'S EVIDENCE MEDIA as a job: every clip its published +// reports cite, cut from the media on disk, and every cited post capture, +// copied into `.export-index/sites/<siteId>/report-media/` before its build. +// The work is publish/reportMedia.ts's, the same `archilyzer reports prepare` +// runs. The job FAILS when any citation lacks its media — the log names each +// one — and the manifest it writes carries the same list. +// +// The spec's `slug` is the SITE id (a spec needs one, and this job belongs to +// no channel); the replay handler reads `params.siteId`. +export async function reportsPrepareAction( + siteId: string, +): Promise<StreamActionResult> { + const paths = getPaths(); + const id = siteId.trim(); + if (!isValidSiteId(id)) { + return { ok: false, error: `"${id}" is not a valid site id` }; + } + if (!listSiteIds(paths).includes(id)) { + return { ok: false, error: `No site "${id}"` }; + } + return runManagedFunction({ + kind: "reports-prepare", + queueKey: REPORTS_PREPARE_QUEUE, + paths, + spec: { kind: "reports-prepare", slug: id, params: { siteId: id } }, + fn: async (onLog, signal) => { + const index = await prepareReportMedia({ siteId: id, paths, onLog, signal }); + if (index.problems.length === 0) return; + for (const line of formatReportMediaProblems(index.problems)) onLog(line); + throw new Error( + `${index.problems.length} problem(s): the site's evidence media is not complete`, + ); + }, + }); +} diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -39,6 +39,7 @@ // pnpm ops deploy-hub --wait // pnpm ops build-homepage --json '{"deploy":true}' --wait // pnpm ops deploy-homepage --json '{"preview":"refresh"}' --wait +// pnpm ops reports-prepare --json '{"siteId":"demo-site"}' --wait // pnpm ops get channel the-quartering // pnpm ops tags --json '{"op":"define","tag":{"id":"eva-collab","label":"Collab"}}' // pnpm ops tag-videos --file ids.json @@ -124,6 +125,9 @@ const ACTIONS = [ "relocate", "relocate-back", "evict-clips", + // A report site's evidence media: cut every cited clip and copy every cited + // post capture into the site's report-media cache ({siteId}). + "reports-prepare", "lane", // The curated-tag writers. `tags` edits the vocabulary (define/remove); // `tag-videos` pins, unpins, suppresses or unsuppresses one tag over a batch @@ -317,6 +321,10 @@ export function usage() { 'build-site, build-deploy and deploy-site all take "siteId" (one) or', ' "siteIds" (a list).', "", + 'reports-prepare cuts every clip and copies every post capture a site\'s', + ' published reports cite into its report-media cache, before its build:', + ' {"siteId"}. The job fails, naming each one, when a citation lacks media.', + "", 'retry-bucket runs one bucket of a channel\'s report as one job, past any', ' lane hold: {"slug", "bucket"}. "ids": [...] runs only those videos, and', " every one must be in the bucket (a stray id is refused, named); a job run", diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -408,6 +408,16 @@ test("capture-posts is a POST to its route, named in the usage", () => { assert.match(usage(), /unless\s+"articles": false/); }); +// A report site's evidence media: a POST to its route, the body passed through +// untouched — the route judges the site id. +test("reports-prepare is a POST to its route, named in the usage", () => { + const p = parseArgs(["reports-prepare", "--json", '{"siteId":"demo-site"}']); + assert.equal(p.method, "POST"); + assert.equal(p.path, "/api/ops/reports-prepare"); + assert.deepEqual(p.body, { siteId: "demo-site" }); + assert.match(usage(), /reports-prepare cuts every clip/); +}); + test("persist-videos is a POST to its route, named in the usage", () => { const p = parseArgs([ "persist-videos", diff --git a/umtool/report-to-video/build-video.mjs b/umtool/report-to-video/build-video.mjs @@ -92,6 +92,7 @@ import { cardWidth, contentWidth, reservedFooterHeight, } from "./render-cards.mjs"; import { DEFAULT_CHANNELS_DIR, createCueSource, siteOriginFromManifest } from "./cues.mjs"; +import { cutArgs } from "yt-dlp-transcript-common/lib/evidenceClip.mjs"; // Where a clip's media is ALREADY on disk -- the build's raw cache, the // editor's corpus windows, the saved source -- asked before anything fetches. import { @@ -648,9 +649,9 @@ export function clipFetchArgs({ url, from, to, fmt, dest, extra = [] }) { * @param {number} a seconds INTO that file where the clip starts * @param {number} b seconds into it where the clip ends */ -export function cutArgs(raw, a, b) { - return ["-ss", a.toFixed(3), "-to", b.toFixed(3), "-i", raw]; -} +// Common's (lib/evidenceClip.mjs) since a report site cuts its evidence clips +// with the same arguments: re-exported, never re-spelled. +export { cutArgs }; /** * The span a clip's source is needed over: its extent, widened by the fetch diff --git a/umtool/report-to-video/sources.mjs b/umtool/report-to-video/sources.mjs @@ -46,27 +46,52 @@ // links before it is returned, and a dangling one -- an unmounted drive -- is // "not here", never an error: the build falls through to the next tier. // +// THE CORPUS HALF IS COMMON'S (common/lib/evidenceClip.mjs): tiers 2-4, the +// window predicates and the probe live there, once, because the report-site +// prepare step (common/lib/evidenceClip-server.ts) asks the same question of +// the same disk and must get the same answer. This module adds the build's own +// raw cache in front of them and re-exports the rest, so every importer here +// keeps its names. +// // A LEAF. Plain ESM with no imports from build-video.mjs or umtool's lib, so // the build, the clip bench (lib/projects/report.mjs) and lib/report/raw-cache // can all import it without a cycle, and so the bench and the render cannot // disagree about what is local. It never computes a path from the cwd: every // root arrives as an argument (see umtool/lib/paths.mjs on why that matters to // the Next build). -import { execFile } from "node:child_process"; -import { readdir, readFile, stat } from "node:fs/promises"; import path from "node:path"; -import { promisify } from "node:util"; - -const execFileP = promisify(execFile); +import { + AUDIO_ONLY_PLATFORMS, + CORPUS_TIERS, + WINDOW_RE, + listNames, + present, + resolveFromTiers, + windowContains, + tightestContaining, + ffprobeSource, +} from "yt-dlp-transcript-common/lib/evidenceClip.mjs"; + +export { + AUDIO_ONLY_PLATFORMS, + AUDIO_PREFERENCE, + CORPUS_WINDOW_EXTS, + SAVED_VIDEO_POINTER, + WIN_EPS, + audioFilesOf, + corpusClipsDir, + ffprobeSource, + present, + readSavedVideoPointer, + tightestContaining, + videoDirOf, + wholeContainersOf, + windowContains, + windowsFromBareNames, +} from "yt-dlp-transcript-common/lib/evidenceClip.mjs"; /** The source kinds, in the order they are consulted. */ -export const SOURCE_KINDS = ["raw-cache", "corpus-window", "saved-video", "audio"]; - -/** - * Platforms whose records have no picture: a feed of episodes. A clip from one - * reads its local audio BEFORE the network, since a fetch has no video to get. - */ -export const AUDIO_ONLY_PLATFORMS = new Set(["podcast", "feed", "rss"]); +export const SOURCE_KINDS = ["raw-cache", ...CORPUS_TIERS.map((t) => t.kind)]; /** * May the audio tier serve this clip, and as what? (See the note at the top.) @@ -91,13 +116,6 @@ export function audioUse({ entry = null, meta = null, render = null, network = t /** audioUse's yes or no: should the audio tier be consulted at all? */ export const audioAllowed = (args = {}) => audioUse(args) !== null; -// A window read back from a 2 dp name can sit a hair outside the request that -// produced it; the same tolerance resolve-windows.mjs and common's -// lib/clipWindow.ts use, for the same reason. -export const WIN_EPS = 0.02; - -const WINDOW_RE = /^(\d+(?:\.\d+)?)-(\d+(?:\.\d+)?)$/; - // ---- the build's raw cache (tier 1) ---------------------------------------- // A raw clip's window is IN ITS NAME, which makes the file immutable and the // cache content-addressed. TWO DECIMALS, always: the name is a function of the @@ -109,13 +127,7 @@ export function rawWindowName(video, from, to) { } /** A directory listing, or [] for a directory that is not there. */ -export async function listRawNames(rawDir) { - try { - return await readdir(rawDir); - } catch { - return []; - } -} +export const listRawNames = listNames; /** The windows THIS video's files hold, parsed out of a clips-raw listing. */ export function windowsFromNames(names, rawDir, video) { @@ -132,21 +144,6 @@ export function windowsFromNames(names, rawDir, video) { return out; } -/** Does this window hold [from, to] whole, to the manifest's tolerance? */ -export function windowContains(w, from, to) { - return !(w.from > from + WIN_EPS || w.to < to - WIN_EPS); -} - -/** The tightest of `windows` containing [from, to], or null. */ -export function tightestContaining(windows, from, to) { - let best = null; - for (const w of windows) { - if (!windowContains(w, from, to)) continue; - if (!best || w.to - w.from < best.to - best.from) best = w; - } - return best; -} - export async function cachedWindowsFor(rawDir, video) { return windowsFromNames(await listRawNames(rawDir), rawDir, video); } @@ -156,145 +153,6 @@ export async function findContainingWindow(rawDir, video, from, to) { return tightestContaining(await cachedWindowsFor(rawDir, video), from, to); } -// ---- the corpus (tiers 2 and 3) -------------------------------------------- - -/** A video's directory in a channels tree: `<channelsDir>/<slug>/data/<id>`. */ -export function videoDirOf(channelsDir, slug, video) { - return path.join(/* turbopackIgnore: true */ channelsDir, slug, "data", video); -} - -/** Where the editor's fetch-window puts a video's windows (lib/clipWindow.ts). */ -export function corpusClipsDir(videoDir) { - return path.join(/* turbopackIgnore: true */ videoDir, "clips"); -} - -// The extensions a corpus window may wear -- common/lib/clipWindow.ts's -// CLIP_WINDOW_EXTS. The editor writes `.mp4`; the other two are what an older -// fetch left. Never `.json` (the sidecar) and never `.part.mp4` (in flight: -// its stem is not `a-b`, so the pattern below refuses it). -export const CORPUS_WINDOW_EXTS = [".mp4", ".mkv", ".webm"]; - -/** - * The windows in a corpus clips dir: un-prefixed, because the directory is - * already per video -- `<from>-<to>.<ext>`. The same numbers as clips-raw's - * names, one predicate over both. - */ -export function windowsFromBareNames(names, dir) { - const out = []; - for (const name of names) { - const ext = CORPUS_WINDOW_EXTS.find((x) => name.endsWith(x)); - if (!ext) continue; - const m = WINDOW_RE.exec(name.slice(0, -ext.length)); - if (!m) continue; - const from = Number(m[1]); - const to = Number(m[2]); - if (!(to > from)) continue; - out.push({ name, path: path.join(/* turbopackIgnore: true */ dir, name), from, to }); - } - return out; -} - -export const SAVED_VIDEO_POINTER = "saved-video.json"; - -// common/lib/mediaFiles.ts's anchored `source-media.<ext>`, over its -// VIDEO_CONTAINER_EXTS: a picture is the point here, so an audio-only -// container is not a source for this tier. -const SOURCE_MEDIA_RE = /^source-media\.(?:mp4|webm|mkv|mov|m4v|ogv|avi)$/i; - -/** - * The saved-video store's pointer for a video, read the way - * common/lib/savedVideo.ts parses it (`dir` and `file`, both non-empty), or - * null. Re-implemented rather than imported: this runs under bare node. - * `height` is the format the persist recorded taking, when it recorded one. - */ -export async function readSavedVideoPointer(videoDir) { - let raw; - try { - raw = JSON.parse(await readFile(path.join(/* turbopackIgnore: true */ videoDir, SAVED_VIDEO_POINTER), "utf8")); - } catch { - return null; - } - if (!raw || typeof raw !== "object") return null; - if (typeof raw.dir !== "string" || raw.dir === "") return null; - if (typeof raw.file !== "string" || raw.file === "") return null; - const height = Number(raw.format?.height); - return { - dir: raw.dir, - file: raw.file, - path: path.join(/* turbopackIgnore: true */ raw.dir, raw.file), - ...(Number.isInteger(height) && height > 0 ? { height } : {}), - }; -} - -/** - * The whole-source containers a video has: the store's (through the pointer) - * first, then any `source-media.<ext>` still in the video dir. Not yet checked - * for existence -- that is `present`'s job, once, for every tier. - */ -export async function wholeContainersOf(videoDir) { - const out = []; - const pointer = await readSavedVideoPointer(videoDir); - if (pointer) out.push({ name: pointer.file, path: pointer.path, height: pointer.height }); - const local = (await listRawNames(videoDir)).filter((n) => SOURCE_MEDIA_RE.test(n)).sort(); - for (const name of local) { - const p = path.join(/* turbopackIgnore: true */ videoDir, name); - if (!out.some((c) => c.path === p)) out.push({ name, path: p }); - } - return out; -} - -// The sound files a video dir may hold: common/lib/mediaFiles.ts's anchored -// `audio.<ext>` over AUDIO_EXTS, plus the `.webm`/`.mp4` an extract-to-mp3 that -// failed leaves behind (sound only). Read in AUDIO_READ_PREFERENCE's order -- -// the file the transcribe path reads -- then by name. -const AUDIO_FILE_RE = /^audio\.(?:mp3|m4a|aac|ogg|oga|opus|wav|flac|webm|mp4)$/i; -export const AUDIO_PREFERENCE = ["audio.mp3", "audio.m4a", "audio.opus"]; - -/** A video dir's audio files, best first. Not yet checked for existence. */ -export async function audioFilesOf(videoDir) { - const names = (await listRawNames(videoDir)).filter((n) => AUDIO_FILE_RE.test(n)); - const rank = (n) => { - const i = AUDIO_PREFERENCE.indexOf(n); - return i < 0 ? AUDIO_PREFERENCE.length : i; - }; - names.sort((a, b) => rank(a) - rank(b) || a.localeCompare(b)); - return names.map((name) => ({ name, path: path.join(/* turbopackIgnore: true */ videoDir, name) })); -} - -/** - * Is there a readable file at `p`, through every link on the way? A dangling - * link, a missing file and an unanswering drive are all "no", never a throw. - */ -export async function present(p) { - try { - return (await stat(p)).isFile(); - } catch { - return false; - } -} - -/** - * The default probe: one ffprobe for the container's duration and its first - * video stream's height. Null when ffprobe cannot read it -- which makes the - * container "not a source", not a failed build. - */ -export async function ffprobeSource(file) { - try { - const { stdout } = await execFileP(process.env.FFPROBE_BIN ?? "ffprobe", [ - "-v", "error", "-select_streams", "v:0", - "-show_entries", "stream=height:format=duration", - "-of", "json", file, - ]); - const doc = JSON.parse(stdout); - const duration = Number(doc?.format?.duration); - if (!Number.isFinite(duration) || duration <= 0) return null; - const height = Number(doc?.streams?.[0]?.height); - return { duration, ...(Number.isInteger(height) && height > 0 ? { height } : {}) }; - } catch { - return null; - } -} - // ---- the tiers --------------------------------------------------------------- // Each lists a video's candidate windows `{name, path, from, to, height?}` for // one context `{video, slug, rawDir, channelsDir, probe}`. A tier that cannot @@ -305,58 +163,8 @@ async function rawCacheWindows({ video, rawDir }) { return cachedWindowsFor(rawDir, video); } -async function corpusWindows({ video, slug, channelsDir }) { - if (!channelsDir || !slug) return []; - const dir = corpusClipsDir(videoDirOf(channelsDir, slug, video)); - return windowsFromBareNames(await listRawNames(dir), dir); -} - -async function savedVideoWindows({ video, slug, channelsDir, probe = ffprobeSource }) { - if (!channelsDir || !slug) return []; - const out = []; - for (const c of await wholeContainersOf(videoDirOf(channelsDir, slug, video))) { - // Existence BEFORE the probe: an unmounted drive must not cost a timeout. - if (!(await present(c.path))) continue; - const info = await probe(c.path); - const duration = Number(info?.duration); - // No duration is no entry rather than a guess: claiming a span a file may - // not cover is the one failure worse than a miss. - if (!Number.isFinite(duration) || duration <= 0) continue; - const height = c.height ?? info?.height; - out.push({ name: c.name, path: c.path, from: 0, to: duration, ...(height ? { height } : {}) }); - } - return out; -} - -// The ONE best audio file, not every one: they are all the same recording, and -// probing three formats of it would buy nothing. -async function audioWindows({ video, slug, channelsDir, probe = ffprobeSource }) { - if (!channelsDir || !slug) return []; - for (const f of await audioFilesOf(videoDirOf(channelsDir, slug, video))) { - if (!(await present(f.path))) continue; - const duration = Number((await probe(f.path))?.duration); - if (!Number.isFinite(duration) || duration <= 0) continue; - return [{ name: f.name, path: f.path, from: 0, to: duration }]; - } - return []; -} - // `audio: true` marks a tier consulted only when the caller allows it. -export const TIERS = [ - { kind: "raw-cache", windows: rawCacheWindows }, - { kind: "corpus-window", windows: corpusWindows }, - { kind: "saved-video", windows: savedVideoWindows }, - { kind: "audio", windows: audioWindows, audio: true }, -]; - -const asSource = (kind, w) => ({ - kind, - path: w.path, - name: w.name, - windowStart: w.from, - windowEnd: w.to, - ...(w.height ? { height: w.height } : {}), -}); +export const TIERS = [{ kind: "raw-cache", windows: rawCacheWindows }, ...CORPUS_TIERS]; /** * The local source for one span of one video, or null when nothing on disk @@ -382,26 +190,18 @@ export async function resolveLocalSource(want, config = {}) { if (rawDir) { const name = rawWindowName(video, from, to); const p = path.join(/* turbopackIgnore: true */ rawDir, name); - if (await present(p)) return asSource("raw-cache", { name, path: p, from, to }); + if (await present(p)) return { kind: "raw-cache", path: p, name, windowStart: from, windowEnd: to }; } // `--no-reuse` re-cuts no cached window, but the sound is not one. if (!audio) return null; } - const ctx = { video, slug, rawDir, channelsDir, probe }; - for (const tier of TIERS) { - if (kinds && !kinds.includes(tier.kind)) continue; - if (tier.audio ? !audio : exact) continue; - const windows = (await tier.windows(ctx)).filter((w) => windowContains(w, from, to)); - // The raw cache's EXACT name first, as it always was; then tightest. - const exactName = tier.kind === "raw-cache" ? rawWindowName(video, from, to) : null; - windows.sort((a, b) => - (b.name === exactName) - (a.name === exactName) || (a.to - a.from) - (b.to - b.from)); - for (const w of windows) { - if (await present(w.path)) return asSource(tier.kind, w); - } - } - return null; + // The raw cache's EXACT name first, as it always was; then tightest. + const exactName = rawWindowName(video, from, to); + return resolveFromTiers(TIERS, { video, slug, rawDir, channelsDir, probe }, from, to, { + allow: (tier) => (!kinds || kinds.includes(tier.kind)) && (tier.audio ? audio : !exact), + preferName: (tier) => (tier.kind === "raw-cache" ? exactName : null), + }); } /** @@ -419,8 +219,8 @@ export async function resolveLocalSource(want, config = {}) { export async function corpusWindowsOf({ video, slug, channelsDir, probe = ffprobeSource }) { const ctx = { video, slug, channelsDir, probe }; const out = []; - for (const tier of TIERS) { - if (tier.kind === "raw-cache" || tier.audio) continue; + for (const tier of CORPUS_TIERS) { + if (tier.audio) continue; for (const w of await tier.windows(ctx)) { if (await present(w.path)) out.push({ ...w, kind: tier.kind }); }