// `archilyzer run [ids…]` over a temp corpus. // // Run with: // pnpm --filter yt-dlp-transcript-common test // // Its own file for the SETTINGS SEAM, like operationBatchRelocation.test.ts: // getPaths() memoizes its first answer, so TRANSCRIPTS_DIR and SETTINGS_FILE are // set before anything imports it. Settings are re-read on every call, so each // test writes the file it needs. // // Every test that could start a job carries a timeout. A lane whose gate is // shut HOLDS rather than failing: with only unref'd poll timers left, node // cancels the remaining tests (exit 1), and the timeout bounds the case where // something else keeps the loop alive. // // The operation that RUNS here is diarization over videos that have a // transcript and no audio: each classifies `missing-input` (with re-download // off), which is counted and never dispatched — no engine, no model, no // network — and still goes through the whole job: the record, the log, the // batch, the summary line and the snapshot. import { mkdtempSync, readdirSync, readFileSync, writeFileSync } from "node:fs"; import { mkdir, rm, symlink, writeFile } from "node:fs/promises"; import os from "node:os"; import path from "node:path"; import { test, after } from "node:test"; import assert from "node:assert/strict"; const ROOT = mkdtempSync(path.join(os.tmpdir(), "run-operation-")); process.env.TRANSCRIPTS_DIR = ROOT; const SETTINGS_FILE = path.join(ROOT, "settings.json"); process.env.SETTINGS_FILE = SETTINGS_FILE; const { runOperation, offlineRefusal } = await import("./run-operation"); const { getPaths } = await import("../lib/paths"); const { operationCatalog } = await import("../lib/operations"); after(() => rm(ROOT, { recursive: true, force: true })); function settings(over: Record = {}): void { writeFileSync( SETTINGS_FILE, JSON.stringify({ autoQueue: { backfill: { enabled: true, held: false } }, backfill: { allowRedownload: false, concurrency: 1 }, diarization: { enabled: true, segModel: "/models/seg.onnx", embModel: "/models/emb.onnx" }, attribution: { enabled: false, diarizedEnabled: false, textOnlyEnabled: false }, ...over, }), ); } async function seed(slug: string, ids: string[]): Promise { const channelDir = path.join(getPaths().channelsDir, slug); await mkdir(channelDir, { recursive: true }); await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ handling: "transcribe", url: "https://example.com/c" }), ); for (const id of ids) { const videoDir = path.join(channelDir, "data", id); await mkdir(videoDir, { recursive: true }); await writeFile( path.join(videoDir, "transcript.json"), JSON.stringify({ transcription: [{ text: "hello", offsets: {} }] }), ); } } function capture() { const o = { stdout: "", stderr: "" }; return { o, out: { log: (s: string) => (o.stdout += `${s}\n`), error: (s: string) => (o.stderr += `${s}\n`), write: (s: string) => (o.stdout += s), }, }; } const jobRecords = () => { try { return readdirSync(getPaths().jobsDir).filter((n) => n.endsWith(".json")); } catch { return []; } }; test("an unknown operation is refused with the list", async () => { const { o, out } = capture(); const code = await runOperation({ operation: "diarisation", channel: "x", ids: [] }, { out }); assert.equal(code, 2); assert.match(o.stderr, /no operation "diarisation"/); assert.match(o.stderr, /Runs here: .*diarization.*digest/); assert.match(o.stderr, /run only by the editor: .*sync.*transcription/); }); test("sync, the scan, downloads and transcription are refused with a sentence, not half-run", async () => { for (const id of ["sync", "metadata-scan", "download"]) { const { o, out } = capture(); assert.equal(await runOperation({ operation: id, channel: "x", ids: [] }, { out }), 1, id); assert.match(o.stderr, /download queue inside the editor/, id); } const { o, out } = capture(); assert.equal(await runOperation({ operation: "transcription", channel: "x", ids: [] }, { out }), 1); assert.match(o.stderr, /worker pool/); // Every catalogued operation has an answer, and exactly the registry's run. const runs = operationCatalog().filter((op) => offlineRefusal(op) === null).map((op) => op.id); assert.deepEqual(runs.sort(), ["attribution-diarized", "attribution-text", "diarization", "digest"]); }); test("an unknown channel, a switched-off operation, a paused lane and a stray --lane are refused", { timeout: 60_000 }, async () => { settings(); await seed("known", ["v1"]); let c = capture(); assert.equal(await runOperation({ operation: "diarization", channel: "nope", ids: [] }, { out: c.out }), 2); assert.match(c.o.stderr, /no channel "nope"/); c = capture(); assert.equal(await runOperation({ operation: "attribution-text", channel: "known", ids: [] }, { out: c.out }), 1); assert.match(c.o.stderr, /switched off in settings\.json/); settings({ autoQueue: { backfill: { enabled: true, held: true } } }); c = capture(); assert.equal(await runOperation({ operation: "diarization", channel: "known", ids: [] }, { out: c.out }), 1); assert.match(c.o.stderr, /backfill lane is paused/); settings(); c = capture(); assert.equal( await runOperation({ operation: "diarization", channel: "known", ids: [], lane: "local" }, { out: c.out }), 2, ); assert.match(c.o.stderr, /--lane is the digest engine lane/); }); test("the media guard refuses an unmounted channel before any job exists", { timeout: 60_000 }, async () => { settings(); const slug = "moved"; const channelDir = path.join(getPaths().channelsDir, slug); await mkdir(channelDir, { recursive: true }); const target = path.join(ROOT, "platter", slug, "media"); // never created await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ handling: "transcribe", url: "https://example.com/m", mediaDir: target }), ); await mkdir(path.join(channelDir, "data")); await symlink(target, path.join(channelDir, "media")); const before = jobRecords(); const { o, out } = capture(); const code = await runOperation({ operation: "diarization", channel: slug, ids: [] }, { out }); assert.equal(code, 1); assert.match(o.stderr, /moved/); assert.match(o.stderr, /not mounted|unreachable|does not exist/); assert.deepEqual(jobRecords(), before, "no job record for a refused run"); }); test("diarization runs through the editor's job body: record, log, summary, snapshot", { timeout: 60_000 }, async () => { settings(); await seed("chan", ["v1", "v2"]); const before = new Set(jobRecords()); const { o, out } = capture(); const code = await runOperation({ operation: "diarization", channel: "chan", ids: [] }, { out }); assert.equal(code, 0, o.stderr + o.stdout); assert.match(o.stdout, /Backfill chan: 0 done, 0 already current, 0 failed; 2 still need their media re-acquired/); const added = jobRecords().filter((n) => !before.has(n)); assert.equal(added.length >= 1, true, "a job record was written"); const records = added.map((n) => JSON.parse(readFileSync(path.join(getPaths().jobsDir, n), "utf8"))); const job = records.find((r) => r.kind === "backfill-channel"); assert.ok(job, JSON.stringify(records)); assert.equal(job.channelSlug, "chan"); assert.equal(job.status, "done"); assert.deepEqual(job.spec?.params?.kindIds, ["diarization"]); // The report the editor would have refreshed at job end. assert.ok( readdirSync(path.join(getPaths().channelsDir, "chan")).some((n) => n.startsWith("snapshot")), "the channel snapshot was regenerated", ); }); test("ids scope the run, and ids with no data dir are named", { timeout: 60_000 }, async () => { settings(); await seed("scoped", ["v1", "v2", "v3"]); const { o, out } = capture(); const code = await runOperation( { operation: "diarization", channel: "scoped", ids: ["v2", "gone", "v2"] }, { out }, ); assert.equal(code, 0, o.stderr); assert.match(o.stderr, /1 of 2 id\(s\) have no data\/\/ on disk and are skipped: gone/); assert.match(o.stdout, /Backfill scoped: 0 done, 0 already current, 0 failed; 1 still need their media re-acquired/); }); test("ids none of which is on disk are refused, and no job is started", { timeout: 60_000 }, async () => { settings(); await seed("empty-ids", ["v1"]); const before = jobRecords(); const { o, out } = capture(); const code = await runOperation({ operation: "diarization", channel: "empty-ids", ids: ["gone", "also-gone"] }, { out }); assert.equal(code, 1); assert.match(o.stderr, /none of the 2 id\(s\) has a data\/\/ on disk/); assert.deepEqual(jobRecords(), before); });