import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdir, mkdtemp, readdir, readFile, rm, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import type { Paths } from "../lib/paths"; import type { ChannelConfig } from "../lib/channelConfig"; import type { ResolvedCookiePolicy } from "../lib/cookiePolicy"; import type { AttemptOutcome } from "../ytdlp/runOneYtdlp"; import { buildRefreshArgs, notFetchedRefusal, refreshSummaryLines, refreshVideoMetadata, resolveRefreshTarget, type RefreshRunner, } from "./refreshVideoMetadata"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test controller/refreshVideoMetadata.test.ts // // One video's metadata re-read. The runner is a stub that records its argv and // writes the info json where the `-o infojson:` template says — no yt-dlp, no // network. Every case is a temp corpus. const SLUG = "demo"; const ID = "H64QQZuw-aA"; const URL = `https://www.youtube.com/watch?v=${ID}`; const CONFIG = { handling: "youtube", url: "https://www.youtube.com/@demo/videos", } as ChannelConfig; // The VOD as it was right after the stream ended: one fragmented audio format, // no captions. const BEFORE = { id: ID, title: "Live: the stream", webpage_url: URL, live_status: "post_live", duration: 3600, formats: [ { format_id: "140", ext: "m4a", protocol: "http_dash_segments", vcodec: "none", acodec: "mp4a.40.2", abr: 144.2, fragments: [{ url: "a" }], }, ], subtitles: {}, automatic_captions: {}, view_count: 10, }; // Hours later: processed formats, auto-captions, `was_live`. const AFTER = { ...BEFORE, live_status: "was_live", duration: 3601, formats: [ { format_id: "139", ext: "m4a", protocol: "https", vcodec: "none", acodec: "mp4a.40.5", abr: 48.8, }, { format_id: "140", ext: "m4a", protocol: "https", vcodec: "none", acodec: "mp4a.40.2", abr: 129.5, }, { format_id: "251", ext: "webm", protocol: "http_dash_segments", vcodec: "none", acodec: "opus", tbr: 135, }, { format_id: "18", ext: "mp4", protocol: "https", vcodec: "avc1.42001E", acodec: "mp4a.40.2", }, ], automatic_captions: { en: [{}], "en-orig": [{}], de: [{}] }, subtitles: { "en-US": [{}] }, view_count: 50, }; type Fixture = { root: string; paths: Paths; channelDir: string; videoDir: string }; async function fixture(opts: { meta?: object; playlist?: string[] } = {}): Promise { const root = await mkdtemp(path.join(tmpdir(), "refresh-meta-")); const channelDir = path.join(root, "channels", SLUG); const videoDir = path.join(channelDir, "data", ID); await mkdir(path.join(channelDir, "data"), { recursive: true }); if (opts.meta) { await mkdir(videoDir, { recursive: true }); await writeFile(path.join(videoDir, "metadata.info.json"), JSON.stringify(opts.meta)); } if (opts.playlist) { await writeFile(path.join(channelDir, "playlist"), opts.playlist.join("\n") + "\n"); } const paths = { channelsDir: path.join(root, "channels"), ytdlpBin: "/nonexistent" } as Paths; return { root, paths, channelDir, videoDir }; } // A stub yt-dlp: each call takes the next scripted outcome; a successful one // writes `write` to the `infojson:` template's path under cwd. function stubRunner( script: Array<{ exitCode: number; stderr?: string; write?: object }>, ): { run: RefreshRunner; calls: Array<{ cwd: string; args: string[] }> } { const calls: Array<{ cwd: string; args: string[] }> = []; const run: RefreshRunner = async (cwd, args) => { calls.push({ cwd, args }); const step = script[calls.length - 1]; if (!step) throw new Error("unexpected spawn"); if (step.write) { const tmpl = args.find((a) => a.startsWith("infojson:")); assert.ok(tmpl, "an infojson output template"); const file = path.join(cwd, `${tmpl.slice("infojson:".length)}.info.json`); await mkdir(path.dirname(file), { recursive: true }); await writeFile(file, JSON.stringify(step.write)); } const out: AttemptOutcome = { exitCode: step.exitCode, stderrTail: step.stderr ?? "", archiveLine: null, }; return out; }; return { run, calls }; } function argAfter(args: string[], flag: string): string | undefined { const i = args.lastIndexOf(flag); return i >= 0 ? args[i + 1] : undefined; } test("the argv: metadata only, cookies and pace from the channel, refusals after its own args", () => { const args = buildRefreshArgs({ videoId: ID, videoUrl: URL, channelConfig: { ...CONFIG, ytdlpExtraArgs: ["--write-subs", "--limit-rate", "1M"] }, cookies: "firefox", paceSeconds: 4, }); assert.ok(args.includes("--skip-download")); assert.ok(args.includes("--write-info-json")); assert.ok(args.includes("--ignore-config")); assert.equal(argAfter(args, "--cookies-from-browser"), "firefox"); // The platform's adaptive pace wins over the floor (the last occurrence). assert.equal(argAfter(args, "--sleep-requests"), "4"); // The channel's own args are kept, and cannot turn the pass into a // subtitle fetch: the negation comes after them. assert.equal(argAfter(args, "--limit-rate"), "1M"); assert.ok(args.lastIndexOf("--no-write-subs") > args.indexOf("--write-subs")); assert.ok(args.includes("--no-write-auto-subs")); assert.ok(args.includes("--no-download-archive")); // Pinned to THIS video's directory. assert.ok(args.includes(`infojson:data/${ID}/metadata`)); assert.deepEqual(args.slice(-2), ["--", URL]); }); test("no cookies in the argv when the policy gives none; the floor pace stands", () => { const args = buildRefreshArgs({ videoId: ID, videoUrl: URL, channelConfig: CONFIG, cookies: undefined, paceSeconds: undefined, }); assert.ok(!args.includes("--cookies-from-browser")); assert.equal(argAfter(args, "--sleep-requests"), "1"); }); test("a refresh rewrites metadata.info.json, records it as `refresh`, and says what it sees", async () => { const f = await fixture({ meta: BEFORE }); try { const { run, calls } = stubRunner([{ exitCode: 0, write: AFTER }]); let log = ""; let cleaned = 0; const result = await refreshVideoMetadata({ paths: f.paths, slug: SLUG, videoId: ID, videoUrl: URL, channelConfig: CONFIG, cookiePolicy: { cookies: "firefox", mode: "always" }, onLog: (s) => { log += s; }, signal: new AbortController().signal, onPlatformClean: () => { cleaned++; return "youtube answered cleanly: its rate-limit backoff is cleared.\n"; }, run, paceSeconds: 1, }); assert.equal(calls.length, 1); assert.equal(calls[0].cwd, f.channelDir); // "always" passes cookies up front. assert.equal(argAfter(calls[0].args, "--cookies-from-browser"), "firefox"); assert.equal(result.usedCookies, true); assert.equal(cleaned, 1); // The file is the new one, and the history has one `refresh` entry. const onDisk = JSON.parse(await readFile(path.join(f.videoDir, "metadata.info.json"), "utf8")); assert.equal(onDisk.live_status, "was_live"); const history = JSON.parse(await readFile(path.join(f.videoDir, "metadata.history.json"), "utf8")); assert.equal(history.entries.length, 1); assert.equal(history.entries[0].by, "refresh"); assert.deepEqual(history.entries[0].changed.live_status, { from: "post_live", to: "was_live" }); // Nothing else was written: no download-outcome.json, no download.log. assert.deepEqual((await readdir(f.videoDir)).sort(), [ "metadata.history.json", "metadata.info.json", ]); // The closing summary, last thing in the log. assert.deepEqual(result.summary, [ `Metadata for ${ID} now says:`, " live_status: was_live", " formats: 4", " audio-only formats: 139 m4a https 49k, 140 m4a https 130k, 251 webm http_dash_segments 135k", " non-fragmented audio: yes (139, 140)", " English subtitles: en-US", " English automatic captions: en, en-orig", " changed: live_status, duration, subtitles_langs, automatic_captions_langs", ]); assert.ok(log.trimEnd().endsWith("changed: live_status, duration, subtitles_langs, automatic_captions_langs")); assert.match(log, /answered cleanly/); } finally { await rm(f.root, { recursive: true, force: true }); } }); test("an auth-gated refresh retries once with cookies in when-required mode, never in defer", async () => { const gate = "ERROR: [youtube] H64QQZuw-aA: Sign in to confirm your age. This video may be inappropriate for some users."; for (const [mode, expectRetry] of [ ["when-required", true], ["defer", false], ] as const) { const f = await fixture({ meta: BEFORE }); try { const { run, calls } = stubRunner([ { exitCode: 1, stderr: gate }, { exitCode: 0, write: AFTER }, ]); const policy: ResolvedCookiePolicy = { cookies: "firefox", mode }; const attempt = refreshVideoMetadata({ paths: f.paths, slug: SLUG, videoId: ID, videoUrl: URL, channelConfig: CONFIG, cookiePolicy: policy, onLog: () => {}, signal: new AbortController().signal, run, paceSeconds: 1, }); if (expectRetry) { const result = await attempt; assert.equal(calls.length, 2, mode); assert.ok(!calls[0].args.includes("--cookies-from-browser"), mode); assert.equal(argAfter(calls[1].args, "--cookies-from-browser"), "firefox", mode); assert.equal(result.usedCookies, true); } else { await assert.rejects(attempt, /could not read the metadata/); assert.equal(calls.length, 1, mode); } } finally { await rm(f.root, { recursive: true, force: true }); } } }); test("a rate-limited refresh records the platform cooldown, rewrites nothing and fails", async () => { const f = await fixture({ meta: BEFORE }); try { const { run, calls } = stubRunner([ { exitCode: 1, stderr: `ERROR: [youtube] ${ID}: Unable to download webpage: HTTP Error 429: Too Many Requests`, }, ]); const backoffs: string[] = []; let cleaned = 0; await assert.rejects( refreshVideoMetadata({ paths: f.paths, slug: SLUG, videoId: ID, videoUrl: URL, channelConfig: CONFIG, cookiePolicy: { cookies: "firefox", mode: "when-required" }, onLog: () => {}, signal: new AbortController().signal, onPlatformBackoff: (c) => { backoffs.push(c); }, onPlatformClean: () => { cleaned++; return null; }, run, paceSeconds: 1, }), /rate-limited the metadata refresh/, ); // No cookie retry into a rate limit, one cooldown, no "clean". assert.equal(calls.length, 1); assert.deepEqual(backoffs, ["rate_limit"]); assert.equal(cleaned, 0); assert.deepEqual(await readdir(f.videoDir), ["metadata.info.json"]); } finally { await rm(f.root, { recursive: true, force: true }); } }); test("the target: an id with no directory is refused, even one the playlist lists — none is created", async () => { const f = await fixture({ playlist: [`https://www.youtube.com/watch?v=listed00001`] }); try { for (const id of ["nope0000000", "listed00001"]) { assert.deepEqual(await resolveRefreshTarget(f.paths, SLUG, id, CONFIG), { ok: false, error: notFetchedRefusal(id, SLUG), }); } assert.equal( notFetchedRefusal("listed00001", SLUG), `"listed00001" has not been fetched into ${SLUG} yet — sync, import or download it first; a refresh only re-reads a video already archived.`, ); // One path segment only. for (const bad of ["..", "a/b", "."]) { const r = await resolveRefreshTarget(f.paths, SLUG, bad, CONFIG); assert.equal(r.ok, false, bad); } // The pass itself refuses too, BEFORE any spawn, and makes no directory. const { run, calls } = stubRunner([{ exitCode: 0, write: AFTER }]); await assert.rejects( refreshVideoMetadata({ paths: f.paths, slug: SLUG, videoId: "listed00001", videoUrl: "https://www.youtube.com/watch?v=listed00001", channelConfig: CONFIG, onLog: () => {}, signal: new AbortController().signal, run, paceSeconds: 1, }), /has not been fetched into demo yet/, ); assert.equal(calls.length, 0); assert.deepEqual(await readdir(path.join(f.channelDir, "data")), []); } finally { await rm(f.root, { recursive: true, force: true }); } }); test("the target: a downloaded video resolves to its own webpage_url", async () => { const f = await fixture({ meta: BEFORE }); try { assert.deepEqual(await resolveRefreshTarget(f.paths, SLUG, ID, CONFIG), { ok: true, url: URL, }); } finally { await rm(f.root, { recursive: true, force: true }); } }); test("the target: metadata yt-dlp did not write is refused (archive.org, Wayback, feed-completed)", async () => { for (const [meta, history, why] of [ [{ ...BEFORE, webpage_url: "https://archive.org/details/example-item" }, null, /archive\.org record/], [ { ...BEFORE, webpage_url: `https://web.archive.org/web/20200101000000/${URL}` }, null, /Wayback Machine copy/, ], [BEFORE, "feed-backfill", /completed by feed-backfill/], ] as const) { const f = await fixture({ meta }); try { if (history) { await writeFile( path.join(f.videoDir, "metadata.history.json"), JSON.stringify({ entries: [ { at: "2026-10-01T00:00:00.000Z", by: history, from: { sha256: "a", bytes: 1 }, to: { sha256: "b", bytes: 2 }, changed: {}, added: { title: "Episode 1" }, removed: {}, counters: {}, volatile: [], }, ], }), ); } const r = await resolveRefreshTarget(f.paths, SLUG, ID, CONFIG); assert.equal(r.ok, false); assert.match(!r.ok ? r.error : "", why); assert.match(!r.ok ? r.error : "", /would overwrite it/); } finally { await rm(f.root, { recursive: true, force: true }); } } }); test("the summary: a first write, a byte-identical one, and one where only formats moved", () => { const first = refreshSummaryLines({ videoId: ID, meta: BEFORE, entries: [], hadBefore: false }); assert.equal(first.at(-1), " changed: first metadata for this video (nothing to compare)"); assert.equal(first[3], " audio-only formats: 140 m4a http_dash_segments 144k"); assert.equal(first[4], " non-fragmented audio: no"); assert.equal(first[5], " English subtitles: none"); const same = refreshSummaryLines({ videoId: ID, meta: BEFORE, entries: [], hadBefore: true }); assert.equal(same.at(-1), " changed: nothing (the file is byte-identical)"); const formatsOnly = refreshSummaryLines({ videoId: ID, meta: BEFORE, hadBefore: true, entries: [ { at: "2026-10-06T00:00:00.000Z", by: "refresh", from: { sha256: "a", bytes: 1 }, to: { sha256: "b", bytes: 1 }, changed: {}, added: {}, removed: {}, counters: { view_count: [10, 11] }, volatile: ["formats"], }, ], }); assert.equal(formatsOnly.at(-1), " changed: no content keys (formats, URLs or counters only)"); });