commit 4bfe6ef1750b406455f15496bf4583a2dc52e117
parent c65c3df271a76212daac1cba4d9c735c66bb6213
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 15:20:00 -0400
sources: the provenance sidecar is the prefetch's own file; a test that the step runs only after a prefetch that succeeded
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
2 files changed, 83 insertions(+), 0 deletions(-)
diff --git a/common/ytdlp/archiveOrgProvenanceHook.test.ts b/common/ytdlp/archiveOrgProvenanceHook.test.ts
@@ -0,0 +1,80 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { chmod, mkdtemp, readFile, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import type { ChannelConfig } from "../lib/channelConfig";
+import { downloadOneManaged } from "./downloadOneManaged";
+import { archiveOrgDetailsUrl, archiveOrgVideoId } from "../lib/archiveOrgId";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test ytdlp/archiveOrgProvenanceHook.test.ts
+//
+// THE archive.org PROVENANCE STEP RUNS RIGHT AFTER A PREFETCH THAT SUCCEEDED,
+// and never after one that failed. The yt-dlp here is a temp script: the
+// prefetch (`--skip-download`) writes a metadata.info.json shaped like the
+// ArchiveOrg extractor's (the ITEM's page for one file of many) and exits 0 or
+// 1; the download pass fails. The step itself is a spy. Every name is invented.
+
+const ITEM = "example-item";
+const FILE = "Clip One.mp4";
+const URL = archiveOrgDetailsUrl({ identifier: ITEM, file: FILE });
+const ID = archiveOrgVideoId({ identifier: ITEM, file: FILE });
+
+async function runWith(prefetchExit: 0 | 1) {
+ const root = await mkdtemp(path.join(tmpdir(), "archiveorg-hook-"));
+ try {
+ const bin = path.join(root, "fake-ytdlp.mjs");
+ await writeFile(
+ bin,
+ `#!/usr/bin/env node
+import { mkdirSync, writeFileSync } from "node:fs";
+import path from "node:path";
+const args = process.argv.slice(2);
+if (!args.includes("--skip-download")) { console.error("ERROR: no download in this test"); process.exit(1); }
+const o = args.find((a) => a.startsWith("infojson:"));
+const file = o.slice("infojson:".length) + ".info.json";
+mkdirSync(path.dirname(file), { recursive: true });
+writeFileSync(file, JSON.stringify({ id: ${JSON.stringify(`${ITEM}/${FILE}`)}, extractor_key: "ArchiveOrg", title: "Example Archive", webpage_url: "https://archive.org/details/${ITEM}" }));
+process.exit(${prefetchExit});
+`,
+ );
+ await chmod(bin, 0o755);
+ const paths = { channelsDir: path.join(root, "channels"), ytdlpBin: bin } as Paths;
+ const calls: { videoDir: string; videoUrl: string; infoAtCall: string }[] = [];
+ await downloadOneManaged({
+ channelSlug: "c",
+ channelConfig: { handling: "transcribe", platform: "archiveorg" } as ChannelConfig,
+ paths,
+ videoUrl: URL,
+ onLog: () => {},
+ signal: new AbortController().signal,
+ archiveOrgProvenance: async (o) => {
+ calls.push({
+ videoDir: o.videoDir,
+ videoUrl: o.videoUrl,
+ infoAtCall: await readFile(path.join(o.videoDir, "metadata.info.json"), "utf8").catch(() => ""),
+ });
+ return null;
+ },
+ });
+ return { calls, dataDir: path.join(paths.channelsDir, "c", "data") };
+ } finally {
+ await rm(root, { recursive: true, force: true });
+ }
+}
+
+test("after a prefetch that succeeded, the step runs once on the record's own dir", async () => {
+ const { calls, dataDir } = await runWith(0);
+ assert.equal(calls.length, 1);
+ assert.equal(calls[0].videoDir, path.join(dataDir, ID));
+ assert.equal(calls[0].videoUrl, URL);
+ // The prefetch's record is already on disk when it runs.
+ assert.match(calls[0].infoAtCall, /"extractor_key":"ArchiveOrg"/);
+});
+
+test("after a prefetch that failed, it does not run", async () => {
+ const { calls } = await runWith(1);
+ assert.equal(calls.length, 0);
+});
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -500,6 +500,9 @@ const PREFETCH_OWN_FILES: ReadonlySet<string> = new Set([
// one a previous rejection wrote before this rule existed. It records what
// happened, never what is on disk, so it is ours to drop with the rest.
"download-outcome.json",
+ // An archive.org record's provenance, written by this pass right after the
+ // prefetch (lib/archiveOrg-server.ts) and fetchable again.
+ "archiveorg.json",
// NOT metadata.history.json, deliberately: a history means an earlier
// metadata.info.json was here before this pass, and it is the one record of
// what the source used to say (lib/metadataHistory-server.ts).