import { test } from "node:test"; import assert from "node:assert/strict"; import { archiveOrgCitationLinks, archiveOrgMetadataPatch, archiveOrgPlayableUrl, buildArchiveOrgProvenance, coerceArchiveOrgProvenance, withFileNameFields, findArchiveOrgInfoJson, listArchiveOrgMediaFiles, parseArchiveOrgItemMetadata, pickArchiveOrgFiles, type ArchiveOrgItemMetadata, } from "./archiveOrg"; import { archiveOrgVideoId } from "./archiveOrgId"; import { archiveOrgFetchFile, archiveOrgInfoJson, archiveOrgRecordFile, parseArchiveOrgLength, } from "./archiveOrg"; import { platformFromMetadata, summarize } from "./transcripts-server"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrg.test.ts // // A synthetic channel-archive item: three videos uploaded by a mirroring // tool, each with its yt-dlp info.json, plus archive.org's derivatives. Every // name, id and date is invented. const ITEM = "example-item"; const V1 = "First Upload-AbC123xyz_9.mp4"; const V2 = "Second Upload-Def456uvw_8.mp4"; const V3 = "Third Upload-Ghi789rst_7.mkv"; function item(over: Partial = {}): ArchiveOrgItemMetadata { return parseArchiveOrgItemMetadata({ metadata: { identifier: ITEM, title: "Example Channel Archive", date: "2024-01-02", publicdate: "2024-03-04 05:06:07", creator: "Example Creator", uploader: "someone@example.org", collection: ["opensource_movies", "community"], mediatype: "movies", ...over, }, files: [ { name: V1, source: "original", format: "MPEG4" }, { name: "First Upload-AbC123xyz_9.info.json", source: "original", format: "JSON" }, { name: V2, source: "original", format: "MPEG4", title: "Second, as titled on archive.org" }, { name: "Second Upload-Def456uvw_8.info.json", source: "original", format: "JSON" }, { name: V3, source: "original", format: "Matroska" }, { name: "Third Upload-Ghi789rst_7.mp4", source: "derivative", format: "h.264", original: V3 }, { name: "Third Upload-Ghi789rst_7.ogv", source: "derivative", format: "Ogg Video", original: V3 }, { name: `${ITEM}_archive.torrent`, source: "metadata", format: "Archive BitTorrent" }, { name: `${ITEM}_meta.xml`, source: "original", format: "Metadata" }, ], })!; } const MIRROR_INFO = { id: "AbC123xyz_9", extractor_key: "Youtube", title: "First Upload (original title)", upload_date: "20230405", uploader: "Example Creator", channel_url: "https://www.youtube.com/channel/UCexample", description: "The original description.", }; test("the metadata API's answer: an item, or null for an unknown identifier", () => { assert.equal(parseArchiveOrgItemMetadata({}), null); assert.equal(parseArchiveOrgItemMetadata(null), null); assert.equal(item().files.length, 9); }); test("media files are the originals with a media extension", () => { assert.deepEqual(listArchiveOrgMediaFiles(item()).map((f) => f.name), [V1, V2, V3]); }); test("a bulk import picks by exact names or by a case-insensitive regex", () => { assert.deepEqual(pickArchiveOrgFiles(item(), { files: [V2, "nope.mp4", `${ITEM}_meta.xml`] }), { picked: [V2], unknown: ["nope.mp4", `${ITEM}_meta.xml`], }); assert.deepEqual(pickArchiveOrgFiles(item(), { match: "^(first|third)" }).picked, [V1, V3]); assert.deepEqual(pickArchiveOrgFiles(item(), { match: "\\.mkv$" }).picked, [V3]); }); test("a file's info.json is its stem's; a single-media item's is its only one", () => { assert.equal(findArchiveOrgInfoJson(item(), V1), "First Upload-AbC123xyz_9.info.json"); assert.equal(findArchiveOrgInfoJson(item(), V3), null); assert.equal(findArchiveOrgInfoJson(item(), undefined), null); }); test("provenance of a mirror reads the original from its info.json", () => { const prov = buildArchiveOrgProvenance({ ref: { identifier: ITEM, file: V1 }, item: item(), infoJson: MIRROR_INFO, fetchedAt: "2026-01-01T00:00:00.000Z", }); assert.equal(prov.identifier, ITEM); assert.equal(prov.file, V1); assert.equal(prov.itemUrl, `https://archive.org/details/${ITEM}`); assert.equal(prov.fileUrl, `https://archive.org/details/${ITEM}/First%20Upload-AbC123xyz_9.mp4`); assert.equal(prov.downloadUrl, `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`); assert.equal(prov.torrentUrl, `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent`); assert.deepEqual(prov.item, { title: "Example Channel Archive", date: "2024-01-02", publicDate: "2024-03-04 05:06:07", creator: "Example Creator", collections: ["opensource_movies", "community"], mediatype: "movies", }); assert.deepEqual(prov.mirror, { platform: "youtube", id: "AbC123xyz_9", url: "https://www.youtube.com/watch?v=AbC123xyz_9", title: "First Upload (original title)", uploadDate: "20230405", uploader: "Example Creator", channelUrl: "https://www.youtube.com/channel/UCexample", description: "The original description.", from: "info-json", }); // The uploading account's e-mail is never kept. assert.equal(JSON.stringify(prov).includes("@"), false); assert.deepEqual(coerceArchiveOrgProvenance(JSON.parse(JSON.stringify(prov))), prov); assert.equal(coerceArchiveOrgProvenance({ identifier: ITEM }), null); }); test("without an info.json the mirror is known by name only; a plain item is no mirror", () => { const byName = buildArchiveOrgProvenance({ ref: { identifier: ITEM, file: V3 }, item: item(), fetchedAt: "2026-01-01T00:00:00.000Z", }); assert.deepEqual(byName.mirror, { platform: "youtube", id: "Ghi789rst_7", url: "https://www.youtube.com/watch?v=Ghi789rst_7", title: "Third Upload", from: "file-name", }); const byIdent = buildArchiveOrgProvenance({ ref: { identifier: "youtube-Jkl012mno_6" }, item: item({ identifier: "youtube-Jkl012mno_6" }), fetchedAt: "2026-01-01T00:00:00.000Z", }); assert.equal(byIdent.mirror?.from, "identifier"); const plain = buildArchiveOrgProvenance({ ref: { identifier: "example-film" }, item: item({ identifier: "example-film" }), fetchedAt: "2026-01-01T00:00:00.000Z", }); assert.equal(plain.mirror, undefined); assert.equal(plain.file, undefined); // A non-YouTube info.json is not a YouTube mirror. const other = buildArchiveOrgProvenance({ ref: { identifier: "example-film", file: "film.mp4" }, item: item({ identifier: "example-film" }), infoJson: { id: "abc", extractor_key: "Generic" }, fetchedAt: "2026-01-01T00:00:00.000Z", }); assert.equal(other.mirror, undefined); }); test("the record is corrected: the file's page and title, the original's date", () => { const prov = buildArchiveOrgProvenance({ ref: { identifier: ITEM, file: V1 }, item: item(), infoJson: MIRROR_INFO, fetchedAt: "2026-01-01T00:00:00.000Z", }); const info = { id: `${ITEM}/${V1}`, title: "Example Channel Archive", webpage_url: `https://archive.org/details/${ITEM}`, upload_date: "20240304", timestamp: 1709528767, uploader: "someone@example.org", }; const patch = archiveOrgMetadataPatch(prov, info); assert.deepEqual(patch, { webpage_url: prov.fileUrl, title: "First Upload (original title)", upload_date: "20230405", description: "The original description.", uploader: "Example Creator", timestamp: null, }); // Applied, nothing is left to change. assert.deepEqual(archiveOrgMetadataPatch(prov, { ...info, ...patch }), {}); // One file of many, no mirror: the file's own archive.org title. const v2 = buildArchiveOrgProvenance({ ref: { identifier: ITEM, file: V2 }, item: item({ creator: undefined }), fetchedAt: "2026-01-01T00:00:00.000Z", }); const p2 = archiveOrgMetadataPatch({ ...v2, mirror: undefined }, info); assert.equal(p2.title, "Second, as titled on archive.org"); assert.equal(p2.uploader, null); assert.equal("upload_date" in p2, false); }); test("citation links: a mirror cites YouTube at the second, archive.org and the torrent as downloads", () => { const prov = buildArchiveOrgProvenance({ ref: { identifier: ITEM, file: V1 }, item: item(), infoJson: MIRROR_INFO, fetchedAt: "2026-01-01T00:00:00.000Z", }); assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: prov.fileUrl, provenance: prov, seconds: 75.6 }), { original: { label: "YouTube", url: "https://www.youtube.com/watch?v=AbC123xyz_9&t=75s" }, downloads: [ { label: "archive.org", url: prov.fileUrl }, { label: "torrent", url: `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent` }, ], }); }); test("citation links: a plain record is derived from its page alone", () => { assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: "https://archive.org/details/example-film", seconds: 30 }), { original: { label: "archive.org", url: "https://archive.org/details/example-film" }, downloads: [{ label: "torrent", url: "https://archive.org/download/example-film/example-film_archive.torrent" }], }); assert.deepEqual(archiveOrgCitationLinks({ webpageUrl: "https://example.com/x" }), { downloads: [] }); }); test("the player plays a browser-playable file: an mp4 before an mkv original", () => { const formats = [ { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.mkv`, ext: "mkv", format_note: "original" }, { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.ogv`, ext: "ogv", format_note: "derivative" }, { url: `https://archive.org/download/${ITEM}/Third%20Upload-Ghi789rst_7.mp4`, ext: "mp4", format_note: "derivative" }, ]; assert.equal(archiveOrgPlayableUrl({ formats }), formats[2].url); assert.equal( archiveOrgPlayableUrl({ id: `${ITEM}/a b.mp3` }), `https://archive.org/download/${ITEM}/a%20b.mp3`, ); assert.equal(archiveOrgPlayableUrl({ id: ITEM }), undefined); }); test("platformFromMetadata and summarize know an archive.org record", () => { assert.equal(platformFromMetadata({ extractor_key: "ArchiveOrg" }), "archiveorg"); assert.equal(platformFromMetadata({ extractor: "archive.org" }), "archiveorg"); assert.equal(platformFromMetadata({ extractor_key: "YoutubeWebArchive" }), "youtube"); assert.equal(platformFromMetadata({ extractor_key: "Youtube" }), "youtube"); assert.equal( platformFromMetadata({ extractor_key: "Generic", webpage_url: `https://archive.org/details/${ITEM}` }), "archiveorg", ); assert.equal(platformFromMetadata({ extractor_key: "Generic", webpage_url: "https://example.com/a.mp3" }), "youtube"); const id = archiveOrgVideoId({ identifier: ITEM, file: V1 }); const s = summarize("example-channel", id, { id: `${ITEM}/${V1}`, extractor_key: "ArchiveOrg", title: "First Upload (original title)", upload_date: "20230405", webpage_url: `https://archive.org/details/${ITEM}/First%20Upload-AbC123xyz_9.mp4`, formats: [{ url: `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`, ext: "mp4", format_note: "original" }], }); assert.equal(s.platform, "archiveorg"); assert.equal(s.id, id); assert.equal(s.slug, `example-channel/${id}`); assert.equal(s.mediaUrl, `https://archive.org/download/${ITEM}/First%20Upload-AbC123xyz_9.mp4`); // Other platforms gain no key. const yt = summarize("c", "AbC123xyz_9", { id: "AbC123xyz_9", extractor_key: "Youtube" }); assert.equal("mediaUrl" in yt, false); }); // ─── The record without yt-dlp (controller/archiveOrgDownload.ts) ─── const FETCH_ITEM: ArchiveOrgItemMetadata = { metadata: { identifier: "example-tapes", title: "Example Tapes", creator: "Example Archivist", uploader: "someone@example.org", publicdate: "2020-02-03 10:11:12", description: ["Part one.", "Part two."], }, files: [ { name: "tape1.flac", source: "original", length: "61.5" }, { name: "tape1.mp3", source: "derivative", original: "tape1.flac", format: "VBR MP3", length: "61.48" }, { name: "tape1.ogg", source: "derivative", original: "tape1.flac", format: "Ogg Vorbis" }, { name: "talk.avi", source: "original" }, { name: "talk.mp4", source: "derivative", original: "talk.avi", format: "h.264" }, { name: "film.mp4", source: "original" }, { name: "film.ogv", source: "derivative", original: "film.mp4" }, ], }; test("the file fetched: a common container as is, else archive.org's mp4/mp3 of it", () => { assert.equal(archiveOrgFetchFile(FETCH_ITEM, "film.mp4")?.name, "film.mp4"); assert.equal(archiveOrgFetchFile(FETCH_ITEM, "talk.avi")?.name, "talk.mp4"); assert.equal(archiveOrgFetchFile(FETCH_ITEM, "tape1.flac")?.name, "tape1.mp3"); assert.equal(archiveOrgFetchFile(FETCH_ITEM, "missing.mp4"), null); assert.equal(archiveOrgRecordFile(FETCH_ITEM, "talk.avi"), "talk.avi"); assert.equal(archiveOrgRecordFile(FETCH_ITEM, undefined), null, "several media files: none is the item"); assert.equal( archiveOrgRecordFile({ metadata: { identifier: "x" }, files: [{ name: "only.mp3", source: "original" }] }, undefined), "only.mp3", ); }); test("archive.org lengths: seconds or a clock", () => { assert.equal(parseArchiveOrgLength("123.45"), 123.45); assert.equal(parseArchiveOrgLength("02:03"), 123); assert.equal(parseArchiveOrgLength("1:02:03"), 3723); assert.equal(parseArchiveOrgLength("soon"), undefined); }); test("the synthesised record: ArchiveOrg, the canonical id, the file's page, formats, the date last", () => { const prov = buildArchiveOrgProvenance({ ref: { identifier: "example-tapes", file: "tape1.flac" }, item: FETCH_ITEM, fetchedAt: "2026-10-05T00:00:00.000Z", }); const fetched = archiveOrgFetchFile(FETCH_ITEM, "tape1.flac")!; const info = archiveOrgInfoJson({ prov, item: FETCH_ITEM, file: "tape1.flac", fetched }); assert.equal(info.extractor_key, "ArchiveOrg"); assert.equal(platformFromMetadata(info), "archiveorg"); assert.equal(info.id, archiveOrgVideoId({ identifier: "example-tapes", file: "tape1.flac" })); assert.equal(info.webpage_url, "https://archive.org/details/example-tapes/tape1.flac"); // One file of many: its own name, never the item's title. assert.equal(info.title, "tape1"); assert.equal(info.uploader, "Example Archivist"); assert.equal(info.duration, 61.48); assert.equal(info.ext, "mp3"); assert.equal(info.description, "Part one.\n\nPart two."); assert.equal(info.upload_date, "20200203"); assert.equal(Object.keys(info).at(-1), "upload_date"); assert.deepEqual( (info.formats as { format_id: string; format_note: string }[]).map((f) => [f.format_id, f.format_note]), [ ["tape1.flac", "original"], ["tape1.mp3", "derivative"], ["tape1.ogg", "derivative"], ], ); // The player plays a playable one of them. assert.equal(summarize("c", String(info.id), info).mediaUrl, "https://archive.org/download/example-tapes/tape1.mp3"); }); test("a mirror's record is the original's: title, date, uploader from its uploaded info.json", () => { const item: ArchiveOrgItemMetadata = { metadata: { identifier: "youtube-AbC123xyz_9", title: "Mirror", date: "2024-01-01" }, files: [{ name: "Clip.mp4", source: "original" }, { name: "Clip.info.json", source: "original" }], }; const prov = buildArchiveOrgProvenance({ ref: { identifier: "youtube-AbC123xyz_9" }, item, infoJson: { id: "AbC123xyz_9", extractor_key: "Youtube", title: "The Original", upload_date: "20190909", uploader: "Orig Channel" }, fetchedAt: "2026-10-05T00:00:00.000Z", }); const info = archiveOrgInfoJson({ prov, item, file: "Clip.mp4", fetched: item.files[0], durationSec: 10 }); assert.equal(info.id, "youtube-AbC123xyz_9"); assert.equal(info.title, "The Original"); assert.equal(info.upload_date, "20190909"); assert.equal(info.uploader, "Orig Channel"); assert.equal(info.duration, 10); assert.equal(info.webpage_url, "https://archive.org/details/youtube-AbC123xyz_9"); }); test("one file of many with no title of its own is titled and dated from its name, the name kept as `file`", () => { const NAMED = "extras_20210102 A Talk _ Some Show [4321 views]-Pqr678stu_4.mp4"; const many = parseArchiveOrgItemMetadata({ metadata: { identifier: ITEM, title: "Example Channel Archive", date: "2024-01-02" }, files: [ { name: NAMED, source: "original", format: "MPEG4" }, { name: V2, source: "original", format: "MPEG4", title: "Second, as titled on archive.org" }, ], })!; const prov = buildArchiveOrgProvenance({ ref: { identifier: ITEM, file: NAMED }, item: many, fetchedAt: "2026-01-01T00:00:00.000Z", }); assert.equal(prov.file, NAMED); assert.equal(prov.mirror?.from, "file-name"); assert.equal(prov.mirror?.title, "A Talk | Some Show"); // The name's leading date is the original's day, not the item's. assert.equal(prov.mirror?.uploadDate, "20210102"); const rec = archiveOrgInfoJson({ prov, item: many, file: NAMED, fetched: many.files[0] }); assert.equal(rec.title, "A Talk | Some Show"); assert.equal(rec.upload_date, "20210102"); assert.equal(Object.keys(rec).at(-1), "upload_date"); // A record with archive.org's date and a timestamp: the name's day replaces // both, as a mirror's date always has. const patch = archiveOrgMetadataPatch(prov, { title: "x", upload_date: "20240102", timestamp: 1704153600 }); assert.equal(patch.upload_date, "20210102"); assert.equal(patch.timestamp, null); // A file with its own title in the item keeps it; the mirror is not titled from the name. const titled = buildArchiveOrgProvenance({ ref: { identifier: ITEM, file: V2 }, item: many, fetchedAt: "2026-01-01T00:00:00.000Z", }); assert.equal(titled.mirror?.title, undefined); assert.equal(archiveOrgMetadataPatch(titled, {}).title, "Second, as titled on archive.org"); // An older sidecar (no mirror title) is brought up to the rule, once. const old = { ...prov, mirror: { ...prov.mirror! } }; delete old.mirror.title; delete old.mirror.uploadDate; const fixed = withFileNameFields(old); assert.equal(fixed.mirror?.title, "A Talk | Some Show"); assert.equal(fixed.mirror?.uploadDate, "20210102"); assert.equal(withFileNameFields(fixed), fixed); // An info.json's title and date are the original's own and are never replaced. const fromInfo = { ...prov, mirror: { ...prov.mirror!, from: "info-json" as const, title: "Original", uploadDate: "20200101" }, }; assert.equal(withFileNameFields(fromInfo), fromInfo); // An info.json without a date takes the name's. const infoNoDate = { ...fromInfo, mirror: { ...fromInfo.mirror } }; delete (infoNoDate.mirror as { uploadDate?: string }).uploadDate; assert.equal(withFileNameFields(infoNoDate).mirror?.uploadDate, "20210102"); assert.equal(withFileNameFields(infoNoDate).mirror?.title, "Original"); }); test("archiveOrgMetadataPatch: a file that is no mirror is dated by its name or folder", () => { const prov = { version: 1, identifier: "example-item", file: "Show/20190303_The Example Show - A Guest/The Example Show - A Guest.mkv", itemUrl: "https://archive.org/details/example-item", fileUrl: "https://archive.org/details/example-item/Show/x.mkv", item: { collections: [], publicDate: "2019-06-07" }, } as unknown as Parameters[0]; const patch = archiveOrgMetadataPatch(prov, { upload_date: "20190607", timestamp: 1559900000 }); assert.equal(patch.upload_date, "20190303"); assert.equal(patch.timestamp, null); });