import { test } from "node:test"; import assert from "node:assert/strict"; import { archiveOrgDetailsUrl, archiveOrgDownloadUrl, archiveOrgTorrentUrl, archiveOrgVideoId, archiveOrgVideoIdFromNativeId, parseArchiveOrgUrl, dateFromMirrorFileName, titleFromMirrorFileName, youtubeIdFromFileName, youtubeIdFromIdentifier, } from "./archiveOrgId"; import { extractVideoId } from "./videoId"; import { defaultWebpageUrl, detectPlatform, queueKeyForUrl } from "./platform"; import { dataDirIdForUrl } from "../ytdlp/runYtdlp"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test lib/archiveOrgId.test.ts // // Every identifier, file and YouTube id here is invented. const ITEM = "example-item"; const FILE = "Example Talk (Part 1)-AbC123xyz_9.mp4"; test("archive.org item hosts are detected; the Wayback Machine is not", () => { assert.equal(detectPlatform(`https://archive.org/details/${ITEM}`), "archiveorg"); assert.equal(detectPlatform(`https://www.archive.org/details/${ITEM}`), "archiveorg"); assert.equal(detectPlatform("https://web.archive.org/web/2020/https://example.com/"), null); assert.equal(queueKeyForUrl(`https://archive.org/details/${ITEM}`), "platform:archiveorg"); }); test("a whole item's id is its identifier, from details, embed and download URLs", () => { for (const url of [ `https://archive.org/details/${ITEM}`, `https://archive.org/details/${ITEM}/`, `https://archive.org/details/${ITEM}?autoplay=1`, `https://archive.org/embed/${ITEM}`, `https://archive.org/download/${ITEM}`, ]) { assert.equal(extractVideoId(url), ITEM, url); } }); test("a file inside an item gets a stable, filesystem-safe, unique id", () => { const url = archiveOrgDetailsUrl({ identifier: ITEM, file: FILE }); assert.equal(url, `https://archive.org/details/${ITEM}/Example%20Talk%20(Part%201)-AbC123xyz_9.mp4`); const id = extractVideoId(url)!; assert.match(id, /^example-item__Example-Talk-Part-1-AbC123xyz_9-[0-9a-f]{8}$/); // The same file by every URL form, encoded or not (yt-dlp unquotes `+`). assert.equal(extractVideoId(`https://archive.org/embed/${ITEM}/${encodeURIComponent(FILE)}`), id); assert.equal(extractVideoId(archiveOrgDownloadUrl(ITEM, FILE)), id); assert.equal( extractVideoId(`https://archive.org/details/${ITEM}/Example+Talk+(Part+1)-AbC123xyz_9.mp4`), id, ); // yt-dlp's native id for the entry maps to the same id. assert.equal(archiveOrgVideoIdFromNativeId(`${ITEM}/${FILE}`), id); assert.equal(archiveOrgVideoIdFromNativeId(ITEM), ITEM); // The output path pins to it (a safe directory name). assert.equal(dataDirIdForUrl(url), id); // Paths that slug alike still differ. const a = archiveOrgVideoId({ identifier: ITEM, file: "a b.mp4" }); const b = archiveOrgVideoId({ identifier: ITEM, file: "a_b.mp4" }); const c = archiveOrgVideoId({ identifier: ITEM, file: "a b.mkv" }); assert.equal(new Set([a, b, c]).size, 3); // Sub-directories are part of the path. assert.equal( parseArchiveOrgUrl(`https://archive.org/details/${ITEM}/disc1/01%20Intro.mp3`)?.file, "disc1/01 Intro.mp3", ); }); test("URLs that name no item have no id", () => { assert.equal(parseArchiveOrgUrl("https://archive.org/search?query=x"), null); assert.equal(parseArchiveOrgUrl("https://archive.org/details/"), null); assert.equal(extractVideoId("https://archive.org/search?query=x"), null); }); test("the torrent and page URLs", () => { assert.equal(archiveOrgTorrentUrl(ITEM), `https://archive.org/download/${ITEM}/${ITEM}_archive.torrent`); assert.equal(defaultWebpageUrl("archiveorg", ITEM), `https://archive.org/details/${ITEM}`); const fileId = archiveOrgVideoId({ identifier: ITEM, file: FILE }); assert.equal(defaultWebpageUrl("archiveorg", fileId), `https://archive.org/details/${ITEM}`); }); test("YouTube ids read from mirror names", () => { assert.equal(youtubeIdFromIdentifier("youtube-AbC123xyz_9"), "AbC123xyz_9"); assert.equal(youtubeIdFromIdentifier(ITEM), null); assert.equal(youtubeIdFromFileName(FILE), "AbC123xyz_9"); assert.equal(youtubeIdFromFileName("Some Title [Zz9-Qq8_Ww7].webm"), "Zz9-Qq8_Ww7"); assert.equal(youtubeIdFromFileName("dir/Some Title [Zz9-Qq8_Ww7].info.json"), "Zz9-Qq8_Ww7"); // An eleven-letter lowercase word is not an id. assert.equal(youtubeIdFromFileName("interview-performance.mp4"), null); assert.equal(youtubeIdFromFileName("plain.mp4"), null); }); test("a mirrored file's name gives its title: date, view count, id and extension off", () => { const cases: [string, string | null][] = [ // `YYYYMMDD [<n> views]-<id>.<ext>` ["20210102 Example Talk at the Hall [1234 views]-AbC123xyz_9.mp4", "Example Talk at the Hall"], // A `<word>_` before the date, ` _ ` for ` | `. ["extras_20200304 A Long Chat _ Guest Name _ TOPIC _ Some Show [9876543 views]-Def456uvw_8.mp4", "A Long Chat | Guest Name | TOPIC | Some Show"], // No date; a separator inside the title; a dashed id that starts with `-`. ["Speaker - 'A Quoted Line!' & Why! _ Daily Clip--bC123xyz_9.mp4", "Speaker - 'A Quoted Line!' & Why! | Daily Clip"], // The bracketed id, a date then ` - `, commas in the count. ["20191231 - New Year Stream [1,234 views] [Ghi789rst_7].mkv", "New Year Stream"], // A sanitised colon: an underscore glued to a word, a space after it. ["20220505 Part One_ The Beginning-Jkl012mno_6.webm", "Part One: The Beginning"], // An underscore anywhere else stays. ["snake_case and_more [12 views]-Mno345pqr_5.mp4", "snake_case and_more"], // Not a calendar day (no 31st of April): kept. ["20210431 Not a Day-Abc234def_5.mp4", "20210431 Not a Day"], // Not a calendar date: kept. Brackets that are not a count: kept. ["20221340 Numbers First [NEW] Thing-Pqr678stu_4.mp4", "20221340 Numbers First [NEW] Thing"], // A dot inside the title survives; only the one extension goes. ["20180101 Talk with A.B. Someone-Stu901vwx_3.mp4", "Talk with A.B. Someone"], // A path inside the item: only its last part. ["sub/dir/20180101 Inside a Folder-Vwx234yza_2.mp4", "Inside a Folder"], // A name with no id at all is cleaned the same way. ["20180101 Plain Recording.mp3", "Plain Recording"], // Nothing sensible left. ["20180101 [55 views]-Yza567bcd_1.mp4", null], ["___.mp4", null], ]; for (const [name, want] of cases) assert.equal(titleFromMirrorFileName(name), want, name); }); test("a mirrored file's name gives its upload day: the leading [word_]YYYYMMDD, a real calendar day only", () => { const cases: [string, string | null][] = [ ["20210102 Example Talk [1234 views]-AbC123xyz_9.mp4", "20210102"], ["extras_20200304 A Long Chat _ Some Show-Def456uvw_8.mp4", "20200304"], ["20191231 - New Year Stream [Ghi789rst_7].mkv", "20191231"], ["sub/dir/20180101 Inside a Folder-Vwx234yza_2.mp4", "20180101"], ["20180101.mp3", "20180101"], // Leap days: only in a leap year. ["20200229 Leap Day-Jkl012mno_6.mp4", "20200229"], ["20190229 No Leap Day-Jkl012mno_6.mp4", null], ["20210431 Not a Day-Abc234def_5.mp4", null], ["20221340 Numbers First-Pqr678stu_4.mp4", null], // Not at the start, or glued to the title: no date. ["Talk from 20210102-Stu901vwx_3.mp4", null], ["20210102Talk-Stu901vwx_3.mp4", null], ["two_words_20210102 Talk-Stu901vwx_3.mp4", null], ["Plain Recording.mp3", null], ]; for (const [name, want] of cases) assert.equal(dateFromMirrorFileName(name), want, name); }); test("dateFromMirrorFileName: a file with no date of its own takes its folder's", () => { assert.equal(dateFromMirrorFileName("Show/20190303_The Example Show - A Guest/The Example Show - A Guest.mkv"), "20190303"); assert.equal(dateFromMirrorFileName("Show/20190303_Folder/20180101 Own date.mkv"), "20180101", "the file's own date wins"); assert.equal(dateFromMirrorFileName("Show/Folder without date/Plain name.mkv"), null); assert.equal(dateFromMirrorFileName("Show/20191340_Bad date/Plain.mkv"), null, "not a calendar day"); });