// WHICH FILE OF A LOCAL ARCHIVE IS WHICH VIDEO (release 21 D1). // // The pure half of `attach-media` (controller/attachMedia.ts): given the file // list of an archive (a zip's entries, a directory's files), group it into // per-video folders, find each folder's video id and media file, and classify // every folder against what the channel holds. No fs: the controller lists, // this decides, and the tests run over plain arrays. // // THE ID, in order: the folder's trailing `(<11-char id>)` — the archive's own // naming, `NNN - - (<id>)` — else the media file's yt-dlp suffix // (`<title>-<id>.<ext>`, `<title> [<id>].<ext>`, or a bare `<id>.<ext>`), else // an explicit item. The 11-character shape is YouTube's; any other platform's // files are attached by explicit `items`. // // `[LOST]` folders (the archive's mark for a video whose media nobody has) // are listed and never attached, whatever they hold. export const ATTACHABLE_MEDIA_EXTS: readonly string[] = ["mp4", "mkv", "webm", "mov", "m4v"]; const YT_ID = "[A-Za-z0-9_-]{11}"; const FOLDER_ID_RE = new RegExp(`\\((${YT_ID})\\)\\s*$`); const FILE_ID_RES = [ new RegExp(`^(${YT_ID})\\.[A-Za-z0-9]+$`), new RegExp(`-(${YT_ID})\\.[A-Za-z0-9]+$`), new RegExp(`\\[(${YT_ID})\\]\\.[A-Za-z0-9]+$`), ]; export type SourceFile = { // Path inside the archive, "/"-separated, no leading "/". path: string; size: number; }; export type ArchiveFolder = { // The folder's path inside the archive ("" for the archive's root). folder: string; // Its last segment (the root's is ""). name: string; id: string | null; idFrom: "folder" | "file" | null; lost: boolean; // Every video container directly in the folder. media: SourceFile[]; // The sidecars createRecords reads, when present. infoJson?: SourceFile; descriptionTxt?: SourceFile; sourceTxt?: SourceFile; }; function extOf(name: string): string { const dot = name.lastIndexOf("."); return dot < 0 ? "" : name.slice(dot + 1).toLowerCase(); } function baseName(p: string): string { const i = p.lastIndexOf("/"); return i < 0 ? p : p.slice(i + 1); } function dirName(p: string): string { const i = p.lastIndexOf("/"); return i < 0 ? "" : p.slice(0, i); } export function isAttachableMedia(name: string): boolean { return ATTACHABLE_MEDIA_EXTS.includes(extOf(name)); } export function idFromFolderName(name: string): string | null { return FOLDER_ID_RE.exec(name)?.[1] ?? null; } export function idFromMediaName(name: string): string | null { const base = baseName(name); for (const re of FILE_ID_RES) { const m = re.exec(base); if (m) return m[1]; } return null; } export function isLostFolder(name: string): boolean { return /\[LOST\]/i.test(name); } // One folder per directory that holds a file. A media file's folder is its // immediate parent; the id comes from that folder's name, else the file's. export function groupArchiveFolders(files: readonly SourceFile[]): ArchiveFolder[] { const byFolder = new Map<string, SourceFile[]>(); for (const f of files) { if (!f.path || f.path.endsWith("/")) continue; const folder = dirName(f.path); const list = byFolder.get(folder) ?? []; list.push(f); byFolder.set(folder, list); } const out: ArchiveFolder[] = []; for (const [folder, list] of [...byFolder].sort(([a], [b]) => a.localeCompare(b))) { const name = baseName(folder); const media = list.filter((f) => isAttachableMedia(f.path)); const pick = (test: (base: string) => boolean) => list.find((f) => test(baseName(f.path))); const infoJson = pick((b) => b.toLowerCase().endsWith(".info.json")); const descriptionTxt = pick((b) => b.toLowerCase() === "description.txt"); const sourceTxt = pick((b) => /^sources?\.txt$/i.test(b)); const lost = isLostFolder(name); const base = { lost, ...(infoJson ? { infoJson } : {}), ...(descriptionTxt ? { descriptionTxt } : {}), ...(sourceTxt ? { sourceTxt } : {}), }; const folderId = folder ? idFromFolderName(name) : null; if (folderId) { out.push({ folder, name, id: folderId, idFrom: "folder", media, ...base }); continue; } // No id on the folder (or the archive's root): every media file is its own // video, by its file name. Sidecars are shared only when there is one. if (media.length === 0) { out.push({ folder, name, id: null, idFrom: null, media, ...base }); continue; } for (const m of media) { const id = idFromMediaName(m.path); out.push({ folder, name, id, idFrom: id ? "file" : null, media: [m], ...(media.length === 1 ? base : { lost }), }); } } return out; } // The archive's title for a folder named `NNN - <title> - (<id>)`, with any // `[LOST]` mark dropped. export function titleFromFolderName(name: string): string { return name .replace(FOLDER_ID_RE, "") .replace(/\s*-\s*$/, "") .replace(/^\s*\d+\s*(?:\[LOST\]\s*)?-\s*/i, "") .replace(/\[LOST\]/gi, "") .trim(); } const MONTHS = [ "january", "february", "march", "april", "may", "june", "july", "august", "september", "october", "november", "december", ]; // "Published on July 13th 2018" (the archive's description.txt) → "20180713". export function publishedDateFromText(text: string): string | null { const m = /published\s+on\s+([A-Za-z]+)\s+(\d{1,2})(?:st|nd|rd|th)?,?\s+(\d{4})/i.exec(text); if (!m) return null; const month = MONTHS.indexOf(m[1].toLowerCase()); const day = Number(m[2]); if (month < 0 || day < 1 || day > 31) return null; return `${m[3]}${String(month + 1).padStart(2, "0")}${String(day).padStart(2, "0")}`; } // The description a record gets from description.txt: the text without the // "Published on …" line (that is the record's upload_date). export function descriptionFromText(text: string): string { return text .split(/\r?\n/) .filter((l) => !/^\s*published\s+on\s+/i.test(l)) .join("\n") .trim(); } // A metadata.info.json for a video the channel does not hold, built from the // folder's description.txt and name when it has no yt-dlp .info.json. Title // first (a reader scans the head for it), upload_date last (another scans the // tail). export function archiveFolderInfoJson(opts: { id: string; folderName: string; descriptionText?: string; platform?: string; durationSec?: number | null; }): Record<string, unknown> { const youtube = opts.platform === undefined || opts.platform === "youtube"; const date = opts.descriptionText ? publishedDateFromText(opts.descriptionText) : null; const description = opts.descriptionText ? descriptionFromText(opts.descriptionText) : ""; const url = youtube ? `https://www.youtube.com/watch?v=${opts.id}` : undefined; return { id: opts.id, title: titleFromFolderName(opts.folderName) || opts.id, ...(description ? { description } : {}), ...(typeof opts.durationSec === "number" && Number.isFinite(opts.durationSec) && opts.durationSec > 0 ? { duration: Math.round(opts.durationSec * 100) / 100 } : {}), ...(url ? { webpage_url: url, original_url: url } : {}), ...(youtube ? { extractor: "youtube", extractor_key: "Youtube" } : {}), ...(date ? { upload_date: date } : {}), }; } // --- classification --------------------------------------------------------- export const ATTACH_ROW_CLASSES = [ // Held, media found, no pointer (or `replace`): attached. "attach", // Not held, media found, `createRecords`: a record is written, then attached. "create", // Held and already has a saved container: left alone without `replace`. "already-attached", // Media found for a video the channel does not hold (no `createRecords`). "not-held", // A `[LOST]` folder: listed, never attached. "lost", // A folder with an id but no video file. "no-media", // Media with no id the rules can read. "unmatched", // More than one video file for one id: name the file with `items`. "ambiguous", ] as const; export type AttachRowClass = (typeof ATTACH_ROW_CLASSES)[number]; export type AttachRow = { class: AttachRowClass; id: string | null; folder: string; // The file to copy ("attach"/"create"), or the candidates ("ambiguous"). media?: SourceFile; candidates?: SourceFile[]; // The folder, for createRecords. source?: ArchiveFolder; // `attach` over an existing pointer (replace: true). replacing?: boolean; }; export type AttachPlan = { rows: AttachRow[]; // Held videos no folder supplied media for (only when the whole archive was // considered: no `items`, no `match`). heldWithoutMedia: string[] | null; }; export type AttachItem = { id: string; path: string }; export function planAttach(opts: { folders: readonly ArchiveFolder[]; held: ReadonlySet<string>; attached: ReadonlySet<string>; createRecords?: boolean; replace?: boolean; match?: RegExp; // Explicit id → file pairs. When given, they are the whole plan. items?: readonly AttachItem[]; // Every file in the archive, to resolve `items` paths. files?: readonly SourceFile[]; }): AttachPlan { const classify = (id: string, media: SourceFile, folder: string, source?: ArchiveFolder): AttachRow => { if (opts.held.has(id)) { if (opts.attached.has(id) && !opts.replace) { return { class: "already-attached", id, folder, media }; } return { class: "attach", id, folder, media, ...(source ? { source } : {}), ...(opts.attached.has(id) ? { replacing: true } : {}), }; } return { class: opts.createRecords ? "create" : "not-held", id, folder, media, ...(source ? { source } : {}), }; }; if (opts.items) { const byPath = new Map((opts.files ?? []).map((f) => [f.path, f])); const folders = new Map(opts.folders.map((f) => [f.folder, f])); const rows = opts.items.map((item): AttachRow => { const media = byPath.get(item.path); const folder = dirName(item.path); if (!media) return { class: "no-media", id: item.id, folder }; return classify(item.id, media, folder, folders.get(folder)); }); return { rows, heldWithoutMedia: null }; } const considered = opts.match ? opts.folders.filter( (f) => opts.match!.test(f.folder) || f.media.some((m) => opts.match!.test(m.path)), ) : opts.folders; // Two folders claiming one id: neither is attached on a guess. const idCount = new Map<string, number>(); for (const f of considered) { if (f.id && f.media.length > 0 && !f.lost) idCount.set(f.id, (idCount.get(f.id) ?? 0) + 1); } const rows: AttachRow[] = []; for (const f of considered) { if (f.lost) { rows.push({ class: "lost", id: f.id, folder: f.folder }); continue; } if (!f.id) { if (f.media.length > 0) { rows.push({ class: "unmatched", id: null, folder: f.folder, candidates: f.media }); } continue; } if (f.media.length === 0) { rows.push({ class: "no-media", id: f.id, folder: f.folder }); continue; } if (f.media.length > 1 || (idCount.get(f.id) ?? 0) > 1) { rows.push({ class: "ambiguous", id: f.id, folder: f.folder, candidates: f.media }); continue; } rows.push(classify(f.id, f.media[0], f.folder, f)); } let heldWithoutMedia: string[] | null = null; if (!opts.match) { const supplied = new Set( rows.filter((r) => r.media && r.id).map((r) => r.id as string), ); for (const r of rows) if (r.class === "ambiguous" && r.id) supplied.add(r.id); heldWithoutMedia = [...opts.held].filter((id) => !supplied.has(id)).sort(); } return { rows, heldWithoutMedia }; } export function countAttachRows(rows: readonly AttachRow[]): Record<AttachRowClass, number> { const out = Object.fromEntries(ATTACH_ROW_CLASSES.map((c) => [c, 0])) as Record<AttachRowClass, number>; for (const r of rows) out[r.class] += 1; return out; }