// WHICH FILE OF A LOCAL ARCHIVE IS WHICH VIDEO (release 21 D1).
//
// The pure half of `attach-media` (controller/attachMedia.ts): given the file
// list of an archive (a zip's entries, a directory's files), group it into
// per-video folders, find each folder's video id and media file, and classify
// every folder against what the channel holds. No fs: the controller lists,
// this decides, and the tests run over plain arrays.
//
// THE ID, in order: the folder's trailing `(<11-char id>)` — the archive's own
// naming, `NNN -
- ()` — else the media file's yt-dlp suffix
// (`-.`, ` [].`, or a bare `.`), else
// an explicit item. The 11-character shape is YouTube's; any other platform's
// files are attached by explicit `items`.
//
// `[LOST]` folders (the archive's mark for a video whose media nobody has)
// are listed and never attached, whatever they hold.
export const ATTACHABLE_MEDIA_EXTS: readonly string[] = ["mp4", "mkv", "webm", "mov", "m4v"];
const YT_ID = "[A-Za-z0-9_-]{11}";
const FOLDER_ID_RE = new RegExp(`\\((${YT_ID})\\)\\s*$`);
const FILE_ID_RES = [
new RegExp(`^(${YT_ID})\\.[A-Za-z0-9]+$`),
new RegExp(`-(${YT_ID})\\.[A-Za-z0-9]+$`),
new RegExp(`\\[(${YT_ID})\\]\\.[A-Za-z0-9]+$`),
];
export type SourceFile = {
// Path inside the archive, "/"-separated, no leading "/".
path: string;
size: number;
};
export type ArchiveFolder = {
// The folder's path inside the archive ("" for the archive's root).
folder: string;
// Its last segment (the root's is "").
name: string;
id: string | null;
idFrom: "folder" | "file" | null;
lost: boolean;
// Every video container directly in the folder.
media: SourceFile[];
// The sidecars createRecords reads, when present.
infoJson?: SourceFile;
descriptionTxt?: SourceFile;
sourceTxt?: SourceFile;
};
function extOf(name: string): string {
const dot = name.lastIndexOf(".");
return dot < 0 ? "" : name.slice(dot + 1).toLowerCase();
}
function baseName(p: string): string {
const i = p.lastIndexOf("/");
return i < 0 ? p : p.slice(i + 1);
}
function dirName(p: string): string {
const i = p.lastIndexOf("/");
return i < 0 ? "" : p.slice(0, i);
}
export function isAttachableMedia(name: string): boolean {
return ATTACHABLE_MEDIA_EXTS.includes(extOf(name));
}
export function idFromFolderName(name: string): string | null {
return FOLDER_ID_RE.exec(name)?.[1] ?? null;
}
export function idFromMediaName(name: string): string | null {
const base = baseName(name);
for (const re of FILE_ID_RES) {
const m = re.exec(base);
if (m) return m[1];
}
return null;
}
export function isLostFolder(name: string): boolean {
return /\[LOST\]/i.test(name);
}
// One folder per directory that holds a file. A media file's folder is its
// immediate parent; the id comes from that folder's name, else the file's.
export function groupArchiveFolders(files: readonly SourceFile[]): ArchiveFolder[] {
const byFolder = new Map();
for (const f of files) {
if (!f.path || f.path.endsWith("/")) continue;
const folder = dirName(f.path);
const list = byFolder.get(folder) ?? [];
list.push(f);
byFolder.set(folder, list);
}
const out: ArchiveFolder[] = [];
for (const [folder, list] of [...byFolder].sort(([a], [b]) => a.localeCompare(b))) {
const name = baseName(folder);
const media = list.filter((f) => isAttachableMedia(f.path));
const pick = (test: (base: string) => boolean) => list.find((f) => test(baseName(f.path)));
const infoJson = pick((b) => b.toLowerCase().endsWith(".info.json"));
const descriptionTxt = pick((b) => b.toLowerCase() === "description.txt");
const sourceTxt = pick((b) => /^sources?\.txt$/i.test(b));
const lost = isLostFolder(name);
const base = {
lost,
...(infoJson ? { infoJson } : {}),
...(descriptionTxt ? { descriptionTxt } : {}),
...(sourceTxt ? { sourceTxt } : {}),
};
const folderId = folder ? idFromFolderName(name) : null;
if (folderId) {
out.push({ folder, name, id: folderId, idFrom: "folder", media, ...base });
continue;
}
// No id on the folder (or the archive's root): every media file is its own
// video, by its file name. Sidecars are shared only when there is one.
if (media.length === 0) {
out.push({ folder, name, id: null, idFrom: null, media, ...base });
continue;
}
for (const m of media) {
const id = idFromMediaName(m.path);
out.push({
folder,
name,
id,
idFrom: id ? "file" : null,
media: [m],
...(media.length === 1 ? base : { lost }),
});
}
}
return out;
}
// The archive's title for a folder named `NNN - - ()`, with any
// `[LOST]` mark dropped.
export function titleFromFolderName(name: string): string {
return name
.replace(FOLDER_ID_RE, "")
.replace(/\s*-\s*$/, "")
.replace(/^\s*\d+\s*(?:\[LOST\]\s*)?-\s*/i, "")
.replace(/\[LOST\]/gi, "")
.trim();
}
const MONTHS = [
"january", "february", "march", "april", "may", "june",
"july", "august", "september", "october", "november", "december",
];
// "Published on July 13th 2018" (the archive's description.txt) → "20180713".
export function publishedDateFromText(text: string): string | null {
const m = /published\s+on\s+([A-Za-z]+)\s+(\d{1,2})(?:st|nd|rd|th)?,?\s+(\d{4})/i.exec(text);
if (!m) return null;
const month = MONTHS.indexOf(m[1].toLowerCase());
const day = Number(m[2]);
if (month < 0 || day < 1 || day > 31) return null;
return `${m[3]}${String(month + 1).padStart(2, "0")}${String(day).padStart(2, "0")}`;
}
// The description a record gets from description.txt: the text without the
// "Published on …" line (that is the record's upload_date).
export function descriptionFromText(text: string): string {
return text
.split(/\r?\n/)
.filter((l) => !/^\s*published\s+on\s+/i.test(l))
.join("\n")
.trim();
}
// A metadata.info.json for a video the channel does not hold, built from the
// folder's description.txt and name when it has no yt-dlp .info.json. Title
// first (a reader scans the head for it), upload_date last (another scans the
// tail).
export function archiveFolderInfoJson(opts: {
id: string;
folderName: string;
descriptionText?: string;
platform?: string;
durationSec?: number | null;
}): Record {
const youtube = opts.platform === undefined || opts.platform === "youtube";
const date = opts.descriptionText ? publishedDateFromText(opts.descriptionText) : null;
const description = opts.descriptionText ? descriptionFromText(opts.descriptionText) : "";
const url = youtube ? `https://www.youtube.com/watch?v=${opts.id}` : undefined;
return {
id: opts.id,
title: titleFromFolderName(opts.folderName) || opts.id,
...(description ? { description } : {}),
...(typeof opts.durationSec === "number" && Number.isFinite(opts.durationSec) && opts.durationSec > 0
? { duration: Math.round(opts.durationSec * 100) / 100 }
: {}),
...(url ? { webpage_url: url, original_url: url } : {}),
...(youtube ? { extractor: "youtube", extractor_key: "Youtube" } : {}),
...(date ? { upload_date: date } : {}),
};
}
// --- classification ---------------------------------------------------------
export const ATTACH_ROW_CLASSES = [
// Held, media found, no pointer (or `replace`): attached.
"attach",
// Not held, media found, `createRecords`: a record is written, then attached.
"create",
// Held and already has a saved container: left alone without `replace`.
"already-attached",
// Media found for a video the channel does not hold (no `createRecords`).
"not-held",
// A `[LOST]` folder: listed, never attached.
"lost",
// A folder with an id but no video file.
"no-media",
// Media with no id the rules can read.
"unmatched",
// More than one video file for one id: name the file with `items`.
"ambiguous",
] as const;
export type AttachRowClass = (typeof ATTACH_ROW_CLASSES)[number];
export type AttachRow = {
class: AttachRowClass;
id: string | null;
folder: string;
// The file to copy ("attach"/"create"), or the candidates ("ambiguous").
media?: SourceFile;
candidates?: SourceFile[];
// The folder, for createRecords.
source?: ArchiveFolder;
// `attach` over an existing pointer (replace: true).
replacing?: boolean;
};
export type AttachPlan = {
rows: AttachRow[];
// Held videos no folder supplied media for (only when the whole archive was
// considered: no `items`, no `match`).
heldWithoutMedia: string[] | null;
};
export type AttachItem = { id: string; path: string };
export function planAttach(opts: {
folders: readonly ArchiveFolder[];
held: ReadonlySet;
attached: ReadonlySet;
createRecords?: boolean;
replace?: boolean;
match?: RegExp;
// Explicit id → file pairs. When given, they are the whole plan.
items?: readonly AttachItem[];
// Every file in the archive, to resolve `items` paths.
files?: readonly SourceFile[];
}): AttachPlan {
const classify = (id: string, media: SourceFile, folder: string, source?: ArchiveFolder): AttachRow => {
if (opts.held.has(id)) {
if (opts.attached.has(id) && !opts.replace) {
return { class: "already-attached", id, folder, media };
}
return {
class: "attach",
id,
folder,
media,
...(source ? { source } : {}),
...(opts.attached.has(id) ? { replacing: true } : {}),
};
}
return {
class: opts.createRecords ? "create" : "not-held",
id,
folder,
media,
...(source ? { source } : {}),
};
};
if (opts.items) {
const byPath = new Map((opts.files ?? []).map((f) => [f.path, f]));
const folders = new Map(opts.folders.map((f) => [f.folder, f]));
const rows = opts.items.map((item): AttachRow => {
const media = byPath.get(item.path);
const folder = dirName(item.path);
if (!media) return { class: "no-media", id: item.id, folder };
return classify(item.id, media, folder, folders.get(folder));
});
return { rows, heldWithoutMedia: null };
}
const considered = opts.match
? opts.folders.filter(
(f) => opts.match!.test(f.folder) || f.media.some((m) => opts.match!.test(m.path)),
)
: opts.folders;
// Two folders claiming one id: neither is attached on a guess.
const idCount = new Map();
for (const f of considered) {
if (f.id && f.media.length > 0 && !f.lost) idCount.set(f.id, (idCount.get(f.id) ?? 0) + 1);
}
const rows: AttachRow[] = [];
for (const f of considered) {
if (f.lost) {
rows.push({ class: "lost", id: f.id, folder: f.folder });
continue;
}
if (!f.id) {
if (f.media.length > 0) {
rows.push({ class: "unmatched", id: null, folder: f.folder, candidates: f.media });
}
continue;
}
if (f.media.length === 0) {
rows.push({ class: "no-media", id: f.id, folder: f.folder });
continue;
}
if (f.media.length > 1 || (idCount.get(f.id) ?? 0) > 1) {
rows.push({ class: "ambiguous", id: f.id, folder: f.folder, candidates: f.media });
continue;
}
rows.push(classify(f.id, f.media[0], f.folder, f));
}
let heldWithoutMedia: string[] | null = null;
if (!opts.match) {
const supplied = new Set(
rows.filter((r) => r.media && r.id).map((r) => r.id as string),
);
for (const r of rows) if (r.class === "ambiguous" && r.id) supplied.add(r.id);
heldWithoutMedia = [...opts.held].filter((id) => !supplied.has(id)).sort();
}
return { rows, heldWithoutMedia };
}
export function countAttachRows(rows: readonly AttachRow[]): Record {
const out = Object.fromEntries(ATTACH_ROW_CLASSES.map((c) => [c, 0])) as Record;
for (const r of rows) out[r.class] += 1;
return out;
}