commit c0f78d7df78d462df1ad045bfd935ab7db4056c9 parent a2f949e945561a0d69ece4aac455a6a6288e54ad Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st> Date: Tue, 6 Oct 2026 18:56:59 -0400 Merge feat/refresh-video-metadata (refresh one video's metadata from the video page and over ops) Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Diffstat:
20 files changed, 1396 insertions(+), 8 deletions(-)
diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md @@ -249,6 +249,7 @@ export WORKER_TOKEN=<the same secret the editor is running with> pnpm ops sync --json '{"slug":"the-quartering"}' --wait pnpm ops metadata-scan --json '{"slug":"the-quartering"}' +pnpm ops refresh-metadata --json '{"slug":"the-quartering","id":"<videoId>"}' --wait pnpm ops channel-config --json '{"slug":"the-quartering","patch":{"downloadFilterExclude":"rerun"}}' pnpm ops channel-priority --json '{"slugs":["the-quartering"],"operation":"download","tier":"paused"}' pnpm ops lane --json '{"lane":"download","held":true}' diff --git a/common/controller/refreshVideoMetadata.test.ts b/common/controller/refreshVideoMetadata.test.ts @@ -0,0 +1,457 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, readdir, readFile, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import type { ChannelConfig } from "../lib/channelConfig"; +import type { ResolvedCookiePolicy } from "../lib/cookiePolicy"; +import type { AttemptOutcome } from "../ytdlp/runOneYtdlp"; +import { + buildRefreshArgs, + notFetchedRefusal, + refreshSummaryLines, + refreshVideoMetadata, + resolveRefreshTarget, + type RefreshRunner, +} from "./refreshVideoMetadata"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/refreshVideoMetadata.test.ts +// +// One video's metadata re-read. The runner is a stub that records its argv and +// writes the info json where the `-o infojson:` template says — no yt-dlp, no +// network. Every case is a temp corpus. + +const SLUG = "demo"; +const ID = "H64QQZuw-aA"; +const URL = `https://www.youtube.com/watch?v=${ID}`; +const CONFIG = { + handling: "youtube", + url: "https://www.youtube.com/@demo/videos", +} as ChannelConfig; + +// The VOD as it was right after the stream ended: one fragmented audio format, +// no captions. +const BEFORE = { + id: ID, + title: "Live: the stream", + webpage_url: URL, + live_status: "post_live", + duration: 3600, + formats: [ + { + format_id: "140", + ext: "m4a", + protocol: "http_dash_segments", + vcodec: "none", + acodec: "mp4a.40.2", + abr: 144.2, + fragments: [{ url: "a" }], + }, + ], + subtitles: {}, + automatic_captions: {}, + view_count: 10, +}; + +// Hours later: processed formats, auto-captions, `was_live`. +const AFTER = { + ...BEFORE, + live_status: "was_live", + duration: 3601, + formats: [ + { + format_id: "139", + ext: "m4a", + protocol: "https", + vcodec: "none", + acodec: "mp4a.40.5", + abr: 48.8, + }, + { + format_id: "140", + ext: "m4a", + protocol: "https", + vcodec: "none", + acodec: "mp4a.40.2", + abr: 129.5, + }, + { + format_id: "251", + ext: "webm", + protocol: "http_dash_segments", + vcodec: "none", + acodec: "opus", + tbr: 135, + }, + { + format_id: "18", + ext: "mp4", + protocol: "https", + vcodec: "avc1.42001E", + acodec: "mp4a.40.2", + }, + ], + automatic_captions: { en: [{}], "en-orig": [{}], de: [{}] }, + subtitles: { "en-US": [{}] }, + view_count: 50, +}; + +type Fixture = { root: string; paths: Paths; channelDir: string; videoDir: string }; + +async function fixture(opts: { meta?: object; playlist?: string[] } = {}): Promise<Fixture> { + const root = await mkdtemp(path.join(tmpdir(), "refresh-meta-")); + const channelDir = path.join(root, "channels", SLUG); + const videoDir = path.join(channelDir, "data", ID); + await mkdir(path.join(channelDir, "data"), { recursive: true }); + if (opts.meta) { + await mkdir(videoDir, { recursive: true }); + await writeFile(path.join(videoDir, "metadata.info.json"), JSON.stringify(opts.meta)); + } + if (opts.playlist) { + await writeFile(path.join(channelDir, "playlist"), opts.playlist.join("\n") + "\n"); + } + const paths = { channelsDir: path.join(root, "channels"), ytdlpBin: "/nonexistent" } as Paths; + return { root, paths, channelDir, videoDir }; +} + +// A stub yt-dlp: each call takes the next scripted outcome; a successful one +// writes `write` to the `infojson:` template's path under cwd. +function stubRunner( + script: Array<{ exitCode: number; stderr?: string; write?: object }>, +): { run: RefreshRunner; calls: Array<{ cwd: string; args: string[] }> } { + const calls: Array<{ cwd: string; args: string[] }> = []; + const run: RefreshRunner = async (cwd, args) => { + calls.push({ cwd, args }); + const step = script[calls.length - 1]; + if (!step) throw new Error("unexpected spawn"); + if (step.write) { + const tmpl = args.find((a) => a.startsWith("infojson:")); + assert.ok(tmpl, "an infojson output template"); + const file = path.join(cwd, `${tmpl.slice("infojson:".length)}.info.json`); + await mkdir(path.dirname(file), { recursive: true }); + await writeFile(file, JSON.stringify(step.write)); + } + const out: AttemptOutcome = { + exitCode: step.exitCode, + stderrTail: step.stderr ?? "", + archiveLine: null, + }; + return out; + }; + return { run, calls }; +} + +function argAfter(args: string[], flag: string): string | undefined { + const i = args.lastIndexOf(flag); + return i >= 0 ? args[i + 1] : undefined; +} + +test("the argv: metadata only, cookies and pace from the channel, refusals after its own args", () => { + const args = buildRefreshArgs({ + videoId: ID, + videoUrl: URL, + channelConfig: { ...CONFIG, ytdlpExtraArgs: ["--write-subs", "--limit-rate", "1M"] }, + cookies: "firefox", + paceSeconds: 4, + }); + assert.ok(args.includes("--skip-download")); + assert.ok(args.includes("--write-info-json")); + assert.ok(args.includes("--ignore-config")); + assert.equal(argAfter(args, "--cookies-from-browser"), "firefox"); + // The platform's adaptive pace wins over the floor (the last occurrence). + assert.equal(argAfter(args, "--sleep-requests"), "4"); + // The channel's own args are kept, and cannot turn the pass into a + // subtitle fetch: the negation comes after them. + assert.equal(argAfter(args, "--limit-rate"), "1M"); + assert.ok(args.lastIndexOf("--no-write-subs") > args.indexOf("--write-subs")); + assert.ok(args.includes("--no-write-auto-subs")); + assert.ok(args.includes("--no-download-archive")); + // Pinned to THIS video's directory. + assert.ok(args.includes(`infojson:data/${ID}/metadata`)); + assert.deepEqual(args.slice(-2), ["--", URL]); +}); + +test("no cookies in the argv when the policy gives none; the floor pace stands", () => { + const args = buildRefreshArgs({ + videoId: ID, + videoUrl: URL, + channelConfig: CONFIG, + cookies: undefined, + paceSeconds: undefined, + }); + assert.ok(!args.includes("--cookies-from-browser")); + assert.equal(argAfter(args, "--sleep-requests"), "1"); +}); + +test("a refresh rewrites metadata.info.json, records it as `refresh`, and says what it sees", async () => { + const f = await fixture({ meta: BEFORE }); + try { + const { run, calls } = stubRunner([{ exitCode: 0, write: AFTER }]); + let log = ""; + let cleaned = 0; + const result = await refreshVideoMetadata({ + paths: f.paths, + slug: SLUG, + videoId: ID, + videoUrl: URL, + channelConfig: CONFIG, + cookiePolicy: { cookies: "firefox", mode: "always" }, + onLog: (s) => { + log += s; + }, + signal: new AbortController().signal, + onPlatformClean: () => { + cleaned++; + return "youtube answered cleanly: its rate-limit backoff is cleared.\n"; + }, + run, + paceSeconds: 1, + }); + assert.equal(calls.length, 1); + assert.equal(calls[0].cwd, f.channelDir); + // "always" passes cookies up front. + assert.equal(argAfter(calls[0].args, "--cookies-from-browser"), "firefox"); + assert.equal(result.usedCookies, true); + assert.equal(cleaned, 1); + + // The file is the new one, and the history has one `refresh` entry. + const onDisk = JSON.parse(await readFile(path.join(f.videoDir, "metadata.info.json"), "utf8")); + assert.equal(onDisk.live_status, "was_live"); + const history = JSON.parse(await readFile(path.join(f.videoDir, "metadata.history.json"), "utf8")); + assert.equal(history.entries.length, 1); + assert.equal(history.entries[0].by, "refresh"); + assert.deepEqual(history.entries[0].changed.live_status, { from: "post_live", to: "was_live" }); + + // Nothing else was written: no download-outcome.json, no download.log. + assert.deepEqual((await readdir(f.videoDir)).sort(), [ + "metadata.history.json", + "metadata.info.json", + ]); + + // The closing summary, last thing in the log. + assert.deepEqual(result.summary, [ + `Metadata for ${ID} now says:`, + " live_status: was_live", + " formats: 4", + " audio-only formats: 139 m4a https 49k, 140 m4a https 130k, 251 webm http_dash_segments 135k", + " non-fragmented audio: yes (139, 140)", + " English subtitles: en-US", + " English automatic captions: en, en-orig", + " changed: live_status, duration, subtitles_langs, automatic_captions_langs", + ]); + assert.ok(log.trimEnd().endsWith("changed: live_status, duration, subtitles_langs, automatic_captions_langs")); + assert.match(log, /answered cleanly/); + } finally { + await rm(f.root, { recursive: true, force: true }); + } +}); + +test("an auth-gated refresh retries once with cookies in when-required mode, never in defer", async () => { + const gate = "ERROR: [youtube] H64QQZuw-aA: Sign in to confirm your age. This video may be inappropriate for some users."; + for (const [mode, expectRetry] of [ + ["when-required", true], + ["defer", false], + ] as const) { + const f = await fixture({ meta: BEFORE }); + try { + const { run, calls } = stubRunner([ + { exitCode: 1, stderr: gate }, + { exitCode: 0, write: AFTER }, + ]); + const policy: ResolvedCookiePolicy = { cookies: "firefox", mode }; + const attempt = refreshVideoMetadata({ + paths: f.paths, + slug: SLUG, + videoId: ID, + videoUrl: URL, + channelConfig: CONFIG, + cookiePolicy: policy, + onLog: () => {}, + signal: new AbortController().signal, + run, + paceSeconds: 1, + }); + if (expectRetry) { + const result = await attempt; + assert.equal(calls.length, 2, mode); + assert.ok(!calls[0].args.includes("--cookies-from-browser"), mode); + assert.equal(argAfter(calls[1].args, "--cookies-from-browser"), "firefox", mode); + assert.equal(result.usedCookies, true); + } else { + await assert.rejects(attempt, /could not read the metadata/); + assert.equal(calls.length, 1, mode); + } + } finally { + await rm(f.root, { recursive: true, force: true }); + } + } +}); + +test("a rate-limited refresh records the platform cooldown, rewrites nothing and fails", async () => { + const f = await fixture({ meta: BEFORE }); + try { + const { run, calls } = stubRunner([ + { + exitCode: 1, + stderr: `ERROR: [youtube] ${ID}: Unable to download webpage: HTTP Error 429: Too Many Requests`, + }, + ]); + const backoffs: string[] = []; + let cleaned = 0; + await assert.rejects( + refreshVideoMetadata({ + paths: f.paths, + slug: SLUG, + videoId: ID, + videoUrl: URL, + channelConfig: CONFIG, + cookiePolicy: { cookies: "firefox", mode: "when-required" }, + onLog: () => {}, + signal: new AbortController().signal, + onPlatformBackoff: (c) => { + backoffs.push(c); + }, + onPlatformClean: () => { + cleaned++; + return null; + }, + run, + paceSeconds: 1, + }), + /rate-limited the metadata refresh/, + ); + // No cookie retry into a rate limit, one cooldown, no "clean". + assert.equal(calls.length, 1); + assert.deepEqual(backoffs, ["rate_limit"]); + assert.equal(cleaned, 0); + assert.deepEqual(await readdir(f.videoDir), ["metadata.info.json"]); + } finally { + await rm(f.root, { recursive: true, force: true }); + } +}); + +test("the target: an id with no directory is refused, even one the playlist lists — none is created", async () => { + const f = await fixture({ playlist: [`https://www.youtube.com/watch?v=listed00001`] }); + try { + for (const id of ["nope0000000", "listed00001"]) { + assert.deepEqual(await resolveRefreshTarget(f.paths, SLUG, id, CONFIG), { + ok: false, + error: notFetchedRefusal(id, SLUG), + }); + } + assert.equal( + notFetchedRefusal("listed00001", SLUG), + `"listed00001" has not been fetched into ${SLUG} yet — sync, import or download it first; a refresh only re-reads a video already archived.`, + ); + // One path segment only. + for (const bad of ["..", "a/b", "."]) { + const r = await resolveRefreshTarget(f.paths, SLUG, bad, CONFIG); + assert.equal(r.ok, false, bad); + } + // The pass itself refuses too, BEFORE any spawn, and makes no directory. + const { run, calls } = stubRunner([{ exitCode: 0, write: AFTER }]); + await assert.rejects( + refreshVideoMetadata({ + paths: f.paths, + slug: SLUG, + videoId: "listed00001", + videoUrl: "https://www.youtube.com/watch?v=listed00001", + channelConfig: CONFIG, + onLog: () => {}, + signal: new AbortController().signal, + run, + paceSeconds: 1, + }), + /has not been fetched into demo yet/, + ); + assert.equal(calls.length, 0); + assert.deepEqual(await readdir(path.join(f.channelDir, "data")), []); + } finally { + await rm(f.root, { recursive: true, force: true }); + } +}); + +test("the target: a downloaded video resolves to its own webpage_url", async () => { + const f = await fixture({ meta: BEFORE }); + try { + assert.deepEqual(await resolveRefreshTarget(f.paths, SLUG, ID, CONFIG), { + ok: true, + url: URL, + }); + } finally { + await rm(f.root, { recursive: true, force: true }); + } +}); + +test("the target: metadata yt-dlp did not write is refused (archive.org, Wayback, feed-completed)", async () => { + for (const [meta, history, why] of [ + [{ ...BEFORE, webpage_url: "https://archive.org/details/example-item" }, null, /archive\.org record/], + [ + { ...BEFORE, webpage_url: `https://web.archive.org/web/20200101000000/${URL}` }, + null, + /Wayback Machine copy/, + ], + [BEFORE, "feed-backfill", /completed by feed-backfill/], + ] as const) { + const f = await fixture({ meta }); + try { + if (history) { + await writeFile( + path.join(f.videoDir, "metadata.history.json"), + JSON.stringify({ + entries: [ + { + at: "2026-10-01T00:00:00.000Z", + by: history, + from: { sha256: "a", bytes: 1 }, + to: { sha256: "b", bytes: 2 }, + changed: {}, + added: { title: "Episode 1" }, + removed: {}, + counters: {}, + volatile: [], + }, + ], + }), + ); + } + const r = await resolveRefreshTarget(f.paths, SLUG, ID, CONFIG); + assert.equal(r.ok, false); + assert.match(!r.ok ? r.error : "", why); + assert.match(!r.ok ? r.error : "", /would overwrite it/); + } finally { + await rm(f.root, { recursive: true, force: true }); + } + } +}); + +test("the summary: a first write, a byte-identical one, and one where only formats moved", () => { + const first = refreshSummaryLines({ videoId: ID, meta: BEFORE, entries: [], hadBefore: false }); + assert.equal(first.at(-1), " changed: first metadata for this video (nothing to compare)"); + assert.equal(first[3], " audio-only formats: 140 m4a http_dash_segments 144k"); + assert.equal(first[4], " non-fragmented audio: no"); + assert.equal(first[5], " English subtitles: none"); + const same = refreshSummaryLines({ videoId: ID, meta: BEFORE, entries: [], hadBefore: true }); + assert.equal(same.at(-1), " changed: nothing (the file is byte-identical)"); + const formatsOnly = refreshSummaryLines({ + videoId: ID, + meta: BEFORE, + hadBefore: true, + entries: [ + { + at: "2026-10-06T00:00:00.000Z", + by: "refresh", + from: { sha256: "a", bytes: 1 }, + to: { sha256: "b", bytes: 1 }, + changed: {}, + added: {}, + removed: {}, + counters: { view_count: [10, 11] }, + volatile: ["formats"], + }, + ], + }); + assert.equal(formatsOnly.at(-1), " changed: no content keys (formats, URLs or counters only)"); +}); diff --git a/common/controller/refreshVideoMetadata.ts b/common/controller/refreshVideoMetadata.ts @@ -0,0 +1,508 @@ +// RE-READ ONE VIDEO'S METADATA, AND NOTHING ELSE. +// +// The motivating case is a YouTube livestream that has just ended. Its VOD +// starts life with one DASH-fragmented audio format and no captions; hours +// later YouTube finishes processing it and the same URL offers plain https +// formats and auto-captions. The metadata.info.json on disk still says what +// the source said the first time, and everything downstream (the download +// filters, the audio check, the subtitle bucket) decides on that file. Before +// this, the only in-app ways to re-read it were a channel-wide metadata scan +// (which writes the scan store, not the video's file) or a full re-download. +// +// WHAT IT IS: one yt-dlp spawn with `--skip-download --write-info-json`, for +// one video, with exactly what the managed download's metadata prefetch +// (ytdlp/downloadOneManaged.ts, attempt 0) carries — the platform's args at its +// current pace, the cookie policy (`alwaysCookies` up front, one +// `authRetryCookies` retry after an auth failure, none in "defer"), and the +// channel's own extra args. It runs inside `withMetadataHistory` as writer +// "refresh", so what moved lands in metadata.history.json like every other +// rewrite. +// +// WHAT IT IS NOT: a download. No subtitles, no media, no archive line, no +// download-outcome.json (that sidecar records download ATTEMPTS, and the video +// page and the buckets read it as one), no download.log. It never deletes or +// moves anything — and the negations go AFTER the channel's own args, because +// yt-dlp takes the last occurrence of an option and `ytdlpExtraArgs` is free +// text: a channel carrying `--write-subs` must not turn this into a subtitle +// fetch. The output template is pinned to `data/<videoId>/` — the directory +// being refreshed — rather than derived from the URL, so a source whose +// URL-derived id differs from the directory name still rewrites THIS file. +// +// NOT FOR A RECORD yt-dlp DID NOT WRITE. An archive.org record, a Wayback copy +// and a record completed from a podcast feed hold metadata their own writers +// put there; a yt-dlp re-read would replace it with what the bare URL says (the +// hazard the persist prefetch already has with Wayback titles). Those are +// refused, by URL and by the writers in the history. +// +// A RATE LIMIT IS RECORDED, NOT RETRIED: the caller's `onPlatformBackoff` +// writes the shared per-platform cooldown the download lane honours, and the +// job fails. A clean answer settles the platform (`onPlatformClean`), as a +// clean metadata scan does. +// +// The job around it (queue, kind, refusals) is refreshVideoMetadataJob.ts; +// this module needs no registry, so a test drives it with a stub runner. + +import path from "node:path"; +import { readFile, stat } from "node:fs/promises"; +import { + AUTH_RETRY_CLASSES, + classifyDownloadFailure, + parseUnavailableFromStderr, +} from "../lib/availability"; +import type { ChannelConfig } from "../lib/channelConfig"; +import { + DEFAULT_COOKIE_MODE, + alwaysCookies, + authRetryCookies, + type ResolvedCookiePolicy, +} from "../lib/cookiePolicy"; +import type { MetadataHistoryEntry } from "../lib/metadataHistory"; +import { + loadMetadataHistory, + withMetadataHistory, +} from "../lib/metadataHistory-server"; +import { detectPlatform } from "../lib/platform"; +import { parseWaybackUrl } from "../lib/wayback"; +import type { Paths } from "../lib/paths"; +import { channelExtraArgs } from "../ytdlp/channelArgs"; +import { + runOneYtdlp, + type AttemptOutcome, +} from "../ytdlp/runOneYtdlp"; +import { findVideoSourceUrl } from "./undownloadedVideos"; + +const INFO_JSON = "metadata.info.json"; + +// One path segment, the shape of every data/<id>/ name. Checked before the id +// reaches a path.join or an `-o` template. +export function isVideoIdSegment(id: string): boolean { + return ( + id.length > 0 && + id !== "." && + id !== ".." && + !/[/\\\0]/.test(id) + ); +} + +export type RefreshTarget = + | { ok: true; url: string } + | { ok: false; error: string }; + +// The refusal for an id with no directory. Exported for the tests. +export function notFetchedRefusal(videoId: string, slug: string): string { + return ( + `"${videoId}" has not been fetched into ${slug} yet — sync, import or ` + + `download it first; a refresh only re-reads a video already archived.` + ); +} + +// WHICH URL TO RE-READ, OR WHY NOT. Only a video this channel already has a +// directory for: it resolves the way every per-video action does (its +// metadata's webpage_url, else the playlist, else a URL rebuilt from the id +// and the platform). +// +// NO DIRECTORY IS A REFUSAL, never one created. A data/<id>/ holding a +// metadata.info.json is not neutral (see PREFETCH_OWN_FILES in +// ytdlp/downloadOneManaged.ts): buildIndex admits any such directory to the +// index and the published site, and deriveChannelSets reads the name as +// "ever fetched". The metadata scan creates none for the same reason, and +// neither does this — the check is made before yt-dlp is spawned, and again +// by refreshVideoMetadata itself. +export async function resolveRefreshTarget( + paths: Paths, + slug: string, + videoId: string, + config: ChannelConfig, +): Promise<RefreshTarget> { + if (!isVideoIdSegment(videoId)) { + return { ok: false, error: `"${videoId}" is not a video id (one path segment)` }; + } + const videoDir = path.join(paths.channelsDir, slug, "data", videoId); + if (!(await isDirectory(videoDir))) { + return { ok: false, error: notFetchedRefusal(videoId, slug) }; + } + const url = await findVideoSourceUrl(paths, slug, videoId, config); + if (!url) { + return { + ok: false, + error: + "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.", + }; + } + const notYtdlp = await notYtdlpMetadata(videoDir, url); + if (notYtdlp) { + return { + ok: false, + error: + `${videoId}'s metadata is not yt-dlp's to rewrite: ${notYtdlp}. ` + + `A yt-dlp re-read would overwrite it.`, + }; + } + return { ok: true, url }; +} + +async function isDirectory(p: string): Promise<boolean> { + return stat(p) + .then((s) => s.isDirectory()) + .catch(() => false); +} + +// The writers that complete a record yt-dlp cannot describe. A record one of +// them has touched holds what they wrote — a title found for a raw Wayback +// file, an archive.org item's own metadata, a podcast episode's feed entry — +// and a yt-dlp re-read replaces the whole file with what the URL alone says. +const NON_YTDLP_WRITERS: ReadonlySet<string> = new Set([ + "feed-backfill", + "archiveorg-provenance", + "archiveorg-import", + "wayback-provenance", +]); + +// WHY THIS RECORD'S METADATA IS NOT yt-dlp's, or null when it is. Asked of +// the URL first (an archive.org record is written whole from the item API +// with no history entry — a first write is not a rewrite — and a Wayback +// copy's title is the operator's), then of the history. +async function notYtdlpMetadata( + videoDir: string, + url: string, +): Promise<string | null> { + if (detectPlatform(url) === "archiveorg") { + return "an archive.org record is written from the item's metadata"; + } + if (parseWaybackUrl(url)) { + return "a Wayback Machine copy's title and date are set from its provenance"; + } + const history = await loadMetadataHistory(videoDir).catch(() => null); + const by = history?.entries.find((e) => NON_YTDLP_WRITERS.has(e.by))?.by; + return by ? `it was completed by ${by}` : null; +} + +// The argv of the one spawn. Exported for the test, which pins it. +export function buildRefreshArgs(opts: { + videoId: string; + videoUrl: string; + channelConfig: ChannelConfig; + cookies: string | undefined; + // The platform pace; default: its current one (channelExtraArgs). Tests + // pass it. + paceSeconds?: number; +}): string[] { + const dir = `data/${opts.videoId}`; + return [ + "--ignore-config", + "--restrict-filenames", + "--no-playlist", + // One request per second, the floor this repo applies everywhere it + // touches a source; the platform's own (adaptive) pace below raises it. + "--sleep-requests", + "1", + ...channelExtraArgs(opts.channelConfig, opts.cookies, opts.paceSeconds), + // AFTER the channel's own args, so they cannot be undone by them (see the + // header). `-o` likewise: the info json lands in THIS video's directory. + "--skip-download", + "--write-info-json", + "--no-write-subs", + "--no-write-auto-subs", + "--no-write-description", + "--no-write-thumbnail", + "--no-write-comments", + "--no-download-archive", + "--no-write-playlist-metafiles", + "-o", + `${dir}/audio.%(ext)s`, + "-o", + `infojson:${dir}/metadata`, + "--", + opts.videoUrl, + ]; +} + +// --- the closing summary ------------------------------------------------------ + +type Json = Record<string, unknown>; + +function isObject(v: unknown): v is Json { + return typeof v === "object" && v !== null && !Array.isArray(v); +} + +function isAudioOnly(f: Json): boolean { + return ( + f.vcodec === "none" && + typeof f.acodec === "string" && + f.acodec !== "none" + ); +} + +// Fragmented = yt-dlp has to stitch it from pieces (DASH segments, HLS). A +// plain https format is one ranged file, which is what a download wants. +function isFragmented(f: Json): boolean { + const protocol = typeof f.protocol === "string" ? f.protocol : ""; + if (Array.isArray(f.fragments) && f.fragments.length > 0) return true; + return !(protocol === "https" || protocol === "http"); +} + +// `140 m4a http_dash_segments 144k`. +function describeAudioFormat(f: Json): string { + const kbps = + typeof f.abr === "number" && f.abr > 0 + ? f.abr + : typeof f.tbr === "number" && f.tbr > 0 + ? f.tbr + : null; + return [ + String(f.format_id ?? "?"), + String(f.ext ?? "?"), + String(f.protocol ?? "?"), + ...(kbps !== null ? [`${Math.round(kbps)}k`] : []), + ].join(" "); +} + +// The English track names of a caption map (en, en-US, en-orig, …), sorted. +function englishTracks(map: unknown): string[] { + if (!isObject(map)) return []; + return Object.keys(map) + .filter((k) => /^en(-|$)/i.test(k)) + .sort(); +} + +function list(values: string[]): string { + return values.length ? values.join(", ") : "none"; +} + +// The content keys that moved across every entry this refresh appended (an +// auth retry can append a second), in the order they first appear. +function movedKeys(entries: MetadataHistoryEntry[]): { + changed: string[]; + added: string[]; + removed: string[]; +} { + const changed = new Set<string>(); + const added = new Set<string>(); + const removed = new Set<string>(); + for (const e of entries) { + for (const k of Object.keys(e.changed)) changed.add(k); + for (const k of Object.keys(e.added)) added.add(k); + for (const k of Object.keys(e.removed)) removed.add(k); + } + return { changed: [...changed], added: [...added], removed: [...removed] }; +} + +// WHAT THE SOURCE SAYS NOW, in the lines an operator waiting on a VOD reads: +// is it still live, how many formats, which audio formats and whether any of +// them is a plain file, which English captions exist, and what moved. +// +// PURE: the parsed file and the history entries arrive as arguments. +export function refreshSummaryLines(input: { + videoId: string; + meta: Json; + entries: MetadataHistoryEntry[]; + // Was there a metadata.info.json before this pass? With none there is + // nothing to compare against, and the history (correctly) records nothing. + hadBefore: boolean; +}): string[] { + const { meta } = input; + const formats = Array.isArray(meta.formats) ? meta.formats.filter(isObject) : []; + const audio = formats.filter(isAudioOnly); + const plainAudio = audio.filter((f) => !isFragmented(f)); + const moved = movedKeys(input.entries); + let changedLine: string; + if (!input.hadBefore) { + changedLine = "first metadata for this video (nothing to compare)"; + } else if (input.entries.length === 0) { + changedLine = "nothing (the file is byte-identical)"; + } else if ( + moved.changed.length + moved.added.length + moved.removed.length === + 0 + ) { + changedLine = "no content keys (formats, URLs or counters only)"; + } else { + changedLine = [ + ...moved.changed, + ...moved.added.map((k) => `+${k}`), + ...moved.removed.map((k) => `-${k}`), + ].join(", "); + } + return [ + `Metadata for ${input.videoId} now says:`, + ` live_status: ${typeof meta.live_status === "string" ? meta.live_status : "(not set)"}`, + ` formats: ${formats.length}`, + ` audio-only formats: ${list(audio.map(describeAudioFormat))}`, + ` non-fragmented audio: ${ + plainAudio.length + ? `yes (${plainAudio.map((f) => String(f.format_id ?? "?")).join(", ")})` + : "no" + }`, + ` English subtitles: ${list(englishTracks(meta.subtitles))}`, + ` English automatic captions: ${list(englishTracks(meta.automatic_captions))}`, + ` changed: ${changedLine}`, + ]; +} + +// --- the pass ----------------------------------------------------------------- + +export type RefreshRunner = ( + cwd: string, + args: string[], +) => Promise<AttemptOutcome>; + +export type RefreshVideoMetadataOpts = { + paths: Paths; + slug: string; + videoId: string; + videoUrl: string; + channelConfig: ChannelConfig; + // Resolved channel-over-global (resolveCookiePolicy). Omitted = the + // historical default: no prophylactic cookies, retry-only. + cookiePolicy?: ResolvedCookiePolicy; + onLog: (s: string) => void; + signal: AbortSignal; + // The source rate-limited us: record the shared platform cooldown. + onPlatformBackoff?: (failureClass: "rate_limit") => Promise<void> | void; + // The source answered cleanly: settle the platform. Returns the line for the + // log, or null when there was nothing to settle. + onPlatformClean?: () => Promise<string | null> | string | null; + // Who asked, recorded on the history entry ("ops", …). Optional. + requestedBy?: string; + // Test seams: the spawn, and the platform pace in the argv. + run?: RefreshRunner; + paceSeconds?: number; +}; + +export type RefreshVideoMetadataResult = { + // The parsed file after the pass. + meta: Json; + entries: MetadataHistoryEntry[]; + usedCookies: boolean; + summary: string[]; +}; + +function succeeded(exitCode: number | null): boolean { + // yt-dlp: 0 = clean, 101 = a clean stop (break-on-existing / max-downloads). + return exitCode === 0 || exitCode === 101; +} + +function tail(stderr: string): string { + return stderr.trim().split("\n").slice(-3).join(" / "); +} + +// Throws on a failed pass (the job ends `failed` with the reason); on success +// logs the summary and returns it. +export async function refreshVideoMetadata( + opts: RefreshVideoMetadataOpts, +): Promise<RefreshVideoMetadataResult> { + const channelDir = path.join(opts.paths.channelsDir, opts.slug); + const videoDir = path.join(channelDir, "data", opts.videoId); + const infoPath = path.join(videoDir, INFO_JSON); + const policy: ResolvedCookiePolicy = opts.cookiePolicy ?? { + cookies: undefined, + mode: DEFAULT_COOKIE_MODE, + }; + const run: RefreshRunner = + opts.run ?? + ((cwd, args) => + runOneYtdlp( + { ytdlpBin: opts.paths.ytdlpBin, onLog: opts.onLog, signal: opts.signal }, + cwd, + args, + )); + // Never create the directory (see resolveRefreshTarget): refused before + // yt-dlp is spawned, whose `-o` template would otherwise make it. + if (!(await isDirectory(videoDir))) { + throw new Error(notFetchedRefusal(opts.videoId, opts.slug)); + } + const hadBefore = await stat(infoPath) + .then((s) => s.isFile()) + .catch(() => false); + const entries: MetadataHistoryEntry[] = []; + const history = { + by: "refresh" as const, + ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}), + onLog: opts.onLog, + onEntry: (e: MetadataHistoryEntry) => entries.push(e), + }; + const pass = (cookies: string | undefined) => + withMetadataHistory(videoDir, history, () => + run( + channelDir, + buildRefreshArgs({ + videoId: opts.videoId, + videoUrl: opts.videoUrl, + channelConfig: opts.channelConfig, + cookies, + ...(opts.paceSeconds !== undefined ? { paceSeconds: opts.paceSeconds } : {}), + }), + ), + ); + + opts.onLog(`Refreshing the metadata of ${opts.videoId} from ${opts.videoUrl}\n`); + const firstCookies = alwaysCookies(policy); + let usedCookies = Boolean(firstCookies); + let outcome = await pass(firstCookies); + + // The prefetch's auth retry, verbatim in effect: once, with cookies, when + // the failure is an auth/age gate, the mode allows it, and the failed pass + // did not already carry them. "defer" has no retry cookies, so the failure + // stands. + if (!succeeded(outcome.exitCode) && !opts.signal.aborted) { + const cls = parseUnavailableFromStderr(outcome.stderrTail); + const retryCookies = authRetryCookies(policy); + if ( + classifyDownloadFailure(outcome.stderrTail, cls) !== "rate_limit" && + AUTH_RETRY_CLASSES.has(cls) && + retryCookies !== undefined && + !firstCookies + ) { + opts.onLog( + `Metadata refresh auth-required (${cls}); retrying with --cookies-from-browser ${retryCookies}\n`, + ); + usedCookies = true; + outcome = await pass(retryCookies); + } + } + + if (opts.signal.aborted) throw new Error("Cancelled"); + + if (!succeeded(outcome.exitCode)) { + const cls = parseUnavailableFromStderr(outcome.stderrTail); + if (classifyDownloadFailure(outcome.stderrTail, cls) === "rate_limit") { + await opts.onPlatformBackoff?.("rate_limit"); + throw new Error( + `The source rate-limited the metadata refresh of ${opts.videoId}; ` + + `a per-platform cooldown has been recorded. Nothing was rewritten. ` + + `(${tail(outcome.stderrTail)})`, + ); + } + throw new Error( + `yt-dlp could not read the metadata of ${opts.videoId} ` + + `(exit ${outcome.exitCode ?? "null"}): ${tail(outcome.stderrTail)}`, + ); + } + + try { + const line = await opts.onPlatformClean?.(); + if (line) opts.onLog(line); + } catch { + /* shared-state write is best-effort */ + } + + let meta: unknown; + try { + meta = JSON.parse(await readFile(infoPath, "utf8")); + } catch (err) { + throw new Error( + `yt-dlp exited cleanly but ${path.relative(channelDir, infoPath)} ` + + `could not be read: ${(err as Error).message}`, + ); + } + if (!isObject(meta)) { + throw new Error(`${path.relative(channelDir, infoPath)} is not a JSON object`); + } + const summary = refreshSummaryLines({ + videoId: opts.videoId, + meta, + entries, + hadBefore, + }); + opts.onLog(`${summary.join("\n")}\n`); + return { meta, entries, usedCookies, summary }; +} diff --git a/common/controller/refreshVideoMetadataJob.ts b/common/controller/refreshVideoMetadataJob.ts @@ -0,0 +1,121 @@ +// THE METADATA REFRESH AS A JOB — what the video page's "Refresh metadata" and +// `/api/ops/refresh-metadata` enqueue. The work is refreshVideoMetadata.ts; its +// own module, as feedMetadataJob.ts is, so the work can be tested (and run by +// anything else) without loading the job registry. +// +// THE REFUSALS COME BEFORE ANY JOB, in the sentences the rest of the app +// already uses for them, so a click and an ops call get the same answer: +// +// - a channel whose text cannot be read (`assertChannelTextReadable`, the +// guard every `needsText` kind gets — asked first here so its sentence +// wins over "not a video of this channel", which is what an unreadable +// `data/` would otherwise look like); +// - an id with no data/<id>/ — a refresh re-reads a video already +// archived and never creates its directory (resolveRefreshTarget); +// - a HELD platform (`heldPlatformRefusal`), then one in a rate-limit +// cooldown — both `info: true`, as the metadata scan answers them: nothing +// is wrong, the source is resting. +// +// ON THE PLATFORM'S DOWNLOAD QUEUE, like the metadata scan and every per-video +// download: it contends for the source's patience, and the queue serialises it +// with a download of the same video on that queue rewriting the same file. +// `needsText` (jobs/jobKinds.ts): it writes one text file and opens no media, +// so a stalled media drive does not hold it. Replayable — a Retry re-resolves +// the target and re-asks the cooldown. + +import { detectPlatform } from "../lib/platform"; +import { downloadQueueKey, resolveQueueKey } from "../lib/queueKeys"; +import { assertChannelTextReadable } from "../lib/channelMedia"; +import type { ChannelConfig } from "../lib/channelConfig"; +import type { ResolvedCookiePolicy } from "../lib/cookiePolicy"; +import type { Paths } from "../lib/paths"; +import { + heldPlatformRefusal, + platformCooldownRemainingMs, + recordDownloadBackoff, + recordPlatformClean, +} from "../jobs/downloadBackoff"; +import { + runManagedFunction, + type StreamActionResult, +} from "../jobs/streamCommand"; +import { + refreshVideoMetadata, + resolveRefreshTarget, +} from "./refreshVideoMetadata"; + +export const REFRESH_METADATA_JOB_KIND = "refresh-metadata"; + +export type RefreshMetadataJobOpts = { + paths: Paths; + slug: string; + videoId: string; + channelConfig: ChannelConfig; + cookiePolicy: ResolvedCookiePolicy; + // Per-run override of the platform queue key. + queueKey?: string; + requestedBy?: string; + // Called after a successful refresh, inside the job. The editor revalidates + // the video and channel pages here. + afterRun?: () => void | Promise<void>; +}; + +export async function runRefreshMetadataJob( + opts: RefreshMetadataJobOpts, +): Promise<StreamActionResult> { + const { paths, slug, videoId, channelConfig } = opts; + try { + await assertChannelTextReadable(paths, slug, channelConfig); + } catch (err) { + return { ok: false, error: (err as Error).message }; + } + const target = await resolveRefreshTarget(paths, slug, videoId, channelConfig); + if (!target.ok) return { ok: false, error: target.error }; + + // The VIDEO's platform, as the clip fetch keys it: a record whose URL is on + // another host than its channel's (a mirror, a Wayback copy) answers to + // that host's cooldown. + const platform = detectPlatform(target.url) ?? "unknown"; + const held = await heldPlatformRefusal(platform, "The metadata refresh", paths); + if (held) return { ok: false, info: true, error: held }; + const remainingMs = await platformCooldownRemainingMs(platform, paths); + if (remainingMs > 0) { + const secs = Math.ceil(remainingMs / 1000); + return { + ok: false, + info: true, + error: + `${platform} is in a rate-limit cooldown (${secs}s remaining). ` + + `Refresh the metadata once the cooldown lapses.`, + }; + } + + return runManagedFunction({ + kind: REFRESH_METADATA_JOB_KIND, + queueKey: resolveQueueKey(downloadQueueKey(channelConfig), opts.queueKey), + paths, + channelSlug: slug, + videoId, + spec: { + kind: REFRESH_METADATA_JOB_KIND, + slug, + params: { videoId, queueKey: opts.queueKey }, + }, + fn: async (onLog, signal) => { + await refreshVideoMetadata({ + paths, + slug, + videoId, + videoUrl: target.url, + channelConfig, + cookiePolicy: opts.cookiePolicy, + onLog, + signal, + ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}), + onPlatformBackoff: () => recordDownloadBackoff(platform, paths), + onPlatformClean: () => recordPlatformClean(platform, paths), + }); + await opts.afterRun?.(); + }, + }); +} diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts @@ -88,6 +88,8 @@ const ADDED_KINDS: Record<string, { label: string; drainable: boolean }> = { "reports-prepare": { label: "Prepare report media", drainable: false }, // A report site's exports: one pass, cancelled rather than drained. "reports-export": { label: "Export reports", drainable: false }, + // One video's metadata re-read: one spawn, cancelled rather than drained. + "refresh-metadata": { label: "Refresh metadata", drainable: false }, }; test("added kinds carry their pinned label and drainability", () => { @@ -146,6 +148,7 @@ const TEXT_KINDS = [ "fetch-window", "evict-clips", "metadata-scan", + "refresh-metadata", "download-missing-subs", "check-availability", "quick-availability-check", diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -621,6 +621,22 @@ const JOB_KINDS: Record<string, JobKindMeta> = { needsMedia: false, needsText: true, }, + // ONE video's metadata.info.json re-read from its source + // (controller/refreshVideoMetadataJob.ts): a metadata-only yt-dlp pass, no + // subtitles, no media, the rewrite recorded in metadata.history.json. On the + // platform download queue, like the metadata scan, for the same reason. + // `needsText`: it writes one text file and opens no big one. Not drainable + // (one spawn); replayable (the spec carries the video id, and a Retry + // re-resolves the target and re-asks the cooldown). + "refresh-metadata": { + kind: "refresh-metadata", + label: "Refresh metadata", + drainable: false, + replayable: true, + queueKeyStrategy: "platform", + needsMedia: false, + needsText: true, + }, // THE HUB'S AND THE HOMEPAGE'S BUILD AND DEPLOY (release 13 slice W1). They // ran from /sites — the hub since release 7, the homepage since release 11 — // with no entry here, so /jobs showed their raw machine kinds. The labels diff --git a/common/lib/metadataHistory-server.ts b/common/lib/metadataHistory-server.ts @@ -108,10 +108,15 @@ export async function recordMetadataRewrite( // yt-dlp that fails late (a prefetch that wrote the file and then hit a // subtitle error) has still rewritten it. `run`'s own result or error is // returned unchanged; a failure to record is logged through `onLog` and -// swallowed. +// swallowed. `onEntry` hears the entry that was appended, when one was — the +// refresh's closing summary names the keys that moved from it rather than +// reading the sidecar back. export async function withMetadataHistory<T>( videoDir: string, - ctx: MetadataRewriteContext & { onLog?: (line: string) => void }, + ctx: MetadataRewriteContext & { + onLog?: (line: string) => void; + onEntry?: (entry: MetadataHistoryEntry) => void; + }, run: () => Promise<T>, ): Promise<T> { const before = await snapshotMetadata(videoDir); @@ -120,7 +125,8 @@ export async function withMetadataHistory<T>( } finally { if (before) { try { - await recordMetadataRewrite(videoDir, before, ctx); + const entry = await recordMetadataRewrite(videoDir, before, ctx); + if (entry) ctx.onEntry?.(entry); } catch (err) { ctx.onLog?.( `Could not record the metadata history: ${(err as Error).message}\n`, diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts @@ -72,6 +72,12 @@ export const METADATA_HISTORY_WRITERS = [ // for it (`archilyzer wayback refresh --titles`, controller/waybackRefresh.ts) // — a raw media file captured by the Wayback Machine carries no title. "wayback-provenance", + // One video's metadata re-read on request — the video page's "Refresh + // metadata" and `pnpm ops refresh-metadata` (controller/refreshVideoMetadata.ts): + // a metadata-only yt-dlp pass, no subtitles, no media. For a source that + // changes after the first fetch — a livestream VOD that gains its processed + // formats and auto-captions hours after it ends. + "refresh", ] as const; export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number]; diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **One video's metadata can be read again from its source.** A livestream that has just ended offers one fragmented audio format and no captions; hours later the same URL has plain formats and auto-captions, and the video's `metadata.info.json` still said what it said the first time. "Refresh metadata" under the video page's header, and `pnpm ops refresh-metadata --json '{"slug":"…","id":"…"}'`, re-read that one video with no subtitles and no media, on the platform's queue, with the channel's cookie policy and pace; a held platform or one in a rate-limit cooldown is refused, and a rate limit records the cooldown. The rewrite is recorded in the metadata history as `refresh`, and the job's log ends with what the source now says: `live_status`, how many formats, each audio-only format and its protocol, whether any is non-fragmented, the English subtitle and caption tracks, and the keys that changed. Nothing is downloaded or deleted, and no download attempt is recorded. A video not yet fetched into the archive is refused — a refresh never creates its directory — as are an archive.org record, a Wayback copy and a record completed from a podcast feed, whose metadata is not yt-dlp's. Needs a restart of the editor. - **A renamed social channel's posts open again.** Each archived post carries the channel slug it was fetched under, and renaming the channel moves its directory without rewriting them, so every post of a renamed channel named the old slug: the index filed it under the new one, the post page and its thread could not find it there, and MCP links named a channel that no longer existed. A post's channel and slug are now read from the directory it is stored in, wherever it was fetched; the files are not rewritten. The next index build corrects the published records. Needs a restart of the editor. - **A video's other English tracks are readable and searchable where their words differ.** Uploaded captions are not always a transcript of what was said, so the tracks beside the transcript stay: the served `en` beside `en-orig`, a regional or auto-translated track, and the captions a local transcription replaced. One is kept where its words differ from the transcript's and from every track kept before it; identical tracks, most of them, add nothing. The index keeps them in an `alts` sub-DB and writes `track` and `altTracks` onto the transcript record only then, so every other record's page is what it was. A search hit in a word only an alternate holds names the track; one every track says is found once, in the transcript. The video page's **Transcript** card reads the transcript and switches tracks ("Track: original audio captions ▾"); switching changes nothing on disk, and **Set as transcript** stays the way the transcript itself changes. English VTTs are no longer shipped as subtitle tracks. One notion of a track — ids, plain labels, which are kept, how a hit across them is found — lives in `common/lib/captionTracks.ts`. - **The next index build reads the alternate tracks once.** Every record that can hold one — two or more English VTTs, or a transcription beside captions — is re-read from disk, and nothing else; the log says `Alternate tracks v1: N record(s) re-read.` and how many hold a track whose words differ. The version is recorded only when no channel is held. A transcribed video's captions now count toward its change time, so a later caption fetch reaches the index. diff --git a/editor/app/api/ops/_lib.test.ts b/editor/app/api/ops/_lib.test.ts @@ -5,6 +5,7 @@ import { optNonNegativeInt, optPositiveInt, optSubset, + reqVideoId, reqVideoItems, } from "./_lib"; @@ -112,3 +113,14 @@ test("reqVideoItems: a traversing slug or id is refused at the door", () => { assert.throws(() => reqVideoItems({ items: [item] }, "items"), OpsInputError, JSON.stringify(item)); } }); + +test("reqVideoId: one path segment comes back trimmed; a traversing or empty one is refused", () => { + assert.equal(reqVideoId({ id: " H64QQZuw-aA " }, "id"), "H64QQZuw-aA"); + for (const id of ["..", ".", "a/b", "a\\b", "", 7]) { + assert.throws( + () => reqVideoId({ id }, "id"), + OpsInputError, + JSON.stringify(id), + ); + } +}); diff --git a/editor/app/api/ops/_lib.ts b/editor/app/api/ops/_lib.ts @@ -192,14 +192,22 @@ export function reqVideoItems( ); } const slug = reqSlug(entry as OpsBody, "slug"); - const id = reqString(entry as OpsBody, "id"); - if (id === "." || id === ".." || /[/\\\0]/.test(id)) { - throw new OpsInputError(`"${id}" is not a video id (one path segment)`); - } + const id = reqVideoId(entry as OpsBody, "id"); return { slug, id }; }); } +// ONE VIDEO ID, one path segment — checked here for the reason reqSlug is: the +// id reaches a path.join under the channel's `data/`, and a traversing one +// must be refused at the door rather than read as an unknown video. +export function reqVideoId(body: OpsBody, key: string): string { + const id = reqString(body, key); + if (id === "." || id === ".." || /[/\\\0]/.test(id)) { + throw new OpsInputError(`"${id}" is not a video id (one path segment)`); + } + return id; +} + export function reqStringArray(body: OpsBody, key: string): string[] { const v = body[key]; if ( diff --git a/editor/app/api/ops/refresh-metadata/route.test.ts b/editor/app/api/ops/refresh-metadata/route.test.ts @@ -0,0 +1,83 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, readdir, rm, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +// Run with: +// pnpm -C editor exec tsx --test "app/api/ops/refresh-metadata/route.test.ts" +// +// The body's shape and the refusals that come before any job: every case here +// is answered from the disk alone, so a temp corpus with one YouTube channel +// is the whole world — no job is queued and no yt-dlp runs. + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "refresh-metadata-route-")); +const SLUG = "demo-yt"; +const CHANNEL = path.join(ROOT, "channels", SLUG); +// Set before the route (and getPaths, which caches) is first imported. +process.env.WORKER_TOKEN = "test-token"; +process.env.TRANSCRIPTS_DIR = ROOT; +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); +await mkdir(path.join(CHANNEL, "data"), { recursive: true }); +await writeFile( + path.join(CHANNEL, "config.json"), + JSON.stringify({ + handling: "youtube", + name: "Demo", + url: "https://www.youtube.com/@demo/videos", + }), +); +await writeFile( + path.join(CHANNEL, "playlist"), + "https://www.youtube.com/watch?v=listed00001\n", +); +const { POST } = await import("./route"); +test.after(() => rm(ROOT, { recursive: true, force: true })); + +async function post(body: Record<string, unknown>): Promise<{ status: number; error: string }> { + const res = await POST( + new Request("http://localhost/api/ops/refresh-metadata", { + method: "POST", + headers: { + authorization: "Bearer test-token", + "content-type": "application/json", + }, + body: JSON.stringify(body), + }), + ); + return { status: res.status, error: ((await res.json()) as { error?: string }).error ?? "" }; +} + +test("the body is { slug, id, queueKey? }: a missing id and an unknown key are refused", async () => { + const missing = await post({ slug: SLUG }); + assert.equal(missing.status, 400); + assert.match(missing.error, /"id" is required/); + const unknown = await post({ slug: SLUG, id: "listed00001", videoId: "x" }); + assert.equal(unknown.status, 400); + assert.match(unknown.error, /unknown key\(s\): videoId — this route accepts slug, id, queueKey/); +}); + +test("a traversing id or slug is refused at the door", async () => { + const id = await post({ slug: SLUG, id: "../escape" }); + assert.equal(id.status, 400); + assert.match(id.error, /is not a video id \(one path segment\)/); + const slug = await post({ slug: "../../escape", id: "abc" }); + assert.equal(slug.status, 400); + assert.match(slug.error, /is not a valid channel slug/); +}); + +test("an id with no data/<id>/ is refused, even one the playlist lists; no job, no directory", async () => { + for (const id of ["nope0000000", "listed00001"]) { + const stray = await post({ slug: SLUG, id }); + assert.equal(stray.status, 400, id); + assert.equal( + stray.error, + `"${id}" has not been fetched into ${SLUG} yet — sync, import or download it first; a refresh only re-reads a video already archived.`, + ); + } + assert.deepEqual(await readdir(path.join(CHANNEL, "data")), []); + const channel = await post({ slug: "no-such-channel", id: "nope0000000" }); + assert.equal(channel.status, 400); + assert.equal(channel.error, 'Channel "no-such-channel" not found'); + assert.deepEqual(await readdir(path.join(ROOT, ".jobs")).catch(() => []), []); +}); diff --git a/editor/app/api/ops/refresh-metadata/route.ts b/editor/app/api/ops/refresh-metadata/route.ts @@ -0,0 +1,30 @@ +import { refreshVideoMetadataAction } from "../../../channels/[slug]/videos/[id]/videoActions"; +import { jobResponse, ops, optString, reqSlug, reqVideoId } from "../_lib"; + +export const dynamic = "force-dynamic"; + +// POST { slug, id, queueKey? } -> { ok: true, jobId } +// +// Re-read ONE video's metadata.info.json from its source — the video page's +// "Refresh metadata". A metadata-only yt-dlp pass (no subtitles, no media) on +// the platform's queue, with the channel's cookie policy and pace; the rewrite +// is recorded in metadata.history.json as `refresh`, and the job's log ends +// with what the source now says (live_status, formats, audio, English +// captions, the keys that changed). +// +// Every refusal is the action's own sentence: an id with no data/<id>/ (a +// refresh re-reads a video already archived and never creates its directory), +// a channel whose text cannot be read, an archive.org, Wayback or +// feed-completed record, a held platform or one in a rate-limit cooldown +// (`info: true`). +export async function POST(request: Request) { + return ops(request, ["slug", "id", "queueKey"], async (body) => + jobResponse( + await refreshVideoMetadataAction( + reqSlug(body, "slug"), + reqVideoId(body, "id"), + optString(body, "queueKey"), + ), + ), + ); +} diff --git a/editor/app/channels/[slug]/videos/[id]/components/RefreshMetadataControl.tsx b/editor/app/channels/[slug]/videos/[id]/components/RefreshMetadataControl.tsx @@ -0,0 +1,46 @@ +"use client"; + +import { useState } from "react"; +import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog"; +import { QueueControl } from "../../../../../components/QueueControl"; +import { cancelJobAction } from "../../../../../jobs/actions"; +import { refreshVideoMetadataAction } from "../videoActions"; + +// "Refresh metadata": re-read this video's metadata.info.json from its source, +// with no subtitles and no media — for a source that changed after the first +// fetch, like a livestream VOD that gains its processed formats and captions +// hours after it ends. It sits under the header's metadata (and its history), +// which is what it rewrites. The job's log ends with what the source now says; +// the rewrite appears in the history above as `refresh`. +export function RefreshMetadataControl({ + slug, + videoId, + defaultQueueKey, + existingQueues, +}: { + slug: string; + videoId: string; + defaultQueueKey: string; + existingQueues: string[]; +}) { + const [queueKey, setQueueKey] = useState(defaultQueueKey); + const actionLabel = `Refresh metadata for ${videoId}`; + return ( + <StreamActionLog + trigger={() => refreshVideoMetadataAction(slug, videoId, queueKey)} + cancelAction={cancelJobAction} + buttonLabel="Refresh metadata" + runningLabel="Refreshing metadata…" + label={actionLabel} + extraControls={ + <QueueControl + value={queueKey} + onChange={setQueueKey} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + actionLabel={actionLabel} + /> + } + /> + ); +} diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -43,6 +43,7 @@ import { VideoPanel, type VideoFile } from "./components/VideoPanel"; import { OperationPanel } from "./components/OperationPanel"; import { TagsPanel } from "./components/TagsPanel"; import { MetadataHistoryDetails } from "./components/MetadataHistoryDetails"; +import { RefreshMetadataControl } from "./components/RefreshMetadataControl"; import { loadVideoTags } from "./lib/videoTags"; import { loadVideoOperationPanels } from "./lib/videoOperationPanels"; import { channelTextStall } from "yt-dlp-transcript-common/lib/channelMedia"; @@ -303,6 +304,20 @@ export default async function VideoDetailPage({ </details> )} {metadataHistory && <MetadataHistoryDetails view={metadataHistory} />} + {/* Re-read the metadata above from the source (no subtitles, no + media). Offered only for a video already in the archive (a + data/<id>/ with files — a listed, never-fetched video's page has + none, and the action refuses it rather than create one), and not + for an archive.org record or a Wayback copy: their metadata is not + yt-dlp's, and the action refuses them too. */} + {dirData.files.length > 0 && !archiveOrg && !wayback && ( + <RefreshMetadataControl + slug={slug} + videoId={id} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + /> + )} </header> <RunningJobsList jobs={activeJobs} hideChannelSlug hideVideoId /> diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -93,6 +93,7 @@ import { sourceFetchFailure, } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; import { runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; +import { runRefreshMetadataJob } from "yt-dlp-transcript-common/controller/refreshVideoMetadataJob"; import { runManagedFunction, type StreamActionResult, @@ -254,6 +255,35 @@ export async function downloadVideoPipelineAction( }); } +// "Refresh metadata": re-read this one video's metadata.info.json from its +// source — no subtitles, no media — for a source that changed after the first +// fetch (a livestream VOD gaining its processed formats and captions). The job +// and every refusal are common/controller/refreshVideoMetadataJob.ts, so the +// ops route and a Retry from /jobs answer exactly as this click does; what is +// the editor's own is the two revalidations. +export async function refreshVideoMetadataAction( + slug: string, + videoId: string, + queueKey?: string, +): Promise<StreamActionResult> { + const r = await loadConfigOrError(slug); + if (!r.ok) return r; + return runRefreshMetadataJob({ + paths: getPaths(), + slug, + videoId, + channelConfig: r.config, + cookiePolicy: resolveCookiePolicy(getSettings(), r.config), + queueKey, + afterRun: () => { + safeRevalidate([ + `/channels/${slug}/videos/${videoId}`, + `/channels/${slug}`, + ]); + }, + }); +} + // Re-fetch an already-downloaded video purely to archive its SOURCE container, // without disturbing the existing transcript. Forces the persistence rule on // (keepSourceVideoOverride=true) so downloadOneManaged downloads the full video diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -53,7 +53,10 @@ import { redownloadIncompleteBucketAction, redownloadShortAudioBucketAction, } from "../channels/[slug]/incompleteTranscriptActions"; -import { replayFetchWindowAction } from "../channels/[slug]/videos/[id]/videoActions"; +import { + refreshVideoMetadataAction, + replayFetchWindowAction, +} from "../channels/[slug]/videos/[id]/videoActions"; import { reportsPrepareAction } from "../sites/lib/reportsPrepareAction"; import { reportsExportAction } from "../sites/lib/reportsExportAction"; @@ -262,6 +265,19 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { const { queueKey } = params(spec); return runMetadataScanAction(spec.slug, queueKey); }, + // One video's metadata re-read. The action re-resolves the target and + // re-asks the platform's hold and cooldown, as a click does. + "refresh-metadata": (spec) => { + const { p, queueKey } = params(spec); + const videoId = str(p.videoId); + if (!videoId) { + return Promise.resolve({ + ok: false, + error: "Job spec is missing its videoId.", + }); + } + return refreshVideoMetadataAction(spec.slug, videoId, queueKey); + }, "check-post-availability": (spec) => { const { p, queueKey } = params(spec); return checkPostAvailabilityAction( diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts @@ -183,6 +183,7 @@ test("a traversing slug is refused at the door, on every route that takes one", // each. const cases: [string, Record<string, unknown>][] = [ ["metadata-scan", { slug: "../../escape" }], + ["refresh-metadata", { slug: "../../escape", id: "abc123" }], ["sync", { slug: "../../escape" }], ["download-missing", { slug: "../../escape" }], ["import-video", { slug: "../../escape", url: "https://example.com/v" }], diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -26,6 +26,7 @@ // // pnpm ops sync --json '{"slug":"the-quartering"}' --wait // pnpm ops metadata-scan --json '{"slug":"the-quartering"}' +// pnpm ops refresh-metadata --json '{"slug":"the-quartering","id":"<videoId>"}' --wait // pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait // pnpm ops import-video --json '{"slug":"demo-archive","url":"https://archive.org/details/example-item"}' // pnpm ops import-archive-org --json '{"slug":"demo-archive","item":"example-item","match":"\\.mp4$"}' --wait @@ -98,6 +99,9 @@ const ACTIONS = [ "channel-priority", "channel-config", "metadata-scan", + // ONE video's metadata.info.json re-read from its source ({slug, id}): no + // subtitles, no media; the rewrite lands in metadata.history.json. + "refresh-metadata", "import-video", // Chosen media files of ONE archive.org item ({slug, item, files: [...] | // match: "<regex>", dryRun?}): one job, one file at a time, paced. @@ -343,6 +347,13 @@ export function usage() { ' report.md and evidence-pack.zip for the site\'s build to publish:', ' {"siteId", "reportId"?, "formats"?: ["html","pdf","md","zip"]}.', "", + 'refresh-metadata re-reads ONE video\'s metadata.info.json from its source', + ' (no subtitles, no media) on the platform\'s queue: {"slug", "id"}. The', + ' job\'s log ends with what the source now says — live_status, formats,', + ' audio-only formats and whether any is non-fragmented, English captions,', + ' the keys that changed. An id with no data/<id>/ is refused (a refresh', + ' re-reads a video already archived), as are archive.org and Wayback records.', + "", 'retry-bucket runs one bucket of a channel\'s report as one job, past any', ' lane hold: {"slug", "bucket"}. "ids": [...] runs only those videos, and', " every one must be in the bucket (a stray id is refused, named); a job run", diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -451,3 +451,20 @@ test("feed-metadata posts {slug, dryRun} to /api/ops/feed-metadata", () => { assert.equal(p.path, "/api/ops/feed-metadata"); assert.deepEqual(p.body, { slug: "demo-channel", dryRun: true }); }); + +// One video's metadata re-read (the video page's "Refresh metadata"): a POST +// with the body passed through untouched — the route judges it. +test("refresh-metadata is a POST to its route, named in the usage", () => { + const p = parseArgs([ + "refresh-metadata", + "--json", + '{"slug":"c","id":"H64QQZuw-aA"}', + "--wait", + ]); + assert.equal(p.method, "POST"); + assert.equal(p.path, "/api/ops/refresh-metadata"); + assert.deepEqual(p.body, { slug: "c", id: "H64QQZuw-aA" }); + assert.equal(p.wait, true); + assert.match(usage(), /Actions:.*metadata-scan, refresh-metadata/); + assert.match(usage(), /refresh-metadata re-reads ONE video/); +});