commit c3e030f6f7c79f1857c20f0cc596d4bd973c2251 parent a2f949e945561a0d69ece4aac455a6a6288e54ad Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st> Date: Tue, 6 Oct 2026 18:16:25 -0400 metadata: refresh one video's metadata from the video page and over ops A livestream that has just ended offers one fragmented audio format and no captions; hours later the same URL has plain formats and auto-captions, and the video's metadata.info.json still said what it said the first time. The only in-app ways to re-read it were the channel-wide metadata scan (which writes the scan store, not the video's file) or a re-download. controller/refreshVideoMetadata.ts runs one yt-dlp pass with --skip-download --write-info-json, the platform args at its current pace, the cookie policy and the prefetch's one auth retry, pinned to data/<id>/, inside withMetadataHistory as the new writer "refresh". A rate limit records the platform cooldown; a clean answer settles it. No subtitles, no media, no download-outcome.json. The log ends with live_status, formats, the audio-only formats and their protocol, whether any is non-fragmented, the English tracks and the keys that changed. refreshVideoMetadataJob.ts is the job ("refresh-metadata", needsText, replayable, on the platform's queue). It refuses an id with no directory that the playlist does not list, a held platform or one in cooldown, and an archive.org, Wayback or feed-completed record, whose metadata a yt-dlp re-read would overwrite. The video page has a "Refresh metadata" control under its header; /api/ops/refresh-metadata and `pnpm ops refresh-metadata` reach the same action. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Diffstat:
21 files changed, 1384 insertions(+), 14 deletions(-)
diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md @@ -249,6 +249,7 @@ export WORKER_TOKEN=<the same secret the editor is running with> pnpm ops sync --json '{"slug":"the-quartering"}' --wait pnpm ops metadata-scan --json '{"slug":"the-quartering"}' +pnpm ops refresh-metadata --json '{"slug":"the-quartering","id":"<videoId>"}' --wait pnpm ops channel-config --json '{"slug":"the-quartering","patch":{"downloadFilterExclude":"rerun"}}' pnpm ops channel-priority --json '{"slugs":["the-quartering"],"operation":"download","tier":"paused"}' pnpm ops lane --json '{"lane":"download","held":true}' diff --git a/common/controller/refreshVideoMetadata.test.ts b/common/controller/refreshVideoMetadata.test.ts @@ -0,0 +1,442 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, readdir, readFile, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import type { ChannelConfig } from "../lib/channelConfig"; +import type { ResolvedCookiePolicy } from "../lib/cookiePolicy"; +import type { AttemptOutcome } from "../ytdlp/runOneYtdlp"; +import { + buildRefreshArgs, + refreshSummaryLines, + refreshVideoMetadata, + resolveRefreshTarget, + type RefreshRunner, +} from "./refreshVideoMetadata"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/refreshVideoMetadata.test.ts +// +// One video's metadata re-read. The runner is a stub that records its argv and +// writes the info json where the `-o infojson:` template says — no yt-dlp, no +// network. Every case is a temp corpus. + +const SLUG = "demo"; +const ID = "H64QQZuw-aA"; +const URL = `https://www.youtube.com/watch?v=${ID}`; +const CONFIG = { + handling: "youtube", + url: "https://www.youtube.com/@demo/videos", +} as ChannelConfig; + +// The VOD as it was right after the stream ended: one fragmented audio format, +// no captions. +const BEFORE = { + id: ID, + title: "Live: the stream", + webpage_url: URL, + live_status: "post_live", + duration: 3600, + formats: [ + { + format_id: "140", + ext: "m4a", + protocol: "http_dash_segments", + vcodec: "none", + acodec: "mp4a.40.2", + abr: 144.2, + fragments: [{ url: "a" }], + }, + ], + subtitles: {}, + automatic_captions: {}, + view_count: 10, +}; + +// Hours later: processed formats, auto-captions, `was_live`. +const AFTER = { + ...BEFORE, + live_status: "was_live", + duration: 3601, + formats: [ + { + format_id: "139", + ext: "m4a", + protocol: "https", + vcodec: "none", + acodec: "mp4a.40.5", + abr: 48.8, + }, + { + format_id: "140", + ext: "m4a", + protocol: "https", + vcodec: "none", + acodec: "mp4a.40.2", + abr: 129.5, + }, + { + format_id: "251", + ext: "webm", + protocol: "http_dash_segments", + vcodec: "none", + acodec: "opus", + tbr: 135, + }, + { + format_id: "18", + ext: "mp4", + protocol: "https", + vcodec: "avc1.42001E", + acodec: "mp4a.40.2", + }, + ], + automatic_captions: { en: [{}], "en-orig": [{}], de: [{}] }, + subtitles: { "en-US": [{}] }, + view_count: 50, +}; + +type Fixture = { root: string; paths: Paths; channelDir: string; videoDir: string }; + +async function fixture(opts: { meta?: object; playlist?: string[] } = {}): Promise<Fixture> { + const root = await mkdtemp(path.join(tmpdir(), "refresh-meta-")); + const channelDir = path.join(root, "channels", SLUG); + const videoDir = path.join(channelDir, "data", ID); + await mkdir(path.join(channelDir, "data"), { recursive: true }); + if (opts.meta) { + await mkdir(videoDir, { recursive: true }); + await writeFile(path.join(videoDir, "metadata.info.json"), JSON.stringify(opts.meta)); + } + if (opts.playlist) { + await writeFile(path.join(channelDir, "playlist"), opts.playlist.join("\n") + "\n"); + } + const paths = { channelsDir: path.join(root, "channels"), ytdlpBin: "/nonexistent" } as Paths; + return { root, paths, channelDir, videoDir }; +} + +// A stub yt-dlp: each call takes the next scripted outcome; a successful one +// writes `write` to the `infojson:` template's path under cwd. +function stubRunner( + script: Array<{ exitCode: number; stderr?: string; write?: object }>, +): { run: RefreshRunner; calls: Array<{ cwd: string; args: string[] }> } { + const calls: Array<{ cwd: string; args: string[] }> = []; + const run: RefreshRunner = async (cwd, args) => { + calls.push({ cwd, args }); + const step = script[calls.length - 1]; + if (!step) throw new Error("unexpected spawn"); + if (step.write) { + const tmpl = args.find((a) => a.startsWith("infojson:")); + assert.ok(tmpl, "an infojson output template"); + const file = path.join(cwd, `${tmpl.slice("infojson:".length)}.info.json`); + await mkdir(path.dirname(file), { recursive: true }); + await writeFile(file, JSON.stringify(step.write)); + } + const out: AttemptOutcome = { + exitCode: step.exitCode, + stderrTail: step.stderr ?? "", + archiveLine: null, + }; + return out; + }; + return { run, calls }; +} + +function argAfter(args: string[], flag: string): string | undefined { + const i = args.lastIndexOf(flag); + return i >= 0 ? args[i + 1] : undefined; +} + +test("the argv: metadata only, cookies and pace from the channel, refusals after its own args", () => { + const args = buildRefreshArgs({ + videoId: ID, + videoUrl: URL, + channelConfig: { ...CONFIG, ytdlpExtraArgs: ["--write-subs", "--limit-rate", "1M"] }, + cookies: "firefox", + paceSeconds: 4, + }); + assert.ok(args.includes("--skip-download")); + assert.ok(args.includes("--write-info-json")); + assert.ok(args.includes("--ignore-config")); + assert.equal(argAfter(args, "--cookies-from-browser"), "firefox"); + // The platform's adaptive pace wins over the floor (the last occurrence). + assert.equal(argAfter(args, "--sleep-requests"), "4"); + // The channel's own args are kept, and cannot turn the pass into a + // subtitle fetch: the negation comes after them. + assert.equal(argAfter(args, "--limit-rate"), "1M"); + assert.ok(args.lastIndexOf("--no-write-subs") > args.indexOf("--write-subs")); + assert.ok(args.includes("--no-write-auto-subs")); + assert.ok(args.includes("--no-download-archive")); + // Pinned to THIS video's directory. + assert.ok(args.includes(`infojson:data/${ID}/metadata`)); + assert.deepEqual(args.slice(-2), ["--", URL]); +}); + +test("no cookies in the argv when the policy gives none; the floor pace stands", () => { + const args = buildRefreshArgs({ + videoId: ID, + videoUrl: URL, + channelConfig: CONFIG, + cookies: undefined, + paceSeconds: undefined, + }); + assert.ok(!args.includes("--cookies-from-browser")); + assert.equal(argAfter(args, "--sleep-requests"), "1"); +}); + +test("a refresh rewrites metadata.info.json, records it as `refresh`, and says what it sees", async () => { + const f = await fixture({ meta: BEFORE }); + try { + const { run, calls } = stubRunner([{ exitCode: 0, write: AFTER }]); + let log = ""; + let cleaned = 0; + const result = await refreshVideoMetadata({ + paths: f.paths, + slug: SLUG, + videoId: ID, + videoUrl: URL, + channelConfig: CONFIG, + cookiePolicy: { cookies: "firefox", mode: "always" }, + onLog: (s) => { + log += s; + }, + signal: new AbortController().signal, + onPlatformClean: () => { + cleaned++; + return "youtube answered cleanly: its rate-limit backoff is cleared.\n"; + }, + run, + paceSeconds: 1, + }); + assert.equal(calls.length, 1); + assert.equal(calls[0].cwd, f.channelDir); + // "always" passes cookies up front. + assert.equal(argAfter(calls[0].args, "--cookies-from-browser"), "firefox"); + assert.equal(result.usedCookies, true); + assert.equal(cleaned, 1); + + // The file is the new one, and the history has one `refresh` entry. + const onDisk = JSON.parse(await readFile(path.join(f.videoDir, "metadata.info.json"), "utf8")); + assert.equal(onDisk.live_status, "was_live"); + const history = JSON.parse(await readFile(path.join(f.videoDir, "metadata.history.json"), "utf8")); + assert.equal(history.entries.length, 1); + assert.equal(history.entries[0].by, "refresh"); + assert.deepEqual(history.entries[0].changed.live_status, { from: "post_live", to: "was_live" }); + + // Nothing else was written: no download-outcome.json, no download.log. + assert.deepEqual((await readdir(f.videoDir)).sort(), [ + "metadata.history.json", + "metadata.info.json", + ]); + + // The closing summary, last thing in the log. + assert.deepEqual(result.summary, [ + `Metadata for ${ID} now says:`, + " live_status: was_live", + " formats: 4", + " audio-only formats: 139 m4a https 49k, 140 m4a https 130k, 251 webm http_dash_segments 135k", + " non-fragmented audio: yes (139, 140)", + " English subtitles: en-US", + " English automatic captions: en, en-orig", + " changed: live_status, duration, subtitles_langs, automatic_captions_langs", + ]); + assert.ok(log.trimEnd().endsWith("changed: live_status, duration, subtitles_langs, automatic_captions_langs")); + assert.match(log, /answered cleanly/); + } finally { + await rm(f.root, { recursive: true, force: true }); + } +}); + +test("an auth-gated refresh retries once with cookies in when-required mode, never in defer", async () => { + const gate = "ERROR: [youtube] H64QQZuw-aA: Sign in to confirm your age. This video may be inappropriate for some users."; + for (const [mode, expectRetry] of [ + ["when-required", true], + ["defer", false], + ] as const) { + const f = await fixture({ meta: BEFORE }); + try { + const { run, calls } = stubRunner([ + { exitCode: 1, stderr: gate }, + { exitCode: 0, write: AFTER }, + ]); + const policy: ResolvedCookiePolicy = { cookies: "firefox", mode }; + const attempt = refreshVideoMetadata({ + paths: f.paths, + slug: SLUG, + videoId: ID, + videoUrl: URL, + channelConfig: CONFIG, + cookiePolicy: policy, + onLog: () => {}, + signal: new AbortController().signal, + run, + paceSeconds: 1, + }); + if (expectRetry) { + const result = await attempt; + assert.equal(calls.length, 2, mode); + assert.ok(!calls[0].args.includes("--cookies-from-browser"), mode); + assert.equal(argAfter(calls[1].args, "--cookies-from-browser"), "firefox", mode); + assert.equal(result.usedCookies, true); + } else { + await assert.rejects(attempt, /could not read the metadata/); + assert.equal(calls.length, 1, mode); + } + } finally { + await rm(f.root, { recursive: true, force: true }); + } + } +}); + +test("a rate-limited refresh records the platform cooldown, rewrites nothing and fails", async () => { + const f = await fixture({ meta: BEFORE }); + try { + const { run, calls } = stubRunner([ + { + exitCode: 1, + stderr: `ERROR: [youtube] ${ID}: Unable to download webpage: HTTP Error 429: Too Many Requests`, + }, + ]); + const backoffs: string[] = []; + let cleaned = 0; + await assert.rejects( + refreshVideoMetadata({ + paths: f.paths, + slug: SLUG, + videoId: ID, + videoUrl: URL, + channelConfig: CONFIG, + cookiePolicy: { cookies: "firefox", mode: "when-required" }, + onLog: () => {}, + signal: new AbortController().signal, + onPlatformBackoff: (c) => { + backoffs.push(c); + }, + onPlatformClean: () => { + cleaned++; + return null; + }, + run, + paceSeconds: 1, + }), + /rate-limited the metadata refresh/, + ); + // No cookie retry into a rate limit, one cooldown, no "clean". + assert.equal(calls.length, 1); + assert.deepEqual(backoffs, ["rate_limit"]); + assert.equal(cleaned, 0); + assert.deepEqual(await readdir(f.videoDir), ["metadata.info.json"]); + } finally { + await rm(f.root, { recursive: true, force: true }); + } +}); + +test("the target: an id with neither a directory nor a playlist entry is refused", async () => { + const f = await fixture({ playlist: [`https://www.youtube.com/watch?v=listed00001`] }); + try { + const stray = await resolveRefreshTarget(f.paths, SLUG, "nope0000000", CONFIG); + assert.deepEqual(stray, { + ok: false, + error: + `"nope0000000" is not a video of ${SLUG}: there is no data/nope0000000/ ` + + `directory and the channel's playlist does not list it.`, + }); + // Listed but never downloaded: accepted, and the directory is new. + assert.deepEqual(await resolveRefreshTarget(f.paths, SLUG, "listed00001", CONFIG), { + ok: true, + url: "https://www.youtube.com/watch?v=listed00001", + hasDir: false, + }); + // One path segment only. + for (const bad of ["..", "a/b", "."]) { + const r = await resolveRefreshTarget(f.paths, SLUG, bad, CONFIG); + assert.equal(r.ok, false, bad); + } + } finally { + await rm(f.root, { recursive: true, force: true }); + } +}); + +test("the target: a downloaded video resolves to its own webpage_url", async () => { + const f = await fixture({ meta: BEFORE }); + try { + assert.deepEqual(await resolveRefreshTarget(f.paths, SLUG, ID, CONFIG), { + ok: true, + url: URL, + hasDir: true, + }); + } finally { + await rm(f.root, { recursive: true, force: true }); + } +}); + +test("the target: metadata yt-dlp did not write is refused (archive.org, Wayback, feed-completed)", async () => { + for (const [meta, history, why] of [ + [{ ...BEFORE, webpage_url: "https://archive.org/details/example-item" }, null, /archive\.org record/], + [ + { ...BEFORE, webpage_url: `https://web.archive.org/web/20200101000000/${URL}` }, + null, + /Wayback Machine copy/, + ], + [BEFORE, "feed-backfill", /completed by feed-backfill/], + ] as const) { + const f = await fixture({ meta }); + try { + if (history) { + await writeFile( + path.join(f.videoDir, "metadata.history.json"), + JSON.stringify({ + entries: [ + { + at: "2026-10-01T00:00:00.000Z", + by: history, + from: { sha256: "a", bytes: 1 }, + to: { sha256: "b", bytes: 2 }, + changed: {}, + added: { title: "Episode 1" }, + removed: {}, + counters: {}, + volatile: [], + }, + ], + }), + ); + } + const r = await resolveRefreshTarget(f.paths, SLUG, ID, CONFIG); + assert.equal(r.ok, false); + assert.match(!r.ok ? r.error : "", why); + assert.match(!r.ok ? r.error : "", /would overwrite it/); + } finally { + await rm(f.root, { recursive: true, force: true }); + } + } +}); + +test("the summary: a first write, a byte-identical one, and one where only formats moved", () => { + const first = refreshSummaryLines({ videoId: ID, meta: BEFORE, entries: [], hadBefore: false }); + assert.equal(first.at(-1), " changed: first metadata for this video (nothing to compare)"); + assert.equal(first[3], " audio-only formats: 140 m4a http_dash_segments 144k"); + assert.equal(first[4], " non-fragmented audio: no"); + assert.equal(first[5], " English subtitles: none"); + const same = refreshSummaryLines({ videoId: ID, meta: BEFORE, entries: [], hadBefore: true }); + assert.equal(same.at(-1), " changed: nothing (the file is byte-identical)"); + const formatsOnly = refreshSummaryLines({ + videoId: ID, + meta: BEFORE, + hadBefore: true, + entries: [ + { + at: "2026-10-06T00:00:00.000Z", + by: "refresh", + from: { sha256: "a", bytes: 1 }, + to: { sha256: "b", bytes: 1 }, + changed: {}, + added: {}, + removed: {}, + counters: { view_count: [10, 11] }, + volatile: ["formats"], + }, + ], + }); + assert.equal(formatsOnly.at(-1), " changed: no content keys (formats, URLs or counters only)"); +}); diff --git a/common/controller/refreshVideoMetadata.ts b/common/controller/refreshVideoMetadata.ts @@ -0,0 +1,493 @@ +// RE-READ ONE VIDEO'S METADATA, AND NOTHING ELSE. +// +// The motivating case is a YouTube livestream that has just ended. Its VOD +// starts life with one DASH-fragmented audio format and no captions; hours +// later YouTube finishes processing it and the same URL offers plain https +// formats and auto-captions. The metadata.info.json on disk still says what +// the source said the first time, and everything downstream (the download +// filters, the audio check, the subtitle bucket) decides on that file. Before +// this, the only in-app ways to re-read it were a channel-wide metadata scan +// (which writes the scan store, not the video's file) or a full re-download. +// +// WHAT IT IS: one yt-dlp spawn with `--skip-download --write-info-json`, for +// one video, with exactly what the managed download's metadata prefetch +// (ytdlp/downloadOneManaged.ts, attempt 0) carries — the platform's args at its +// current pace, the cookie policy (`alwaysCookies` up front, one +// `authRetryCookies` retry after an auth failure, none in "defer"), and the +// channel's own extra args. It runs inside `withMetadataHistory` as writer +// "refresh", so what moved lands in metadata.history.json like every other +// rewrite. +// +// WHAT IT IS NOT: a download. No subtitles, no media, no archive line, no +// download-outcome.json (that sidecar records download ATTEMPTS, and the video +// page and the buckets read it as one), no download.log. It never deletes or +// moves anything — and the negations go AFTER the channel's own args, because +// yt-dlp takes the last occurrence of an option and `ytdlpExtraArgs` is free +// text: a channel carrying `--write-subs` must not turn this into a subtitle +// fetch. The output template is pinned to `data/<videoId>/` — the directory +// being refreshed — rather than derived from the URL, so a source whose +// URL-derived id differs from the directory name still rewrites THIS file. +// +// NOT FOR A RECORD yt-dlp DID NOT WRITE. An archive.org record, a Wayback copy +// and a record completed from a podcast feed hold metadata their own writers +// put there; a yt-dlp re-read would replace it with what the bare URL says (the +// hazard the persist prefetch already has with Wayback titles). Those are +// refused, by URL and by the writers in the history. +// +// A RATE LIMIT IS RECORDED, NOT RETRIED: the caller's `onPlatformBackoff` +// writes the shared per-platform cooldown the download lane honours, and the +// job fails. A clean answer settles the platform (`onPlatformClean`), as a +// clean metadata scan does. +// +// The job around it (queue, kind, refusals) is refreshVideoMetadataJob.ts; +// this module needs no registry, so a test drives it with a stub runner. + +import path from "node:path"; +import { readFile, stat } from "node:fs/promises"; +import { + AUTH_RETRY_CLASSES, + classifyDownloadFailure, + parseUnavailableFromStderr, +} from "../lib/availability"; +import type { ChannelConfig } from "../lib/channelConfig"; +import { + DEFAULT_COOKIE_MODE, + alwaysCookies, + authRetryCookies, + type ResolvedCookiePolicy, +} from "../lib/cookiePolicy"; +import type { MetadataHistoryEntry } from "../lib/metadataHistory"; +import { + loadMetadataHistory, + withMetadataHistory, +} from "../lib/metadataHistory-server"; +import { detectPlatform } from "../lib/platform"; +import { parseWaybackUrl } from "../lib/wayback"; +import type { Paths } from "../lib/paths"; +import { channelExtraArgs } from "../ytdlp/channelArgs"; +import { + runOneYtdlp, + type AttemptOutcome, +} from "../ytdlp/runOneYtdlp"; +import { findPlaylistUrl, findVideoSourceUrl } from "./undownloadedVideos"; + +const INFO_JSON = "metadata.info.json"; + +// One path segment, the shape of every data/<id>/ name. Checked before the id +// reaches a path.join or an `-o` template. +export function isVideoIdSegment(id: string): boolean { + return ( + id.length > 0 && + id !== "." && + id !== ".." && + !/[/\\\0]/.test(id) + ); +} + +export type RefreshTarget = + | { ok: true; url: string; hasDir: boolean } + | { ok: false; error: string }; + +// WHICH URL TO RE-READ, OR WHY NOT. A video this channel has a directory for +// resolves the way every per-video action does (its metadata's webpage_url, +// else the playlist, else a URL rebuilt from the id and the platform). One +// with NO directory is accepted only when the channel's playlist lists it: an +// id that is neither is a typo, and rebuilding a URL from a typo would create +// a directory for a video that is not this channel's. +export async function resolveRefreshTarget( + paths: Paths, + slug: string, + videoId: string, + config: ChannelConfig, +): Promise<RefreshTarget> { + if (!isVideoIdSegment(videoId)) { + return { ok: false, error: `"${videoId}" is not a video id (one path segment)` }; + } + const videoDir = path.join(paths.channelsDir, slug, "data", videoId); + const hasDir = await stat(videoDir) + .then((s) => s.isDirectory()) + .catch(() => false); + const url = hasDir + ? await findVideoSourceUrl(paths, slug, videoId, config) + : await findPlaylistUrl(paths, slug, videoId); + if (!url && !hasDir) { + return { + ok: false, + error: + `"${videoId}" is not a video of ${slug}: there is no data/${videoId}/ ` + + `directory and the channel's playlist does not list it.`, + }; + } + if (!url) { + return { + ok: false, + error: + "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.", + }; + } + const notYtdlp = await notYtdlpMetadata(videoDir, url); + if (notYtdlp) { + return { + ok: false, + error: + `${videoId}'s metadata is not yt-dlp's to rewrite: ${notYtdlp}. ` + + `A yt-dlp re-read would overwrite it.`, + }; + } + return { ok: true, url, hasDir }; +} + +// The writers that complete a record yt-dlp cannot describe. A record one of +// them has touched holds what they wrote — a title found for a raw Wayback +// file, an archive.org item's own metadata, a podcast episode's feed entry — +// and a yt-dlp re-read replaces the whole file with what the URL alone says. +const NON_YTDLP_WRITERS: ReadonlySet<string> = new Set([ + "feed-backfill", + "archiveorg-provenance", + "archiveorg-import", + "wayback-provenance", +]); + +// WHY THIS RECORD'S METADATA IS NOT yt-dlp's, or null when it is. Asked of +// the URL first (an archive.org record is written whole from the item API +// with no history entry — a first write is not a rewrite — and a Wayback +// copy's title is the operator's), then of the history. +async function notYtdlpMetadata( + videoDir: string, + url: string, +): Promise<string | null> { + if (detectPlatform(url) === "archiveorg") { + return "an archive.org record is written from the item's metadata"; + } + if (parseWaybackUrl(url)) { + return "a Wayback Machine copy's title and date are set from its provenance"; + } + const history = await loadMetadataHistory(videoDir).catch(() => null); + const by = history?.entries.find((e) => NON_YTDLP_WRITERS.has(e.by))?.by; + return by ? `it was completed by ${by}` : null; +} + +// The argv of the one spawn. Exported for the test, which pins it. +export function buildRefreshArgs(opts: { + videoId: string; + videoUrl: string; + channelConfig: ChannelConfig; + cookies: string | undefined; + // The platform pace; default: its current one (channelExtraArgs). Tests + // pass it. + paceSeconds?: number; +}): string[] { + const dir = `data/${opts.videoId}`; + return [ + "--ignore-config", + "--restrict-filenames", + "--no-playlist", + // One request per second, the floor this repo applies everywhere it + // touches a source; the platform's own (adaptive) pace below raises it. + "--sleep-requests", + "1", + ...channelExtraArgs(opts.channelConfig, opts.cookies, opts.paceSeconds), + // AFTER the channel's own args, so they cannot be undone by them (see the + // header). `-o` likewise: the info json lands in THIS video's directory. + "--skip-download", + "--write-info-json", + "--no-write-subs", + "--no-write-auto-subs", + "--no-write-description", + "--no-write-thumbnail", + "--no-write-comments", + "--no-download-archive", + "--no-write-playlist-metafiles", + "-o", + `${dir}/audio.%(ext)s`, + "-o", + `infojson:${dir}/metadata`, + "--", + opts.videoUrl, + ]; +} + +// --- the closing summary ------------------------------------------------------ + +type Json = Record<string, unknown>; + +function isObject(v: unknown): v is Json { + return typeof v === "object" && v !== null && !Array.isArray(v); +} + +function isAudioOnly(f: Json): boolean { + return ( + f.vcodec === "none" && + typeof f.acodec === "string" && + f.acodec !== "none" + ); +} + +// Fragmented = yt-dlp has to stitch it from pieces (DASH segments, HLS). A +// plain https format is one ranged file, which is what a download wants. +function isFragmented(f: Json): boolean { + const protocol = typeof f.protocol === "string" ? f.protocol : ""; + if (Array.isArray(f.fragments) && f.fragments.length > 0) return true; + return !(protocol === "https" || protocol === "http"); +} + +// `140 m4a http_dash_segments 144k`. +function describeAudioFormat(f: Json): string { + const kbps = + typeof f.abr === "number" && f.abr > 0 + ? f.abr + : typeof f.tbr === "number" && f.tbr > 0 + ? f.tbr + : null; + return [ + String(f.format_id ?? "?"), + String(f.ext ?? "?"), + String(f.protocol ?? "?"), + ...(kbps !== null ? [`${Math.round(kbps)}k`] : []), + ].join(" "); +} + +// The English track names of a caption map (en, en-US, en-orig, …), sorted. +function englishTracks(map: unknown): string[] { + if (!isObject(map)) return []; + return Object.keys(map) + .filter((k) => /^en(-|$)/i.test(k)) + .sort(); +} + +function list(values: string[]): string { + return values.length ? values.join(", ") : "none"; +} + +// The content keys that moved across every entry this refresh appended (an +// auth retry can append a second), in the order they first appear. +function movedKeys(entries: MetadataHistoryEntry[]): { + changed: string[]; + added: string[]; + removed: string[]; +} { + const changed = new Set<string>(); + const added = new Set<string>(); + const removed = new Set<string>(); + for (const e of entries) { + for (const k of Object.keys(e.changed)) changed.add(k); + for (const k of Object.keys(e.added)) added.add(k); + for (const k of Object.keys(e.removed)) removed.add(k); + } + return { changed: [...changed], added: [...added], removed: [...removed] }; +} + +// WHAT THE SOURCE SAYS NOW, in the lines an operator waiting on a VOD reads: +// is it still live, how many formats, which audio formats and whether any of +// them is a plain file, which English captions exist, and what moved. +// +// PURE: the parsed file and the history entries arrive as arguments. +export function refreshSummaryLines(input: { + videoId: string; + meta: Json; + entries: MetadataHistoryEntry[]; + // Was there a metadata.info.json before this pass? With none there is + // nothing to compare against, and the history (correctly) records nothing. + hadBefore: boolean; +}): string[] { + const { meta } = input; + const formats = Array.isArray(meta.formats) ? meta.formats.filter(isObject) : []; + const audio = formats.filter(isAudioOnly); + const plainAudio = audio.filter((f) => !isFragmented(f)); + const moved = movedKeys(input.entries); + let changedLine: string; + if (!input.hadBefore) { + changedLine = "first metadata for this video (nothing to compare)"; + } else if (input.entries.length === 0) { + changedLine = "nothing (the file is byte-identical)"; + } else if ( + moved.changed.length + moved.added.length + moved.removed.length === + 0 + ) { + changedLine = "no content keys (formats, URLs or counters only)"; + } else { + changedLine = [ + ...moved.changed, + ...moved.added.map((k) => `+${k}`), + ...moved.removed.map((k) => `-${k}`), + ].join(", "); + } + return [ + `Metadata for ${input.videoId} now says:`, + ` live_status: ${typeof meta.live_status === "string" ? meta.live_status : "(not set)"}`, + ` formats: ${formats.length}`, + ` audio-only formats: ${list(audio.map(describeAudioFormat))}`, + ` non-fragmented audio: ${ + plainAudio.length + ? `yes (${plainAudio.map((f) => String(f.format_id ?? "?")).join(", ")})` + : "no" + }`, + ` English subtitles: ${list(englishTracks(meta.subtitles))}`, + ` English automatic captions: ${list(englishTracks(meta.automatic_captions))}`, + ` changed: ${changedLine}`, + ]; +} + +// --- the pass ----------------------------------------------------------------- + +export type RefreshRunner = ( + cwd: string, + args: string[], +) => Promise<AttemptOutcome>; + +export type RefreshVideoMetadataOpts = { + paths: Paths; + slug: string; + videoId: string; + videoUrl: string; + channelConfig: ChannelConfig; + // Resolved channel-over-global (resolveCookiePolicy). Omitted = the + // historical default: no prophylactic cookies, retry-only. + cookiePolicy?: ResolvedCookiePolicy; + onLog: (s: string) => void; + signal: AbortSignal; + // The source rate-limited us: record the shared platform cooldown. + onPlatformBackoff?: (failureClass: "rate_limit") => Promise<void> | void; + // The source answered cleanly: settle the platform. Returns the line for the + // log, or null when there was nothing to settle. + onPlatformClean?: () => Promise<string | null> | string | null; + // Who asked, recorded on the history entry ("ops", …). Optional. + requestedBy?: string; + // Test seams: the spawn, and the platform pace in the argv. + run?: RefreshRunner; + paceSeconds?: number; +}; + +export type RefreshVideoMetadataResult = { + // The parsed file after the pass. + meta: Json; + entries: MetadataHistoryEntry[]; + usedCookies: boolean; + summary: string[]; +}; + +function succeeded(exitCode: number | null): boolean { + // yt-dlp: 0 = clean, 101 = a clean stop (break-on-existing / max-downloads). + return exitCode === 0 || exitCode === 101; +} + +function tail(stderr: string): string { + return stderr.trim().split("\n").slice(-3).join(" / "); +} + +// Throws on a failed pass (the job ends `failed` with the reason); on success +// logs the summary and returns it. +export async function refreshVideoMetadata( + opts: RefreshVideoMetadataOpts, +): Promise<RefreshVideoMetadataResult> { + const channelDir = path.join(opts.paths.channelsDir, opts.slug); + const videoDir = path.join(channelDir, "data", opts.videoId); + const infoPath = path.join(videoDir, INFO_JSON); + const policy: ResolvedCookiePolicy = opts.cookiePolicy ?? { + cookies: undefined, + mode: DEFAULT_COOKIE_MODE, + }; + const run: RefreshRunner = + opts.run ?? + ((cwd, args) => + runOneYtdlp( + { ytdlpBin: opts.paths.ytdlpBin, onLog: opts.onLog, signal: opts.signal }, + cwd, + args, + )); + const hadBefore = await stat(infoPath) + .then((s) => s.isFile()) + .catch(() => false); + const entries: MetadataHistoryEntry[] = []; + const history = { + by: "refresh" as const, + ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}), + onLog: opts.onLog, + onEntry: (e: MetadataHistoryEntry) => entries.push(e), + }; + const pass = (cookies: string | undefined) => + withMetadataHistory(videoDir, history, () => + run( + channelDir, + buildRefreshArgs({ + videoId: opts.videoId, + videoUrl: opts.videoUrl, + channelConfig: opts.channelConfig, + cookies, + ...(opts.paceSeconds !== undefined ? { paceSeconds: opts.paceSeconds } : {}), + }), + ), + ); + + opts.onLog(`Refreshing the metadata of ${opts.videoId} from ${opts.videoUrl}\n`); + const firstCookies = alwaysCookies(policy); + let usedCookies = Boolean(firstCookies); + let outcome = await pass(firstCookies); + + // The prefetch's auth retry, verbatim in effect: once, with cookies, when + // the failure is an auth/age gate, the mode allows it, and the failed pass + // did not already carry them. "defer" has no retry cookies, so the failure + // stands. + if (!succeeded(outcome.exitCode) && !opts.signal.aborted) { + const cls = parseUnavailableFromStderr(outcome.stderrTail); + const retryCookies = authRetryCookies(policy); + if ( + classifyDownloadFailure(outcome.stderrTail, cls) !== "rate_limit" && + AUTH_RETRY_CLASSES.has(cls) && + retryCookies !== undefined && + !firstCookies + ) { + opts.onLog( + `Metadata refresh auth-required (${cls}); retrying with --cookies-from-browser ${retryCookies}\n`, + ); + usedCookies = true; + outcome = await pass(retryCookies); + } + } + + if (opts.signal.aborted) throw new Error("Cancelled"); + + if (!succeeded(outcome.exitCode)) { + const cls = parseUnavailableFromStderr(outcome.stderrTail); + if (classifyDownloadFailure(outcome.stderrTail, cls) === "rate_limit") { + await opts.onPlatformBackoff?.("rate_limit"); + throw new Error( + `The source rate-limited the metadata refresh of ${opts.videoId}; ` + + `a per-platform cooldown has been recorded. Nothing was rewritten. ` + + `(${tail(outcome.stderrTail)})`, + ); + } + throw new Error( + `yt-dlp could not read the metadata of ${opts.videoId} ` + + `(exit ${outcome.exitCode ?? "null"}): ${tail(outcome.stderrTail)}`, + ); + } + + try { + const line = await opts.onPlatformClean?.(); + if (line) opts.onLog(line); + } catch { + /* shared-state write is best-effort */ + } + + let meta: unknown; + try { + meta = JSON.parse(await readFile(infoPath, "utf8")); + } catch (err) { + throw new Error( + `yt-dlp exited cleanly but ${path.relative(channelDir, infoPath)} ` + + `could not be read: ${(err as Error).message}`, + ); + } + if (!isObject(meta)) { + throw new Error(`${path.relative(channelDir, infoPath)} is not a JSON object`); + } + const summary = refreshSummaryLines({ + videoId: opts.videoId, + meta, + entries, + hadBefore, + }); + opts.onLog(`${summary.join("\n")}\n`); + return { meta, entries, usedCookies, summary }; +} diff --git a/common/controller/refreshVideoMetadataJob.ts b/common/controller/refreshVideoMetadataJob.ts @@ -0,0 +1,127 @@ +// THE METADATA REFRESH AS A JOB — what the video page's "Refresh metadata" and +// `/api/ops/refresh-metadata` enqueue. The work is refreshVideoMetadata.ts; its +// own module, as feedMetadataJob.ts is, so the work can be tested (and run by +// anything else) without loading the job registry. +// +// THE REFUSALS COME BEFORE ANY JOB, in the sentences the rest of the app +// already uses for them, so a click and an ops call get the same answer: +// +// - a channel whose text cannot be read (`assertChannelTextReadable`, the +// guard every `needsText` kind gets — asked first here so its sentence +// wins over "not a video of this channel", which is what an unreadable +// `data/` would otherwise look like); +// - an id that is neither a directory nor a playlist entry +// (resolveRefreshTarget); +// - a HELD platform (`heldPlatformRefusal`), then one in a rate-limit +// cooldown — both `info: true`, as the metadata scan answers them: nothing +// is wrong, the source is resting. +// +// ON THE PLATFORM'S DOWNLOAD QUEUE, like the metadata scan and every per-video +// download: it contends for the source's patience, and the queue serialises it +// with a download of the same video on that queue rewriting the same file. +// `needsText` (jobs/jobKinds.ts): it writes one text file and opens no media, +// so a stalled media drive does not hold it. Replayable — a Retry re-resolves +// the target and re-asks the cooldown. + +import { detectPlatform } from "../lib/platform"; +import { downloadQueueKey, resolveQueueKey } from "../lib/queueKeys"; +import { assertChannelTextReadable } from "../lib/channelMedia"; +import type { ChannelConfig } from "../lib/channelConfig"; +import type { ResolvedCookiePolicy } from "../lib/cookiePolicy"; +import type { Paths } from "../lib/paths"; +import { + heldPlatformRefusal, + platformCooldownRemainingMs, + recordDownloadBackoff, + recordPlatformClean, +} from "../jobs/downloadBackoff"; +import { + runManagedFunction, + type StreamActionResult, +} from "../jobs/streamCommand"; +import { + refreshVideoMetadata, + resolveRefreshTarget, +} from "./refreshVideoMetadata"; + +export const REFRESH_METADATA_JOB_KIND = "refresh-metadata"; + +export type RefreshMetadataJobOpts = { + paths: Paths; + slug: string; + videoId: string; + channelConfig: ChannelConfig; + cookiePolicy: ResolvedCookiePolicy; + // Per-run override of the platform queue key. + queueKey?: string; + requestedBy?: string; + // Called after a successful refresh, inside the job. The editor revalidates + // the video and channel pages here. + afterRun?: () => void | Promise<void>; +}; + +export async function runRefreshMetadataJob( + opts: RefreshMetadataJobOpts, +): Promise<StreamActionResult> { + const { paths, slug, videoId, channelConfig } = opts; + try { + await assertChannelTextReadable(paths, slug, channelConfig); + } catch (err) { + return { ok: false, error: (err as Error).message }; + } + const target = await resolveRefreshTarget(paths, slug, videoId, channelConfig); + if (!target.ok) return { ok: false, error: target.error }; + + // The VIDEO's platform, as the clip fetch keys it: a record whose URL is on + // another host than its channel's (a mirror, a Wayback copy) answers to + // that host's cooldown. + const platform = detectPlatform(target.url) ?? "unknown"; + const held = await heldPlatformRefusal(platform, "The metadata refresh", paths); + if (held) return { ok: false, info: true, error: held }; + const remainingMs = await platformCooldownRemainingMs(platform, paths); + if (remainingMs > 0) { + const secs = Math.ceil(remainingMs / 1000); + return { + ok: false, + info: true, + error: + `${platform} is in a rate-limit cooldown (${secs}s remaining). ` + + `Refresh the metadata once the cooldown lapses.`, + }; + } + + return runManagedFunction({ + kind: REFRESH_METADATA_JOB_KIND, + queueKey: resolveQueueKey(downloadQueueKey(channelConfig), opts.queueKey), + paths, + channelSlug: slug, + videoId, + spec: { + kind: REFRESH_METADATA_JOB_KIND, + slug, + params: { videoId, queueKey: opts.queueKey }, + }, + fn: async (onLog, signal) => { + if (!target.hasDir) { + onLog( + `${videoId} has no data/${videoId}/ yet; the channel's playlist lists it, ` + + `so this creates the directory with its metadata.info.json.\n`, + ); + } + await refreshVideoMetadata({ + paths, + slug, + videoId, + videoUrl: target.url, + channelConfig, + cookiePolicy: opts.cookiePolicy, + onLog, + signal, + ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}), + onPlatformBackoff: () => recordDownloadBackoff(platform, paths), + onPlatformClean: () => recordPlatformClean(platform, paths), + }); + await opts.afterRun?.(); + }, + }); +} diff --git a/common/controller/undownloadedVideos.ts b/common/controller/undownloadedVideos.ts @@ -35,12 +35,8 @@ export async function findVideoSourceUrl( } catch { // fall through to playlist scan } - // Post-reconcile a video's dir name is its canonical id, so match the - // requested videoId against each URL's canonical id directly. - const urls = await readPlaylistUrls(path.join(channelRoot, "playlist")); - for (const url of urls) { - if (extractVideoId(url) === videoId) return url; - } + const listed = await findPlaylistUrl(paths, slug, videoId); + if (listed) return listed; // Last resort: re-create the URL from the canonical id (the dir name) and the // channel's platform. Only fires when both metadata.info.json and the // playlist came up empty, so the exact original URL always wins when present. @@ -49,3 +45,21 @@ export async function findVideoSourceUrl( if (platform) return defaultWebpageUrl(platform, videoId); return null; } + +// The channel's playlist entry for one video, or null. Post-reconcile a +// video's dir name is its canonical id, so the requested videoId is matched +// against each URL's canonical id directly. Exported for the metadata refresh +// (controller/refreshVideoMetadata.ts), which accepts an id with no directory +// only when the listing names it — never a URL rebuilt from a guess. +export async function findPlaylistUrl( + paths: Paths, + slug: string, + videoId: string, +): Promise<string | null> { + const channelRoot = path.join(paths.channelsDir, slug); + const urls = await readPlaylistUrls(path.join(channelRoot, "playlist")); + for (const url of urls) { + if (extractVideoId(url) === videoId) return url; + } + return null; +} diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts @@ -88,6 +88,8 @@ const ADDED_KINDS: Record<string, { label: string; drainable: boolean }> = { "reports-prepare": { label: "Prepare report media", drainable: false }, // A report site's exports: one pass, cancelled rather than drained. "reports-export": { label: "Export reports", drainable: false }, + // One video's metadata re-read: one spawn, cancelled rather than drained. + "refresh-metadata": { label: "Refresh metadata", drainable: false }, }; test("added kinds carry their pinned label and drainability", () => { @@ -146,6 +148,7 @@ const TEXT_KINDS = [ "fetch-window", "evict-clips", "metadata-scan", + "refresh-metadata", "download-missing-subs", "check-availability", "quick-availability-check", diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -621,6 +621,22 @@ const JOB_KINDS: Record<string, JobKindMeta> = { needsMedia: false, needsText: true, }, + // ONE video's metadata.info.json re-read from its source + // (controller/refreshVideoMetadataJob.ts): a metadata-only yt-dlp pass, no + // subtitles, no media, the rewrite recorded in metadata.history.json. On the + // platform download queue, like the metadata scan, for the same reason. + // `needsText`: it writes one text file and opens no big one. Not drainable + // (one spawn); replayable (the spec carries the video id, and a Retry + // re-resolves the target and re-asks the cooldown). + "refresh-metadata": { + kind: "refresh-metadata", + label: "Refresh metadata", + drainable: false, + replayable: true, + queueKeyStrategy: "platform", + needsMedia: false, + needsText: true, + }, // THE HUB'S AND THE HOMEPAGE'S BUILD AND DEPLOY (release 13 slice W1). They // ran from /sites — the hub since release 7, the homepage since release 11 — // with no entry here, so /jobs showed their raw machine kinds. The labels diff --git a/common/lib/metadataHistory-server.ts b/common/lib/metadataHistory-server.ts @@ -108,10 +108,15 @@ export async function recordMetadataRewrite( // yt-dlp that fails late (a prefetch that wrote the file and then hit a // subtitle error) has still rewritten it. `run`'s own result or error is // returned unchanged; a failure to record is logged through `onLog` and -// swallowed. +// swallowed. `onEntry` hears the entry that was appended, when one was — the +// refresh's closing summary names the keys that moved from it rather than +// reading the sidecar back. export async function withMetadataHistory<T>( videoDir: string, - ctx: MetadataRewriteContext & { onLog?: (line: string) => void }, + ctx: MetadataRewriteContext & { + onLog?: (line: string) => void; + onEntry?: (entry: MetadataHistoryEntry) => void; + }, run: () => Promise<T>, ): Promise<T> { const before = await snapshotMetadata(videoDir); @@ -120,7 +125,8 @@ export async function withMetadataHistory<T>( } finally { if (before) { try { - await recordMetadataRewrite(videoDir, before, ctx); + const entry = await recordMetadataRewrite(videoDir, before, ctx); + if (entry) ctx.onEntry?.(entry); } catch (err) { ctx.onLog?.( `Could not record the metadata history: ${(err as Error).message}\n`, diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts @@ -72,6 +72,12 @@ export const METADATA_HISTORY_WRITERS = [ // for it (`archilyzer wayback refresh --titles`, controller/waybackRefresh.ts) // — a raw media file captured by the Wayback Machine carries no title. "wayback-provenance", + // One video's metadata re-read on request — the video page's "Refresh + // metadata" and `pnpm ops refresh-metadata` (controller/refreshVideoMetadata.ts): + // a metadata-only yt-dlp pass, no subtitles, no media. For a source that + // changes after the first fetch — a livestream VOD that gains its processed + // formats and auto-captions hours after it ends. + "refresh", ] as const; export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number]; diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **One video's metadata can be read again from its source.** A livestream that has just ended offers one fragmented audio format and no captions; hours later the same URL has plain formats and auto-captions, and the video's `metadata.info.json` still said what it said the first time. "Refresh metadata" under the video page's header, and `pnpm ops refresh-metadata --json '{"slug":"…","id":"…"}'`, re-read that one video with no subtitles and no media, on the platform's queue, with the channel's cookie policy and pace; a held platform or one in a rate-limit cooldown is refused, and a rate limit records the cooldown. The rewrite is recorded in the metadata history as `refresh`, and the job's log ends with what the source now says: `live_status`, how many formats, each audio-only format and its protocol, whether any is non-fragmented, the English subtitle and caption tracks, and the keys that changed. Nothing is downloaded or deleted, and no download attempt is recorded. An id with no directory that the channel's playlist does not list is refused, as are an archive.org record, a Wayback copy and a record completed from a podcast feed, whose metadata is not yt-dlp's. Needs a restart of the editor. - **A renamed social channel's posts open again.** Each archived post carries the channel slug it was fetched under, and renaming the channel moves its directory without rewriting them, so every post of a renamed channel named the old slug: the index filed it under the new one, the post page and its thread could not find it there, and MCP links named a channel that no longer existed. A post's channel and slug are now read from the directory it is stored in, wherever it was fetched; the files are not rewritten. The next index build corrects the published records. Needs a restart of the editor. - **A video's other English tracks are readable and searchable where their words differ.** Uploaded captions are not always a transcript of what was said, so the tracks beside the transcript stay: the served `en` beside `en-orig`, a regional or auto-translated track, and the captions a local transcription replaced. One is kept where its words differ from the transcript's and from every track kept before it; identical tracks, most of them, add nothing. The index keeps them in an `alts` sub-DB and writes `track` and `altTracks` onto the transcript record only then, so every other record's page is what it was. A search hit in a word only an alternate holds names the track; one every track says is found once, in the transcript. The video page's **Transcript** card reads the transcript and switches tracks ("Track: original audio captions ▾"); switching changes nothing on disk, and **Set as transcript** stays the way the transcript itself changes. English VTTs are no longer shipped as subtitle tracks. One notion of a track — ids, plain labels, which are kept, how a hit across them is found — lives in `common/lib/captionTracks.ts`. - **The next index build reads the alternate tracks once.** Every record that can hold one — two or more English VTTs, or a transcription beside captions — is re-read from disk, and nothing else; the log says `Alternate tracks v1: N record(s) re-read.` and how many hold a track whose words differ. The version is recorded only when no channel is held. A transcribed video's captions now count toward its change time, so a later caption fetch reaches the index. diff --git a/editor/app/api/ops/_lib.test.ts b/editor/app/api/ops/_lib.test.ts @@ -5,6 +5,7 @@ import { optNonNegativeInt, optPositiveInt, optSubset, + reqVideoId, reqVideoItems, } from "./_lib"; @@ -112,3 +113,14 @@ test("reqVideoItems: a traversing slug or id is refused at the door", () => { assert.throws(() => reqVideoItems({ items: [item] }, "items"), OpsInputError, JSON.stringify(item)); } }); + +test("reqVideoId: one path segment comes back trimmed; a traversing or empty one is refused", () => { + assert.equal(reqVideoId({ id: " H64QQZuw-aA " }, "id"), "H64QQZuw-aA"); + for (const id of ["..", ".", "a/b", "a\\b", "", 7]) { + assert.throws( + () => reqVideoId({ id }, "id"), + OpsInputError, + JSON.stringify(id), + ); + } +}); diff --git a/editor/app/api/ops/_lib.ts b/editor/app/api/ops/_lib.ts @@ -192,14 +192,22 @@ export function reqVideoItems( ); } const slug = reqSlug(entry as OpsBody, "slug"); - const id = reqString(entry as OpsBody, "id"); - if (id === "." || id === ".." || /[/\\\0]/.test(id)) { - throw new OpsInputError(`"${id}" is not a video id (one path segment)`); - } + const id = reqVideoId(entry as OpsBody, "id"); return { slug, id }; }); } +// ONE VIDEO ID, one path segment — checked here for the reason reqSlug is: the +// id reaches a path.join under the channel's `data/`, and a traversing one +// must be refused at the door rather than read as an unknown video. +export function reqVideoId(body: OpsBody, key: string): string { + const id = reqString(body, key); + if (id === "." || id === ".." || /[/\\\0]/.test(id)) { + throw new OpsInputError(`"${id}" is not a video id (one path segment)`); + } + return id; +} + export function reqStringArray(body: OpsBody, key: string): string[] { const v = body[key]; if ( diff --git a/editor/app/api/ops/refresh-metadata/route.test.ts b/editor/app/api/ops/refresh-metadata/route.test.ts @@ -0,0 +1,80 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, readdir, rm, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +// Run with: +// pnpm -C editor exec tsx --test "app/api/ops/refresh-metadata/route.test.ts" +// +// The body's shape and the refusals that come before any job: every case here +// is answered from the disk alone, so a temp corpus with one YouTube channel +// is the whole world — no job is queued and no yt-dlp runs. + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "refresh-metadata-route-")); +const SLUG = "demo-yt"; +const CHANNEL = path.join(ROOT, "channels", SLUG); +// Set before the route (and getPaths, which caches) is first imported. +process.env.WORKER_TOKEN = "test-token"; +process.env.TRANSCRIPTS_DIR = ROOT; +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); +await mkdir(path.join(CHANNEL, "data"), { recursive: true }); +await writeFile( + path.join(CHANNEL, "config.json"), + JSON.stringify({ + handling: "youtube", + name: "Demo", + url: "https://www.youtube.com/@demo/videos", + }), +); +await writeFile( + path.join(CHANNEL, "playlist"), + "https://www.youtube.com/watch?v=listed00001\n", +); +const { POST } = await import("./route"); +test.after(() => rm(ROOT, { recursive: true, force: true })); + +async function post(body: Record<string, unknown>): Promise<{ status: number; error: string }> { + const res = await POST( + new Request("http://localhost/api/ops/refresh-metadata", { + method: "POST", + headers: { + authorization: "Bearer test-token", + "content-type": "application/json", + }, + body: JSON.stringify(body), + }), + ); + return { status: res.status, error: ((await res.json()) as { error?: string }).error ?? "" }; +} + +test("the body is { slug, id, queueKey? }: a missing id and an unknown key are refused", async () => { + const missing = await post({ slug: SLUG }); + assert.equal(missing.status, 400); + assert.match(missing.error, /"id" is required/); + const unknown = await post({ slug: SLUG, id: "listed00001", videoId: "x" }); + assert.equal(unknown.status, 400); + assert.match(unknown.error, /unknown key\(s\): videoId — this route accepts slug, id, queueKey/); +}); + +test("a traversing id or slug is refused at the door", async () => { + const id = await post({ slug: SLUG, id: "../escape" }); + assert.equal(id.status, 400); + assert.match(id.error, /is not a video id \(one path segment\)/); + const slug = await post({ slug: "../../escape", id: "abc" }); + assert.equal(slug.status, 400); + assert.match(slug.error, /is not a valid channel slug/); +}); + +test("an id that is neither a directory nor in the playlist is refused, and no job is written", async () => { + const stray = await post({ slug: SLUG, id: "nope0000000" }); + assert.equal(stray.status, 400); + assert.equal( + stray.error, + `"nope0000000" is not a video of ${SLUG}: there is no data/nope0000000/ directory and the channel's playlist does not list it.`, + ); + const channel = await post({ slug: "no-such-channel", id: "nope0000000" }); + assert.equal(channel.status, 400); + assert.equal(channel.error, 'Channel "no-such-channel" not found'); + assert.deepEqual(await readdir(path.join(ROOT, ".jobs")).catch(() => []), []); +}); diff --git a/editor/app/api/ops/refresh-metadata/route.ts b/editor/app/api/ops/refresh-metadata/route.ts @@ -0,0 +1,28 @@ +import { refreshVideoMetadataAction } from "../../../channels/[slug]/videos/[id]/videoActions"; +import { jobResponse, ops, optString, reqSlug, reqVideoId } from "../_lib"; + +export const dynamic = "force-dynamic"; + +// POST { slug, id, queueKey? } -> { ok: true, jobId } +// +// Re-read ONE video's metadata.info.json from its source — the video page's +// "Refresh metadata". A metadata-only yt-dlp pass (no subtitles, no media) on +// the platform's queue, with the channel's cookie policy and pace; the rewrite +// is recorded in metadata.history.json as `refresh`, and the job's log ends +// with what the source now says (live_status, formats, audio, English +// captions, the keys that changed). +// +// Every refusal is the action's own sentence: an id with no data/<id>/ that +// the channel's playlist does not list, a channel whose text cannot be read, +// a held platform or one in a rate-limit cooldown (`info: true`). +export async function POST(request: Request) { + return ops(request, ["slug", "id", "queueKey"], async (body) => + jobResponse( + await refreshVideoMetadataAction( + reqSlug(body, "slug"), + reqVideoId(body, "id"), + optString(body, "queueKey"), + ), + ), + ); +} diff --git a/editor/app/channels/[slug]/videos/[id]/components/RefreshMetadataControl.tsx b/editor/app/channels/[slug]/videos/[id]/components/RefreshMetadataControl.tsx @@ -0,0 +1,46 @@ +"use client"; + +import { useState } from "react"; +import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog"; +import { QueueControl } from "../../../../../components/QueueControl"; +import { cancelJobAction } from "../../../../../jobs/actions"; +import { refreshVideoMetadataAction } from "../videoActions"; + +// "Refresh metadata": re-read this video's metadata.info.json from its source, +// with no subtitles and no media — for a source that changed after the first +// fetch, like a livestream VOD that gains its processed formats and captions +// hours after it ends. It sits under the header's metadata (and its history), +// which is what it rewrites. The job's log ends with what the source now says; +// the rewrite appears in the history above as `refresh`. +export function RefreshMetadataControl({ + slug, + videoId, + defaultQueueKey, + existingQueues, +}: { + slug: string; + videoId: string; + defaultQueueKey: string; + existingQueues: string[]; +}) { + const [queueKey, setQueueKey] = useState(defaultQueueKey); + const actionLabel = `Refresh metadata for ${videoId}`; + return ( + <StreamActionLog + trigger={() => refreshVideoMetadataAction(slug, videoId, queueKey)} + cancelAction={cancelJobAction} + buttonLabel="Refresh metadata" + runningLabel="Refreshing metadata…" + label={actionLabel} + extraControls={ + <QueueControl + value={queueKey} + onChange={setQueueKey} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + actionLabel={actionLabel} + /> + } + /> + ); +} diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -43,6 +43,7 @@ import { VideoPanel, type VideoFile } from "./components/VideoPanel"; import { OperationPanel } from "./components/OperationPanel"; import { TagsPanel } from "./components/TagsPanel"; import { MetadataHistoryDetails } from "./components/MetadataHistoryDetails"; +import { RefreshMetadataControl } from "./components/RefreshMetadataControl"; import { loadVideoTags } from "./lib/videoTags"; import { loadVideoOperationPanels } from "./lib/videoOperationPanels"; import { channelTextStall } from "yt-dlp-transcript-common/lib/channelMedia"; @@ -303,6 +304,17 @@ export default async function VideoDetailPage({ </details> )} {metadataHistory && <MetadataHistoryDetails view={metadataHistory} />} + {/* Re-read the metadata above from the source (no subtitles, no + media). Not offered for an archive.org record or a Wayback copy: + their metadata is not yt-dlp's, and the action refuses them. */} + {!archiveOrg && !wayback && ( + <RefreshMetadataControl + slug={slug} + videoId={id} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + /> + )} </header> <RunningJobsList jobs={activeJobs} hideChannelSlug hideVideoId /> diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -93,6 +93,7 @@ import { sourceFetchFailure, } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; import { runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; +import { runRefreshMetadataJob } from "yt-dlp-transcript-common/controller/refreshVideoMetadataJob"; import { runManagedFunction, type StreamActionResult, @@ -254,6 +255,35 @@ export async function downloadVideoPipelineAction( }); } +// "Refresh metadata": re-read this one video's metadata.info.json from its +// source — no subtitles, no media — for a source that changed after the first +// fetch (a livestream VOD gaining its processed formats and captions). The job +// and every refusal are common/controller/refreshVideoMetadataJob.ts, so the +// ops route and a Retry from /jobs answer exactly as this click does; what is +// the editor's own is the two revalidations. +export async function refreshVideoMetadataAction( + slug: string, + videoId: string, + queueKey?: string, +): Promise<StreamActionResult> { + const r = await loadConfigOrError(slug); + if (!r.ok) return r; + return runRefreshMetadataJob({ + paths: getPaths(), + slug, + videoId, + channelConfig: r.config, + cookiePolicy: resolveCookiePolicy(getSettings(), r.config), + queueKey, + afterRun: () => { + safeRevalidate([ + `/channels/${slug}/videos/${videoId}`, + `/channels/${slug}`, + ]); + }, + }); +} + // Re-fetch an already-downloaded video purely to archive its SOURCE container, // without disturbing the existing transcript. Forces the persistence rule on // (keepSourceVideoOverride=true) so downloadOneManaged downloads the full video diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -53,7 +53,10 @@ import { redownloadIncompleteBucketAction, redownloadShortAudioBucketAction, } from "../channels/[slug]/incompleteTranscriptActions"; -import { replayFetchWindowAction } from "../channels/[slug]/videos/[id]/videoActions"; +import { + refreshVideoMetadataAction, + replayFetchWindowAction, +} from "../channels/[slug]/videos/[id]/videoActions"; import { reportsPrepareAction } from "../sites/lib/reportsPrepareAction"; import { reportsExportAction } from "../sites/lib/reportsExportAction"; @@ -262,6 +265,19 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { const { queueKey } = params(spec); return runMetadataScanAction(spec.slug, queueKey); }, + // One video's metadata re-read. The action re-resolves the target and + // re-asks the platform's hold and cooldown, as a click does. + "refresh-metadata": (spec) => { + const { p, queueKey } = params(spec); + const videoId = str(p.videoId); + if (!videoId) { + return Promise.resolve({ + ok: false, + error: "Job spec is missing its videoId.", + }); + } + return refreshVideoMetadataAction(spec.slug, videoId, queueKey); + }, "check-post-availability": (spec) => { const { p, queueKey } = params(spec); return checkPostAvailabilityAction( diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts @@ -183,6 +183,7 @@ test("a traversing slug is refused at the door, on every route that takes one", // each. const cases: [string, Record<string, unknown>][] = [ ["metadata-scan", { slug: "../../escape" }], + ["refresh-metadata", { slug: "../../escape", id: "abc123" }], ["sync", { slug: "../../escape" }], ["download-missing", { slug: "../../escape" }], ["import-video", { slug: "../../escape", url: "https://example.com/v" }], diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -26,6 +26,7 @@ // // pnpm ops sync --json '{"slug":"the-quartering"}' --wait // pnpm ops metadata-scan --json '{"slug":"the-quartering"}' +// pnpm ops refresh-metadata --json '{"slug":"the-quartering","id":"<videoId>"}' --wait // pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait // pnpm ops import-video --json '{"slug":"demo-archive","url":"https://archive.org/details/example-item"}' // pnpm ops import-archive-org --json '{"slug":"demo-archive","item":"example-item","match":"\\.mp4$"}' --wait @@ -98,6 +99,9 @@ const ACTIONS = [ "channel-priority", "channel-config", "metadata-scan", + // ONE video's metadata.info.json re-read from its source ({slug, id}): no + // subtitles, no media; the rewrite lands in metadata.history.json. + "refresh-metadata", "import-video", // Chosen media files of ONE archive.org item ({slug, item, files: [...] | // match: "<regex>", dryRun?}): one job, one file at a time, paced. @@ -343,6 +347,13 @@ export function usage() { ' report.md and evidence-pack.zip for the site\'s build to publish:', ' {"siteId", "reportId"?, "formats"?: ["html","pdf","md","zip"]}.', "", + 'refresh-metadata re-reads ONE video\'s metadata.info.json from its source', + ' (no subtitles, no media) on the platform\'s queue: {"slug", "id"}. The', + ' job\'s log ends with what the source now says — live_status, formats,', + ' audio-only formats and whether any is non-fragmented, English captions,', + ' the keys that changed. An id with no data/<id>/ that the playlist does', + ' not list is refused, as are archive.org and Wayback records.', + "", 'retry-bucket runs one bucket of a channel\'s report as one job, past any', ' lane hold: {"slug", "bucket"}. "ids": [...] runs only those videos, and', " every one must be in the bucket (a stray id is refused, named); a job run", diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -451,3 +451,20 @@ test("feed-metadata posts {slug, dryRun} to /api/ops/feed-metadata", () => { assert.equal(p.path, "/api/ops/feed-metadata"); assert.deepEqual(p.body, { slug: "demo-channel", dryRun: true }); }); + +// One video's metadata re-read (the video page's "Refresh metadata"): a POST +// with the body passed through untouched — the route judges it. +test("refresh-metadata is a POST to its route, named in the usage", () => { + const p = parseArgs([ + "refresh-metadata", + "--json", + '{"slug":"c","id":"H64QQZuw-aA"}', + "--wait", + ]); + assert.equal(p.method, "POST"); + assert.equal(p.path, "/api/ops/refresh-metadata"); + assert.deepEqual(p.body, { slug: "c", id: "H64QQZuw-aA" }); + assert.equal(p.wait, true); + assert.match(usage(), /Actions:.*metadata-scan, refresh-metadata/); + assert.match(usage(), /refresh-metadata re-reads ONE video/); +});