Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 1a703d562abab0383be5a69784e9964eed231f23
parent c3e030f6f7c79f1857c20f0cc596d4bd973c2251
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  6 Oct 2026 18:22:11 -0400

metadata: a refresh refuses a video with no data/<id>/ rather than create one

A data/<id>/ holding a metadata.info.json is not neutral: buildIndex
admits any such directory to the index and the published site, and
deriveChannelSets reads the name as "ever fetched". The metadata scan
creates none, and the refresh no longer does either: an id with no
directory is refused, playlist-listed or not, before yt-dlp is spawned
(resolveRefreshTarget, and again in refreshVideoMetadata). The video
page offers the control only when the video's directory has files.
findPlaylistUrl is no longer needed and is reverted.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/refreshVideoMetadata.test.ts | 45++++++++++++++++++++++++++++++---------------
Mcommon/controller/refreshVideoMetadata.ts | 59+++++++++++++++++++++++++++++++++++++----------------------
Mcommon/controller/refreshVideoMetadataJob.ts | 10++--------
Mcommon/controller/undownloadedVideos.ts | 26++++++--------------------
Meditor/CHANGELOG.md | 2+-
Meditor/app/api/ops/refresh-metadata/route.test.ts | 17++++++++++-------
Meditor/app/api/ops/refresh-metadata/route.ts | 8+++++---
Meditor/app/channels/[slug]/videos/[id]/page.tsx | 9++++++---
Mscripts/archilyzer-ops.mjs | 4++--
9 files changed, 99 insertions(+), 81 deletions(-)

diff --git a/common/controller/refreshVideoMetadata.test.ts b/common/controller/refreshVideoMetadata.test.ts @@ -9,6 +9,7 @@ import type { ResolvedCookiePolicy } from "../lib/cookiePolicy"; import type { AttemptOutcome } from "../ytdlp/runOneYtdlp"; import { buildRefreshArgs, + notFetchedRefusal, refreshSummaryLines, refreshVideoMetadata, resolveRefreshTarget, @@ -331,27 +332,42 @@ test("a rate-limited refresh records the platform cooldown, rewrites nothing and } }); -test("the target: an id with neither a directory nor a playlist entry is refused", async () => { +test("the target: an id with no directory is refused, even one the playlist lists — none is created", async () => { const f = await fixture({ playlist: [`https://www.youtube.com/watch?v=listed00001`] }); try { - const stray = await resolveRefreshTarget(f.paths, SLUG, "nope0000000", CONFIG); - assert.deepEqual(stray, { - ok: false, - error: - `"nope0000000" is not a video of ${SLUG}: there is no data/nope0000000/ ` + - `directory and the channel's playlist does not list it.`, - }); - // Listed but never downloaded: accepted, and the directory is new. - assert.deepEqual(await resolveRefreshTarget(f.paths, SLUG, "listed00001", CONFIG), { - ok: true, - url: "https://www.youtube.com/watch?v=listed00001", - hasDir: false, - }); + for (const id of ["nope0000000", "listed00001"]) { + assert.deepEqual(await resolveRefreshTarget(f.paths, SLUG, id, CONFIG), { + ok: false, + error: notFetchedRefusal(id, SLUG), + }); + } + assert.equal( + notFetchedRefusal("listed00001", SLUG), + `"listed00001" has not been fetched into ${SLUG} yet — sync, import or download it first; a refresh only re-reads a video already archived.`, + ); // One path segment only. for (const bad of ["..", "a/b", "."]) { const r = await resolveRefreshTarget(f.paths, SLUG, bad, CONFIG); assert.equal(r.ok, false, bad); } + // The pass itself refuses too, BEFORE any spawn, and makes no directory. + const { run, calls } = stubRunner([{ exitCode: 0, write: AFTER }]); + await assert.rejects( + refreshVideoMetadata({ + paths: f.paths, + slug: SLUG, + videoId: "listed00001", + videoUrl: "https://www.youtube.com/watch?v=listed00001", + channelConfig: CONFIG, + onLog: () => {}, + signal: new AbortController().signal, + run, + paceSeconds: 1, + }), + /has not been fetched into demo yet/, + ); + assert.equal(calls.length, 0); + assert.deepEqual(await readdir(path.join(f.channelDir, "data")), []); } finally { await rm(f.root, { recursive: true, force: true }); } @@ -363,7 +379,6 @@ test("the target: a downloaded video resolves to its own webpage_url", async () assert.deepEqual(await resolveRefreshTarget(f.paths, SLUG, ID, CONFIG), { ok: true, url: URL, - hasDir: true, }); } finally { await rm(f.root, { recursive: true, force: true }); diff --git a/common/controller/refreshVideoMetadata.ts b/common/controller/refreshVideoMetadata.ts @@ -69,7 +69,7 @@ import { runOneYtdlp, type AttemptOutcome, } from "../ytdlp/runOneYtdlp"; -import { findPlaylistUrl, findVideoSourceUrl } from "./undownloadedVideos"; +import { findVideoSourceUrl } from "./undownloadedVideos"; const INFO_JSON = "metadata.info.json"; @@ -85,15 +85,29 @@ export function isVideoIdSegment(id: string): boolean { } export type RefreshTarget = - | { ok: true; url: string; hasDir: boolean } + | { ok: true; url: string } | { ok: false; error: string }; -// WHICH URL TO RE-READ, OR WHY NOT. A video this channel has a directory for -// resolves the way every per-video action does (its metadata's webpage_url, -// else the playlist, else a URL rebuilt from the id and the platform). One -// with NO directory is accepted only when the channel's playlist lists it: an -// id that is neither is a typo, and rebuilding a URL from a typo would create -// a directory for a video that is not this channel's. +// The refusal for an id with no directory. Exported for the tests. +export function notFetchedRefusal(videoId: string, slug: string): string { + return ( + `"${videoId}" has not been fetched into ${slug} yet — sync, import or ` + + `download it first; a refresh only re-reads a video already archived.` + ); +} + +// WHICH URL TO RE-READ, OR WHY NOT. Only a video this channel already has a +// directory for: it resolves the way every per-video action does (its +// metadata's webpage_url, else the playlist, else a URL rebuilt from the id +// and the platform). +// +// NO DIRECTORY IS A REFUSAL, never one created. A data/<id>/ holding a +// metadata.info.json is not neutral (see PREFETCH_OWN_FILES in +// ytdlp/downloadOneManaged.ts): buildIndex admits any such directory to the +// index and the published site, and deriveChannelSets reads the name as +// "ever fetched". The metadata scan creates none for the same reason, and +// neither does this — the check is made before yt-dlp is spawned, and again +// by refreshVideoMetadata itself. export async function resolveRefreshTarget( paths: Paths, slug: string, @@ -104,20 +118,10 @@ export async function resolveRefreshTarget( return { ok: false, error: `"${videoId}" is not a video id (one path segment)` }; } const videoDir = path.join(paths.channelsDir, slug, "data", videoId); - const hasDir = await stat(videoDir) - .then((s) => s.isDirectory()) - .catch(() => false); - const url = hasDir - ? await findVideoSourceUrl(paths, slug, videoId, config) - : await findPlaylistUrl(paths, slug, videoId); - if (!url && !hasDir) { - return { - ok: false, - error: - `"${videoId}" is not a video of ${slug}: there is no data/${videoId}/ ` + - `directory and the channel's playlist does not list it.`, - }; + if (!(await isDirectory(videoDir))) { + return { ok: false, error: notFetchedRefusal(videoId, slug) }; } + const url = await findVideoSourceUrl(paths, slug, videoId, config); if (!url) { return { ok: false, @@ -134,7 +138,13 @@ export async function resolveRefreshTarget( `A yt-dlp re-read would overwrite it.`, }; } - return { ok: true, url, hasDir }; + return { ok: true, url }; +} + +async function isDirectory(p: string): Promise<boolean> { + return stat(p) + .then((s) => s.isDirectory()) + .catch(() => false); } // The writers that complete a record yt-dlp cannot describe. A record one of @@ -395,6 +405,11 @@ export async function refreshVideoMetadata( cwd, args, )); + // Never create the directory (see resolveRefreshTarget): refused before + // yt-dlp is spawned, whose `-o` template would otherwise make it. + if (!(await isDirectory(videoDir))) { + throw new Error(notFetchedRefusal(opts.videoId, opts.slug)); + } const hadBefore = await stat(infoPath) .then((s) => s.isFile()) .catch(() => false); diff --git a/common/controller/refreshVideoMetadataJob.ts b/common/controller/refreshVideoMetadataJob.ts @@ -10,8 +10,8 @@ // guard every `needsText` kind gets — asked first here so its sentence // wins over "not a video of this channel", which is what an unreadable // `data/` would otherwise look like); -// - an id that is neither a directory nor a playlist entry -// (resolveRefreshTarget); +// - an id with no data/<id>/ — a refresh re-reads a video already +// archived and never creates its directory (resolveRefreshTarget); // - a HELD platform (`heldPlatformRefusal`), then one in a rate-limit // cooldown — both `info: true`, as the metadata scan answers them: nothing // is wrong, the source is resting. @@ -102,12 +102,6 @@ export async function runRefreshMetadataJob( params: { videoId, queueKey: opts.queueKey }, }, fn: async (onLog, signal) => { - if (!target.hasDir) { - onLog( - `${videoId} has no data/${videoId}/ yet; the channel's playlist lists it, ` + - `so this creates the directory with its metadata.info.json.\n`, - ); - } await refreshVideoMetadata({ paths, slug, diff --git a/common/controller/undownloadedVideos.ts b/common/controller/undownloadedVideos.ts @@ -35,8 +35,12 @@ export async function findVideoSourceUrl( } catch { // fall through to playlist scan } - const listed = await findPlaylistUrl(paths, slug, videoId); - if (listed) return listed; + // Post-reconcile a video's dir name is its canonical id, so match the + // requested videoId against each URL's canonical id directly. + const urls = await readPlaylistUrls(path.join(channelRoot, "playlist")); + for (const url of urls) { + if (extractVideoId(url) === videoId) return url; + } // Last resort: re-create the URL from the canonical id (the dir name) and the // channel's platform. Only fires when both metadata.info.json and the // playlist came up empty, so the exact original URL always wins when present. @@ -45,21 +49,3 @@ export async function findVideoSourceUrl( if (platform) return defaultWebpageUrl(platform, videoId); return null; } - -// The channel's playlist entry for one video, or null. Post-reconcile a -// video's dir name is its canonical id, so the requested videoId is matched -// against each URL's canonical id directly. Exported for the metadata refresh -// (controller/refreshVideoMetadata.ts), which accepts an id with no directory -// only when the listing names it — never a URL rebuilt from a guess. -export async function findPlaylistUrl( - paths: Paths, - slug: string, - videoId: string, -): Promise<string | null> { - const channelRoot = path.join(paths.channelsDir, slug); - const urls = await readPlaylistUrls(path.join(channelRoot, "playlist")); - for (const url of urls) { - if (extractVideoId(url) === videoId) return url; - } - return null; -} diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,7 +1,7 @@ # Changelog ## [Unreleased] -- **One video's metadata can be read again from its source.** A livestream that has just ended offers one fragmented audio format and no captions; hours later the same URL has plain formats and auto-captions, and the video's `metadata.info.json` still said what it said the first time. "Refresh metadata" under the video page's header, and `pnpm ops refresh-metadata --json '{"slug":"…","id":"…"}'`, re-read that one video with no subtitles and no media, on the platform's queue, with the channel's cookie policy and pace; a held platform or one in a rate-limit cooldown is refused, and a rate limit records the cooldown. The rewrite is recorded in the metadata history as `refresh`, and the job's log ends with what the source now says: `live_status`, how many formats, each audio-only format and its protocol, whether any is non-fragmented, the English subtitle and caption tracks, and the keys that changed. Nothing is downloaded or deleted, and no download attempt is recorded. An id with no directory that the channel's playlist does not list is refused, as are an archive.org record, a Wayback copy and a record completed from a podcast feed, whose metadata is not yt-dlp's. Needs a restart of the editor. +- **One video's metadata can be read again from its source.** A livestream that has just ended offers one fragmented audio format and no captions; hours later the same URL has plain formats and auto-captions, and the video's `metadata.info.json` still said what it said the first time. "Refresh metadata" under the video page's header, and `pnpm ops refresh-metadata --json '{"slug":"…","id":"…"}'`, re-read that one video with no subtitles and no media, on the platform's queue, with the channel's cookie policy and pace; a held platform or one in a rate-limit cooldown is refused, and a rate limit records the cooldown. The rewrite is recorded in the metadata history as `refresh`, and the job's log ends with what the source now says: `live_status`, how many formats, each audio-only format and its protocol, whether any is non-fragmented, the English subtitle and caption tracks, and the keys that changed. Nothing is downloaded or deleted, and no download attempt is recorded. A video not yet fetched into the archive is refused — a refresh never creates its directory — as are an archive.org record, a Wayback copy and a record completed from a podcast feed, whose metadata is not yt-dlp's. Needs a restart of the editor. - **A renamed social channel's posts open again.** Each archived post carries the channel slug it was fetched under, and renaming the channel moves its directory without rewriting them, so every post of a renamed channel named the old slug: the index filed it under the new one, the post page and its thread could not find it there, and MCP links named a channel that no longer existed. A post's channel and slug are now read from the directory it is stored in, wherever it was fetched; the files are not rewritten. The next index build corrects the published records. Needs a restart of the editor. - **A video's other English tracks are readable and searchable where their words differ.** Uploaded captions are not always a transcript of what was said, so the tracks beside the transcript stay: the served `en` beside `en-orig`, a regional or auto-translated track, and the captions a local transcription replaced. One is kept where its words differ from the transcript's and from every track kept before it; identical tracks, most of them, add nothing. The index keeps them in an `alts` sub-DB and writes `track` and `altTracks` onto the transcript record only then, so every other record's page is what it was. A search hit in a word only an alternate holds names the track; one every track says is found once, in the transcript. The video page's **Transcript** card reads the transcript and switches tracks ("Track: original audio captions ▾"); switching changes nothing on disk, and **Set as transcript** stays the way the transcript itself changes. English VTTs are no longer shipped as subtitle tracks. One notion of a track — ids, plain labels, which are kept, how a hit across them is found — lives in `common/lib/captionTracks.ts`. - **The next index build reads the alternate tracks once.** Every record that can hold one — two or more English VTTs, or a transcription beside captions — is re-read from disk, and nothing else; the log says `Alternate tracks v1: N record(s) re-read.` and how many hold a track whose words differ. The version is recorded only when no channel is held. A transcribed video's captions now count toward its change time, so a later caption fetch reaches the index. diff --git a/editor/app/api/ops/refresh-metadata/route.test.ts b/editor/app/api/ops/refresh-metadata/route.test.ts @@ -66,13 +66,16 @@ test("a traversing id or slug is refused at the door", async () => { assert.match(slug.error, /is not a valid channel slug/); }); -test("an id that is neither a directory nor in the playlist is refused, and no job is written", async () => { - const stray = await post({ slug: SLUG, id: "nope0000000" }); - assert.equal(stray.status, 400); - assert.equal( - stray.error, - `"nope0000000" is not a video of ${SLUG}: there is no data/nope0000000/ directory and the channel's playlist does not list it.`, - ); +test("an id with no data/<id>/ is refused, even one the playlist lists; no job, no directory", async () => { + for (const id of ["nope0000000", "listed00001"]) { + const stray = await post({ slug: SLUG, id }); + assert.equal(stray.status, 400, id); + assert.equal( + stray.error, + `"${id}" has not been fetched into ${SLUG} yet — sync, import or download it first; a refresh only re-reads a video already archived.`, + ); + } + assert.deepEqual(await readdir(path.join(CHANNEL, "data")), []); const channel = await post({ slug: "no-such-channel", id: "nope0000000" }); assert.equal(channel.status, 400); assert.equal(channel.error, 'Channel "no-such-channel" not found'); diff --git a/editor/app/api/ops/refresh-metadata/route.ts b/editor/app/api/ops/refresh-metadata/route.ts @@ -12,9 +12,11 @@ export const dynamic = "force-dynamic"; // with what the source now says (live_status, formats, audio, English // captions, the keys that changed). // -// Every refusal is the action's own sentence: an id with no data/<id>/ that -// the channel's playlist does not list, a channel whose text cannot be read, -// a held platform or one in a rate-limit cooldown (`info: true`). +// Every refusal is the action's own sentence: an id with no data/<id>/ (a +// refresh re-reads a video already archived and never creates its directory), +// a channel whose text cannot be read, an archive.org, Wayback or +// feed-completed record, a held platform or one in a rate-limit cooldown +// (`info: true`). export async function POST(request: Request) { return ops(request, ["slug", "id", "queueKey"], async (body) => jobResponse( diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -305,9 +305,12 @@ export default async function VideoDetailPage({ )} {metadataHistory && <MetadataHistoryDetails view={metadataHistory} />} {/* Re-read the metadata above from the source (no subtitles, no - media). Not offered for an archive.org record or a Wayback copy: - their metadata is not yt-dlp's, and the action refuses them. */} - {!archiveOrg && !wayback && ( + media). Offered only for a video already in the archive (a + data/<id>/ with files — a listed, never-fetched video's page has + none, and the action refuses it rather than create one), and not + for an archive.org record or a Wayback copy: their metadata is not + yt-dlp's, and the action refuses them too. */} + {dirData.files.length > 0 && !archiveOrg && !wayback && ( <RefreshMetadataControl slug={slug} videoId={id} diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -351,8 +351,8 @@ export function usage() { ' (no subtitles, no media) on the platform\'s queue: {"slug", "id"}. The', ' job\'s log ends with what the source now says — live_status, formats,', ' audio-only formats and whether any is non-fragmented, English captions,', - ' the keys that changed. An id with no data/<id>/ that the playlist does', - ' not list is refused, as are archive.org and Wayback records.', + ' the keys that changed. An id with no data/<id>/ is refused (a refresh', + ' re-reads a video already archived), as are archive.org and Wayback records.', "", 'retry-bucket runs one bucket of a channel\'s report as one job, past any', ' lane hold: {"slug", "bucket"}. "ids": [...] runs only those videos, and',