Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit fb9050a94e0ff14257df997094ae589d6bf8a3cf
parent 3751f19e76a1a186483fb18482cc9dc819c59b6b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 05:25:17 -0400

feed metadata backfill: archilyzer feeds backfill-metadata, POST /api/ops/feed-metadata, pnpm ops feed-metadata; [Unreleased] bullet

The CLI row runs the controller offline; the editor action answers a held
host or one in its cooldown with a sentence, as the metadata scan does, and
enqueues the feed-metadata job; the ops route is its adapter.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/bin/archilyzer.ts | 20++++++++++++++++++++
Acommon/bin/feeds-backfill-metadata.ts | 41+++++++++++++++++++++++++++++++++++++++++
Meditor/CHANGELOG.md | 1+
Aeditor/app/api/ops/feed-metadata/route.ts | 21+++++++++++++++++++++
Meditor/app/channels/[slug]/pipelineActions.ts | 48++++++++++++++++++++++++++++++++++++++++++++++++
Mscripts/archilyzer-ops.mjs | 4++++
Mscripts/archilyzer-ops.test.mjs | 7+++++++
7 files changed, 142 insertions(+), 0 deletions(-)

diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -308,6 +308,26 @@ export const COMMANDS: Command[] = [ ); }, }, + { + path: ["feeds", "backfill-metadata"], + usage: + "<slug> [--feed <url>] [--dry-run] complete a podcast channel's records (title, date, description, duration) from its RSS feed: one fetch of the feed (default: the channel's url), no media; --dry-run counts matched / unmatched / already complete and writes nothing", + flags: { feed: "string", "dry-run": "boolean" }, + maxPositionals: 1, + run: async ({ positionals, flags }) => { + const [slug] = positionals; + if (!slug) { + console.error("feeds backfill-metadata: which channel? Pass its slug."); + return 2; + } + return (await import("./feeds-backfill-metadata")).main({ + slug, + ...(typeof flags.feed === "string" ? { feedUrl: flags.feed } : {}), + dryRun: flags["dry-run"] === true, + signal: interrupted(), + }); + }, + }, // The bins that parse their own flags, run as children with their argv // verbatim (_spawnBin.ts says why). Each row is the whole integration. script(["duplicates"], "duplicate-shorts.ts", diff --git a/common/bin/feeds-backfill-metadata.ts b/common/bin/feeds-backfill-metadata.ts @@ -0,0 +1,41 @@ +// `archilyzer feeds backfill-metadata <slug> [--feed <url>] [--dry-run]` — +// a podcast channel's records completed from its RSS feed, offline. The work +// and its rules are controller/feedMetadataBackfill.ts; the editor runs the +// same function as the `feed-metadata` job (`POST /api/ops/feed-metadata`). +// +// One fetch of the feed and nothing else on the network. It writes +// metadata.info.json files through the history, atomically; it does not see +// the editor's queues, so not beside a download on the same channel. + +import { isValidChannelSlug } from "../controller/channels"; +import { + backfillFeedMetadata, + feedBackfillSummary, +} from "../controller/feedMetadataBackfill"; + +export async function main(opts: { + slug: string; + feedUrl?: string; + dryRun: boolean; + signal?: AbortSignal; +}): Promise<number> { + if (!isValidChannelSlug(opts.slug)) { + console.error(`feeds backfill-metadata: "${opts.slug}" is not a channel slug`); + return 2; + } + try { + const result = await backfillFeedMetadata({ + slug: opts.slug, + ...(opts.feedUrl ? { feedUrl: opts.feedUrl } : {}), + dryRun: opts.dryRun, + requestedBy: "cli", + onLog: (line) => console.log(line.replace(/\n$/, "")), + ...(opts.signal ? { signal: opts.signal } : {}), + }); + console.log(feedBackfillSummary(result)); + return result.failed.length > 0 || opts.signal?.aborted ? 1 : 0; + } catch (err) { + console.error(`feeds backfill-metadata: ${(err as Error).message}`); + return 1; + } +} diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **A podcast channel's episodes imported by their enclosure URL get their titles and dates back from the feed.** Such a record carried only its file name as its title and no upload date, so the index skipped it and every surface showed the file id. `archilyzer feeds backfill-metadata <slug> [--feed <url>] [--dry-run]`, `POST /api/ops/feed-metadata {slug, dryRun?}` and `pnpm ops feed-metadata` (the new `feed-metadata` job, on the channel's download queue) fetch the channel's RSS feed once — no media — and match each record lacking a date or a real title to an item by guid, enclosure URL or enclosure file name; a record more than one item fits is reported, not guessed. A match fills in what the record lacks: the title, the description as plain text, the duration (a measured one is kept), the upload date and timestamp in UTC, and `webpage_url` only when the new URL keeps the record's id. Every change is a `feed-backfill` entry in the video's metadata history; a dry run counts matched, unmatched and already complete and writes nothing. Rebuild the index afterwards for the titles and dates to reach search and the sites. - **A site's build composes its reports, and a site that publishes only its reports ships nothing else.** Every site's compose now writes the reports its `site.json` publishes: each report's page and its citations as `citations.json` and `citations.csv` under `/reports/<id>/`, its cited stills, a page per cited moment with the record, the transcript lines around the span and every report that cites it, and the clips and post captures `archilyzer reports prepare` made for it, only the cited ones. Each quote is checked against the record as it is composed (a span's against its cues within 5 s either side, read from `en-orig` when the `en` track has no cues; a post's against its text) and the score, time and method are written into the citation, replacing any typed by hand. The build stops with the list of every problem before anything is written: an invalid report, a citation of a channel outside the site or of a post the site may not carry, a missing record, still or post, a quote that matches less than 60 % of what the record says, and a citation without prepared media or with media cut for another span (`--allow-missing-media` on `archilyzer compose site` and `build site` lets those two through, without a clip). A site with `publish: "cited"` removes everything corpus-shaped from `export/public` before it writes its reports, and its built `out/` is checked against what a cited site may hold: anything else, a file over 25 MiB or more than 20,000 files fails the build, and every deploy path (the Publish tab, `deploy site`, Build & deploy, Build & deploy all, the container build) refuses it, as it refuses a site set to cited whose last build was a full one. The hub's compose removes a report site's files too. - **A site has a Reports tab.** `/sites/<site>/reports` lists every report under the site's `reports/` directory — the published ones in their order, then the drafts — with its kind, dates, sections, claims, citations by kind and, for a fact-check, how many claims carry each verdict. Each report's problems, from the same checker the prepare step and the build use, open under it. A draft with no problems can be published, and a published report moved up or down or unpublished; each writes only the site's `reports` list, applied to the list as it is on disk at that moment, so it never overwrites another change to the site. "Prepare evidence media" queues the `reports-prepare` job, and beside it the tab shows the last prepared media (moments by kind, total size, problems by kind) and links the last prepare job. What the site publishes (full or cited) is shown with a link to Settings, where it is changed. - **A site can say what it publishes, and which reports.** `site.json` takes `publish` — `"full"`, the searchable corpus every site has been (the default, never written), or `"cited"`, only the site's reports and the moments they cite — and `reports`, the ordered ids of its published reports (each a slug; invalid and repeated ids are dropped). The site form has a Publish control and lists the site's reports read-only; saving the form keeps the stored list. A cited site still builds as a full one until the reports pipeline applies the scope. SITE.md documents both keys. diff --git a/editor/app/api/ops/feed-metadata/route.ts b/editor/app/api/ops/feed-metadata/route.ts @@ -0,0 +1,21 @@ +import { feedMetadataBackfillAction } from "../../../channels/[slug]/pipelineActions"; +import { jobResponse, ops, optBool, reqSlug } from "../_lib"; + +export const dynamic = "force-dynamic"; + +// POST { slug: string, dryRun?: boolean } -> { ok: true, jobId } +// +// A podcast channel's records completed from its RSS feed: one fetch of the +// channel's url, then title, date, description and duration into each record +// that lacks them. `dryRun` logs matched / unmatched / already complete and +// writes nothing. Follow the job's log for the counts. +export async function POST(request: Request) { + return ops(request, ["slug", "dryRun"], async (body) => + jobResponse( + await feedMetadataBackfillAction( + reqSlug(body, "slug"), + optBool(body, "dryRun"), + ), + ), + ); +} diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts @@ -28,6 +28,7 @@ import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdl import { mergeRosterFile } from "yt-dlp-transcript-common/controller/rosterStore"; import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; import { runMetadataScanJob } from "yt-dlp-transcript-common/controller/metadataScanJob"; +import { runFeedMetadataJob } from "yt-dlp-transcript-common/controller/feedMetadataJob"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { isGateHeld } from "yt-dlp-transcript-common/lib/pauseGates"; import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy"; @@ -555,3 +556,50 @@ export async function importVideoAction( }, }); } + +// A PODCAST CHANNEL'S RECORDS COMPLETED FROM ITS RSS FEED (the +// `feed-metadata` job, controller/feedMetadataBackfill.ts): one fetch of the +// channel's url, then title, date, description and duration written into each +// record that lacks them, through the metadata history. `dryRun` counts +// matched / unmatched / already complete and writes nothing. +// +// The feed is one request to the channel's host, so a held host or one in its +// rate-limit cooldown is answered with a sentence, as the metadata scan is. +export async function feedMetadataBackfillAction( + slug: string, + dryRun?: boolean, +): Promise<StreamActionResult> { + const paths = getPaths(); + const channelConfig = await readChannelConfig(paths, slug); + if (!channelConfig) { + return { ok: false, error: `Channel "${slug}" not found` }; + } + if (!channelConfig.url) { + return { ok: false, error: "Channel has no `url` configured" }; + } + const platform = detectPlatform(channelConfig.url) ?? "unknown"; + const held = await heldPlatformRefusal(platform, "The feed backfill", paths); + if (held) return { ok: false, info: true, error: held }; + const remainingMs = await platformCooldownRemainingMs(platform, paths); + if (remainingMs > 0) { + const secs = Math.ceil(remainingMs / 1000); + return { + ok: false, + info: true, + error: `${platform} is in a rate-limit cooldown (${secs}s remaining). Run the feed backfill once it lapses.`, + }; + } + return runFeedMetadataJob({ + paths, + slug, + ...(dryRun ? { dryRun: true } : {}), + requestedBy: "editor", + afterRun: () => { + safeRevalidate([ + `/channels/${slug}`, + "/channels", + ["/channels/[slug]/videos/[id]", "page"], + ]); + }, + }); +} diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -26,6 +26,7 @@ // // pnpm ops sync --json '{"slug":"the-quartering"}' --wait // pnpm ops metadata-scan --json '{"slug":"the-quartering"}' +// pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait // pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}' // pnpm ops channel-priority --json '{"slugs":["x"],"operation":"download","tier":"paused"}' // pnpm ops lane --json '{"lane":"download","held":true}' @@ -95,6 +96,9 @@ const ACTIONS = [ "channel-config", "metadata-scan", "import-video", + // A podcast channel's records completed from its RSS feed ({slug, dryRun?}): + // one fetch of the feed, no media. + "feed-metadata", "refresh-report", "sync", "download-missing", diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -436,3 +436,10 @@ test("persist-videos is a POST to its route, named in the usage", () => { assert.equal(parseArgs(["persist-videos", "--file", "list.json"]).bodyFile, "list.json"); assert.match(usage(), /persist-videos/); }); + +test("feed-metadata posts {slug, dryRun} to /api/ops/feed-metadata", () => { + const p = parseArgs(["feed-metadata", "--json", '{"slug":"demo-channel","dryRun":true}']); + assert.equal(p.method, "POST"); + assert.equal(p.path, "/api/ops/feed-metadata"); + assert.deepEqual(p.body, { slug: "demo-channel", dryRun: true }); +});