commit ea1e503cadc761a7e5788f44689c863480c7b823
parent 458e992307d0303c995846ba73fc8fee2a36ae1b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 10 Oct 2026 03:01:18 -0400
Merge r20/a6-a9 (release 19 A6–A9: archive.org imports and remote listings, cues without an index, publish edges, MCP archival writes) into r20/integration
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
45 files changed, 3768 insertions(+), 225 deletions(-)
diff --git a/COMMANDS.md b/COMMANDS.md
@@ -18,7 +18,7 @@ Run from the repo root (in the container: `docker compose exec editor pnpm archi
| `archilyzer compose hub` | | compose the hub's export/public (hub-sites.json, corpus.json, …) |
| `archilyzer compose homepage` | | compose homepage/public (whole-pool stats + landing summary) |
| `archilyzer publish index` | | update the index: the LMDB index, the stats datasets and the chart templates in one child (8 GB heap), then the index stamp every build reads |
-| `archilyzer publish build` | `<id\|all> [--runner local\|docker\|auto] [--force] [--skip-archives]` | build a site (or every stale one) into its bundle <exportBuildsDir>/<id>/out from the current index; --runner docker builds every site in containers (host only); a fresh site is a no-op without --force |
+| `archilyzer publish build` | `<id\|all> [--runner local\|docker\|auto] [--force] [--skip-archives] [--allow-missing-media] [--out <dir>]` | build a site (or every stale one) into its bundle <exportBuildsDir>/<id>/out from the current index; --runner docker builds every site in containers (host only); a fresh site is a no-op without --force; --allow-missing-media lets a report citation with no prepared media through compose; --out <dir> then lays the site's bundle out in <dir> for a server that is not Pages (hard links where it can, else copies; emptied first and guarded as a local deploy's destination; not a deploy, nothing recorded) |
| `archilyzer publish deploy` | `<id\|all> [--preview <branch>] [--to local] [--force]` | ship a site's bundle to its Pages project (a preview with --preview), or with --to local into ARCHILYZER\_SITE\_OUT; a bundle already deployed there is a no-op without --force |
| `archilyzer publish hub` | `[--deploy \| --deploy-only] [--preview <branch>] [--force]` | build the hub into its bundle <exportBuildsDir>/\_hub/out, then (--deploy) ship it; --deploy-only ships the bundle as built |
| `archilyzer publish homepage` | `[--deploy \| --deploy-only] [--preview <branch>] [--to local] [--force]` | build homepage/out (source mirror included), then (--deploy) ship it; --deploy-only ships it as built |
@@ -78,39 +78,41 @@ Drives a running editor over HTTP (`/api/ops/*`, the same actions its pages run)
| action | what it does | example |
|---|---|---|
-| `channel-priority` | — | `pnpm ops channel-priority --json '{"slugs":["x"],"operation":"download","tier":"paused"}'` |
-| `channel-config` | channel-config changes a channel as its Configure form does: {"slug"} and any of "patch" (form field names; "" clears one), "sites" (the WHOLE membership set: \[{"siteId", "groupId"? \| "newGroupName"?}\], \[\] = on no site; an unknown site id is refused), "excludeFromBuild" and "excludeFromCleanup" (set to the value given, not toggled). | `pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}'`<br>`pnpm ops channel-config --json '{"slug":"x","sites":[{"siteId":"anilyzer"}]}'`<br>`pnpm ops channel-config --json '{"slug":"x","sites":[],"excludeFromBuild":true}'` |
+| `channel-priority` | channel-priority sets channels' priority, as the /channels deck's tier control does: {"slugs", "tier": "normal" \| "low" \| "paused"} sets the base tier; with "operation" (sync, transcription, download, digest, backfill) it pins that operation's tier, "tier": null clearing it back to the base; "preset": "sync-only" \| "clear" is the deck's two shortcuts and takes no tier. A manual change clears an automatic pause. | `pnpm ops channel-priority --json '{"slugs":["x"],"operation":"download","tier":"paused"}'` |
+| `channel-config` | channel-config changes a channel as its Configure form does: {"slug"} and any of "patch" (form field names; "" clears one), "sites" (the WHOLE membership set: \[{"siteId", "groupId"? \| "newGroupName"?}\], \[\] = on no site; an unknown site id is refused), "excludeFromBuild" and "excludeFromCleanup" (set to the value given, not toggled). A VOD mirror's recorded date: "patch": {"recordedDateTitlePattern": "<regex with named groups year, month, day>"} (CHANNEL.md, recordedDate). | `pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}'`<br>`pnpm ops channel-config --json '{"slug":"x","sites":[{"siteId":"anilyzer"}]}'`<br>`pnpm ops channel-config --json '{"slug":"x","sites":[],"excludeFromBuild":true}'` |
| `create-channel` | create-channel is the New channel form: {"fields": {"name", "handling": "youtube"\|"transcribe", "url"?, "platform"?, "sourceKind"?, "postFetcher"?, "socialHandle"?, …}} with channel-config's patch keys; "slug"? (else derived from the name), "sites"? (absent = on no site). "fetchPlaylist", "fetchPostsNow" and "prioritizeDownload" are the form's checkboxes, OFF unless true; a job they start comes back as jobId(s), so --wait follows it. | `pnpm ops create-channel --json '{"fields":{"name":"Example (X)","handling":"transcribe","url":"https://x.com/example"}}'` |
| `rename-channel` | rename-channel moves a channel to a new slug, as Danger → Rename does: {"slug", "newSlug"}. Refused while the channel is busy (a job, a lane unit, media in transition) or when the new slug is taken. Old links break. | `pnpm ops rename-channel --json '{"slug":"old-slug","newSlug":"new-slug"}'` |
| `delete-channel` | delete-channel removes a channel's whole directory, as Danger → Delete does: {"slug", "confirm"} — "confirm" must repeat the slug. No undo outside the transcripts/ repo's own history. | `pnpm ops delete-channel --json '{"slug":"x","confirm":"x"}'` |
-| `metadata-scan` | — | `pnpm ops metadata-scan --json '{"slug":"the-quartering"}'` |
+| `metadata-scan` | metadata-scan reads every listed video's title, date and description into the channel's scan file, fetching no media: {"slug"}. Not held by the download pause and needs no disk floor; one request per video, so it respects the platform's cooldown and records one when pushed back. | `pnpm ops metadata-scan --json '{"slug":"the-quartering"}'` |
| `refresh-metadata` | refresh-metadata re-reads ONE video's metadata.info.json from its source (no subtitles, no media) on the platform's queue: {"slug", "id"}. The job's log ends with what the source now says — live\_status, formats, audio-only formats and whether any is non-fragmented, English captions, the keys that changed. An id with no data/<id>/ is refused (a refresh re-reads a video already archived), as are archive.org and Wayback records. | `pnpm ops refresh-metadata --json '{"slug":"the-quartering","id":"<videoId>"}' --wait` |
-| `import-video` | — | `pnpm ops import-video --json '{"slug":"demo-archive","url":"https://archive.org/details/example-item"}'` |
-| `import-archive-org` | — | `pnpm ops import-archive-org --json '{"slug":"demo-archive","item":"example-item","match":"\\.mp4$"}' --wait` |
+| `import-video` | import-video imports one video by URL into a channel: {"slug", "url"}. An archive.org URL becomes its canonical item or file (an item of several media files is refused — see import-archive-org); a BitChute or Odysee URL runs on that platform's queue at its pace, refused while it is held or cooling down, and refused when already on disk; a Wayback capture is named by what it copies. | `pnpm ops import-video --json '{"slug":"demo-archive","url":"https://archive.org/details/example-item"}'` |
+| `import-archive-org` | import-archive-org imports archive.org media into a channel, as one job on archive.org's queue: {"slug", "item", "files": \[...\] \| "match": "<regex>"} for one item; {"slug", "items": \["<id>" \| {"item", "files"? \| "match"?}, ...\]} for many (a bare id takes "match", else every media original); or {"slug", "query": "<archive.org search>", "limit"?: 100} for the first items a search finds (at most 500). One file at a time, a jittered pause between files and between items; a record already held (on disk, or in the saved-video store) is skipped. "dryRun": true lists each file as held, RESTRICTED (archive.org marks it not for download: a fetch would answer 401/403) or would get, and fetches nothing. A 401/403 skips the rest of its item; three such items in a row, a 429 or three failures in a row stop the job. The log ends with a summary: line; a re-run resumes. | `pnpm ops import-archive-org --json '{"slug":"demo-archive","item":"example-item","match":"\\.mp4$"}' --wait`<br>`pnpm ops import-archive-org --json '{"slug":"demo-archive","query":"collection:example-collection","dryRun":true}' --wait` |
+| `remote-listing` | remote-listing lists an Odysee or BitChute channel upstream and diffs it against what it holds: {"slug"}. One flat-playlist read on the platform's queue (refused while it is held or cooling down; a 429 backs it off), nothing written. `get remote-listing <slug>` waits for the job and prints {listed, held, notHeld: \[{id, url}\], heldNotListed: \[id\], ...} on stdout. | |
| `attach-media` | attach-media copies each held video's file out of a LOCAL archive into the saved-video store, as its source container (nothing fetched): {"slug", "source"} — an absolute path to a directory, a .zip (read in place) or a .7z. The id is the folder's trailing "(<id>)", else the file's yt-dlp suffix; "items": \[{"id", "path"}\] names exact files (path inside the source), "match" narrows the folders by regex. "createRecords": true writes a record for a video the channel does not hold; "replace": true re-attaches over a saved container. "dryRun": true logs each folder's class (attach, not-held, already-attached, lost, no-media, unmatched, ambiguous) and the held videos with no media in it, and writes nothing. The log ends with a summary: line; a re-run resumes. | |
| `prepare-playable` | prepare-playable remuxes each of a channel's saved containers, losslessly (-c copy), into a browser-playable mp4 (+faststart) or webm, and makes one single-file torrent per copy, under playable/ beside the saved-video store: {"slug"}. "ids" narrows it; "trackers": \[...\] is each .torrent's announce list (none by default; the infohash does not depend on it); "root" names another playable root. A video prepared from the same source (by sha256) is skipped; "dryRun": true logs each decision. | |
-| `feed-metadata` | — | `pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait` |
-| `refresh-report` | — | `pnpm ops refresh-report --json '{"all":true}'` |
-| `sync` | — | `pnpm ops sync --json '{"slug":"the-quartering"}' --wait` |
-| `download-missing` | — | |
+| `build-cues` | build-cues writes each video's transcript.cues.json from its raw transcript (the caption-track rule for VTTs), as the digest card's Normalize button does: {"slug"}, "ids": \[...\] for those videos only (every one held), "force": true to rewrite a fresh one. A job on the channel's queue. The file every reader without an index build wants. | `pnpm ops build-cues --json '{"slug":"demo-yt","ids":["<videoId>"]}' --wait` |
+| `feed-metadata` | feed-metadata completes a podcast channel's records from its RSS feed: one fetch of the channel's url, then title, date, description and duration into each record that lacks them: {"slug"}. "dryRun": true counts matched, unmatched and already complete, and writes nothing. | `pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait` |
+| `refresh-report` | refresh-report regenerates a channel's report (the buckets and counts its page shows) on the one serial report queue: {"slug"} answers the job queued, or the one already waiting ("started": false); {"all": true} queues every channel and answers {queued, jobIds, skipped}. | `pnpm ops refresh-report --json '{"all":true}'` |
+| `sync` | sync lists a channel and downloads what is new, as its Sync button does: {"slug"}. "full": true runs the periodic whole-listing sweep now (else the configured cadence decides); "queueKey" picks another queue. On the channel's platform queue, paced like its downloads. | `pnpm ops sync --json '{"slug":"the-quartering"}' --wait` |
+| `download-missing` | download-missing downloads every listed video the channel does not hold: {"slug"}. "ignoreArchive": true drops the download archive so listed videos it names are fetched again (recovery from a stale archive); "abortOnError": false carries on past a failure (default: stop at the first that is not one video's own). Paced on the platform's queue. | |
| `retry-bucket` | retry-bucket runs one bucket of a channel's report as one job, past any lane hold: {"slug", "bucket"}. "ids": \[...\] runs only those videos, and every one must be in the bucket (a stray id is refused, named); a job run with ids is not replayable, as a checkbox selection in the UI is not. | |
| `transcribe-bucket` | transcribe-bucket transcribes a channel's "downloaded, not transcribed" bucket on the transcription queue, as the channel page's Transcribe button does: {"slug"}. "ids": \[...\] narrows it the same way as on retry-bucket. | |
| `fetch-posts` | fetch-posts fetches a social channel's new posts: {"slug"}. "full": true re-walks the whole timeline; "older": true walks back from the oldest archived post through search (X; needs a login), saving its place for the next run, down to "floor": "YYYY-MM-DD" when given; "from": "YYYY-MM-DD" starts the walk afresh there, replacing its saved place (and a "complete") — for a gap above one surviving old post. "limit": N caps the posts one run reads; "pages": N caps the pages (a forum thread: its latest N pages). "full" and "older" together are refused. An older walk over an account that shows no posts (nothing archived, and the last timeline fetch read none) is refused unless "force": true. A drained fetch stops at its next resume point and the next run resumes. | |
| `capture-posts` | capture-posts captures archived posts of a social channel (X, forum): a screenshot of each through the connected X profile (a forum thread: its host's forum profile), and its attached media through gallery-dl (a forum thread: the same profile), into the channel's posts-media/<id>/: {"slug", "ids": \[...\]}. Every id must be in the channel's posts archive. "shots": false or "media": false skips that half; posts already captured are skipped unless "force": true. A post that links to an X Article also gets the article (article.json, .md, .png and its images) unless "articles": false; both halves off with "articles": true reads only the articles. Paced like a post fetch, on its queue. | |
-| `publish` | publish runs publish stages on the editor's publish queue, one at a time, under one run id: {"verb": …}. "index" updates the index; "build" builds "siteId"/"siteIds" (forced; the index first when stale; "runner": "docker" builds every site in containers); "deploy" ships their built bundles (production, "preview": "<branch>", or "to": "local"; "force" redeploys a bundle already shipped there); "hub" / "homepage" build them, {"deploy": true} deploys after; "now" is Publish now (the stale index, then each policy target); "stale" builds every stale site. The answer lists every job ({target, kind, jobId}) and --wait follows them all. A site never built is refused: "no build of <id> in <dir> — archilyzer publish build <id>". `get publish` is the status. | |
+| `publish` | publish runs publish stages on the editor's publish queue, one at a time, under one run id: {"verb": …}. "index" updates the index; "build" builds "siteId"/"siteIds" (forced; the index first when stale; "runner": "docker" builds every site in containers; "allowMissingMedia": true lets a report citation with no prepared media through compose, local builds only); "deploy" ships their built bundles (production, "preview": "<branch>", or "to": "local"; "force" redeploys a bundle already shipped there); "hub" / "homepage" build them, {"deploy": true} deploys after; "now" is Publish now (the stale index, then each policy target); "stale" builds every stale site. The answer lists every job ({target, kind, jobId}) and --wait follows them all. A site never built is refused: "no build of <id> in <dir> — archilyzer publish build <id>". `get publish` is the status. | |
| `build-index`, `build-site`, `build-deploy`, `deploy-site`, `build-hub`, `deploy-hub`, `build-homepage`, `deploy-homepage` | build-index, build-site, build-deploy, deploy-site, build-hub, deploy-hub, build-homepage and deploy-homepage are publish's aliases, with their old bodies and answers ("skipData" is accepted and ignored). | `pnpm ops build-deploy --json '{"siteIds":["anilyzer","jeralyzer"]}' --wait`<br>`pnpm ops build-site --json '{"siteId":"anilyzer"}' --wait`<br>`pnpm ops deploy-site --json '{"siteId":"anilyzer","preview":"tags-exclude"}' --wait`<br>`pnpm ops build-hub --wait`<br>`pnpm ops build-hub --json '{"deploy":true}' --wait`<br>`pnpm ops deploy-hub --wait`<br>`pnpm ops build-homepage --json '{"deploy":true}' --wait`<br>`pnpm ops deploy-homepage --json '{"preview":"refresh"}' --wait` |
| `build-site`, `build-deploy`, `deploy-site` | build-site, build-deploy and deploy-site all take "siteId" (one) or "siteIds" (a list). | |
| `build-hub` | build-hub builds the hub into its bundle; {"deploy": true} deploys it after, and deploy-hub ships the one already built. Both deploy to the Pages project set on /sites under Hub, and take "preview" too. | |
| `build-homepage` | build-homepage builds the homepage package into homepage/out; {"deploy": true} deploys it after (only if the build succeeded), and deploy-homepage ships the one already built. Both deploy to the Pages project archilyzer (https://archilyzer.pages.dev), production unless "preview" is given. | |
| `relocate` | relocate moves channels' media to a location: {"slugs", "locationId" \| "root"}. "dryRun": true answers each channel's preview (bytes to copy, free space both sides) and moves nothing — though, as on the Storage panel, a channel never tiered is first tiered in place on the corpus disk. | `pnpm ops relocate --json '{"slugs":["x"],"locationId":"platter"}'`<br>`pnpm ops relocate --json '{"slugs":["x"],"locationId":"platter","dryRun":true}'` |
-| `relocate-back` | — | |
-| `evict-clips` | — | |
+| `relocate-back` | relocate-back moves channels' media back onto the corpus disk, one job per channel on the relocation queue: {"slugs"}. A busy channel is answered in "skipped"; --wait follows the jobs started. | |
+| `evict-clips` | evict-clips deletes cached clip windows (data/<id>/clips/) older than a number of days, as /storage's button does: {"olderThanDays", "slug"?, "dryRun"?}. BY AGE: nothing knows whether a umtool report still cites a window; an evicted window is fetched again when asked for. | |
| `reports-prepare` | reports-prepare cuts every clip and copies every post capture a site's published reports cite into its report-media cache, before its build: {"siteId"}. The job fails, naming each one, when a citation lacks media. When nothing is missing it then exports the reports, as reports-export. | `pnpm ops reports-prepare --json '{"siteId":"demo-site"}' --wait` |
| `reports-export` | reports-export writes each published report as report.html, report.pdf, report.md, slides.html, slides.pdf and evidence-pack.zip for the site's build to publish: {"siteId", "reportId"?, "formats"?: \["html","pdf","md", "slides","slides-pdf","zip"\]}. | `pnpm ops reports-export --json '{"siteId":"demo-site","formats":["html","md"]}' --wait` |
| `lane` | lane takes {"lane": "transcription"\|"download"\|"digest"\|"backfill"\| "publish", "enabled"?, "held"?, "action"?: "start"\|"stop"\|"drain"}. | `pnpm ops lane --json '{"lane":"download","held":true}'`<br>`pnpm ops lane --json '{"lane":"publish","action":"drain"}'` |
-| `tags` | — | `pnpm ops tags --json '{"op":"define","tag":{"id":"eva-collab","label":"Collab"}}'` |
-| `tag-videos` | — | `pnpm ops tag-videos --file ids.json` |
-| `keep-videos` | — | |
+| `tags` | tags edits the curated-tag vocabulary through its one writer: {"op": "define", "tag": {...}} (the WHOLE definition: id, label, rules and "sites", the sites it exists on, absent = every site) or {"op": "remove", "tag": "<id>"}. `get tags [<id>]` reads it. | `pnpm ops tags --json '{"op":"define","tag":{"id":"eva-collab","label":"Collab"}}'` |
+| `tag-videos` | tag-videos pins, unpins, suppresses or unsuppresses one tag on many videos in ONE write: {"tag", "op": "add" \| "remove" \| "suppress" \| "unsuppress", "videos": \[{"slug", "id"}, ...\]}. "remove" unpins only — a rule's hit survives; "suppress" rejects it. The write records who asked (ARCHILYZER\_AGENT); a big list goes in --file. | `pnpm ops tag-videos --file ids.json` |
+| `keep-videos` | keep-videos sets the "Do not clean" marker on every held video of a channel whose title or description matches a download-filter pattern: {"slug", "match", "fields"?: \["title" \| "description"\], "note"?, "dryRun"?}. A match not held is reported in "notDownloaded", never created; a fresh channel wants metadata-scan first. | |
| `persist-videos` | persist-videos saves specific videos, across channels, to the saved-video store: {"items": \[{"slug", "id"}, ...\]}. "format": "original" \| "video\_720" (default: each channel's own). "replace": "above-height" also re-fetches a saved one whose height is unknown or above that quality (default "never"). "gapMs" pauses between downloads (default the batch gap), "minFreeMemMb" waits for that much free memory before each. "dryRun": true answers with the buckets (saved, wrongHeight, toFetch, noUrl, unknown) and starts nothing. One job per channel, on its download queue; a low disk or a rate limit stops it, and running the same body again resumes — saved videos are skipped. | `pnpm ops persist-videos --file list.json --wait` |
| `fetch-windows` | fetch-windows fetches clip windows, one paced job per platform queue (YouTube and Rumble side by side): {"siteId"} fetches every window the site's published reports cite and the disk does not hold; {"items": \[{"slug", "id", "from", "to", "clipId"?, "reason"?, "pad"?, "webpageUrl"?}, ...\], "requestedBy", "manifest"?} fetches a list. "maxHeight" caps the source height (default 720). "dryRun": true lists the windows per platform, the ones already on disk ("cached") and the ones no fetch can fill ("unfetchable": deleted, off the site) and starts nothing. A platform cooling down or held is refused for its group; a 429, or two 403s in a row, backs the platform off and stops its job. Running the same body again resumes — fetched windows are cached, and a window a queued or running job will already write is answered in "inFlight" with that job, whose id joins "jobIds" (so --wait follows it) and no second job is queued. Windows run on the platform's clip queue (clips:youtube), never behind its long downloads. | `pnpm ops fetch-windows --json '{"siteId":"demo-site","dryRun":true}'`<br>`pnpm ops fetch-windows --file windows.json --wait` |
| `cut-release` | cut-release turns a changelog's \[Unreleased\] into "## \[<version>\] - <date>": {"workspace": "editor" \| "export" \| "all", "version": "X.Y.Z" \| "next" \| "next-minor", "commit": boolean (default false), "date": "YYYY-MM-DD" (default today)}. "all" cuts both with ONE version and commits each ("Release <workspace> <version>") — or neither: every check runs before either file is written. Only an editor built from release 10 or later has the route (an older one answers 404); with no editor running, `archilyzer release cut` does the same locally. | `pnpm ops cut-release --json '{"workspace":"all","version":"next","commit":true}'` |
@@ -141,6 +143,9 @@ Usage: pnpm ops <action> [--json '<body>' | --file <path>] [--wait]
pnpm ops get settings [<key>]
pnpm ops get storage | sites | workers | auto-queue | scheduler
pnpm ops get cleanup <slug>
+ pnpm ops get remote-listing <slug> [--wait-timeout <seconds>]
+ pnpm ops get transcript <videoId> [--slug <slug>]
+ pnpm ops get coverage <slug>
pnpm ops list
--wait follows the job's log and survives a poll that fails (a busy
@@ -177,6 +182,17 @@ get cleanup <slug> is one channel's /cleanup row: what each sweep would
reclaim (they overlap — never add them), what holds the rest, and the
failed-transcriptions count. measured: false means unknown, not zero.
+get coverage <slug> is what a channel holds by date: held, dated (by
+ recorded date — the channel's recordedDate title rule — else by upload
+ date), undated, first and last day, per year and month, and every gap
+ over 30 days with nothing held. Off disk; nothing written. (The MCP's
+ channel_coverage takes gap days, a date window and a video list.)
+
+get transcript <videoId> [--slug <slug>] reads one video's cues off disk
+ with no index: a fresh cues.json, else what build-cues would write, else
+ the English VTT alone — {source, cuesJson, title?, ..., cues}. Nothing is
+ written. Without --slug, the channel holding data/<id>/ is found.
+
"preview": "<branch>" on deploy-site or build-deploy makes it a Cloudflare
Pages PREVIEW instead of production: the same bundle goes to a branch
alias, https://<branch>.<project>.pages.dev, and the live site is left
diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md
@@ -53,7 +53,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| Variable | Default | What it does | Read by |
|---|---|---|---|
-| `WORKER_TOKEN` | unset (both surfaces off) | Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`, the MCP's `fetch_clip`). Set the same value on both ends. | common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts |
+| `WORKER_TOKEN` | unset (both surfaces off) | Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`; the MCP's `fetch_clip`, `enqueue`, `get_job`, `channel_coverage` and `get_transcript`'s editor fallback). Set the same value on both ends. | common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, mcp/src/editorOps.ts |
| `SYNC_HEARTBEAT_SECONDS` | `settings.syncScheduler.heartbeatSeconds` | Overrides the editor's in-process sync heartbeat. `0` = no internal timer (tick from cron instead). | editor/app/scheduler/heartbeat.ts |
| `SYNC_TICK_URL` | `http://127.0.0.1:3001/api/scheduler/tick` | Where `archilyzer sync tick` (cron's heartbeat) posts. | common/bin/sync-tick.ts |
| `SYNC_TICK_TOKEN` | unset (no auth) | Bearer token for the tick endpoint; set on both the editor and the cron job. | common/bin/sync-tick.ts, editor/app/scheduler/auth.ts |
@@ -82,7 +82,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| `CLAUDE_DIGEST_MODEL` | the CLI's default | The model the metered digest lane asks `claude` for when settings name none. | common/lib/digestApps.ts |
| `NITTER_INSTANCES` | a built-in list | Comma-separated Nitter instances for the X fallback fetcher, in order of preference. | common/social/xNitterFetcher.ts |
| `ARCHILYZER_X_BROWSER` | the first of `chromium`, `google-chrome`, `google-chrome-stable`, `chrome` on PATH, else Playwright's bundled Chromium | The Chromium-family browser /settings' "Connect X account" opens (a path, or a name looked up on PATH), and a forum-thread channel's "Connect forum session" too. It is launched without the automation signals, in the X session profile (or the forum host's profile); a value that is not an executable refuses the connect rather than opening another browser. | common/social/xBrowser.ts |
-| `UMTOOL_URL` | unset (no link) | umtool's front door; when set, the video page links to it. | editor/app/channels/[slug]/videos/[id]/page.tsx |
+| `UMTOOL_URL` | unset (no link; the MCP's `notes` off) | umtool's front door; when set, the video page links to it, and the MCP's `notes` tool reads the operator's notes there (e.g. `http://localhost:3050`). | editor/app/channels/[slug]/videos/[id]/page.tsx, mcp/src/archivalTools.ts |
| `TRANSCRIPT_SITE_URL` | — | MCP server: one published archive to read over HTTP. | mcp/src/sources.ts |
| `TRANSCRIPT_HUB_URL` | — | MCP server: a hub, federating every archive it lists. | mcp/src/sources.ts |
| `TRANSCRIPT_LOCAL_DIR` | — | MCP server: a composed public dir on disk. | mcp/src/sources.ts |
@@ -93,7 +93,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| `ARCHILYZER_INDEX_ALLOW_HELD` | off | `1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel. | common/controller/buildIndex.ts |
| `UV_THREADPOOL_SIZE` | `16` for the editor (`4` is Node's own) | Threads in Node's pool for filesystem calls. A call on a stalled drive holds one until the drive answers, so the editor starts with 16. It buys time for calls already in flight and isolates nothing: the storage health probe and its gate keep new calls off a stalled drive. | Node's libuv (set by editor/package.json `start` and docker/entrypoint.sh) |
| `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts |
-| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to, and whose `/api/pulse` `archilyzer storage migrate-tier` asks before it refuses to run beside it. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool, common/bin/migrate-media-tier.ts |
+| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's editor-backed tools (`fetch_clip`, `enqueue`, `get_job`, `channel_coverage`, `get_transcript`'s fallback) talk to, and whose `/api/pulse` `archilyzer storage migrate-tier` asks before it refuses to run beside it. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, mcp/src/editorOps.ts, umtool, common/bin/migrate-media-tier.ts |
| `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs |
| `DIARIZE_ENGINE_KIND` | `sherpa-onnx` | The diarization engine: `sherpa-onnx` or `sortformer`. | scripts/diarize.mjs |
| `DIARIZE_ENGINE_CMD` | the bundled sherpa script | The engine command the wrapper runs. | scripts/diarize.mjs |
diff --git a/OPERATING.md b/OPERATING.md
@@ -60,13 +60,21 @@ pnpm ops persist-videos --json '{"items":[{"slug":"example","id":"<videoId>"}]
```sh
pnpm ops import-archive-org --json '{"slug":"example-archive","item":"<item>","match":"\\.mp4$","dryRun":true}' --wait
pnpm ops import-archive-org --json '{"slug":"example-archive","item":"<item>","match":"\\.mp4$"}' --wait
+pnpm ops import-archive-org --json '{"slug":"example-archive","items":["<item>","<item>"],"dryRun":true}' --wait
+pnpm ops import-archive-org --json '{"slug":"example-archive","query":"collection:<collection>","limit":50,"dryRun":true}' --wait
+pnpm ops get remote-listing example-odysee
pnpm ops import-video --json '{"slug":"example","url":"https://www.bitchute.com/video/<id>/"}' --wait
pnpm ops import-video --json '{"slug":"example","url":"https://odysee.com/@example:0/<video>:0"}' --wait
pnpm ops import-video --json '{"slug":"example","url":"https://web.archive.org/web/<timestamp>/<original-url>"}' --wait
```
-- `import-archive-org` takes one item's files (`files` exact, or `match` a regex), one at a time on archive.org's
- queue; files already held are skipped. archive.org is fetched over BitTorrent when it can be, never by yt-dlp.
+- `import-archive-org` takes one item's files (`files` exact, or `match` a regex), many items (`items`), or the
+ items an archive.org search finds (`query`, the first `limit`), one file at a time on archive.org's queue with a
+ pause between files and between items; records already held (on disk, or in the saved-video store) are skipped.
+ A dry run lists each file as held, RESTRICTED (archive.org will answer 401/403) or would-get. archive.org is
+ fetched over BitTorrent when it can be, never by yt-dlp.
+- `get remote-listing <slug>` lists an Odysee or BitChute channel upstream (one paced read on the platform's queue)
+ and prints what it lists that the channel does not hold, and what it holds that is no longer listed.
- A one-off Odysee or BitChute import is paced like that platform's own downloads. A whole Odysee or BitChute
channel is a channel with that URL, then `sync`.
- A Wayback capture is named by what it copies; `wayback.json` beside it records the capture.
@@ -97,6 +105,9 @@ pnpm ops transcribe-bucket --json '{"slug":"example"}' --wait
pnpm ops transcribe --json '{"path":"/abs/clip.mp4","start":120,"end":150,"words":true}' --wait
```
+- A transcript the index has not caught up with: `pnpm ops get transcript <videoId> --slug <slug>` reads its cues
+ off disk (the MCP's `get_transcript` does the same through the editor), and
+ `pnpm ops build-cues --json '{"slug":"<slug>","ids":["<videoId>"]}' --wait` writes its `transcript.cues.json`.
- The transcription lane takes every channel's downloaded, untranscribed audio; `transcribe-bucket` runs one
channel's now.
- `transcribe` runs one local file (or a window of it) through the corpus's own engine and model; the result JSON
@@ -145,11 +156,17 @@ pnpm ops publish --json '{"verb":"now"}' --wait
pnpm ops publish --json '{"verb":"build","siteId":"example-site"}' --wait
pnpm ops publish --json '{"verb":"deploy","siteId":"example-site","preview":"check"}' --wait
pnpm archilyzer publish status
+pnpm archilyzer publish build example-site --out /srv/example-private
```
- Publishing is stages on one queue: update the index, build a site into its bundle, deploy the bundle, each
checked live. `now` runs what Publish now runs (the stale index, then each site's policy). Offline, under the same
lock: `pnpm archilyzer publish index`, `publish build <id>`, `publish deploy <id> --preview <branch>`.
+- A private build served by something other than Pages: `publish build <id> --out <dir>` lays the bundle out in
+ `<dir>` (hard links where it can; the directory is emptied first and guarded like a local deploy's). A report
+ citation whose media is not prepared yet passes with `--allow-missing-media` (`"allowMissingMedia": true` on the
+ ops `build` verb).
+- The publish lane: `pnpm ops lane --json '{"lane":"publish","action":"start"}'` (`"held": true` pauses it).
- Details, policies and the hub and homepage: [PUBLISH.md](PUBLISH.md).
## Research with the MCP
@@ -157,4 +174,8 @@ pnpm archilyzer publish status
- `/ask` answers a question with citations in the conversation; `/sweep` writes a cited report to a file.
- Search first, then pull only the cited seconds with `fetch_clip`. The editor fetches only for a channel it
already archives.
+- With the editor configured, the MCP also queues archival work (`enqueue`: sync, download-missing, retry-bucket,
+ transcribe-bucket, fetch-posts, import-video), follows it (`get_job`), reports what a channel holds by date and
+ where the gaps are (`channel_coverage`), and reads a video not yet published (`get_transcript`). The operator's
+ notes: `notes` (with `UMTOOL_URL`). Settings, storage and deletes stay `pnpm ops`.
- Tools and their arguments: [mcp/README.md](mcp/README.md).
diff --git a/common/bin/_cli.test.ts b/common/bin/_cli.test.ts
@@ -436,7 +436,14 @@ test("the stage row: kind and target, the flags the child's argv carries, nothin
test("the publish rows and their flags (status and now are S3's)", () => {
const row = (...p: string[]) => resolveCommand(COMMANDS, p)?.command;
assert.deepEqual(row("publish", "index")?.flags ?? {}, {});
- assert.deepEqual(row("publish", "build", "jer")?.flags, { runner: "string", force: "boolean", "skip-archives": "boolean" });
+ assert.deepEqual(row("publish", "build", "jer")?.flags, {
+ runner: "string",
+ force: "boolean",
+ "skip-archives": "boolean",
+ // release 19 A8
+ "allow-missing-media": "boolean",
+ out: "string",
+ });
assert.equal(row("publish", "build", "all")?.maxPositionals, 1);
assert.deepEqual(row("publish", "deploy", "all")?.flags, { preview: "string", to: "string", force: "boolean" });
assert.deepEqual(row("publish", "hub")?.flags, {
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -86,8 +86,14 @@ export const COMMANDS: Command[] = [
{
path: ["publish", "build"],
usage:
- "<id|all> [--runner local|docker|auto] [--force] [--skip-archives] build a site (or every stale one) into its bundle <exportBuildsDir>/<id>/out from the current index; --runner docker builds every site in containers (host only); a fresh site is a no-op without --force",
- flags: { runner: "string", force: "boolean", "skip-archives": "boolean" },
+ "<id|all> [--runner local|docker|auto] [--force] [--skip-archives] [--allow-missing-media] [--out <dir>] build a site (or every stale one) into its bundle <exportBuildsDir>/<id>/out from the current index; --runner docker builds every site in containers (host only); a fresh site is a no-op without --force; --allow-missing-media lets a report citation with no prepared media through compose; --out <dir> then lays the site's bundle out in <dir> for a server that is not Pages (hard links where it can, else copies; emptied first and guarded as a local deploy's destination; not a deploy, nothing recorded)",
+ flags: {
+ runner: "string",
+ force: "boolean",
+ "skip-archives": "boolean",
+ "allow-missing-media": "boolean",
+ out: "string",
+ },
maxPositionals: 1,
run: async ({ positionals, flags }) => {
const [target] = positionals;
@@ -96,11 +102,17 @@ export const COMMANDS: Command[] = [
console.error("publish build: give <id|all> [--runner local|docker|auto]");
return 2;
}
+ if (flags.out !== undefined && (typeof flags.out !== "string" || !flags.out.trim())) {
+ console.error("publish build: --out needs a directory");
+ return 2;
+ }
return (await import("./publish")).publishBuild({
target,
runner: runner as "local" | "docker" | "auto" | undefined,
force: flags.force === true,
skipArchives: flags["skip-archives"] === true,
+ allowMissingMedia: flags["allow-missing-media"] === true,
+ ...(typeof flags.out === "string" ? { out: flags.out } : {}),
});
},
},
diff --git a/common/bin/publish-out.test.ts b/common/bin/publish-out.test.ts
@@ -0,0 +1,97 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, readFileSync, statSync, symlinkSync, writeFileSync } from "node:fs";
+import os from "node:os";
+import path from "node:path";
+
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test bin/publish-out.test.ts
+//
+// `archilyzer publish build <id> --out <dir>` (release 19 A8): the row's
+// flags, the refusals that come before any build, and the lay-out itself over
+// a hand-made bundle — hard links into it, its symlinks kept, the destination
+// emptied, nothing recorded. The build stage is the stage tests'.
+
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "publish-out-"));
+process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
+process.env.EXPORT_BUILDS_DIR = path.join(ROOT, "builds");
+mkdirSync(path.join(ROOT, "transcripts", "sites", "demo-site"), { recursive: true });
+writeFileSync(path.join(ROOT, "transcripts", "sites", "demo-site", "site.json"), JSON.stringify({ siteId: "demo-site" }));
+
+const { COMMANDS } = await import("./archilyzer");
+const { resolveCommand } = await import("./_cli");
+const { layOutBundle, publishBuild } = await import("./publish");
+const { getPaths } = await import("../lib/paths");
+const { writeBuiltStamp } = await import("../publish/stamps");
+
+function capture() {
+ const lines: string[] = [];
+ const errors: string[] = [];
+ return { out: { log: (s: string) => lines.push(s), error: (s: string) => errors.push(s) }, lines, errors };
+}
+
+test("the publish build row takes --allow-missing-media and --out <dir>", () => {
+ const row = resolveCommand(COMMANDS, ["publish", "build"])?.command;
+ assert.equal(row?.flags?.["allow-missing-media"], "boolean");
+ assert.equal(row?.flags?.out, "string");
+});
+
+test("--out with all, an unknown site, or a guarded destination is refused before any build", async () => {
+ const paths = getPaths();
+ const all = capture();
+ assert.equal(await publishBuild({ target: "all", out: path.join(ROOT, "x"), paths }, all.out), 2);
+ assert.match(all.errors.join(""), /--out lays out ONE site's bundle/);
+ const unknown = capture();
+ assert.equal(await publishBuild({ target: "nope", out: path.join(ROOT, "x"), paths }, unknown.out), 2);
+ assert.match(unknown.errors.join(""), /no site "nope"/);
+ const guarded = capture();
+ assert.equal(await publishBuild({ target: "demo-site", out: paths.transcriptsDir, paths }, guarded.out), 2);
+ assert.match(guarded.errors.join(""), /--out .* holds the corpus .* --out empties its destination/);
+ const notBundle = path.join(ROOT, "busy");
+ mkdirSync(notBundle);
+ writeFileSync(path.join(notBundle, "keep.txt"), "x");
+ const busy = capture();
+ assert.equal(await publishBuild({ target: "demo-site", out: notBundle, paths }, busy.out), 2);
+ assert.match(busy.errors.join(""), /holds no index\.html/);
+});
+
+test("the lay-out hard-links the bundle into an emptied destination and records nothing", async () => {
+ const paths = getPaths();
+ const empty = capture();
+ assert.equal(await layOutBundle(paths, "demo-site", path.join(ROOT, "none"), empty.out), 3);
+ assert.match(empty.errors.join(""), /no build of demo-site/);
+
+ const bundle = path.join(paths.exportBuildsDir, "demo-site", "out");
+ mkdirSync(path.join(bundle, "channels"), { recursive: true });
+ writeFileSync(path.join(bundle, "index.html"), "<html></html>");
+ writeFileSync(path.join(bundle, "channels", "a.json"), "{}");
+ symlinkSync("index.html", path.join(bundle, "home.html"));
+ await writeBuiltStamp(paths, {
+ v: 1,
+ stampId: "b-1",
+ target: "demo-site",
+ kind: "site",
+ indexStampId: null,
+ inputSig: "sig",
+ builtAt: 1,
+ commit: null,
+ branch: null,
+ runner: "local",
+ audience: "private",
+ corpusGeneratedAt: null,
+ files: 2,
+ bytes: 10,
+ archivesStaged: 0,
+ } as never);
+ const dest = path.join(ROOT, "serve");
+ mkdirSync(dest);
+ writeFileSync(path.join(dest, "index.html"), "old");
+ writeFileSync(path.join(dest, "stale.html"), "old");
+ const c = capture();
+ assert.equal(await layOutBundle(paths, "demo-site", dest, c.out), 0);
+ assert.match(c.lines.join(""), /build b-1 laid out in .* — 2 hard-linked, 0 copied, 1 symlinks \(not a deploy: nothing recorded\)/);
+ assert.equal(statSync(path.join(dest, "index.html")).ino, statSync(path.join(bundle, "index.html")).ino);
+ assert.equal(readFileSync(path.join(dest, "channels", "a.json"), "utf8"), "{}");
+ assert.equal(readFileSync(path.join(dest, "home.html"), "utf8"), "<html></html>");
+ assert.throws(() => statSync(path.join(dest, "stale.html")));
+ assert.throws(() => statSync(path.join(paths.exportBuildsDir, "demo-site", "deployed.json")));
+});
diff --git a/common/bin/publish.ts b/common/bin/publish.ts
@@ -4,6 +4,7 @@
//
// publish index update-index (as a child: its heap cap)
// publish build <id|all> [--runner local|docker|auto] [--force] [--skip-archives]
+// [--allow-missing-media] [--out <dir>]
// publish deploy <id|all> [--preview <b>] [--to local] [--force]
// publish hub [--deploy] [--preview <b>] [--force]
// publish homepage [--deploy] [--preview <b>] [--to local] [--force]
@@ -23,6 +24,8 @@ import { newStampId } from "../publish/stamps";
import { STAGE_EXIT, runStage, stageCommand, stageMain } from "../publish/stageRun";
import { parseStageArgs, type StageRequest } from "../publish/stages";
import type { CommandContext } from "./_cli";
+import path from "node:path";
+import { copyFile, link, lstat, mkdir, readdir, readlink, rm, symlink } from "node:fs/promises";
type Out = { log: (s: string) => void; error: (s: string) => void };
@@ -109,6 +112,9 @@ export type BuildArgs = {
force?: boolean;
skipArchives?: boolean;
allowMissingMedia?: boolean;
+ // A private build served elsewhere (release 19 A8): after the build (or the
+ // no-op of a fresh one), the bundle is laid out in this directory.
+ out?: string;
runId?: string;
paths?: Paths;
signal?: AbortSignal;
@@ -116,7 +122,25 @@ export type BuildArgs = {
export async function publishBuild(a: BuildArgs, out: Out = console): Promise<number> {
const all = a.target === "all";
+ if (all && a.out !== undefined) {
+ out.error("publish build: --out lays out ONE site's bundle — give a site id, not all");
+ return STAGE_EXIT.usage;
+ }
if (!all && !knownSite(a.target, "publish build", out, a.paths)) return STAGE_EXIT.usage;
+ if (a.out !== undefined) {
+ // Refused before the build, so a typo'd destination costs no build.
+ const paths = a.paths ?? getPaths();
+ const { bundleDir } = await import("../publish/build");
+ const { localDestProblem } = await import("../publish/deployStage");
+ const problem = await localDestProblem(path.resolve(a.out), bundleDir(paths, a.target), paths);
+ if (problem) {
+ out.error(`publish build: --out ${problem.replace(/a local deploy empties/g, "--out empties")}`);
+ return STAGE_EXIT.usage;
+ }
+ const code = await publishBuild({ ...a, out: undefined }, out);
+ if (code !== 0) return code;
+ return layOutBundle(paths, a.target, path.resolve(a.out), out);
+ }
const signal = a.signal ?? interrupted();
let runner: "local" | "docker" = "local";
if (a.runner === "docker" || a.runner === "auto") {
@@ -146,6 +170,64 @@ export async function publishBuild(a: BuildArgs, out: Out = console): Promise<nu
);
}
+// --- publish build --out --------------------------------------------------------
+
+// THE SITE'S BUNDLE, LAID OUT IN `dest` for a server that is not Pages (a
+// private build served elsewhere). Cheap: every file is a HARD LINK into the
+// bundle when the two share a filesystem — a build never edits a bundle in
+// place, it installs a new one (publish/build.ts installBundle), so the links
+// keep the files of the build they were made from — else a copy. `dest` is
+// emptied first (its contents, never the directory: it may be a mount) and
+// guarded as a local deploy's destination is (localDestProblem). Not a deploy:
+// nothing is recorded in deployed.json.
+export async function layOutBundle(paths: Paths, target: string, dest: string, out: Out = console): Promise<number> {
+ const { bundleDir } = await import("../publish/build");
+ const { readBuiltStamp } = await import("../publish/stamps");
+ const src = bundleDir(paths, target);
+ const built = await readBuiltStamp(paths, target);
+ if (!built) {
+ out.error(`publish build: no build of ${target} in ${src} — nothing to lay out`);
+ return STAGE_EXIT.precondition;
+ }
+ await mkdir(dest, { recursive: true });
+ for (const e of await readdir(dest)) await rm(path.join(dest, e), { recursive: true, force: true });
+ const counts = { linked: 0, copied: 0, links: 0 };
+ const walk = async (from: string, to: string): Promise<void> => {
+ for (const e of await readdir(from, { withFileTypes: true })) {
+ const s = path.join(from, e.name);
+ const d = path.join(to, e.name);
+ if (e.isDirectory()) {
+ await mkdir(d, { recursive: true });
+ await walk(s, d);
+ } else if (e.isSymbolicLink()) {
+ await symlink(await readlink(s), d);
+ counts.links++;
+ } else {
+ try {
+ await link(s, d);
+ counts.linked++;
+ } catch (err) {
+ const code = (err as NodeJS.ErrnoException).code;
+ if (code !== "EXDEV" && code !== "EPERM" && code !== "EMLINK") throw err;
+ await copyFile(s, d);
+ counts.copied++;
+ }
+ }
+ }
+ };
+ const st = await lstat(src).catch(() => null);
+ if (!st?.isDirectory()) {
+ out.error(`publish build: the bundle ${src} is missing`);
+ return STAGE_EXIT.precondition;
+ }
+ await walk(src, dest);
+ out.log(
+ `[out] ${target}: build ${built.stampId} laid out in ${dest} — ${counts.linked} hard-linked, ` +
+ `${counts.copied} copied${counts.links ? `, ${counts.links} symlinks` : ""} (not a deploy: nothing recorded).`,
+ );
+ return 0;
+}
+
export type DeployArgs = {
target: string; // a site id or "all"
preview?: string;
diff --git a/common/controller/archiveOrgImport.test.ts b/common/controller/archiveOrgImport.test.ts
@@ -24,10 +24,16 @@ writeFileSync(
mkdirSync(path.join(ROOT, "channels", "c", "data"), { recursive: true });
const {
+ ARCHIVE_ORG_ITEM_GAP_SECONDS,
ARCHIVE_ORG_MIN_GAP_SECONDS,
archiveOrgGapMs,
+ archiveOrgRestriction,
+ archiveOrgSearchUrl,
resolveArchiveOrgImportUrl,
+ runArchiveOrgBatchImport,
runArchiveOrgImport,
+ searchArchiveOrgItems,
+ summarizeArchiveOrgBatch,
} = await import("./archiveOrgImport");
const { ArchiveOrgClient } = await import("../lib/archiveOrgClient");
const { getPaths } = await import("../lib/paths");
@@ -221,3 +227,184 @@ test("bulk: a dry run fetches nothing", async () => {
assert.equal(n, 0);
assert.equal(result.planned, 3);
});
+
+// ─── Many items (release 19 A6) ───
+
+// A scripted archive.org that answers by URL: each item's metadata, a search,
+// and `{}` (no such item) for anything else. Every request is recorded.
+function routedClient(items: Record<string, unknown>, search?: unknown) {
+ const asked: string[] = [];
+ const c = new ArchiveOrgClient(
+ {
+ sleep: async () => {},
+ fetch: async (url: string) => {
+ asked.push(url);
+ if (url.includes("advancedsearch.php")) return new Response(JSON.stringify(search ?? {}), { status: 200 });
+ const id = /\/metadata\/([^/?]+)/.exec(url)?.[1] ?? "";
+ return new Response(JSON.stringify(items[decodeURIComponent(id)] ?? {}), { status: 200 });
+ },
+ },
+ { minGapMs: 0 },
+ );
+ return { client: c, asked };
+}
+
+const one = (identifier: string, extra: Record<string, unknown> = {}, file: Record<string, unknown> = {}) => ({
+ metadata: { identifier, title: identifier, ...extra },
+ files: [{ name: `${identifier}.mp4`, source: "original", ...file }],
+});
+
+test("restriction: an access-restricted item restricts every file; else a private file only", () => {
+ const whole = archiveOrgRestriction(one("r1", { "access-restricted-item": "true" }) as never);
+ assert.equal(whole.item, true);
+ assert.deepEqual([...whole.files], ["r1.mp4"]);
+ const priv = archiveOrgRestriction({
+ metadata: { identifier: "r2" },
+ files: [
+ { name: "a.mp4", source: "original", private: "true" },
+ { name: "b.mp4", source: "original" },
+ ],
+ } as never);
+ assert.equal(priv.item, false);
+ assert.deepEqual([...priv.files], ["a.mp4"]);
+});
+
+test("search: one request through the client, identifiers in order; the URL sorts and caps", async () => {
+ const url = archiveOrgSearchUrl("collection:example AND mediatype:movies", 50);
+ assert.match(url, /^https:\/\/archive\.org\/advancedsearch\.php\?/);
+ const q = new URL(url).searchParams;
+ assert.equal(q.get("q"), "collection:example AND mediatype:movies");
+ assert.deepEqual(q.getAll("fl[]"), ["identifier", "title"]);
+ assert.equal(q.get("sort[]"), "identifier asc");
+ assert.equal(q.get("rows"), "50");
+ assert.equal(q.get("output"), "json");
+ const { client: c, asked } = routedClient(
+ {},
+ { response: { numFound: 7, docs: [{ identifier: "a-1", title: ["A"] }, { identifier: "b-2" }, { title: "no id" }] } },
+ );
+ const r = await searchArchiveOrgItems("x", { rows: 9999, client: c });
+ assert.equal(asked.length, 1);
+ assert.equal(new URL(asked[0]).searchParams.get("rows"), "500");
+ assert.equal(r.found, 7);
+ assert.deepEqual(r.items, [{ identifier: "a-1", title: "A" }, { identifier: "b-2" }]);
+});
+
+test("batch dry run: held (on disk or saved), restricted and missing items are told apart; nothing fetched", async () => {
+ const dataDir = path.join(ROOT, "channels", "c", "data");
+ // h-disk holds a transcript; h-saved has only its saved-video pointer (the
+ // saved-container tier) — both are held.
+ mkdirSync(path.join(dataDir, "h-disk"), { recursive: true });
+ writeFileSync(path.join(dataDir, "h-disk", "transcript.json"), "{}");
+ mkdirSync(path.join(dataDir, "h-saved"), { recursive: true });
+ writeFileSync(
+ path.join(dataDir, "h-saved", "saved-video.json"),
+ JSON.stringify({ storedAt: "2026-01-01T00:00:00Z", dir: "/store/c/h-saved", file: "source.mp4", bytes: 10 }),
+ );
+ const { client: c } = routedClient({
+ "h-disk": one("h-disk"),
+ "h-saved": one("h-saved"),
+ "r-item": one("r-item", { "access-restricted-item": true }),
+ "r-file": one("r-file", {}, { private: "true" }),
+ fresh: one("fresh"),
+ });
+ const sleeps: number[] = [];
+ const lines: string[] = [];
+ let n = 0;
+ const b = await runArchiveOrgBatchImport({
+ paths: getPaths(),
+ slug: "c",
+ channelConfig: CONFIG,
+ items: ["h-disk", "h-saved", "r-item", "r-file", "gone", "fresh"].map((identifier) => ({
+ identifier,
+ selection: { match: "." },
+ })),
+ onLog: (l) => lines.push(l),
+ signal: new AbortController().signal,
+ dryRun: true,
+ deps: {
+ client: c,
+ downloadOne: async () => {
+ n++;
+ return outcome("ok");
+ },
+ sleep: async (ms) => {
+ sleeps.push(ms);
+ },
+ random: () => 0,
+ },
+ });
+ assert.equal(n, 0);
+ // The inter-item gap before every item after the first, and no download gap.
+ assert.deepEqual(sleeps, Array(5).fill(ARCHIVE_ORG_ITEM_GAP_SECONDS * 1000));
+ const s = summarizeArchiveOrgBatch(b);
+ assert.equal(s.dryRun, true);
+ assert.equal(s.items, 5);
+ assert.equal(s.held, 2);
+ assert.deepEqual(s.restricted, [
+ { identifier: "r-item", files: ["r-item.mp4"] },
+ { identifier: "r-file", files: ["r-file.mp4"] },
+ ]);
+ assert.deepEqual(s.missing, ["gone"]);
+ assert.match(lines.join(""), /RESTRICTED r-item/);
+ assert.match(lines.join(""), /would get fresh/);
+});
+
+test("batch import: a 401 skips the rest of its item; three refused items in a row stop the batch", async () => {
+ const two = (id: string) => ({
+ metadata: { identifier: id },
+ files: [`${id}-a.mp4`, `${id}-b.mp4`].map((name) => ({ name, source: "original" })),
+ });
+ const { client: c } = routedClient({ p1: two("p1"), p2: two("p2"), p3: two("p3"), p4: two("p4") });
+ const urls: string[] = [];
+ const b = await runArchiveOrgBatchImport({
+ paths: getPaths(),
+ slug: "c",
+ channelConfig: CONFIG,
+ items: ["p1", "p2", "p3", "p4"].map((identifier) => ({ identifier, selection: { match: "." } })),
+ onLog: () => {},
+ signal: new AbortController().signal,
+ deps: {
+ client: c,
+ downloadOne: async (o) => {
+ urls.push(o.videoUrl);
+ return {
+ ...outcome("failed"),
+ attempts: [{ n: 1, error: `archive.org answered HTTP 401 for ${o.videoUrl}` }],
+ } as unknown as DownloadOutcomeRecord;
+ },
+ sleep: async () => {},
+ },
+ });
+ // One attempt per item (its second file skipped as restricted), three items, then the stop.
+ assert.equal(urls.length, 3);
+ assert.equal(b.refusedStorm, true);
+ assert.match(b.stopped ?? "", /refused 3 items in a row/);
+ assert.deepEqual(b.notReached, ["p4"]);
+ assert.deepEqual(b.items[0].restricted, ["p1-a.mp4", "p1-b.mp4"]);
+ assert.equal(b.items[0].failed.length, 0);
+});
+
+test("batch import: a rate limit stops everything; the rest are not reached", async () => {
+ const { client: c } = routedClient({ q1: one("q1"), q2: one("q2"), q3: one("q3") });
+ let n = 0;
+ const b = await runArchiveOrgBatchImport({
+ paths: getPaths(),
+ slug: "c",
+ channelConfig: CONFIG,
+ items: ["q1", "q2", "q3"].map((identifier) => ({ identifier, selection: { match: "." } })),
+ onLog: () => {},
+ signal: new AbortController().signal,
+ deps: {
+ client: c,
+ downloadOne: async () => {
+ n++;
+ return n === 1 ? outcome("ok") : outcome("failed", "rate_limit");
+ },
+ sleep: async () => {},
+ },
+ });
+ assert.equal(n, 2);
+ assert.equal(b.rateLimited, true);
+ assert.deepEqual(b.notReached, ["q3"]);
+ assert.deepEqual(summarizeArchiveOrgBatch(b).imported, 1);
+});
diff --git a/common/controller/archiveOrgImport.ts b/common/controller/archiveOrgImport.ts
@@ -27,6 +27,15 @@
// backoff takes over), and ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES failures in
// a row stop it too. Re-running the same command resumes: what landed is
// skipped.
+//
+// MANY ITEMS (release 19 A6): `{items: [...]}` or `{query}` (archive.org's
+// advanced search, one request) is one job over many items, the same polite
+// rules shared across the batch plus a jittered pause between items. A file
+// archive.org marks as not for download (`access-restricted-item` on the item,
+// `private` on the file) is RESTRICTED: a dry run lists it, an import skips it;
+// a download answered 401/403 marks the rest of its item restricted, and three
+// such items in a row stop the batch (the platform backs off). A record already
+// HELD — on disk, or its media in the saved-video store — is never fetched.
import path from "node:path";
import type { ChannelConfig } from "../lib/channelConfig";
@@ -43,7 +52,12 @@ import {
archiveOrgVideoId,
parseArchiveOrgUrl,
} from "../lib/archiveOrgId";
-import { archiveOrgClient, type ArchiveOrgClient } from "../lib/archiveOrgClient";
+import {
+ ArchiveOrgRequestError,
+ archiveOrgClient,
+ type ArchiveOrgClient,
+} from "../lib/archiveOrgClient";
+import { isSavedVideo } from "../lib/savedVideo-server";
import { getSettings } from "../lib/settings";
import { diskGate } from "../lib/diskSpace";
import { resolveCookiePolicy } from "../lib/cookiePolicy";
@@ -123,8 +137,12 @@ export type ArchiveOrgImportPlanEntry = {
file: string;
url: string;
id: string;
- // Already downloaded: skipped, never fetched again.
+ // HELD: already downloaded, or its media kept in the saved-video store (the
+ // saved-container tier) — skipped, never fetched again.
onDisk: boolean;
+ // archive.org marks the file (or the whole item) as not for download: a
+ // fetch answers 401/403. Listed by a dry run, skipped by an import.
+ restricted: boolean;
};
export type ArchiveOrgImportPlan = {
@@ -134,8 +152,41 @@ export type ArchiveOrgImportPlan = {
entries: ArchiveOrgImportPlanEntry[];
// Named in `files` but not a media original of the item.
unknown: string[];
+ // The whole item is access-restricted (archive.org's lending flag).
+ restrictedItem: boolean;
};
+const truthy = (v: unknown): boolean => v === true || v === "true" || v === "1";
+
+// What archive.org will not hand out, read off the item's own metadata: an
+// item carrying `access-restricted-item` (every file), else each file marked
+// `private`. Either answers a download with 401/403; the metadata says so
+// before anything is asked for.
+export function archiveOrgRestriction(item: ArchiveOrgItemMetadata): {
+ item: boolean;
+ files: Set<string>;
+} {
+ const meta = item.metadata as Record<string, unknown>;
+ const whole = truthy(meta["access-restricted-item"]);
+ const files = new Set<string>();
+ for (const f of item.files) {
+ if (whole || truthy((f as Record<string, unknown>).private)) files.add(f.name);
+ }
+ return { item: whole, files };
+}
+
+// HELD: the record has its transcript or (on a transcribe channel) its audio —
+// `destinationExists`, what every download path asks — or its media lives in
+// the saved-video store, which a download must never bypass.
+export async function isHeldArchiveOrgRecord(
+ dataDir: string,
+ id: string,
+ handling: ChannelConfig["handling"],
+): Promise<boolean> {
+ if (await destinationExists(dataDir, id, handling)) return true;
+ return isSavedVideo(path.join(dataDir, id));
+}
+
export async function planArchiveOrgImport(opts: {
identifier: string;
selection: ArchiveOrgFileSelection;
@@ -148,6 +199,7 @@ export async function planArchiveOrgImport(opts: {
const item = await client.itemMetadata(opts.identifier, opts.signal);
const identifier = item.metadata.identifier || opts.identifier;
const media = listArchiveOrgMediaFiles(item);
+ const restriction = archiveOrgRestriction(item);
const { picked, unknown } = pickArchiveOrgFiles(item, opts.selection);
const entries: ArchiveOrgImportPlanEntry[] = [];
for (const file of picked) {
@@ -159,7 +211,8 @@ export async function planArchiveOrgImport(opts: {
file,
url,
id,
- onDisk: await destinationExists(opts.dataDir, id, opts.handling),
+ onDisk: await isHeldArchiveOrgRecord(opts.dataDir, id, opts.handling),
+ restricted: restriction.files.has(file),
});
}
const title = item.metadata.title;
@@ -169,9 +222,51 @@ export async function planArchiveOrgImport(opts: {
mediaFiles: media.length,
entries,
unknown,
+ restrictedItem: restriction.item,
};
}
+// ─── A search ───
+
+export const ARCHIVE_ORG_SEARCH_DEFAULT_ROWS = 100;
+export const ARCHIVE_ORG_SEARCH_MAX_ROWS = 500;
+
+// archive.org's advanced search, ONE request through the polite client (its
+// chain, its gap, its backoff): the identifiers of the first `rows` items the
+// query matches, in identifier order so a re-run walks the same list.
+export function archiveOrgSearchUrl(query: string, rows: number): string {
+ const q = new URLSearchParams();
+ q.set("q", query);
+ q.append("fl[]", "identifier");
+ q.append("fl[]", "title");
+ q.append("sort[]", "identifier asc");
+ q.set("rows", String(rows));
+ q.set("page", "1");
+ q.set("output", "json");
+ return `https://archive.org/advancedsearch.php?${q.toString()}`;
+}
+
+export async function searchArchiveOrgItems(
+ query: string,
+ opts: { rows?: number; client?: ArchiveOrgClient; signal?: AbortSignal } = {},
+): Promise<{ found: number; items: { identifier: string; title?: string }[] }> {
+ const rows = Math.min(ARCHIVE_ORG_SEARCH_MAX_ROWS, Math.max(1, opts.rows ?? ARCHIVE_ORG_SEARCH_DEFAULT_ROWS));
+ const client = opts.client ?? archiveOrgClient;
+ const raw = (await client.getJson(archiveOrgSearchUrl(query, rows), opts.signal)) as {
+ response?: { numFound?: unknown; docs?: unknown };
+ } | null;
+ const docs = Array.isArray(raw?.response?.docs) ? (raw!.response!.docs as unknown[]) : [];
+ const items: { identifier: string; title?: string }[] = [];
+ for (const d of docs) {
+ const r = d as { identifier?: unknown; title?: unknown };
+ if (typeof r?.identifier !== "string" || !r.identifier) continue;
+ const title = Array.isArray(r.title) ? r.title.find((t) => typeof t === "string") : r.title;
+ items.push({ identifier: r.identifier, ...(typeof title === "string" ? { title } : {}) });
+ }
+ const found = typeof raw?.response?.numFound === "number" ? raw.response.numFound : items.length;
+ return { found, items };
+}
+
// The gap before the next file: the configured pause, floored, plus up to
// half again at random so a batch never settles into a fixed beat.
export function archiveOrgGapMs(sleepBetweenDownloadsSeconds: number, random: number): number {
@@ -179,15 +274,39 @@ export function archiveOrgGapMs(sleepBetweenDownloadsSeconds: number, random: nu
return Math.round(base * (1 + 0.5 * Math.min(1, Math.max(0, random))) * 1000);
}
+// THE INTER-ITEM GAP: before the next item of a batch is asked about, a
+// jittered pause on top of the client's own 2 s between requests, so a dry run
+// over a search of hundreds of items reads as a person paging, not a crawl.
+// (A download after the first is paced by archiveOrgGapMs as well.)
+export const ARCHIVE_ORG_ITEM_GAP_SECONDS = 3;
+export function archiveOrgItemGapMs(random: number): number {
+ return Math.round(ARCHIVE_ORG_ITEM_GAP_SECONDS * (1 + 0.5 * Math.min(1, Math.max(0, random))) * 1000);
+}
+
+// Items in a row whose downloads archive.org refused with 401/403 before the
+// batch stops: one restricted item is that item; three in a row is archive.org
+// refusing us, and the platform backs off.
+export const ARCHIVE_ORG_MAX_REFUSED_ITEMS = 3;
+
function isOk(rec: DownloadOutcomeRecord): boolean {
return rec.status.startsWith("ok");
}
+// A download error that is archive.org saying "not for you" (lib/archiveOrgClient
+// words every non-retryable answer "archive.org answered HTTP <n> for <url>").
+export function isRefusalError(error: string): boolean {
+ return /\bHTTP 40[13]\b/.test(error);
+}
+
export type ArchiveOrgImportResult = {
identifier: string;
planned: number;
imported: string[];
+ // Held: on disk or in the saved-video store.
skipped: string[];
+ // Not for download (archive.org's metadata says so, or a fetch answered
+ // 401/403): never fetched.
+ restricted: string[];
failed: { file: string; error: string }[];
unknown: string[];
// Why the batch ended before its last file, when it did.
@@ -219,18 +338,47 @@ function abortableSleep(ms: number, signal: AbortSignal): Promise<void> {
});
}
-export async function runArchiveOrgImport(opts: {
+export type ArchiveOrgItemRequest = {
+ identifier: string;
+ selection: ArchiveOrgFileSelection;
+};
+
+export type ArchiveOrgBatchResult = {
+ // One per item reached, in order.
+ items: ArchiveOrgImportResult[];
+ // Items archive.org has no record of (or whose metadata it would not give).
+ missing: { identifier: string; error: string }[];
+ // Items never reached: the batch stopped first.
+ notReached: string[];
+ stopped?: string;
+ rateLimited?: boolean;
+ // Three items in a row refused with 401/403: the caller backs the platform off.
+ refusedStorm?: boolean;
+ dryRun: boolean;
+};
+
+type BatchOpts = {
paths: Paths;
slug: string;
channelConfig: ChannelConfig;
- identifier: string;
- selection: ArchiveOrgFileSelection;
onLog: (line: string) => void;
signal: AbortSignal;
drainSignal?: AbortSignal;
dryRun?: boolean;
deps?: ArchiveOrgImportDeps;
-}): Promise<ArchiveOrgImportResult> {
+};
+
+// MANY ITEMS, ONE JOB (`import-archive-org {items | query}`). Each item is
+// planned (one cached metadata request) and imported as a single-item import
+// is, with the state that makes it polite shared across the batch: the gap
+// before every download after the first, the consecutive-failure count, the
+// rate-limit stop. Between items, the inter-item gap. An item archive.org has
+// no record of is listed and passed over; a rate limit stops everything (the
+// rest are `notReached`, and re-running the same body resumes — held records
+// are skipped); three items in a row refused with 401/403 stop it too.
+export async function runArchiveOrgBatchImport(
+ opts: BatchOpts & { items: ArchiveOrgItemRequest[] },
+): Promise<ArchiveOrgBatchResult> {
const deps = opts.deps ?? {};
const downloadOne = deps.downloadOne ?? downloadOneManaged;
const sleep = deps.sleep ?? abortableSleep;
@@ -238,124 +386,269 @@ export async function runArchiveOrgImport(opts: {
const settings = getSettings();
const dataDir = path.join(opts.paths.channelsDir, opts.slug, "data");
const log = opts.onLog;
-
- const plan = await planArchiveOrgImport({
- identifier: opts.identifier,
- selection: opts.selection,
- dataDir,
- handling: opts.channelConfig.handling,
- client: deps.client,
- signal: opts.signal,
- });
- const result: ArchiveOrgImportResult = {
- identifier: plan.identifier,
- planned: plan.entries.length,
- imported: [],
- skipped: [],
- failed: [],
- unknown: plan.unknown,
- };
- log(
- `archive.org item ${plan.identifier}${plan.title ? ` ("${plan.title}")` : ""}: ` +
- `${plan.mediaFiles} media files, ${plan.entries.length} chosen, ` +
- `${plan.entries.filter((e) => e.onDisk).length} already downloaded.\n`,
- );
- if (plan.unknown.length > 0) {
- log(`Not media files of the item (ignored): ${plan.unknown.map((n) => JSON.stringify(n)).join(", ")}\n`);
- }
- if (opts.dryRun) {
- for (const e of plan.entries) log(` ${e.onDisk ? "on disk " : "would get"} ${e.id} ${e.file}\n`);
- result.skipped = plan.entries.filter((e) => e.onDisk).map((e) => e.file);
- return result;
- }
-
const sleepSeconds =
opts.channelConfig.sleepBetweenDownloadsSeconds ?? settings.sleepBetweenDownloadsSeconds;
+ const batch: ArchiveOrgBatchResult = { items: [], missing: [], notReached: [], dryRun: opts.dryRun === true };
let fetched = 0;
let consecutiveFailures = 0;
- for (const entry of plan.entries) {
+ let refusedItemsInARow = 0;
+ const stop = (why: string) => {
+ batch.stopped = why;
+ };
+
+ for (let i = 0; i < opts.items.length; i++) {
+ const req = opts.items[i];
+ if (batch.stopped) {
+ batch.notReached.push(req.identifier);
+ continue;
+ }
if (opts.signal.aborted) {
- result.stopped = "cancelled";
- break;
+ stop("cancelled");
+ batch.notReached.push(req.identifier);
+ continue;
}
if (opts.drainSignal?.aborted) {
- result.stopped = "drained";
- break;
- }
- if (entry.onDisk || (await destinationExists(dataDir, entry.id, opts.channelConfig.handling))) {
- result.skipped.push(entry.file);
+ stop("drained");
+ batch.notReached.push(req.identifier);
continue;
}
- const gate = await diskGate(opts.paths, settings, { dir: dataDir });
- if (!gate.ok) {
- result.stopped = gate.message;
- log(`Stopping: ${gate.message}.\n`);
- break;
- }
- if (fetched > 0) {
- const gap = archiveOrgGapMs(sleepSeconds, random());
- log(`Waiting ${(gap / 1000).toFixed(1)}s before the next file (archive.org pacing)...\n`);
- await sleep(gap, opts.signal);
+ if (i > 0) {
+ await sleep(archiveOrgItemGapMs(random()), opts.signal);
if (opts.signal.aborted) {
- result.stopped = "cancelled";
- break;
+ stop("cancelled");
+ batch.notReached.push(req.identifier);
+ continue;
}
}
- fetched++;
- log(`[${fetched}] ${entry.file} → data/${entry.id}/\n`);
- let rec: DownloadOutcomeRecord | null = null;
- let error = "";
+
+ let plan: ArchiveOrgImportPlan;
try {
- rec = await downloadOne({
- channelSlug: opts.slug,
- channelConfig: opts.channelConfig,
- paths: opts.paths,
- videoUrl: entry.url,
- onLog: log,
+ plan = await planArchiveOrgImport({
+ identifier: req.identifier,
+ selection: req.selection,
+ dataDir,
+ handling: opts.channelConfig.handling,
+ client: deps.client,
signal: opts.signal,
- cookiePolicy: resolveCookiePolicy(settings, opts.channelConfig),
- inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback,
- globalSkipLiveDownloads: settings.skipLiveDownloads,
- appendArchive: true,
});
} catch (err) {
- error = (err as Error).message;
+ const e = err as ArchiveOrgRequestError;
+ if (e instanceof ArchiveOrgRequestError && e.rateLimited) {
+ batch.rateLimited = true;
+ stop(`archive.org did not answer for item ${req.identifier}; stopping (re-run later — held records are skipped)`);
+ log(`${batch.stopped}.\n`);
+ batch.notReached.push(req.identifier);
+ continue;
+ }
+ if (opts.signal.aborted) {
+ stop("cancelled");
+ batch.notReached.push(req.identifier);
+ continue;
+ }
+ batch.missing.push({ identifier: req.identifier, error: e.message });
+ log(`archive.org item ${req.identifier}: ${e.message} — passed over.\n`);
+ continue;
}
- if (rec && isOk(rec)) {
- consecutiveFailures = 0;
- result.imported.push(entry.file);
- await mergeRosterFile(
- opts.paths,
- opts.slug,
- [{ id: entry.id, url: entry.url }],
- new Date().toISOString(),
- "import",
- ).catch(() => {
- /* the download succeeded; a roster write failure must not fail it */
- });
- deps.onImported?.(entry.id);
+
+ const result: ArchiveOrgImportResult = {
+ identifier: plan.identifier,
+ planned: plan.entries.length,
+ imported: [],
+ skipped: [],
+ restricted: [],
+ failed: [],
+ unknown: plan.unknown,
+ };
+ batch.items.push(result);
+ const held = plan.entries.filter((e) => e.onDisk).length;
+ const restricted = plan.entries.filter((e) => e.restricted && !e.onDisk).length;
+ log(
+ `archive.org item ${plan.identifier}${plan.title ? ` ("${plan.title}")` : ""}: ` +
+ `${plan.mediaFiles} media files, ${plan.entries.length} chosen, ` +
+ `${held} already held` +
+ `${restricted ? `, ${restricted} restricted${plan.restrictedItem ? " (the item is access-restricted)" : ""}` : ""}.\n`,
+ );
+ if (plan.unknown.length > 0) {
+ log(`Not media files of the item (ignored): ${plan.unknown.map((n) => JSON.stringify(n)).join(", ")}\n`);
+ }
+ if (opts.dryRun) {
+ for (const e of plan.entries) {
+ const state = e.onDisk ? "held " : e.restricted ? "RESTRICTED" : "would get";
+ log(` ${state} ${e.id} ${e.file}\n`);
+ }
+ result.skipped = plan.entries.filter((e) => e.onDisk).map((e) => e.file);
+ result.restricted = plan.entries.filter((e) => !e.onDisk && e.restricted).map((e) => e.file);
continue;
}
- consecutiveFailures++;
- const why = error || rec?.attempts.at(-1)?.error || rec?.status || "failed";
- result.failed.push({ file: entry.file, error: why });
- log(` failed: ${why}\n`);
- if (rec?.failureClass === "rate_limit") {
- result.rateLimited = true;
- result.stopped = "archive.org rate-limited the download; stopping (re-run later — files on disk are skipped)";
- log(`${result.stopped}.\n`);
- break;
+
+ let refusedHere = false;
+ for (const entry of plan.entries) {
+ if (opts.signal.aborted) {
+ stop("cancelled");
+ break;
+ }
+ if (opts.drainSignal?.aborted) {
+ stop("drained");
+ break;
+ }
+ if (entry.onDisk || (await isHeldArchiveOrgRecord(dataDir, entry.id, opts.channelConfig.handling))) {
+ result.skipped.push(entry.file);
+ continue;
+ }
+ if (entry.restricted || refusedHere) {
+ result.restricted.push(entry.file);
+ continue;
+ }
+ const gate = await diskGate(opts.paths, settings, { dir: dataDir });
+ if (!gate.ok) {
+ stop(gate.message);
+ log(`Stopping: ${gate.message}.\n`);
+ break;
+ }
+ if (fetched > 0) {
+ const gap = archiveOrgGapMs(sleepSeconds, random());
+ log(`Waiting ${(gap / 1000).toFixed(1)}s before the next file (archive.org pacing)...\n`);
+ await sleep(gap, opts.signal);
+ if (opts.signal.aborted) {
+ stop("cancelled");
+ break;
+ }
+ }
+ fetched++;
+ log(`[${fetched}] ${entry.file} → data/${entry.id}/\n`);
+ let rec: DownloadOutcomeRecord | null = null;
+ let error = "";
+ try {
+ rec = await downloadOne({
+ channelSlug: opts.slug,
+ channelConfig: opts.channelConfig,
+ paths: opts.paths,
+ videoUrl: entry.url,
+ onLog: log,
+ signal: opts.signal,
+ cookiePolicy: resolveCookiePolicy(settings, opts.channelConfig),
+ inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback,
+ globalSkipLiveDownloads: settings.skipLiveDownloads,
+ appendArchive: true,
+ });
+ } catch (err) {
+ error = (err as Error).message;
+ }
+ if (rec && isOk(rec)) {
+ consecutiveFailures = 0;
+ refusedItemsInARow = 0;
+ result.imported.push(entry.file);
+ await mergeRosterFile(
+ opts.paths,
+ opts.slug,
+ [{ id: entry.id, url: entry.url }],
+ new Date().toISOString(),
+ "import",
+ ).catch(() => {
+ /* the download succeeded; a roster write failure must not fail it */
+ });
+ deps.onImported?.(entry.id);
+ continue;
+ }
+ const why = error || rec?.attempts.at(-1)?.error || rec?.status || "failed";
+ if (isRefusalError(why)) {
+ // archive.org will not hand this item's files to us: the rest of the
+ // item is skipped as restricted, and it is not a failure streak.
+ refusedHere = true;
+ result.restricted.push(entry.file);
+ log(` refused (${why}) — the rest of ${plan.identifier} is skipped as restricted\n`);
+ continue;
+ }
+ consecutiveFailures++;
+ result.failed.push({ file: entry.file, error: why });
+ log(` failed: ${why}\n`);
+ if (rec?.failureClass === "rate_limit") {
+ result.rateLimited = true;
+ batch.rateLimited = true;
+ stop("archive.org rate-limited the download; stopping (re-run later — held records are skipped)");
+ log(`${batch.stopped}.\n`);
+ break;
+ }
+ if (consecutiveFailures >= ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES) {
+ stop(`${consecutiveFailures} failures in a row; stopping`);
+ log(`${batch.stopped}.\n`);
+ break;
+ }
+ }
+ if (batch.stopped) result.stopped = batch.stopped;
+ if (refusedHere) {
+ refusedItemsInARow++;
+ if (refusedItemsInARow >= ARCHIVE_ORG_MAX_REFUSED_ITEMS && !batch.stopped) {
+ batch.refusedStorm = true;
+ stop(`archive.org refused ${refusedItemsInARow} items in a row (401/403); stopping`);
+ result.stopped = batch.stopped;
+ log(`${batch.stopped}.\n`);
+ }
}
- if (consecutiveFailures >= ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES) {
- result.stopped = `${consecutiveFailures} failures in a row; stopping`;
- log(`${result.stopped}.\n`);
- break;
+ log(
+ `archive.org import of ${plan.identifier}: ${result.imported.length} imported, ` +
+ `${result.skipped.length} already held, ${result.restricted.length} restricted, ${result.failed.length} failed` +
+ `${result.stopped ? ` — stopped: ${result.stopped}` : ""}.\n`,
+ );
+ }
+ return batch;
+}
+
+// The batch's totals, for its last log line (`summary: {…}`) and the action's
+// verdict.
+export function summarizeArchiveOrgBatch(b: ArchiveOrgBatchResult): {
+ dryRun: boolean;
+ items: number;
+ planned: number;
+ imported: number;
+ held: number;
+ restricted: { identifier: string; files: string[] }[];
+ failed: number;
+ missing: string[];
+ notReached: string[];
+ stopped?: string;
+} {
+ const sum = (f: (r: ArchiveOrgImportResult) => number) => b.items.reduce((n, r) => n + f(r), 0);
+ return {
+ dryRun: b.dryRun,
+ items: b.items.length,
+ planned: sum((r) => r.planned),
+ imported: sum((r) => r.imported.length),
+ held: sum((r) => r.skipped.length),
+ restricted: b.items
+ .filter((r) => r.restricted.length > 0)
+ .map((r) => ({ identifier: r.identifier, files: r.restricted })),
+ failed: sum((r) => r.failed.length),
+ missing: b.missing.map((m) => m.identifier),
+ notReached: b.notReached,
+ ...(b.stopped ? { stopped: b.stopped } : {}),
+ };
+}
+
+// ONE ITEM — the original `import-archive-org {item}` — as a batch of one.
+export async function runArchiveOrgImport(
+ opts: BatchOpts & { identifier: string; selection: ArchiveOrgFileSelection },
+): Promise<ArchiveOrgImportResult> {
+ const { identifier, selection, ...rest } = opts;
+ const batch = await runArchiveOrgBatchImport({ ...rest, items: [{ identifier, selection }] });
+ // A single item's metadata failure is the import's error, as it always was.
+ if (batch.missing.length > 0) throw new Error(batch.missing[0].error);
+ const result = batch.items[0];
+ if (!result) {
+ if (batch.stopped && batch.stopped !== "cancelled" && batch.stopped !== "drained") {
+ throw new Error(batch.stopped);
}
+ return {
+ identifier,
+ planned: 0,
+ imported: [],
+ skipped: [],
+ restricted: [],
+ failed: [],
+ unknown: [],
+ ...(batch.stopped ? { stopped: batch.stopped } : {}),
+ ...(batch.rateLimited ? { rateLimited: true } : {}),
+ };
}
- log(
- `archive.org import of ${plan.identifier}: ${result.imported.length} imported, ` +
- `${result.skipped.length} already on disk, ${result.failed.length} failed` +
- `${result.stopped ? ` — stopped: ${result.stopped}` : ""}.\n`,
- );
- return result;
+ return { ...result, ...(batch.rateLimited ? { rateLimited: true } : {}) };
}
diff --git a/common/controller/channelCoverage.test.ts b/common/controller/channelCoverage.test.ts
@@ -0,0 +1,95 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs";
+import os from "node:os";
+import path from "node:path";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/channelCoverage.test.ts
+//
+// Coverage by date (release 19 A9): the recorded date wins when the channel's
+// title rule yields one (release 20 D1's coverageDate), a big metadata file is
+// read at both ends, gaps are the spans past gapDays. Every id is invented.
+
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "coverage-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+writeFileSync(process.env.SETTINGS_FILE, "{}\n");
+
+const { channelCoverage, readDateAndTitle, summarizeCoverage } = await import("./channelCoverage");
+const { getPaths } = await import("../lib/paths");
+
+function channel(slug: string, config: Record<string, unknown>, videos: Record<string, Record<string, unknown> | string | null>) {
+ const dir = path.join(ROOT, "channels", slug);
+ mkdirSync(path.join(dir, "data"), { recursive: true });
+ writeFileSync(path.join(dir, "config.json"), JSON.stringify({ handling: "youtube", name: slug, ...config }));
+ for (const [id, meta] of Object.entries(videos)) {
+ mkdirSync(path.join(dir, "data", id), { recursive: true });
+ if (meta === null) continue;
+ writeFileSync(path.join(dir, "data", id, "metadata.info.json"), typeof meta === "string" ? meta : JSON.stringify(meta));
+ }
+}
+
+test("gaps are the spans longer than gapDays, oldest first; months and years counted", () => {
+ const s = summarizeCoverage(
+ [
+ { id: "a", date: "20240101", uploadDate: "20240101" },
+ { id: "b", date: "20240115", uploadDate: "20240115" },
+ { id: "c", date: "20240401", uploadDate: "20240401" },
+ { id: "d", date: "20250101", uploadDate: "20250101" },
+ ],
+ { gapDays: 30 },
+ );
+ assert.equal(s.first, "20240101");
+ assert.equal(s.last, "20250101");
+ assert.deepEqual(s.byYear, { "2024": 3, "2025": 1 });
+ assert.deepEqual(s.byMonth, { "2024-01": 2, "2024-04": 1, "2025-01": 1 });
+ assert.deepEqual(s.gaps, [
+ { after: "20240115", before: "20240401", days: 77 },
+ { after: "20240401", before: "20250101", days: 275 },
+ ]);
+});
+
+test("a big metadata file is read at both ends: title at the head, upload_date at the tail", async () => {
+ const dir = path.join(ROOT, "big");
+ mkdirSync(dir, { recursive: true });
+ const formats = Array.from({ length: 6000 }, (_, i) => ({ format_id: String(i), url: `https://example.invalid/${"x".repeat(40)}` }));
+ writeFileSync(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({ id: "big", title: "Stream — März 5 \"live\"", formats, upload_date: "20240306" }),
+ );
+ assert.deepEqual(await readDateAndTitle(dir), { uploadDate: "20240306", title: 'Stream — März 5 "live"' });
+ assert.deepEqual(await readDateAndTitle(path.join(ROOT, "nowhere")), {});
+});
+
+test("a mirror's videos are dated by the title rule; the rest by upload; no metadata is undated", async () => {
+ channel(
+ "vod-mirror",
+ { recordedDate: { titlePattern: "(?<year>\\d{4})-(?<month>\\d{2})-(?<day>\\d{2})" } },
+ {
+ m1: { title: "Stream 2023-05-01", upload_date: "20240110" },
+ m2: { title: "Stream 2023-05-03", upload_date: "20240110" },
+ m3: { title: "No date in this one", upload_date: "20240110" },
+ // A title date after the upload is not a recording date.
+ m4: { title: "Preview of 2030-01-01", upload_date: "20240111" },
+ bare: null,
+ },
+ );
+ const c = await channelCoverage(getPaths(), "vod-mirror", { gapDays: 60, list: true });
+ assert.equal(c.titlePattern, "(?<year>\\d{4})-(?<month>\\d{2})-(?<day>\\d{2})");
+ assert.equal(c.held, 5);
+ assert.equal(c.dated, 4);
+ assert.equal(c.byRecordedDate, 2);
+ assert.equal(c.byUploadDate, 2);
+ assert.deepEqual(c.undated, ["bare"]);
+ assert.equal(c.first, "20230501");
+ assert.deepEqual(c.gaps, [{ after: "20230503", before: "20240110", days: 252 }]);
+ assert.deepEqual(
+ c.videos?.map((v) => [v.id, v.date]),
+ [["m1", "20230501"], ["m2", "20230503"], ["m3", "20240110"], ["m4", "20240111"]],
+ );
+ const window = await channelCoverage(getPaths(), "vod-mirror", { from: "20240101" });
+ assert.deepEqual(window.byYear, { "2024": 2 });
+ assert.equal(window.videos, undefined);
+ await assert.rejects(channelCoverage(getPaths(), "no-such"), /Channel "no-such" not found/);
+});
diff --git a/common/controller/channelCoverage.ts b/common/controller/channelCoverage.ts
@@ -0,0 +1,196 @@
+// A CHANNEL'S COVERAGE: WHAT IT HOLDS, BY DATE, AND WHERE THE GAPS ARE
+// (release 19 A9; the MCP's `channel_coverage`, `GET /api/ops/coverage`).
+//
+// Every held video (a data/<id>/ directory) is dated with `coverageDate`
+// (lib/recordedDate.ts, release 20 D1): its RECORDED date when the channel has
+// a title rule (`config.recordedDate`) and the title yields one, else its
+// upload date. A VOD mirror's coverage is then the streams' days, not the
+// copies'. The dates come off disk, not the index, so a video imported an hour
+// ago counts.
+//
+// CHEAP ON A BIG CHANNEL: metadata.info.json can run to megabytes (formats), so
+// a small file is parsed whole and a big one is read at both ends — `title`
+// sits near the head of yt-dlp's JSON, `upload_date` near the tail (the same
+// tail read recencyIndex.ts measured at 0.19 ms a file). A video with neither
+// is `undated`, never guessed. Nothing is written.
+
+import path from "node:path";
+import { open, readFile, stat } from "node:fs/promises";
+import type { Paths } from "../lib/paths";
+import { assertChannelTextReadable } from "../lib/channelMedia";
+import { mapConcurrent } from "../lib/concurrency";
+import { compileRecordedDateRule, coverageDate, deriveRecordedDate } from "../lib/recordedDate";
+import { readChannelConfig } from "./channels";
+import { heldVideoIds } from "./videoCues";
+
+const WHOLE_FILE_BYTES = 256 * 1024;
+const HEAD_BYTES = 16 * 1024;
+const TAIL_BYTES = 8 * 1024;
+
+export const DEFAULT_GAP_DAYS = 30;
+
+export type DatedVideo = {
+ id: string;
+ date: string; // YYYYMMDD
+ uploadDate: string;
+ recordedDate?: string;
+ title?: string;
+};
+
+export type ChannelCoverage = {
+ slug: string;
+ // The channel's recorded-date title rule, when it has one.
+ titlePattern: string | null;
+ held: number;
+ dated: number;
+ // Dated by the title rule / by the upload date.
+ byRecordedDate: number;
+ byUploadDate: number;
+ undated: string[];
+ first: string | null;
+ last: string | null;
+ // Videos per year and per month ("2024", "2024-03"), only those with any.
+ byYear: Record<string, number>;
+ byMonth: Record<string, number>;
+ // Spans longer than `gapDays` with no held video, oldest first: the last
+ // held date before, the first after, and the days between.
+ gapDays: number;
+ gaps: { after: string; before: string; days: number }[];
+ // With `list`: every dated video in the window, oldest first.
+ videos?: DatedVideo[];
+};
+
+const UPLOAD_RE = /"upload_date"\s*:\s*"(\d{8})"/;
+const TITLE_RE = /"title"\s*:\s*("(?:[^"\\]|\\.)*")/;
+
+function fromText(text: string): { uploadDate?: string; title?: string } {
+ const up = UPLOAD_RE.exec(text)?.[1];
+ const rawTitle = TITLE_RE.exec(text)?.[1];
+ let title: string | undefined;
+ if (rawTitle) {
+ try {
+ title = JSON.parse(rawTitle) as string;
+ } catch {
+ /* a title cut by the read boundary is no title */
+ }
+ }
+ return { ...(up ? { uploadDate: up } : {}), ...(title ? { title } : {}) };
+}
+
+// The upload date and title of one video, from its metadata.info.json.
+export async function readDateAndTitle(videoDir: string): Promise<{ uploadDate?: string; title?: string }> {
+ const file = path.join(videoDir, "metadata.info.json");
+ const st = await stat(file).catch(() => null);
+ if (!st?.isFile()) return {};
+ if (st.size <= WHOLE_FILE_BYTES) {
+ try {
+ const j = JSON.parse(await readFile(file, "utf8")) as { upload_date?: unknown; title?: unknown };
+ return {
+ ...(typeof j.upload_date === "string" && /^\d{8}$/.test(j.upload_date) ? { uploadDate: j.upload_date } : {}),
+ ...(typeof j.title === "string" && j.title ? { title: j.title } : {}),
+ };
+ } catch {
+ return {};
+ }
+ }
+ const fh = await open(file, "r");
+ try {
+ const head = Buffer.alloc(HEAD_BYTES);
+ const tail = Buffer.alloc(TAIL_BYTES);
+ await fh.read(head, 0, HEAD_BYTES, 0);
+ await fh.read(tail, 0, TAIL_BYTES, st.size - TAIL_BYTES);
+ // A multi-byte sequence split by a read boundary decodes as U+FFFD; the
+ // regexes only take a whole quoted string, so a cut title is no title.
+ const h = fromText(head.toString("utf8"));
+ const t = fromText(tail.toString("utf8"));
+ const uploadDate = t.uploadDate ?? h.uploadDate;
+ const title = h.title ?? t.title;
+ return { ...(uploadDate ? { uploadDate } : {}), ...(title ? { title } : {}) };
+ } finally {
+ await fh.close();
+ }
+}
+
+function daysBetween(a: string, b: string): number {
+ const toMs = (d: string) => Date.UTC(Number(d.slice(0, 4)), Number(d.slice(4, 6)) - 1, Number(d.slice(6, 8)));
+ return Math.round((toMs(b) - toMs(a)) / 86_400_000);
+}
+
+// The pure half: dated videos → the summary (byYear, byMonth, gaps).
+export function summarizeCoverage(
+ videos: DatedVideo[],
+ opts: { gapDays?: number } = {},
+): Pick<ChannelCoverage, "first" | "last" | "byYear" | "byMonth" | "gapDays" | "gaps"> {
+ const gapDays = opts.gapDays ?? DEFAULT_GAP_DAYS;
+ const sorted = [...videos].sort((a, b) => a.date.localeCompare(b.date) || a.id.localeCompare(b.id));
+ const byYear: Record<string, number> = {};
+ const byMonth: Record<string, number> = {};
+ const gaps: ChannelCoverage["gaps"] = [];
+ for (let i = 0; i < sorted.length; i++) {
+ const d = sorted[i].date;
+ byYear[d.slice(0, 4)] = (byYear[d.slice(0, 4)] ?? 0) + 1;
+ const m = `${d.slice(0, 4)}-${d.slice(4, 6)}`;
+ byMonth[m] = (byMonth[m] ?? 0) + 1;
+ if (i > 0) {
+ const prev = sorted[i - 1].date;
+ const days = daysBetween(prev, d);
+ if (days > gapDays) gaps.push({ after: prev, before: d, days });
+ }
+ }
+ return {
+ first: sorted[0]?.date ?? null,
+ last: sorted.at(-1)?.date ?? null,
+ byYear,
+ byMonth,
+ gapDays,
+ gaps,
+ };
+}
+
+export async function channelCoverage(
+ paths: Paths,
+ slug: string,
+ opts: { gapDays?: number; from?: string; to?: string; list?: boolean } = {},
+): Promise<ChannelCoverage> {
+ const config = await readChannelConfig(paths, slug);
+ if (!config) throw new Error(`Channel "${slug}" not found`);
+ // The text guard: an unmounted or moving drive is not an empty channel.
+ await assertChannelTextReadable(paths, slug, config);
+ const dataDir = path.join(paths.channelsDir, slug, "data");
+ const ids = [...(await heldVideoIds(dataDir))].sort();
+ const rule = config.recordedDate ? compileRecordedDateRule(config.recordedDate) : null;
+ const read = await mapConcurrent(ids, 32, async (id) => ({ id, ...(await readDateAndTitle(path.join(dataDir, id))) }));
+ const dated: DatedVideo[] = [];
+ const undated: string[] = [];
+ let byRecordedDate = 0;
+ for (const r of read) {
+ if (!r.uploadDate) {
+ undated.push(r.id);
+ continue;
+ }
+ const recordedDate = rule && r.title ? deriveRecordedDate(rule, r.title, r.uploadDate) : undefined;
+ if (recordedDate) byRecordedDate++;
+ dated.push({
+ id: r.id,
+ date: coverageDate({ uploadDate: r.uploadDate, ...(recordedDate ? { recordedDate } : {}) }),
+ uploadDate: r.uploadDate,
+ ...(recordedDate ? { recordedDate } : {}),
+ ...(r.title ? { title: r.title } : {}),
+ });
+ }
+ const inWindow = dated.filter((v) => (!opts.from || v.date >= opts.from) && (!opts.to || v.date <= opts.to));
+ const summary = summarizeCoverage(inWindow, { gapDays: opts.gapDays });
+ return {
+ slug,
+ titlePattern: config.recordedDate?.titlePattern ?? null,
+ held: ids.length,
+ dated: dated.length,
+ byRecordedDate,
+ byUploadDate: dated.length - byRecordedDate,
+ undated,
+ ...summary,
+ ...(opts.list
+ ? { videos: [...inWindow].sort((a, b) => a.date.localeCompare(b.date) || a.id.localeCompare(b.id)) }
+ : {}),
+ };
+}
diff --git a/common/controller/normalizeAll.ts b/common/controller/normalizeAll.ts
@@ -24,6 +24,12 @@ export type NormalizeAllOptions = {
// says in its own header that it was modelled on this one, so this is the
// symmetry being completed rather than a new pattern.
channelSlugs?: string[];
+ // Restrict to these video ids within the channels walked (release 19 A7,
+ // `pnpm ops build-cues {slug, ids}`). An id the channel does not hold is
+ // passed over silently here; the action refuses it before any job.
+ videoIds?: string[];
+ // Rewrite cues.json even when it is fresh.
+ force?: boolean;
onLog?: (msg: string) => void;
signal?: AbortSignal;
concurrency?: number;
@@ -68,7 +74,10 @@ export async function normalizeAllTranscripts(
result.failed++;
continue;
}
- const videoIds = await readdir(dataDir).catch(() => [] as string[]);
+ const onlyIds = opts.videoIds ? new Set(opts.videoIds) : null;
+ const videoIds = (await readdir(dataDir).catch(() => [] as string[])).filter(
+ (id) => !onlyIds || onlyIds.has(id),
+ );
log(`Normalize ${ch.slug}: ${videoIds.length} videos`);
let wrote = 0;
let fresh = 0;
@@ -83,6 +92,7 @@ export async function normalizeAllTranscripts(
videoDir: path.join(dataDir, id),
channelSlug: ch.slug,
configName: ch.config.name,
+ ...(opts.force ? { force: true } : {}),
});
if (outcome.status === "wrote") wrote++;
else if (outcome.status === "fresh") fresh++;
diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts
@@ -86,9 +86,6 @@ export async function normalizeTranscript(
const picked = pickIndexTranscript(files);
if (!picked) return { status: "skipped", reason: "no-raw-transcript" };
- const metaPath = path.join(opts.videoDir, META_FILENAME);
- const transcriptPath = path.join(opts.videoDir, picked.filename);
-
// The same question every reader of cues.json asks (isCuesJsonFresh), so a
// normalize pass rewrites exactly the files they refuse.
const freshness = await isCuesJsonFresh(opts.videoDir);
@@ -97,6 +94,37 @@ export async function normalizeTranscript(
return { status: "fresh", cuesPath };
}
+ const built = await buildNormalizedTranscript(opts, files);
+ if (built.status === "skipped") return built;
+ const out = built.transcript;
+
+ // Compact, no trailing newline: transcript.cues.json's historical bytes.
+ await writeJsonAtomic(cuesPath, out, { indent: 0, newline: false });
+ opts.log?.(
+ `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${out.vttFile ?? out.transcriptFormat}, ${out.cues?.length ?? 0} cues)`,
+ );
+ return { status: "wrote", cuesPath };
+}
+
+// WHAT normalizeTranscript WOULD WRITE, built in memory and not written: the
+// summary from metadata.info.json and the cues of the transcript the index
+// would pick (the caption-track rule for VTTs). Shared by normalize and by a
+// reader that must not write (the ops transcript read, release 19 A7).
+export async function buildNormalizedTranscript(
+ opts: Pick<NormalizeOptions, "videoDir" | "channelSlug" | "configName" | "formatHint">,
+ known?: Awaited<ReturnType<typeof readVideoFiles>>,
+): Promise<
+ | { status: "built"; transcript: NormalizedTranscript }
+ | { status: "skipped"; reason: "no-raw-transcript" | "no-metadata" }
+> {
+ const files = known ?? (await readVideoFiles(opts.videoDir));
+ if (!files.hasMeta) return { status: "skipped", reason: "no-metadata" };
+ const picked = pickIndexTranscript(files);
+ if (!picked) return { status: "skipped", reason: "no-raw-transcript" };
+ const metaPath = path.join(opts.videoDir, META_FILENAME);
+ const transcriptPath = path.join(opts.videoDir, picked.filename);
+ const cuesPath = path.join(opts.videoDir, CUES_JSON_FILENAME);
+
const metaRaw = await readFile(metaPath, "utf8");
const parsedMeta = JSON.parse(metaRaw) as RawMetadata;
const summary = summarize(
@@ -135,23 +163,19 @@ export async function normalizeTranscript(
);
}
- const out: NormalizedTranscript = {
- version: CUES_FILE_VERSION,
- source: picked.kind,
- ...(vttFile !== undefined
- ? { vttFile, captionTrackRule: CAPTION_TRACK_RULE_VERSION }
- : {}),
- transcriptFormat,
- ...summary,
- cues,
+ return {
+ status: "built",
+ transcript: {
+ version: CUES_FILE_VERSION,
+ source: picked.kind,
+ ...(vttFile !== undefined
+ ? { vttFile, captionTrackRule: CAPTION_TRACK_RULE_VERSION }
+ : {}),
+ transcriptFormat,
+ ...summary,
+ cues,
+ },
};
-
- // Compact, no trailing newline: transcript.cues.json's historical bytes.
- await writeJsonAtomic(cuesPath, out, { indent: 0, newline: false });
- opts.log?.(
- `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${vttFile ?? transcriptFormat}, ${cues.length} cues)`,
- );
- return { status: "wrote", cuesPath };
}
// Read the format recorded in a prior transcript.cues.json, if any. This is the
diff --git a/common/controller/remoteListing.test.ts b/common/controller/remoteListing.test.ts
@@ -0,0 +1,87 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import type { ChannelConfig } from "../lib/channelConfig";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/remoteListing.test.ts
+//
+// The diff of a listing against the held ids, and one run with the listing
+// injected. Every id here is invented.
+
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "remote-listing-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+writeFileSync(process.env.SETTINGS_FILE, "{}\n");
+
+const { diffRemoteListing, heldVideoIds, isRemoteListingPlatform, runRemoteListing } = await import(
+ "./remoteListing"
+);
+const { getPaths } = await import("../lib/paths");
+
+test("only Odysee and BitChute are listed this way", () => {
+ assert.equal(isRemoteListingPlatform("odysee"), true);
+ assert.equal(isRemoteListingPlatform("bitchute"), true);
+ assert.equal(isRemoteListingPlatform("youtube"), false);
+ assert.equal(isRemoteListingPlatform(null), false);
+});
+
+test("the diff keeps listing order, collapses duplicates, and counts entries with no id", () => {
+ const d = diffRemoteListing(
+ [
+ "https://www.bitchute.com/video/AbC123/",
+ "https://www.bitchute.com/video/held01/",
+ "https://www.bitchute.com/channel/somebody/",
+ "https://www.bitchute.com/embed/AbC123/",
+ "https://odysee.com/@demo:0/a-video:7",
+ ],
+ new Set(["held01", "gone99"]),
+ );
+ assert.equal(d.listed, 5);
+ assert.equal(d.unparsed, 1);
+ assert.equal(d.held, 1);
+ assert.deepEqual(d.notHeld, [
+ { id: "AbC123", url: "https://www.bitchute.com/video/AbC123/" },
+ { id: "7", url: "https://odysee.com/@demo:0/a-video:7" },
+ ]);
+ assert.deepEqual(d.heldNotListed, ["gone99"]);
+});
+
+test("held ids are data/ directories; an absent data/ holds nothing", async () => {
+ assert.deepEqual([...(await heldVideoIds(path.join(ROOT, "nowhere")))], []);
+ const dir = path.join(ROOT, "h", "data");
+ mkdirSync(path.join(dir, "v1"), { recursive: true });
+ mkdirSync(path.join(dir, ".hidden"), { recursive: true });
+ writeFileSync(path.join(dir, "stray.txt"), "");
+ assert.deepEqual([...(await heldVideoIds(dir))], ["v1"]);
+});
+
+test("a run lists through the injected reader and answers the diff", async () => {
+ const slug = "demo-bitchute";
+ mkdirSync(path.join(ROOT, "channels", slug, "data", "held01"), { recursive: true });
+ const config = { handling: "transcribe", platform: "bitchute", url: "https://www.bitchute.com/channel/demo/" } as ChannelConfig;
+ writeFileSync(path.join(ROOT, "channels", slug, "config.json"), JSON.stringify(config));
+ const lines: string[] = [];
+ let asked = 0;
+ const r = await runRemoteListing({
+ paths: getPaths(),
+ slug,
+ channelConfig: config,
+ platform: "bitchute",
+ onLog: (l) => lines.push(l),
+ signal: new AbortController().signal,
+ deps: {
+ list: async () => {
+ asked++;
+ return ["https://www.bitchute.com/video/held01/", "https://www.bitchute.com/video/new001/"];
+ },
+ now: () => new Date("2026-10-10T00:00:00Z"),
+ },
+ });
+ assert.equal(asked, 1);
+ assert.equal(r.listedAt, "2026-10-10T00:00:00.000Z");
+ assert.deepEqual(r.notHeld, [{ id: "new001", url: "https://www.bitchute.com/video/new001/" }]);
+ assert.match(lines.join(""), /2 listed, 1 held, 1 not held, 0 held but not listed/);
+});
diff --git a/common/controller/remoteListing.ts b/common/controller/remoteListing.ts
@@ -0,0 +1,122 @@
+// AN ODYSEE OR BITCHUTE CHANNEL'S REMOTE LISTING, DIFFED AGAINST WHAT IS HELD
+// (release 19 A6, `pnpm ops get remote-listing <slug>`).
+//
+// One flat-playlist enumeration of the channel's own URL — the same one-spawn
+// read the quick availability check and a sync's full sweep make, with the
+// platform's paced args (`configArgs`) — on the platform's own queue, so it is
+// ONE STREAM with every other request to that platform. Nothing is written: no
+// roster merge, no maybe-missing, no download. The answer is the diff:
+//
+// notHeld listed upstream, no data/<id>/ here — what a download would get
+// heldNotListed held here, absent from the listing (deleted, unlisted, or a
+// record imported from elsewhere)
+//
+// POLITE: the editor refuses it while the platform is held or cooling down,
+// waits out the platform's import floor before the read (jobs/platformGap.ts),
+// and a rate-limited read backs the platform off (the caller records it).
+// Ids are the URL-derived canonical ids, which ARE the data-dir names on both
+// platforms (lib/videoId.ts).
+
+import path from "node:path";
+import type { ChannelConfig } from "../lib/channelConfig";
+import type { Paths } from "../lib/paths";
+import { assertChannelTextReadable } from "../lib/channelMedia";
+import { extractVideoId } from "../lib/videoId";
+import { fetchFlatPlaylistUrls } from "../ytdlp/runYtdlp";
+import { heldVideoIds } from "./videoCues";
+
+export { heldVideoIds };
+
+export const REMOTE_LISTING_PLATFORMS = ["odysee", "bitchute"] as const;
+export type RemoteListingPlatform = (typeof REMOTE_LISTING_PLATFORMS)[number];
+
+export function isRemoteListingPlatform(p: string | null | undefined): p is RemoteListingPlatform {
+ return (REMOTE_LISTING_PLATFORMS as readonly string[]).includes(p ?? "");
+}
+
+// The log line a remote-listing job ends with: this marker, then the result as
+// compact JSON (`pnpm ops get remote-listing` prints it on stdout). The same
+// string is REMOTE_LISTING_RESULT_MARKER in scripts/archilyzer-ops.mjs.
+export const REMOTE_LISTING_RESULT_MARKER = "@@remote-listing ";
+
+export type RemoteListingResult = {
+ slug: string;
+ platform: RemoteListingPlatform;
+ url: string;
+ listedAt: string;
+ // Entries the enumeration printed, and how many named no video id.
+ listed: number;
+ unparsed: number;
+ held: number;
+ notHeld: { id: string; url: string }[];
+ heldNotListed: string[];
+};
+
+// The pure half: a listing's URLs against the held ids. Listing order is kept
+// (newest first, as the platform lists); duplicates collapse.
+export function diffRemoteListing(
+ urls: string[],
+ heldIds: ReadonlySet<string>,
+): Pick<RemoteListingResult, "listed" | "unparsed" | "held" | "notHeld" | "heldNotListed"> {
+ const seen = new Set<string>();
+ const notHeld: { id: string; url: string }[] = [];
+ let unparsed = 0;
+ let held = 0;
+ for (const url of urls) {
+ const id = extractVideoId(url);
+ if (!id) {
+ unparsed++;
+ continue;
+ }
+ if (seen.has(id)) continue;
+ seen.add(id);
+ if (heldIds.has(id)) held++;
+ else notHeld.push({ id, url });
+ }
+ const heldNotListed = [...heldIds].filter((id) => !seen.has(id)).sort();
+ return { listed: urls.length, unparsed, held, notHeld, heldNotListed };
+}
+
+export type RemoteListingDeps = {
+ list?: typeof fetchFlatPlaylistUrls;
+ now?: () => Date;
+};
+
+export async function runRemoteListing(opts: {
+ paths: Paths;
+ slug: string;
+ channelConfig: ChannelConfig;
+ platform: RemoteListingPlatform;
+ onLog: (line: string) => void;
+ signal: AbortSignal;
+ deps?: RemoteListingDeps;
+}): Promise<RemoteListingResult> {
+ const list = opts.deps?.list ?? fetchFlatPlaylistUrls;
+ const now = opts.deps?.now ?? (() => new Date());
+ const url = opts.channelConfig.url;
+ if (!url) throw new Error(`Channel "${opts.slug}" has no URL configured`);
+ // The text guard: an unmounted or moving drive is not "nothing held".
+ await assertChannelTextReadable(opts.paths, opts.slug, opts.channelConfig);
+ const urls = await list({
+ channelConfig: opts.channelConfig,
+ paths: opts.paths,
+ channelSlug: opts.slug,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ });
+ const held = await heldVideoIds(path.join(opts.paths.channelsDir, opts.slug, "data"));
+ const diff = diffRemoteListing(urls, held);
+ const result: RemoteListingResult = {
+ slug: opts.slug,
+ platform: opts.platform,
+ url,
+ listedAt: now().toISOString(),
+ ...diff,
+ };
+ opts.onLog(
+ `Remote listing of ${opts.slug} (${opts.platform}): ${diff.listed} listed, ${diff.held} held, ` +
+ `${diff.notHeld.length} not held, ${diff.heldNotListed.length} held but not listed` +
+ `${diff.unparsed ? `, ${diff.unparsed} entries with no video id` : ""}.\n`,
+ );
+ return result;
+}
diff --git a/common/controller/videoCues.ts b/common/controller/videoCues.ts
@@ -0,0 +1,152 @@
+// ONE VIDEO'S CUES OFF DISK, WITHOUT AN INDEX (release 19 A7).
+//
+// A video imported or transcribed since the last index build — or one whose
+// transcript.cues.json was never written (a youtube-handling channel's
+// captions are only normalized by an explicit pass) — is invisible to every
+// reader of the published shards. This answers it from the video directory,
+// and writes nothing:
+//
+// cues.json transcript.cues.json, when it is fresh (isCuesJsonFresh — the
+// question every reader of it asks)
+// built what normalize WOULD write (buildNormalizedTranscript): the
+// metadata summary and the cues of the transcript the index
+// would pick, the caption-track rule for VTTs
+// vtt no metadata.info.json: the English VTT the rule picks, cues
+// only
+//
+// The ops transcript route serves it; the MCP's get_transcript falls back to it
+// through the editor when the archive has no such video.
+
+import path from "node:path";
+import { readdir, stat } from "node:fs/promises";
+import type { Paths } from "../lib/paths";
+import type { Cue } from "../lib/vtt";
+import { assertChannelTextReadable } from "../lib/channelMedia";
+import { readEnglishVttCues, readVideoFiles, CUES_JSON_FILENAME } from "../lib/videoStatus";
+import { readChannelConfig } from "./channels";
+import {
+ buildNormalizedTranscript,
+ isCuesJsonFresh,
+ readNormalizedTranscript,
+ type NormalizedTranscript,
+} from "./normalizeTranscript";
+
+export type VideoCues = {
+ slug: string;
+ id: string;
+ source: "cues.json" | "built" | "vtt";
+ // What transcript.cues.json says about itself: absent, stale, or fresh.
+ cuesJson: "fresh" | "stale" | "missing";
+ title?: string;
+ channel?: string;
+ uploadDate?: string;
+ duration?: number;
+ webpageUrl?: string;
+ platform?: string;
+ description?: string;
+ // The file the cues came from, when they came from a raw transcript.
+ file?: string;
+ cues: Cue[];
+};
+
+export type VideoCuesError = { error: string; status: 400 | 404 | 409 | 503 };
+
+const ONE_SEGMENT = (s: string) => s !== "." && s !== ".." && !/[/\\\0]/.test(s) && s.length > 0;
+
+// The channel's held ids: its data/<id>/ directories. An absent data/ is a
+// channel with nothing held; any other read error is thrown, never read as
+// "nothing held" (the text guard runs first).
+export async function heldVideoIds(dataDir: string): Promise<Set<string>> {
+ try {
+ const entries = await readdir(dataDir, { withFileTypes: true });
+ return new Set(entries.filter((d) => d.isDirectory() && !d.name.startsWith(".")).map((d) => d.name));
+ } catch (err) {
+ if ((err as NodeJS.ErrnoException).code === "ENOENT") return new Set();
+ throw err;
+ }
+}
+
+// The channels holding data/<id>/, for a read that was not told the channel.
+export async function channelsHoldingVideo(paths: Paths, id: string): Promise<string[]> {
+ if (!ONE_SEGMENT(id)) return [];
+ const out: string[] = [];
+ const entries = await readdir(paths.channelsDir, { withFileTypes: true }).catch(() => []);
+ for (const e of entries) {
+ if (!e.isDirectory() || e.name.startsWith(".")) continue;
+ const st = await stat(path.join(paths.channelsDir, e.name, "data", id)).catch(() => null);
+ if (st?.isDirectory()) out.push(e.name);
+ }
+ return out.sort();
+}
+
+function fromNormalized(
+ slug: string,
+ id: string,
+ t: NormalizedTranscript,
+ source: VideoCues["source"],
+ cuesJson: VideoCues["cuesJson"],
+): VideoCues {
+ const r = t as NormalizedTranscript & Record<string, unknown>;
+ const str = (v: unknown) => (typeof v === "string" && v ? v : undefined);
+ return {
+ slug,
+ id,
+ source,
+ cuesJson,
+ ...(str(r.title) ? { title: str(r.title) } : {}),
+ ...(str(r.channel) ? { channel: str(r.channel) } : {}),
+ ...(str(r.uploadDate) ? { uploadDate: str(r.uploadDate) } : {}),
+ ...(typeof r.duration === "number" ? { duration: r.duration } : {}),
+ ...(str(r.webpageUrl) ? { webpageUrl: str(r.webpageUrl) } : {}),
+ ...(str(r.platform) ? { platform: str(r.platform) } : {}),
+ ...(str(r.description) ? { description: str(r.description) } : {}),
+ ...(t.vttFile ? { file: t.vttFile } : t.source === "whisper" ? { file: "transcript.json" } : {}),
+ cues: t.cues ?? [],
+ };
+}
+
+export async function readVideoCues(
+ paths: Paths,
+ opts: { slug?: string; id: string },
+): Promise<VideoCues | VideoCuesError> {
+ const id = opts.id;
+ if (!ONE_SEGMENT(id)) return { error: `"${id}" is not a video id (one path segment)`, status: 400 };
+ let slug = opts.slug;
+ if (!slug) {
+ const holders = await channelsHoldingVideo(paths, id);
+ if (holders.length === 0) return { error: `no channel holds a video "${id}"`, status: 404 };
+ if (holders.length > 1) {
+ return { error: `${holders.length} channels hold a video "${id}" (${holders.join(", ")}) — name the channel`, status: 409 };
+ }
+ slug = holders[0];
+ }
+ const config = await readChannelConfig(paths, slug);
+ if (!config) return { error: `Channel "${slug}" not found`, status: 404 };
+ try {
+ await assertChannelTextReadable(paths, slug, config);
+ } catch (err) {
+ return { error: (err as Error).message, status: 503 };
+ }
+ const videoDir = path.join(paths.channelsDir, slug, "data", id);
+ const st = await stat(videoDir).catch(() => null);
+ if (!st?.isDirectory()) return { error: `Channel "${slug}" holds no video "${id}"`, status: 404 };
+
+ const files = await readVideoFiles(videoDir);
+ const freshness = await isCuesJsonFresh(videoDir);
+ const cuesJson: VideoCues["cuesJson"] = files.entries.includes(CUES_JSON_FILENAME)
+ ? freshness.fresh
+ ? "fresh"
+ : "stale"
+ : "missing";
+ if (cuesJson === "fresh") {
+ const t = await readNormalizedTranscript(freshness.cuesPath);
+ if (t) return fromNormalized(slug, id, t, "cues.json", cuesJson);
+ }
+ const built = await buildNormalizedTranscript({ videoDir, channelSlug: slug, configName: config.name }, files);
+ if (built.status === "built") return fromNormalized(slug, id, built.transcript, "built", cuesJson);
+ if (built.reason === "no-metadata") {
+ const read = await readEnglishVttCues(videoDir, files.entries);
+ if (read) return { slug, id, source: "vtt", cuesJson, file: read.filename, cues: read.cues };
+ }
+ return { error: `${slug}/${id} has no transcript to read (no transcript.json and no English VTT)`, status: 404 };
+}
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -293,6 +293,19 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
queueKeyStrategy: "platform",
needsMedia: true,
},
+ // AN ODYSEE OR BITCHUTE CHANNEL'S LISTING, diffed against what is held
+ // (release 19 A6, controller/remoteListing.ts): one flat-playlist read on the
+ // platform's queue, nothing written. Reads data/ (the text tier) for the
+ // held ids, so the text guard covers it.
+ "remote-listing": {
+ kind: "remote-listing",
+ label: "Remote listing",
+ drainable: false,
+ replayable: false,
+ queueKeyStrategy: "platform",
+ needsMedia: false,
+ needsText: true,
+ },
// ONE WINDOW of a video's source media, fetched into data/<id>/clips/ for a
// tool that asked for it by name (umtool's clip bench). On its platform's
// CLIP queue (`clips:<platform>`, lib/queueKeys.ts clipWindowQueueKey —
diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts
@@ -89,7 +89,7 @@ const DECLARED: EnvVarDecl[] = [
paths("STAGIT_BIN", "`stagit` on PATH, then `~/.local/bin/stagit`", "stagit, which renders the source's history pages (`/source/git/`: the log and a page per commit with its diff). Optional: without it the source is published without them. See [PUBLISH.md](PUBLISH.md)."),
// ── runtime ────────────────────────────────────────────────────────────
- { name: "WORKER_TOKEN", audience: "runtime", default: "unset (both surfaces off)", readBy: "common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts", doc: "Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`, the MCP's `fetch_clip`). Set the same value on both ends." },
+ { name: "WORKER_TOKEN", audience: "runtime", default: "unset (both surfaces off)", readBy: "common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, mcp/src/editorOps.ts", doc: "Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`; the MCP's `fetch_clip`, `enqueue`, `get_job`, `channel_coverage` and `get_transcript`'s editor fallback). Set the same value on both ends." },
{ name: "SYNC_HEARTBEAT_SECONDS", audience: "runtime", default: "`settings.syncScheduler.heartbeatSeconds`", readBy: "editor/app/scheduler/heartbeat.ts", doc: "Overrides the editor's in-process sync heartbeat. `0` = no internal timer (tick from cron instead)." },
{ name: "SYNC_TICK_URL", audience: "runtime", default: "`http://127.0.0.1:3001/api/scheduler/tick`", readBy: "common/bin/sync-tick.ts", doc: "Where `archilyzer sync tick` (cron's heartbeat) posts." },
{ name: "SYNC_TICK_TOKEN", audience: "runtime", default: "unset (no auth)", readBy: "common/bin/sync-tick.ts, editor/app/scheduler/auth.ts", doc: "Bearer token for the tick endpoint; set on both the editor and the cron job." },
@@ -118,7 +118,7 @@ const DECLARED: EnvVarDecl[] = [
{ name: "CLAUDE_DIGEST_MODEL", audience: "runtime", default: "the CLI's default", readBy: "common/lib/digestApps.ts", doc: "The model the metered digest lane asks `claude` for when settings name none." },
{ name: "NITTER_INSTANCES", audience: "runtime", default: "a built-in list", readBy: "common/social/xNitterFetcher.ts", doc: "Comma-separated Nitter instances for the X fallback fetcher, in order of preference." },
{ name: "ARCHILYZER_X_BROWSER", audience: "runtime", default: "the first of `chromium`, `google-chrome`, `google-chrome-stable`, `chrome` on PATH, else Playwright's bundled Chromium", readBy: "common/social/xBrowser.ts", doc: "The Chromium-family browser /settings' \"Connect X account\" opens (a path, or a name looked up on PATH), and a forum-thread channel's \"Connect forum session\" too. It is launched without the automation signals, in the X session profile (or the forum host's profile); a value that is not an executable refuses the connect rather than opening another browser." },
- { name: "UMTOOL_URL", audience: "runtime", default: "unset (no link)", readBy: "editor/app/channels/[slug]/videos/[id]/page.tsx", doc: "umtool's front door; when set, the video page links to it." },
+ { name: "UMTOOL_URL", audience: "runtime", default: "unset (no link; the MCP's `notes` off)", readBy: "editor/app/channels/[slug]/videos/[id]/page.tsx, mcp/src/archivalTools.ts", doc: "umtool's front door; when set, the video page links to it, and the MCP's `notes` tool reads the operator's notes there (e.g. `http://localhost:3050`)." },
{ name: "TRANSCRIPT_SITE_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: one published archive to read over HTTP." },
{ name: "TRANSCRIPT_HUB_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a hub, federating every archive it lists." },
{ name: "TRANSCRIPT_LOCAL_DIR", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a composed public dir on disk." },
@@ -129,7 +129,7 @@ const DECLARED: EnvVarDecl[] = [
{ name: "ARCHILYZER_INDEX_ALLOW_HELD", audience: "runtime", default: "off", readBy: "common/controller/buildIndex.ts", doc: "`1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel." },
{ name: "UV_THREADPOOL_SIZE", audience: "runtime", default: "`16` for the editor (`4` is Node's own)", readBy: "Node's libuv (set by editor/package.json `start` and docker/entrypoint.sh)", doc: "Threads in Node's pool for filesystem calls. A call on a stalled drive holds one until the drive answers, so the editor starts with 16. It buys time for calls already in flight and isolates nothing: the storage health probe and its gate keep new calls off a stalled drive." },
{ name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." },
- { name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool, common/bin/migrate-media-tier.ts", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to, and whose `/api/pulse` `archilyzer storage migrate-tier` asks before it refuses to run beside it." },
+ { name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, mcp/src/editorOps.ts, umtool, common/bin/migrate-media-tier.ts", doc: "Which editor `pnpm ops` and the MCP's editor-backed tools (`fetch_clip`, `enqueue`, `get_job`, `channel_coverage`, `get_transcript`'s fallback) talk to, and whose `/api/pulse` `archilyzer storage migrate-tier` asks before it refuses to run beside it." },
{ name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." },
{ name: "DIARIZE_ENGINE_KIND", audience: "runtime", default: "`sherpa-onnx`", readBy: "scripts/diarize.mjs", doc: "The diarization engine: `sherpa-onnx` or `sortformer`." },
{ name: "DIARIZE_ENGINE_CMD", audience: "runtime", default: "the bundled sherpa script", readBy: "scripts/diarize.mjs", doc: "The engine command the wrapper runs." },
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,12 @@
# Changelog
## [Unreleased]
+- **archive.org imports take many items, or a search.** `pnpm ops import-archive-org` now takes `"items"` (identifiers, or `{item, files | match}`) or `"query"` (an archive.org search, its first `"limit"` items) as one job, paced between files and between items. Records already held — on disk or in the saved-video store — are skipped; a dry run lists every file as held, RESTRICTED (archive.org will not hand it out) or would-get; three items in a row refused by archive.org stop the job and back the platform off.
+- **What an Odysee or BitChute channel lists that it does not hold.** `pnpm ops get remote-listing <slug>` reads the channel's listing once, on the platform's own queue at its pace, and prints the videos listed but not held and the ones held but no longer listed. Nothing is written.
+- **Cues without an index build.** `pnpm ops build-cues` writes `transcript.cues.json` for a channel (or chosen ids); `pnpm ops get transcript <id>` reads one video's cues off disk; the MCP's `get_transcript` reads a video the archive has not published yet through the editor, marked as such.
+- **Publish a private build elsewhere.** `archilyzer publish build <id> --out <dir>` lays the site's bundle out in a directory another server serves (hard links where it can); `--allow-missing-media` on it, and `"allowMissingMedia": true` on the ops publish `build` verb, let a report citation with no prepared media through.
+- **The MCP can ask the editor for archival work.** With the editor configured, `enqueue` queues a sync, download-missing, retry-bucket (chosen ids), transcribe-bucket, post fetch or video import; `get_job` follows it; `channel_coverage` shows what a channel holds by date and its gaps (a VOD mirror's videos by the day they were recorded); `notes` lists and reads the operator's umtool notes (`UMTOOL_URL`). Settings, storage and deletes stay `pnpm ops`. `pnpm ops get coverage <slug>` is the same coverage from a shell.
+- **Every `pnpm ops` action has its help paragraph** (twelve had none), so COMMANDS.md describes each.
- **A channel that mirrors another's streams can date its videos by the stream, not the upload.** A new channel setting, "Recorded date from the title (regex)" on the Configure form (`recordedDate.titlePattern` in config.json; `pnpm ops channel-config` with `recordedDateTitlePattern`), names where a title carries the recording's date — a regex with the groups `year`, `month` (a number or a month name) and `day`. The index then gives each of that channel's videos a `recordedDate`, which coverage reads before the upload date; a title without a date, or with one after the upload, keeps the upload date. Changing the pattern re-dates the channel's videos at the next index build.
- **A clip window of a video whose source is saved is cut from it, not fetched.** `fetch_clip`, `POST /api/media/fetch-window` and `pnpm ops fetch-windows` now cut a window out of the video's saved container (a persisted source, a full-source fetch, or media attached from a local archive) when it covers the seconds asked for, and answer at once as a cached window — no request to the platform, so a deleted channel's held videos are clippable. The window's sidecar records `source: "saved-video"`, and the video page marks it "cut from the saved video". A batch runs such windows as their own job on `clips:saved-video`, outside every platform's queue, hold and cooldown. A saved video whose file cannot be read (its drive unplugged, the file gone) is refused with the media guard's sentence rather than fetched; one that ends before the window is fetched as before.
- **`pnpm e2e` runs against a production build, and rebuilds it when the code changed.** The editor and umtool suites now run under `next start` by default (release 19's full editor suite: 24 min, against 71 under `next dev`). Before a run starts, the build's stamp — a fingerprint of the files the build reads, uncommitted edits included — is checked against the tree, and a stale build is rebuilt first through the heavy slot under a 5 GB cap, with the reason printed (`e2e build: rebuilding editor — common changed since the build …`). The test build has its own directory (`editor/.next/e2e`, `umtool/.next-e2e-start`), so it never replaces the build a running editor or umtool serves. `E2E_MODE=dev` runs `next dev` for iterating on one spec. The export and homepage suites stay on `next dev` (they are static exports). Every suite also writes `test-results/timings.json`, and `node scripts/e2e-timings.mjs` prints each spec file's time against the branch's last run. Four slow tests no longer wait on real clocks: the first rate-limit cooldown is 20 s and the clip-window gap 2 s on the test server only (`E2E_BACKOFF_BASE_MS`, `E2E_CLIP_WINDOW_GAP_MS`).
diff --git a/editor/app/api/ops/build-cues/route.test.ts b/editor/app/api/ops/build-cues/route.test.ts
@@ -0,0 +1,99 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, readFile, stat, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { setupOpsCorpus, callPost, callGet } from "../_testCorpus";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/api/ops/build-cues/route.test.ts"
+//
+// Cues without an index (release 19 A7): `build-cues` writes
+// transcript.cues.json for the named videos only, and `GET transcript` reads a
+// video's cues off disk — from a fresh cues.json, from what normalize would
+// write, or from the VTT alone — writing nothing.
+
+const corpus = await setupOpsCorpus(null);
+await corpus.writeChannelConfig("demo-yt", { url: "https://www.youtube.com/@demo" });
+await corpus.writeChannelConfig("other-yt", { url: "https://www.youtube.com/@other" });
+const DATA = path.join(corpus.transcripts, "channels", "demo-yt", "data");
+const VTT = (text: string) => `WEBVTT\n\n00:00:01.000 --> 00:00:03.000\n${text}\n\n00:00:04.000 --> 00:00:06.000\nsecond line\n`;
+async function video(dir: string, opts: { meta?: boolean; text: string }) {
+ await mkdir(dir, { recursive: true });
+ if (opts.meta) {
+ await writeFile(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({ id: path.basename(dir), title: `Title ${path.basename(dir)}`, upload_date: "20240102", duration: 6 }),
+ );
+ }
+ await writeFile(path.join(dir, "transcript.en.vtt"), VTT(opts.text));
+}
+await video(path.join(DATA, "vid00000001"), { meta: true, text: "first video" });
+await video(path.join(DATA, "vid00000002"), { meta: true, text: "second video" });
+await video(path.join(DATA, "bare0000001"), { text: "no metadata here" });
+await video(path.join(corpus.transcripts, "channels", "other-yt", "data", "vid00000002"), { meta: true, text: "a copy" });
+
+const { POST } = await import("./route");
+const { GET } = await import("../transcript/route");
+const { getRegistry } = await import("yt-dlp-transcript-common/jobs/registry");
+test.after(() => corpus.cleanup());
+const exists = (p: string) => stat(p).then(() => true, () => false);
+
+test("build-cues refusals: keys, ids, a stray id, a channel that does not exist", async () => {
+ const cases: [Record<string, unknown>, RegExp][] = [
+ [{ slug: "demo-yt", id: "x" }, /unknown key\(s\): id/],
+ [{ slug: "demo-yt", ids: [] }, /"ids" must be a non-empty array/],
+ [{ slug: "demo-yt", ids: ["../x"] }, /not a video id/],
+ [{ slug: "demo-yt", ids: ["vid00000001", "nope"] }, /1 of "ids" not held by demo-yt: nope/],
+ [{ slug: "no-such", ids: ["a"] }, /Channel "no-such" not found/],
+ [{ slug: "demo-yt", force: "yes" }, /"force" must be a boolean/],
+ ];
+ for (const [body, re] of cases) {
+ const r = await callPost(POST, body);
+ assert.equal(r.status, 400, JSON.stringify(body));
+ assert.match(r.body.error ?? "", re, JSON.stringify(body));
+ }
+});
+
+test("transcript reads cues with no index: built from the VTT + metadata, or the VTT alone; nothing written", async () => {
+ const url = (q: string) => `http://localhost/api/ops/transcript?${q}`;
+ const built = await callGet(GET, url("id=vid00000001&slug=demo-yt"));
+ assert.equal(built.status, 200, JSON.stringify(built.body));
+ assert.equal(built.body.source, "built");
+ assert.equal(built.body.cuesJson, "missing");
+ assert.equal(built.body.title, "Title vid00000001");
+ assert.equal(built.body.file, "transcript.en.vtt");
+ assert.deepEqual((built.body.cues as { text: string }[]).map((c) => c.text), ["first video", "second line"]);
+ assert.equal(await exists(path.join(DATA, "vid00000001", "transcript.cues.json")), false);
+
+ const bare = await callGet(GET, url("id=bare0000001"));
+ assert.equal(bare.status, 200, JSON.stringify(bare.body));
+ assert.equal(bare.body.slug, "demo-yt");
+ assert.equal(bare.body.source, "vtt");
+ assert.equal((bare.body.cues as unknown[]).length, 2);
+
+ const two = await callGet(GET, url("id=vid00000002"));
+ assert.equal(two.status, 409);
+ assert.match(two.body.error ?? "", /2 channels hold a video "vid00000002" \(demo-yt, other-yt\) — name the channel/);
+ assert.equal((await callGet(GET, url("id=nothing"))).status, 404);
+ assert.match((await callGet(GET, url("slug=demo-yt"))).body.error ?? "", /"id" is required/);
+ assert.match((await callGet(GET, url("id=x&track=en"))).body.error ?? "", /unknown query key\(s\): track/);
+ assert.equal((await callGet(GET, url("id=x"), {}, {})).status, 401);
+});
+
+test("build-cues writes cues.json for the named ids only; transcript then reads it as fresh", async () => {
+ const r = await callPost(POST, { slug: "demo-yt", ids: ["vid00000001"] });
+ assert.equal(r.status, 200, JSON.stringify(r.body));
+ const jobId = r.body.jobId as string;
+ for (let i = 0; i < 200; i++) {
+ const s = getRegistry().get(jobId)?.status;
+ if (s === "done" || s === "failed") break;
+ await new Promise((res) => setTimeout(res, 50));
+ }
+ assert.equal(getRegistry().get(jobId)?.status, "done");
+ const cues = JSON.parse(await readFile(path.join(DATA, "vid00000001", "transcript.cues.json"), "utf8"));
+ assert.equal(cues.cues.length, 2);
+ assert.equal(await exists(path.join(DATA, "vid00000002", "transcript.cues.json")), false);
+ const fresh = await callGet(GET, "http://localhost/api/ops/transcript?id=vid00000001&slug=demo-yt");
+ assert.equal(fresh.body.source, "cues.json");
+ assert.equal(fresh.body.cuesJson, "fresh");
+});
diff --git a/editor/app/api/ops/build-cues/route.ts b/editor/app/api/ops/build-cues/route.ts
@@ -0,0 +1,32 @@
+import { normalizeChannelAction } from "../../../channels/[slug]/normalizeActions";
+import { OpsInputError, jobResponse, ops, optBool, optString, reqSlug, reqVideoId, type OpsBody } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, ids?: string[], force?, queueKey? } -> { ok: true, jobId }
+//
+// Write each video's transcript.cues.json from its raw transcript (the
+// caption-track rule for VTTs), as the digest card's Normalize button does
+// (normalizeActions.ts, controller/normalizeAll.ts) — the file every reader
+// without an index build wants. `ids` narrows it to those videos, every one
+// held by the channel; `force` rewrites a cues.json that is already fresh. On
+// the channel's own queue.
+function optIds(body: OpsBody): string[] | undefined {
+ const v = body.ids;
+ if (v === undefined) return undefined;
+ if (!Array.isArray(v) || v.length === 0) {
+ throw new OpsInputError('"ids" must be a non-empty array of video ids');
+ }
+ return [...new Set(v.map((id, i) => reqVideoId({ [`ids[${i}]`]: id }, `ids[${i}]`)))];
+}
+
+export async function POST(request: Request) {
+ return ops(request, ["slug", "ids", "force", "queueKey"], async (body) =>
+ jobResponse(
+ await normalizeChannelAction(reqSlug(body, "slug"), optString(body, "queueKey"), {
+ ids: optIds(body),
+ force: optBool(body, "force") ?? false,
+ }),
+ ),
+ );
+}
diff --git a/editor/app/api/ops/coverage/route.test.ts b/editor/app/api/ops/coverage/route.test.ts
@@ -0,0 +1,49 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { setupOpsCorpus, callGet } from "../_testCorpus";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/api/ops/coverage/route.test.ts"
+//
+// GET /api/ops/coverage (release 19 A9, the MCP's channel_coverage): the query
+// refusals and one channel's coverage off a temp corpus. The dating rules are
+// common/controller/channelCoverage.test.ts.
+
+const corpus = await setupOpsCorpus(null);
+await corpus.writeChannelConfig("demo-yt", { url: "https://www.youtube.com/@demo" });
+for (const [id, date] of [["v1", "20240101"], ["v2", "20240301"]]) {
+ const dir = path.join(corpus.transcripts, "channels", "demo-yt", "data", id);
+ await mkdir(dir, { recursive: true });
+ await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify({ id, title: id, upload_date: date }));
+}
+const { GET } = await import("./route");
+test.after(() => corpus.cleanup());
+const get = (q: string) => callGet(GET, `http://localhost/api/ops/coverage?${q}`);
+
+test("coverage refusals: slug, gapDays, dates, list, unknown keys, the token", async () => {
+ const cases: [string, RegExp][] = [
+ ["", /"slug" is required/],
+ ["slug=../x", /not a valid channel slug/],
+ ["slug=demo-yt&gapDays=0", /"gapDays" must be a whole number/],
+ ["slug=demo-yt&from=2024-01-01", /"from" must be a date YYYYMMDD/],
+ ["slug=demo-yt&list=yes", /"list" is 1 or absent/],
+ ["slug=demo-yt&channel=x", /unknown query key\(s\): channel/],
+ ];
+ for (const [q, re] of cases) {
+ const r = await get(q);
+ assert.equal(r.status, 400, q);
+ assert.match(r.body.error ?? "", re, q);
+ }
+ assert.equal((await get("slug=no-such")).status, 404);
+ assert.equal((await callGet(GET, "http://localhost/api/ops/coverage?slug=demo-yt", {}, {})).status, 401);
+});
+
+test("coverage answers the held videos by date and the gaps", async () => {
+ const r = await get("slug=demo-yt&gapDays=30&list=1");
+ assert.equal(r.status, 200, JSON.stringify(r.body));
+ assert.equal(r.body.held, 2);
+ assert.deepEqual(r.body.gaps, [{ after: "20240101", before: "20240301", days: 60 }]);
+ assert.equal((r.body.videos as unknown[]).length, 2);
+});
diff --git a/editor/app/api/ops/coverage/route.ts b/editor/app/api/ops/coverage/route.ts
@@ -0,0 +1,53 @@
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { isValidChannelSlug } from "yt-dlp-transcript-common/controller/channels";
+import { channelCoverage } from "yt-dlp-transcript-common/controller/channelCoverage";
+import { readRoute } from "../_read";
+import { opsFail } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// GET ?slug=<channel>[&gapDays=<n>][&from=YYYYMMDD][&to=YYYYMMDD][&list=1]
+// -> { ok, slug, titlePattern, held, dated, byRecordedDate, byUploadDate,
+// undated: [id], first, last, byYear, byMonth, gapDays,
+// gaps: [{ after, before, days }], videos?: [{ id, date, uploadDate,
+// recordedDate?, title? }] }
+//
+// What a channel holds, by date, and the spans with nothing held
+// (controller/channelCoverage.ts): each held video dated with coverageDate —
+// its recorded date when the channel has a title rule, else its upload date —
+// off disk, so a video imported an hour ago counts. `gapDays` (30) is the
+// shortest span reported as a gap; `from`/`to` narrow the window; `list`
+// adds every dated video. Nothing is written. The MCP's channel_coverage.
+const DAY = /^\d{8}$/;
+
+export async function GET(request: Request) {
+ return readRoute(request, ["slug", "gapDays", "from", "to", "list"], async (q) => {
+ const slug = q.get("slug")?.trim() ?? "";
+ if (!slug) return opsFail('"slug" is required');
+ if (!isValidChannelSlug(slug)) return opsFail(`"${slug}" is not a valid channel slug`);
+ const rawGap = q.get("gapDays");
+ const gapDays = rawGap === null ? undefined : Number(rawGap);
+ if (gapDays !== undefined && (!Number.isInteger(gapDays) || gapDays < 1)) {
+ return opsFail('"gapDays" must be a whole number of days above zero');
+ }
+ const from = q.get("from") ?? undefined;
+ const to = q.get("to") ?? undefined;
+ for (const [k, v] of [["from", from], ["to", to]] as const) {
+ if (v !== undefined && !DAY.test(v)) return opsFail(`"${k}" must be a date YYYYMMDD`);
+ }
+ const list = q.get("list");
+ if (list !== null && list !== "1" && list !== "true") return opsFail('"list" is 1 or absent');
+ try {
+ const coverage = await channelCoverage(getPaths(), slug, {
+ ...(gapDays !== undefined ? { gapDays } : {}),
+ ...(from ? { from } : {}),
+ ...(to ? { to } : {}),
+ list: list !== null,
+ });
+ return { ok: true, ...coverage };
+ } catch (e) {
+ const message = (e as Error).message;
+ return opsFail(message, /not found/.test(message) ? 404 : 503);
+ }
+ });
+}
diff --git a/editor/app/api/ops/import-archive-org/route.test.ts b/editor/app/api/ops/import-archive-org/route.test.ts
@@ -0,0 +1,52 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { setupOpsCorpus, callPost } from "../_testCorpus";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/api/ops/import-archive-org/route.test.ts"
+//
+// The three body shapes (item / items / query) and every refusal that comes
+// before a job — none of them asks archive.org anything, so no network is
+// touched and no job is queued. The batch itself (held, restricted, the
+// inter-item gap, the 401 storm) is common/controller/archiveOrgImport.test.ts.
+
+const corpus = await setupOpsCorpus(null);
+await corpus.writeChannelConfig("demo-archive", { handling: "transcribe", platform: "archiveorg" });
+const { POST } = await import("./route");
+test.after(() => corpus.cleanup());
+
+test("refusals before any job: keys, modes, items, limit, regex, identifiers, channel", async () => {
+ const many = Array.from({ length: 501 }, (_, i) => `item-${i}`);
+ const cases: [Record<string, unknown>, RegExp][] = [
+ [{ slug: "demo-archive", items: ["a"], force: true }, /unknown key\(s\): force/],
+ [{ slug: "demo-archive" }, /exactly one of "item", "items" \(a list\) or "query"/],
+ [{ slug: "demo-archive", item: "a", query: "x" }, /exactly one of "item", "items"/],
+ [{ slug: "demo-archive", items: ["a"], query: "x" }, /exactly one of "item", "items"/],
+ [{ slug: "demo-archive", item: "a" }, /exactly one of "files" \(a list\) or "match"/],
+ [{ slug: "demo-archive", item: "a", files: ["x.mp4"], match: "." }, /exactly one of "files"/],
+ [{ slug: "demo-archive", item: "../a", match: "." }, /not an archive.org identifier/],
+ [{ slug: "demo-archive", items: [] }, /"items" must be a non-empty array/],
+ [{ slug: "demo-archive", items: many }, /holds 501 items; at most 500/],
+ [{ slug: "demo-archive", items: [{ item: "a", size: 1 }] }, /"items\[0\]" has unknown key\(s\): size/],
+ [{ slug: "demo-archive", items: [{ files: ["x.mp4"] }] }, /"items\[0\].item" is required/],
+ [{ slug: "demo-archive", items: [{ item: "a", files: [] }] }, /"items\[0\].files" must be a non-empty array/],
+ [{ slug: "demo-archive", items: [{ item: "a", files: ["x"], match: "." }] }, /item "a": "files" or "match", not both/],
+ [{ slug: "demo-archive", items: ["ok-1", "bad id"] }, /"bad id" is not an archive.org identifier/],
+ [{ slug: "demo-archive", items: [{ item: "a", match: "(" }] }, /item "a": "match" is not a valid regex/],
+ [{ slug: "demo-archive", items: ["a"], files: ["x.mp4"] }, /with "items", give each its own/],
+ [{ slug: "demo-archive", items: ["a"], limit: 5 }, /"limit" caps a "query"/],
+ [{ slug: "demo-archive", query: "x", limit: 501 }, /"limit" is at most 500/],
+ [{ slug: "demo-archive", query: "x", limit: 0 }, /"limit" must be a whole number above zero/],
+ [{ slug: "demo-archive", query: " " }, /"query" must be a non-empty string/],
+ [{ slug: "demo-archive", query: "x", match: "(" }, /"match" is not a valid regex/],
+ [{ slug: "demo-archive", query: "x", dryRun: "yes" }, /"dryRun" must be a boolean/],
+ [{ slug: "no-such", items: ["a"] }, /Channel "no-such" not found/],
+ ];
+ for (const [body, re] of cases) {
+ const r = await callPost(POST, body);
+ assert.equal(r.status, 400, JSON.stringify(body));
+ assert.match(r.body.error ?? "", re, JSON.stringify(body));
+ assert.equal(r.body.jobId, undefined);
+ }
+ assert.deepEqual(await corpus.listJobIds(), []);
+});
diff --git a/editor/app/api/ops/import-archive-org/route.ts b/editor/app/api/ops/import-archive-org/route.ts
@@ -1,44 +1,102 @@
-import { importArchiveOrgAction } from "../../../channels/[slug]/pipelineActions";
+import {
+ importArchiveOrgAction,
+ type ArchiveOrgImportEntry,
+} from "../../../channels/[slug]/pipelineActions";
import {
OpsInputError,
jobResponse,
ops,
optBool,
+ optPositiveInt,
optString,
reqSlug,
- reqString,
type OpsBody,
} from "../_lib";
export const dynamic = "force-dynamic";
// POST { slug, item, files?: string[], match?: string, dryRun? }
+// | { slug, items: (string | { item, files? | match? })[], match?, dryRun? }
+// | { slug, query: string, limit?: number, match?, dryRun? }
// -> { ok: true, jobId }
//
-// Import chosen media files of ONE archive.org item into an existing channel,
-// as one job on archive.org's own queue: one file at a time, a jittered pause
-// between them, files already downloaded skipped, a rate limit or three
-// failures in a row ending it (controller/archiveOrgImport.ts). Exactly one of
-// `files` (exact paths in the item, untrimmed) and `match` (a case-insensitive
-// regex over them). `dryRun` logs what would be fetched and fetches nothing.
-function optFiles(body: OpsBody): string[] | undefined {
- const v = body.files;
+// Import media files from archive.org into an existing channel, as ONE job on
+// archive.org's own queue: one file at a time, a jittered pause between them
+// and between items, records already held (on disk or in the saved-video
+// store) skipped, a rate limit or three failures in a row ending it, three
+// items in a row refused with 401/403 ending it too (controller/
+// archiveOrgImport.ts). `item` is one item and needs exactly one of `files`
+// (exact paths in the item, untrimmed) and `match` (a case-insensitive regex
+// over them); `items` is many (a bare identifier takes `match`, else every
+// media original); `query` is an archive.org search, its first `limit` (100,
+// at most 500) items. `dryRun` logs each file as held, RESTRICTED (archive.org
+// marks it not for download — a fetch would answer 401/403) or would-get, and
+// fetches nothing. The log ends with `summary: {…}`.
+const MAX_ARCHIVE_ORG_ITEMS = 500;
+
+function optFiles(v: unknown, key: string): string[] | undefined {
if (v === undefined) return undefined;
if (!Array.isArray(v) || v.length === 0 || v.some((s) => typeof s !== "string" || !s)) {
- throw new OpsInputError('"files" must be a non-empty array of file names');
+ throw new OpsInputError(`"${key}" must be a non-empty array of file names`);
}
return v as string[];
}
+function optItems(body: OpsBody): ArchiveOrgImportEntry[] | undefined {
+ const v = body.items;
+ if (v === undefined) return undefined;
+ if (!Array.isArray(v) || v.length === 0) {
+ throw new OpsInputError('"items" must be a non-empty array of identifiers or { "item", "files"? | "match"? }');
+ }
+ if (v.length > MAX_ARCHIVE_ORG_ITEMS) {
+ throw new OpsInputError(`"items" holds ${v.length} items; at most ${MAX_ARCHIVE_ORG_ITEMS} per job`);
+ }
+ return v.map((entry, i) => {
+ if (typeof entry === "string" && entry.trim()) return entry.trim();
+ if (typeof entry !== "object" || entry === null || Array.isArray(entry)) {
+ throw new OpsInputError(`"items[${i}]" must be an identifier or { "item", "files"? | "match"? }`);
+ }
+ const e = entry as Record<string, unknown>;
+ const stray = Object.keys(e).filter((k) => !["item", "files", "match"].includes(k));
+ if (stray.length) {
+ throw new OpsInputError(`"items[${i}]" has unknown key(s): ${stray.join(", ")} — accepted: item, files, match`);
+ }
+ if (typeof e.item !== "string" || !e.item.trim()) {
+ throw new OpsInputError(`"items[${i}].item" is required and must be a non-empty string`);
+ }
+ if (e.match !== undefined && typeof e.match !== "string") {
+ throw new OpsInputError(`"items[${i}].match" must be a string`);
+ }
+ return {
+ item: e.item.trim(),
+ ...(e.files !== undefined ? { files: optFiles(e.files, `items[${i}].files`) } : {}),
+ ...(e.match !== undefined ? { match: e.match as string } : {}),
+ };
+ });
+}
+
export async function POST(request: Request) {
- return ops(request, ["slug", "item", "files", "match", "dryRun"], async (body) =>
- jobResponse(
- await importArchiveOrgAction(reqSlug(body, "slug"), {
- item: reqString(body, "item"),
- files: optFiles(body),
- match: optString(body, "match"),
- dryRun: optBool(body, "dryRun"),
- }),
- ),
+ return ops(
+ request,
+ ["slug", "item", "items", "query", "limit", "files", "match", "dryRun"],
+ async (body) => {
+ const query = optString(body, "query");
+ if (query !== undefined && !query.trim()) throw new OpsInputError('"query" must be a non-empty string');
+ const limit = optPositiveInt(body, "limit");
+ if (limit !== undefined && limit > 500) throw new OpsInputError('"limit" is at most 500');
+ const item = optString(body, "item");
+ if (item !== undefined && !item.trim()) throw new OpsInputError('"item" must be a non-empty string');
+ return jobResponse(
+ await importArchiveOrgAction(reqSlug(body, "slug"), {
+ item: item?.trim(),
+ items: optItems(body),
+ query: query?.trim(),
+ limit,
+ files: optFiles(body.files, "files"),
+ match: optString(body, "match"),
+ dryRun: optBool(body, "dryRun"),
+ }),
+ );
+ },
);
}
diff --git a/editor/app/api/ops/publish/route.test.ts b/editor/app/api/ops/publish/route.test.ts
@@ -66,3 +66,30 @@ test("GET publish is the publish status: the index, the lane, a row per target",
const anon = await callGet(GET as unknown as Parameters<typeof callGet>[0], undefined, {}, {});
assert.equal(anon.status, 401);
});
+
+test("build takes allowMissingMedia (a local build's), carried onto the queued stage", async () => {
+ const docker = await callPost(POST, { verb: "build", runner: "docker", allowMissingMedia: true });
+ assert.equal(docker.status, 400);
+ assert.match(String(docker.body.error), /"allowMissingMedia" is a local build's/);
+ const notBool = await callPost(POST, { verb: "build", siteId: "pubsite", allowMissingMedia: "yes" });
+ assert.equal(notBool.status, 400);
+ assert.match(String(notBool.body.error), /"allowMissingMedia" must be a boolean/);
+ const deploy = await callPost(POST, { verb: "deploy", siteId: "pubsite", allowMissingMedia: true });
+ assert.equal(deploy.status, 400);
+ assert.match(String(deploy.body.error), /verb "deploy" takes .* — not allowMissingMedia/);
+
+ const before = new Set(await corpus.listJobIds());
+ const queued = await callPost(POST, { verb: "build", siteId: "pubsite", allowMissingMedia: true });
+ assert.equal(queued.status, 200, JSON.stringify(queued.body));
+ const jobs = queued.body.jobs as { kind: string; target: string; jobId: string }[];
+ const build = jobs.find((j) => j.target === "pubsite");
+ assert.ok(build, JSON.stringify(jobs));
+ const { readFile } = await import("node:fs/promises");
+ const path = await import("node:path");
+ const fresh = (await corpus.listJobIds()).filter((f) => !before.has(f));
+ const metas = await Promise.all(
+ fresh.map((f) => readFile(path.join(corpus.transcripts, ".jobs", f), "utf8")),
+ );
+ const mine = metas.find((m) => m.includes(build!.jobId)) ?? metas.join("\n");
+ assert.match(mine, /allowMissingMedia|allow-missing-media/, mine);
+});
diff --git a/editor/app/api/ops/publish/route.ts b/editor/app/api/ops/publish/route.ts
@@ -23,11 +23,14 @@ type Verb = (typeof VERBS)[number];
// POST { verb: index | build | deploy | hub | homepage | now | stale,
// siteId | siteIds?, preview?, to?: "pages" | "local", force?,
-// runner?: "local" | "docker", deploy?, skipArchives? }
+// runner?: "local" | "docker", deploy?, skipArchives?,
+// allowMissingMedia? }
//
// index the index update (a no-op when nothing changed)
// build each named site's build, forced — the index first when stale;
-// `runner: "docker"` builds every site in containers (host only)
+// `runner: "docker"` builds every site in containers (host only);
+// `allowMissingMedia: true` lets a report citation with no
+// prepared media through compose (local builds only)
// deploy each named site's built bundle: production, `preview`, or
// `to: "local"` (ARCHILYZER_SITE_OUT); `force` redeploys a
// bundle already shipped there
@@ -44,7 +47,18 @@ type Verb = (typeof VERBS)[number];
//
// GET → the publish status (`readPublishStatus`): the index, the lane, a row
// per target with its chips, and the plan Publish now would run.
-const KEYS = ["verb", "siteId", "siteIds", "preview", "to", "force", "runner", "deploy", "skipArchives"];
+const KEYS = [
+ "verb",
+ "siteId",
+ "siteIds",
+ "preview",
+ "to",
+ "force",
+ "runner",
+ "deploy",
+ "skipArchives",
+ "allowMissingMedia",
+];
function only(body: OpsBody, verb: Verb, allowed: string[]): void {
const extra = Object.keys(body).filter((k) => k !== "verb" && !allowed.includes(k));
@@ -75,9 +89,13 @@ export async function POST(request: Request) {
return publishResponse(await enqueuePlan(plan));
}
case "build": {
- only(body, verb, ["siteId", "siteIds", "force", "runner", "skipArchives"]);
+ only(body, verb, ["siteId", "siteIds", "force", "runner", "skipArchives", "allowMissingMedia"]);
const runner = body.runner === undefined ? "local" : oneOf(body, "runner", ["local", "docker"] as const);
+ const allowMissingMedia = optBool(body, "allowMissingMedia");
if (runner === "docker") {
+ if (allowMissingMedia !== undefined) {
+ throw new OpsInputError('"allowMissingMedia" is a local build\'s — the docker runner does not take it');
+ }
if (body.siteId !== undefined || body.siteIds !== undefined) {
throw new OpsInputError('"runner": "docker" builds every site — send no "siteId"/"siteIds"');
}
@@ -109,7 +127,12 @@ export async function POST(request: Request) {
}
const ids = reqSiteIds(body);
return publishResponse(
- await enqueueAsks(ids.map((target) => ({ target, build: { force: force !== false, skipArchives } }))),
+ await enqueueAsks(
+ ids.map((target) => ({
+ target,
+ build: { force: force !== false, skipArchives, ...(allowMissingMedia ? { allowMissingMedia: true } : {}) },
+ })),
+ ),
);
}
case "deploy": {
diff --git a/editor/app/api/ops/remote-listing/route.test.ts b/editor/app/api/ops/remote-listing/route.test.ts
@@ -0,0 +1,74 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, readFile } from "node:fs/promises";
+import path from "node:path";
+import { setupOpsCorpus, callPost } from "../_testCorpus";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/api/ops/remote-listing/route.test.ts"
+//
+// The refusals before any job, and one listing run end to end against the e2e
+// fake yt-dlp (it lists fake00000001…05 for any channel URL): the job's last
+// line carries the diff, and nothing is written into the channel.
+
+const corpus = await setupOpsCorpus(null);
+await corpus.writeChannelConfig("demo-odysee", {
+ handling: "transcribe",
+ platform: "odysee",
+ url: "https://odysee.com/@demo:0",
+});
+await corpus.writeChannelConfig("demo-yt", { url: "https://www.youtube.com/@demo" });
+await corpus.writeChannelConfig("demo-nourl", {});
+const DATA = path.join(corpus.transcripts, "channels", "demo-odysee", "data");
+// Held: one id the fake lists, one it does not.
+await mkdir(path.join(DATA, "fake00000002"), { recursive: true });
+await mkdir(path.join(DATA, "imported-elsewhere"), { recursive: true });
+const { POST } = await import("./route");
+const { getRegistry } = await import("yt-dlp-transcript-common/jobs/registry");
+const { getPaths } = await import("yt-dlp-transcript-common/lib/paths");
+test.after(() => corpus.cleanup());
+
+test("refusals before any job: keys, slug, channel, platform, url", async () => {
+ const cases: [Record<string, unknown>, RegExp][] = [
+ [{}, /"slug" is required/],
+ [{ slug: "demo-odysee", full: true }, /unknown key\(s\): full/],
+ [{ slug: "../x" }, /not a valid channel slug/],
+ [{ slug: "no-such" }, /Channel "no-such" not found/],
+ [{ slug: "demo-yt" }, /reads an Odysee or BitChute channel; "demo-yt" is on youtube/],
+ [{ slug: "demo-nourl" }, /no `url` configured/],
+ ];
+ for (const [body, re] of cases) {
+ const r = await callPost(POST, body);
+ assert.equal(r.status, 400, JSON.stringify(body));
+ assert.match(r.body.error ?? "", re, JSON.stringify(body));
+ }
+ assert.deepEqual(await corpus.listJobIds(), []);
+});
+
+test("a listing is a job whose last line is the diff against what is held", async () => {
+ const r = await callPost(POST, { slug: "demo-odysee" });
+ assert.equal(r.status, 200, JSON.stringify(r.body));
+ const jobId = r.body.jobId as string;
+ for (let i = 0; i < 200; i++) {
+ const s = getRegistry().get(jobId)?.status;
+ if (s === "done" || s === "failed") break;
+ await new Promise((res) => setTimeout(res, 50));
+ }
+ const log = await readFile(path.join(getPaths().jobsDir, `${jobId}.log`), "utf8");
+ assert.equal(getRegistry().get(jobId)?.status, "done", log);
+ const line = log.split("\n").find((l) => l.startsWith("@@remote-listing "));
+ assert.ok(line, log);
+ const result = JSON.parse(line!.slice("@@remote-listing ".length));
+ assert.equal(result.platform, "odysee");
+ assert.equal(result.listed, 5);
+ assert.equal(result.held, 1);
+ assert.deepEqual(
+ result.notHeld.map((n: { id: string }) => n.id),
+ ["fake00000001", "fake00000003", "fake00000004", "fake00000005"],
+ );
+ assert.deepEqual(result.heldNotListed, ["imported-elsewhere"]);
+ // Read-only: no roster, no playlist, no maybe-missing written.
+ const { readdir } = await import("node:fs/promises");
+ const top = await readdir(path.join(corpus.transcripts, "channels", "demo-odysee"));
+ assert.deepEqual(top.filter((f) => !["config.json", "data", "fake-ytdlp.invocations"].includes(f)), []);
+});
diff --git a/editor/app/api/ops/remote-listing/route.ts b/editor/app/api/ops/remote-listing/route.ts
@@ -0,0 +1,16 @@
+import { remoteListingAction } from "../../../channels/[slug]/pipelineActions";
+import { jobResponse, ops, reqSlug } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug } -> { ok: true, jobId }
+//
+// An Odysee or BitChute channel's listing, diffed against what it holds
+// (controller/remoteListing.ts): one flat-playlist read on the platform's own
+// queue, paced, nothing written. The job's log ends with
+// `@@remote-listing {slug, platform, url, listedAt, listed, unparsed, held,
+// notHeld: [{id, url}], heldNotListed: [id]}`; `pnpm ops get remote-listing
+// <slug>` waits for it and prints that JSON on stdout.
+export async function POST(request: Request) {
+ return ops(request, ["slug"], async (body) => jobResponse(await remoteListingAction(reqSlug(body, "slug"))));
+}
diff --git a/editor/app/api/ops/transcript/route.ts b/editor/app/api/ops/transcript/route.ts
@@ -0,0 +1,33 @@
+import { NextResponse } from "next/server";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { isValidChannelSlug } from "yt-dlp-transcript-common/controller/channels";
+import { readVideoCues } from "yt-dlp-transcript-common/controller/videoCues";
+import { readRoute } from "../_read";
+import { opsFail } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// GET ?id=<videoId>[&slug=<channel>]
+// -> { ok, slug, id, source: "cues.json" | "built" | "vtt",
+// cuesJson: "fresh" | "stale" | "missing", title?, channel?,
+// uploadDate?, duration?, webpageUrl?, platform?, description?, file?,
+// cues: [{ start, end, text }] }
+//
+// One video's cues off disk, with no index (controller/videoCues.ts): a fresh
+// transcript.cues.json, else what normalize would write, else the English VTT
+// alone. Nothing is written. Without `slug` the channel holding data/<id>/ is
+// found (two holders → 409, name it). The MCP's get_transcript asks this when
+// the archive it reads has no such video.
+export async function GET(request: Request) {
+ return readRoute(request, ["id", "slug"], async (q) => {
+ const id = q.get("id")?.trim() ?? "";
+ if (!id) return opsFail('"id" is required');
+ const slug = q.get("slug")?.trim() || undefined;
+ if (slug !== undefined && !isValidChannelSlug(slug)) {
+ return opsFail(`"${slug}" is not a valid channel slug`);
+ }
+ const read = await readVideoCues(getPaths(), { id, ...(slug ? { slug } : {}) });
+ if ("error" in read) return opsFail(read.error, read.status);
+ return NextResponse.json({ ok: true, ...read });
+ });
+}
diff --git a/editor/app/channels/[slug]/normalizeActions.ts b/editor/app/channels/[slug]/normalizeActions.ts
@@ -15,6 +15,7 @@
// putting it anywhere else would make the fix for a stalled digest lane queue
// up BEHIND the digest lane it is meant to unblock.
+import path from "node:path";
import { safeRevalidate } from "../../lib/safeRevalidate";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import {
@@ -27,12 +28,32 @@ import {
} from "yt-dlp-transcript-common/jobs/streamCommand";
import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler";
import { normalizeChannelTranscripts } from "yt-dlp-transcript-common/controller/normalizeAll";
+import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
+import { heldVideoIds } from "yt-dlp-transcript-common/controller/videoCues";
+// `ids` narrows the pass to those videos (`pnpm ops build-cues`, release 19 A7)
+// and every one must be held — a stray is refused, named, before any job.
+// `force` rewrites cues.json even where it is fresh.
export async function normalizeChannelAction(
slug: string,
queueKey?: string,
+ opts: { ids?: string[]; force?: boolean } = {},
): Promise<StreamActionResult> {
const paths = getPaths();
+ if (opts.ids !== undefined || opts.force !== undefined) {
+ const config = await readChannelConfig(paths, slug);
+ if (!config) return { ok: false, error: `Channel "${slug}" not found` };
+ }
+ if (opts.ids !== undefined) {
+ const held = await heldVideoIds(path.join(paths.channelsDir, slug, "data"));
+ const stray = opts.ids.filter((id) => !held.has(id));
+ if (stray.length) {
+ return {
+ ok: false,
+ error: `${stray.length} of "ids" not held by ${slug}: ${stray.join(", ")}`,
+ };
+ }
+ }
return runManagedFunction({
kind: "normalize-transcripts",
queueKey: resolveQueueKey(channelQueueKey(slug), queueKey),
@@ -42,6 +63,8 @@ export async function normalizeChannelAction(
const result = await normalizeChannelTranscripts({
paths,
channelSlug: slug,
+ ...(opts.ids ? { videoIds: opts.ids } : {}),
+ ...(opts.force ? { force: true } : {}),
onLog,
signal,
});
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -17,10 +17,20 @@ import {
platformQueueKey,
} from "yt-dlp-transcript-common/lib/platform";
import {
+ ARCHIVE_ORG_SEARCH_DEFAULT_ROWS,
resolveArchiveOrgImportUrl,
- runArchiveOrgImport,
+ runArchiveOrgBatchImport,
+ searchArchiveOrgItems,
+ summarizeArchiveOrgBatch,
+ type ArchiveOrgItemRequest,
} from "yt-dlp-transcript-common/controller/archiveOrgImport";
import {
+ REMOTE_LISTING_RESULT_MARKER,
+ isRemoteListingPlatform,
+ runRemoteListing,
+} from "yt-dlp-transcript-common/controller/remoteListing";
+import { EnumerationIncompleteError } from "yt-dlp-transcript-common/ytdlp/runYtdlp";
+import {
importPlatformSignal,
resolveBitchuteImportUrl,
} from "yt-dlp-transcript-common/controller/bitchuteImport";
@@ -721,42 +731,105 @@ function archiveOrgRefusal(paths: Paths, what: string): Promise<string | null> {
return platformRefusal(paths, "archiveorg", "archive.org", what);
}
-// IMPORT CHOSEN FILES OF ONE archive.org ITEM (`pnpm ops import-archive-org`):
-// one job on archive.org's own queue that imports the files one at a time,
-// with a jittered pause between them, skipping any already downloaded, and
-// stopping on a rate limit or three failures in a row
-// (controller/archiveOrgImport.ts). `files` names exact paths in the item;
-// `match` is a case-insensitive regex over them. `dryRun` lists what would be
-// fetched and fetches nothing.
+// IMPORT FROM archive.org (`pnpm ops import-archive-org`): one job on
+// archive.org's own queue that imports files one at a time, with a jittered
+// pause between them, skipping any already held, and stopping on a rate limit
+// or three failures in a row (controller/archiveOrgImport.ts).
+//
+// item ONE item; exactly one of `files` (exact paths in the item) and
+// `match` (a case-insensitive regex over them).
+// items MANY: identifiers, or `{item, files? | match?}`; a bare identifier
+// takes the top-level `match`, else every media original of it.
+// query archive.org's advanced search: the first `limit` (100) items it
+// matches, each as a bare identifier in `items`.
+//
+// `dryRun` lists each file as held, RESTRICTED (archive.org will answer
+// 401/403) or "would get", and fetches nothing. The log ends with
+// `summary: {…}`.
+export type ArchiveOrgImportEntry = string | { item: string; files?: string[]; match?: string };
+
+const ARCHIVE_ORG_ID_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*$/;
+
export async function importArchiveOrgAction(
slug: string,
- opts: { item: string; files?: string[]; match?: string; dryRun?: boolean },
+ opts: {
+ item?: string;
+ items?: ArchiveOrgImportEntry[];
+ query?: string;
+ limit?: number;
+ files?: string[];
+ match?: string;
+ dryRun?: boolean;
+ },
): Promise<StreamActionResult> {
const paths = getPaths();
const channelConfig = await readChannelConfig(paths, slug);
if (!channelConfig) {
return { ok: false, error: `Channel "${slug}" not found` };
}
- const item = opts.item.trim();
- if (!/^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(item)) {
- return { ok: false, error: `"${item}" is not an archive.org identifier` };
- }
- if ((opts.files === undefined) === (opts.match === undefined)) {
- return { ok: false, error: 'Name the files: exactly one of "files" (a list) or "match" (a regex)' };
+ const modes = [opts.item !== undefined, opts.items !== undefined, opts.query !== undefined].filter(Boolean).length;
+ if (modes !== 1) {
+ return { ok: false, error: 'Name what to import: exactly one of "item", "items" (a list) or "query" (an archive.org search)' };
}
- if (opts.match !== undefined) {
+ const badRegex = (m: string | undefined): string | null => {
+ if (m === undefined) return null;
try {
- new RegExp(opts.match, "i");
+ new RegExp(m, "i");
+ return null;
} catch (e) {
- return { ok: false, error: `"match" is not a valid regex: ${(e as Error).message}` };
+ return `"match" is not a valid regex: ${(e as Error).message}`;
+ }
+ };
+ const top = badRegex(opts.match);
+ if (top) return { ok: false, error: top };
+ if (opts.item === undefined && opts.files !== undefined) {
+ return { ok: false, error: '"files" names files of one item — with "items", give each its own: {"item", "files"}' };
+ }
+ if (opts.query === undefined && opts.limit !== undefined) {
+ return { ok: false, error: '"limit" caps a "query" — it has nothing to cap here' };
+ }
+ // Every item, validated before any job: one bad identifier is a refusal
+ // naming it, never a batch that stops part-way.
+ const requests: ArchiveOrgItemRequest[] = [];
+ const every = { match: opts.match ?? "." };
+ if (opts.item !== undefined) {
+ const item = opts.item.trim();
+ if (!ARCHIVE_ORG_ID_RE.test(item)) {
+ return { ok: false, error: `"${item}" is not an archive.org identifier` };
+ }
+ if ((opts.files === undefined) === (opts.match === undefined)) {
+ return { ok: false, error: 'Name the files: exactly one of "files" (a list) or "match" (a regex)' };
+ }
+ requests.push({ identifier: item, selection: opts.files !== undefined ? { files: opts.files } : { match: opts.match! } });
+ } else if (opts.items !== undefined) {
+ const seen = new Set<string>();
+ for (const entry of opts.items) {
+ const e = typeof entry === "string" ? { item: entry } : entry;
+ const item = e.item.trim();
+ if (!ARCHIVE_ORG_ID_RE.test(item)) {
+ return { ok: false, error: `"${item}" is not an archive.org identifier` };
+ }
+ if (seen.has(item)) continue;
+ seen.add(item);
+ if (e.files !== undefined && e.match !== undefined) {
+ return { ok: false, error: `item "${item}": "files" or "match", not both` };
+ }
+ const own = badRegex(e.match);
+ if (own) return { ok: false, error: `item "${item}": ${own}` };
+ requests.push({
+ identifier: item,
+ selection: e.files !== undefined ? { files: e.files } : e.match !== undefined ? { match: e.match } : every,
+ });
}
}
if (!opts.dryRun) {
const err = await lowDiskError(paths, slug);
if (err) return err;
- const refused = await archiveOrgRefusal(paths, "The archive.org import");
- if (refused) return { ok: false, info: true, error: refused };
}
+ // A dry run asks archive.org too (metadata, a search), so a hold or a
+ // cooldown refuses it as well.
+ const refused = await archiveOrgRefusal(paths, "The archive.org import");
+ if (refused) return { ok: false, info: true, error: refused };
const downloadConfig =
channelConfig.platform === "archiveorg"
? channelConfig
@@ -767,12 +840,26 @@ export async function importArchiveOrgAction(
paths,
channelSlug: slug,
fn: async (onLog, signal, _setProgress, ctx) => {
- const result = await runArchiveOrgImport({
+ let items = requests;
+ if (opts.query !== undefined) {
+ const rows = opts.limit ?? ARCHIVE_ORG_SEARCH_DEFAULT_ROWS;
+ const found = await searchArchiveOrgItems(opts.query, { rows, signal }).catch(async (err) => {
+ if ((err as { rateLimited?: boolean }).rateLimited) {
+ await recordDownloadBackoff("archiveorg", paths, "rate_limit");
+ }
+ throw err;
+ });
+ onLog(
+ `archive.org search ${JSON.stringify(opts.query)}: ${found.found} item(s) match, ` +
+ `taking the first ${found.items.length} (identifier order).\n`,
+ );
+ items = found.items.map((it) => ({ identifier: it.identifier, selection: every }));
+ }
+ const batch = await runArchiveOrgBatchImport({
paths,
slug,
channelConfig: downloadConfig,
- identifier: item,
- selection: opts.files !== undefined ? { files: opts.files } : { match: opts.match! },
+ items,
onLog,
signal,
drainSignal: ctx.drainSignal,
@@ -781,16 +868,86 @@ export async function importArchiveOrgAction(
onImported: (id) => safeRevalidate([`/channels/${slug}/videos/${id}`]),
},
});
+ const summary = summarizeArchiveOrgBatch(batch);
+ onLog(`summary: ${JSON.stringify(summary)}\n`);
safeRevalidate([`/channels/${slug}`, "/channels"]);
// The platform's shared pacing state learns what archive.org said, so
// the next import (and any other archive.org job) backs off or settles.
- if (result.rateLimited) {
+ // Three items in a row refused with 401/403 back it off as a rate limit
+ // does: archive.org is refusing us, not one item.
+ if (batch.rateLimited || batch.refusedStorm) {
await recordDownloadBackoff("archiveorg", paths, "rate_limit");
- } else if (result.imported.length > 0 && result.failed.length === 0) {
+ } else if (summary.imported > 0 && summary.failed === 0) {
await recordPlatformClean("archiveorg", paths);
}
- if (result.failed.length > 0 && result.imported.length === 0 && !opts.dryRun) {
- throw new Error(result.stopped ?? `${result.failed.length} file(s) failed`);
+ if (!opts.dryRun && summary.failed > 0 && summary.imported === 0) {
+ throw new Error(batch.stopped ?? `${summary.failed} file(s) failed`);
+ }
+ if (opts.item !== undefined && batch.missing.length > 0) {
+ throw new Error(batch.missing[0].error);
+ }
+ },
+ });
+}
+
+// AN ODYSEE OR BITCHUTE CHANNEL'S REMOTE LISTING, diffed against what it
+// holds (`pnpm ops get remote-listing <slug>`, controller/remoteListing.ts):
+// one flat-playlist read of the channel's URL on the platform's own queue —
+// one stream with every other request there — refused while the platform is
+// held or cooling down, after the platform's import floor, and a rate-limited
+// read backs the platform off. Nothing is written. The job's log ends with
+// the result as `@@remote-listing {…}`.
+export async function remoteListingAction(slug: string): Promise<StreamActionResult> {
+ const paths = getPaths();
+ const channelConfig = await readChannelConfig(paths, slug);
+ if (!channelConfig) {
+ return { ok: false, error: `Channel "${slug}" not found` };
+ }
+ if (!channelConfig.url) {
+ return { ok: false, error: "Channel has no `url` configured" };
+ }
+ const platform = detectPlatform(channelConfig.url);
+ if (!isRemoteListingPlatform(platform)) {
+ return {
+ ok: false,
+ error: `A remote listing reads an Odysee or BitChute channel; "${slug}" is on ${platform ?? "an unknown platform"} — its sync lists it`,
+ };
+ }
+ const label = PACED_LABELS[platform] ?? platform;
+ const refused = await platformRefusal(paths, platform, label, "The remote listing");
+ if (refused) return { ok: false, info: true, error: refused };
+ const settings = getSettings();
+ return runManagedFunction({
+ kind: "remote-listing",
+ queueKey: platformQueueKey(platform),
+ paths,
+ channelSlug: slug,
+ fn: async (onLog, signal) => {
+ await waitForPlatformGap({
+ label,
+ remainingMs: async () =>
+ Math.max(
+ platformGapRemainingMs(platform),
+ await platformCooldownRemainingMs(platform, paths).catch(() => 0),
+ ),
+ signal,
+ onLog,
+ });
+ try {
+ const result = await runRemoteListing({ paths, slug, channelConfig, platform, onLog, signal });
+ onLog(`${REMOTE_LISTING_RESULT_MARKER}${JSON.stringify(result)}\n`);
+ } catch (err) {
+ if (err instanceof EnumerationIncompleteError) {
+ await recordDownloadBackoff(platform, paths, "rate_limit").catch(() => {});
+ }
+ throw err;
+ } finally {
+ notePlatformGap(
+ platform,
+ downloadGapMs(settings.sleepBetweenDownloadsSeconds, 0, 0, {
+ minSeconds: platformImportMinGapSeconds(platform),
+ }),
+ );
}
},
});
diff --git a/editor/app/sites/lib/publishCore.ts b/editor/app/sites/lib/publishCore.ts
@@ -43,7 +43,10 @@ export type DeployWhere = "production" | "preview" | "local";
export type TargetAsk = {
// A site id, "_hub" or "_homepage".
target: string;
- build?: { force?: boolean; skipArchives?: boolean };
+ // allowMissingMedia: a report citation with no prepared media is let
+ // through compose (`archilyzer publish build --allow-missing-media`; the ops
+ // publish route's build verb, release 19 A8).
+ build?: { force?: boolean; skipArchives?: boolean; allowMissingMedia?: boolean };
deploy?: { where: DeployWhere; preview?: string; force?: boolean };
};
@@ -135,6 +138,7 @@ export function wantedPlan(
reason: "asked",
...(ask.build.force !== false ? { force: true } : {}),
...(ask.build.skipArchives ? { skipArchives: true } : {}),
+ ...(ask.build.allowMissingMedia ? { allowMissingMedia: true } : {}),
...(indexStep ? { indexAfter: runStart } : {}),
});
}
diff --git a/mcp/README.md b/mcp/README.md
@@ -7,8 +7,13 @@ Desktop, Cursor, and any other MCP client.
It is a **local tool you run yourself**. It does not change the archive: it only
reads the site's already-published static JSON shards (`corpus.json` +
`transcripts/<slug>/…`), either from disk or over HTTP. Nothing is hosted for you.
-The one exception is `fetch_clip`, which asks a local Archilyzer editor to fetch a
-clip window; the MCP itself still writes nothing.
+The exceptions all ask a local Archilyzer editor (`ARCHILYZER_EDITOR_URL` +
+`WORKER_TOKEN`, the editor's own) to do the work — `fetch_clip` fetches a clip
+window, `enqueue` queues archival work, `get_job` / `channel_coverage` read the
+editor's jobs and disk, and `get_transcript` falls back to the editor's disk for a
+video not yet published; `notes` reads umtool (`UMTOOL_URL`). The MCP itself still
+writes nothing. Settings, storage and deletes are not reachable from it (`pnpm ops`
+only).
## Tools
@@ -20,10 +25,14 @@ clip window; the MCP itself still writes nothing.
| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets — **each `[mm:ss]` is a clickable link to that exact moment** (or a compact `[mm:ss\|sec]` with `link_style:"base"`). Alias-aware, pageable, and **filterable** (`states`, `date_from`/`date_to`, `media_type`, `age`, `exclude`, `scopes`). A page that isn't the whole match set is flagged **above** the hits. |
| `enumerate_matches` | A query's **complete** match set as a worklist (id/title/channel/date + batch count) in **one scan**. Takes the **same filters** as `search_transcripts`, so the two can never disagree about coverage. The tool to use whenever you need to count or cover everything. |
| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around one query or up to 8 (`queries`), with per-query counts; or full transcripts without a query. Reads posts too. |
-| `get_transcript` | One video's full transcript as clean markdown (metadata + **linked** timestamped captions). |
+| `get_transcript` | One video's full transcript as clean markdown (metadata + **linked** timestamped captions). A video the archive does not hold yet (imported or transcribed since the last build) is read off the **local editor's disk** when one is configured (`GET /api/ops/transcript`: a fresh `transcript.cues.json`, else the raw transcript normalized in memory, else the English VTT), marked as such, with no moment links. |
| `get_post` / `get_thread` | One archived social post, or its whole thread. Posts have no timeline — cite them with no `@ mm:ss`. |
| `get_video_metadata` | Everything known about one video without the transcript body: metadata, plus **view/like counts, cue count and transcript coverage** (`stats/`), **other archived copies of the same recording** with an explicit timings-aligned verdict (`duplicates.json`), and **AI chapters/tags** where they exist (`digests/`). |
| `fetch_clip` | The media behind a cited moment, **fetched by the local editor** (`POST /api/media/fetch-window`) through its paced, cookie-aware, provenanced job — never a yt-dlp run by hand. Needs `ARCHILYZER_EDITOR_URL` (default `http://localhost:3001`) and `WORKER_TOKEN` (the editor's own) in this server's env; without them it says so and fetches nothing. The editor must already archive the cited channel (a channel dir under its `transcripts/`), else it answers 404 `Channel "<slug>" not found`: an MCP pointed at a public site with a fresh editor gets that on every clip. A window is the cited span ± `pad` (default 3 s), at most 15 min, and lands at `channels/<slug>/data/<id>/clips/`; `full: true` fetches the whole recording into the saved-video store (needs a video the editor already knows). `maxHeight` (144–2160) caps the source height: a window is fetched at or under it (default 720); a whole recording at 720 or less is saved as the editor's 720p H.264 preset and above 720 at the original quality (omitted, the channel's source-video quality applies). A file already on disk is returned as it is, never re-fetched for a different cap, and the answer gives its height and says when it is taller than asked. Waits up to `wait_seconds` (default 90, max 300), then returns the job id to resume with `job`; a client with a 60 s default request timeout must raise it or pass `wait_seconds` ≤ 50 — the fetch continues on the editor either way; resume it with `job`, and once it has finished the same request finds it cached. While it waits it sends one progress notification per poll to a client that asked for progress (a `progressToken`), which keeps a reset-on-progress timeout alive. A Rumble embed id is mapped to the editor's slug id through the record's `webpageUrl`, so pass the citing corpus as `source`; a video not in `source` is passed through as cited (known limitation). The file is a read-only corpus artifact. |
+| `get_job` | One editor job's state — kind, channel, status, times, exit code, where it waits in its queue — and the last `tail` (default 40) log lines (`GET /api/ops/job/<id>`). For the job id `enqueue` or `fetch_clip` returned. Needs the editor. |
+| `enqueue` | Queue archival work on a channel the editor archives, as its page's buttons do (`POST /api/ops/<kind>`): `sync` (`full`), `download-missing`, `retry-bucket` (`bucket`, `ids` — the way to download chosen videos), `transcribe-bucket` (`ids`), `fetch-posts` (`full` / `older`), `import-video` (`url`). Answers the job id; the editor runs it on its own queues at each platform's pace. Only when the operator asks. Needs the editor. |
+| `channel_coverage` | What a channel **holds** on the editor's disk by date (`GET /api/ops/coverage`): count, first and last day, per year, and every gap longer than `gap_days` (30); `date_from` / `date_to` narrow it, `list` lists the videos. Each video is dated by its **recorded date** when the channel has a title rule (a VOD mirror, release 20), else its upload date. Needs the editor. |
+| `notes` | The operator's notes in umtool on articles and report-video projects: `list` (every open note, by target), `read` (`target` as `list` names it — the digest with anchors resolved and the source file to edit). `reply` answers with the `umtool notes reply` command, because umtool records a reply sent over HTTP as the operator's. Needs `UMTOOL_URL`. |
| `open_link` | Paste an archilyzer viewer **share link** to re-run that exact search here (query tree + every filter, at full fidelity) — plan, results and corpus handle in **one** call. `dry_run:true` for the plan alone. |
| `list_sources` | Show the **default** corpus and, with a hub, its member sites as ready-to-paste handles. A site with reports says how many; a cited-only site says it is one. |
| `resolve_source` | Turn a URL or site name into the canonical `source` handle and check it can be read. Changes nothing. |
@@ -496,8 +505,8 @@ nothing to reset).
**Still read-only.** A `source` handle only changes *which* already-published
static shards are read — the same capability the startup flags already grant
-this locally-run tool. This server never writes to any corpus; `fetch_clip` asks
-the editor, and the editor writes.
+this locally-run tool. This server never writes to any corpus; `fetch_clip` and
+`enqueue` ask the editor, and the editor writes.
## Protocol
diff --git a/mcp/src/archivalTools.test.ts b/mcp/src/archivalTools.test.ts
@@ -0,0 +1,204 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { connectWith, fakeEditor, isError, textOf } from "./editorOps.testkit";
+import { enqueueBody, notesTarget, notesTool } from "./archivalTools";
+
+// ─── The archival tools (release 19 A9), through tools/call ───
+//
+// Each maps to one ops route on the fake editor (or umtool's notes routes);
+// what is pinned is the request sent and the words the agent reads.
+
+const EDITOR = "http://editor.test";
+
+test("enqueue: each kind posts to its own route with only the keys it takes", () => {
+ assert.deepEqual(enqueueBody({ kind: "sync", channel: "demo", full: true }), {
+ ok: true,
+ kind: "sync",
+ route: "/api/ops/sync",
+ body: { slug: "demo", full: true },
+ });
+ assert.deepEqual(enqueueBody({ kind: "retry-bucket", channel: "demo", bucket: "noTranscript", ids: ["a1"] }), {
+ ok: true,
+ kind: "retry-bucket",
+ route: "/api/ops/retry-bucket",
+ body: { slug: "demo", bucket: "noTranscript", ids: ["a1"] },
+ });
+ assert.deepEqual(enqueueBody({ kind: "import-video", channel: "demo", url: "https://example.invalid/v/1" }), {
+ ok: true,
+ kind: "import-video",
+ route: "/api/ops/import-video",
+ body: { slug: "demo", url: "https://example.invalid/v/1" },
+ });
+ const refusals: [Record<string, unknown>, RegExp][] = [
+ [{ kind: "delete-channel", channel: "demo" }, /kind must be one of sync, download-missing/],
+ [{ kind: "sync" }, /channel is required/],
+ [{ kind: "sync", channel: "../x" }, /channel is required/],
+ [{ kind: "download-missing", channel: "demo", ids: ["a"] }, /download-missing does not take ids$/],
+ [{ kind: "sync", channel: "demo", url: "https://x" }, /sync does not take url \(it takes full\)/],
+ [{ kind: "retry-bucket", channel: "demo" }, /retry-bucket needs bucket/],
+ [{ kind: "transcribe-bucket", channel: "demo", ids: ["../a"] }, /ids must be a non-empty list of video ids/],
+ [{ kind: "import-video", channel: "demo", url: "ftp://x" }, /import-video needs url/],
+ [{ kind: "fetch-posts", channel: "demo", older: "yes" }, /older must be true or false/],
+ ];
+ for (const [args, re] of refusals) {
+ const r = enqueueBody(args);
+ assert.equal(r.ok, false, JSON.stringify(args));
+ assert.match(!r.ok ? r.error : "", re, JSON.stringify(args));
+ }
+});
+
+test("enqueue through tools/call: the job id comes back with how to follow it", async () => {
+ const { deps, sent } = fakeEditor((s) =>
+ s.url.endsWith("/api/ops/transcribe-bucket")
+ ? { status: 200, body: { ok: true, jobId: "j-77" } }
+ : { status: 400, body: { ok: false, error: 'Channel "nope" not found' } },
+ );
+ const client = await connectWith(deps);
+ const ok = await client.callTool({ name: "enqueue", arguments: { kind: "transcribe-bucket", channel: "demo" } });
+ assert.equal(isError(ok), false, textOf(ok));
+ assert.equal(sent[0].url, `${EDITOR}/api/ops/transcribe-bucket`);
+ assert.equal(sent[0].method, "POST");
+ assert.deepEqual(sent[0].body, { slug: "demo" });
+ assert.match(textOf(ok), /queued as job j-77.*follow it with get_job \(job: "j-77"\)/s);
+ const refused = await client.callTool({ name: "enqueue", arguments: { kind: "sync", channel: "nope" } });
+ assert.equal(isError(refused), true);
+ assert.match(textOf(refused), /enqueue sync: the editor refused \(HTTP 400\): Channel "nope" not found/);
+ const none = await (await connectWith(fakeEditor(() => ({ status: 200, body: {} }), {}).deps)).callTool({
+ name: "enqueue",
+ arguments: { kind: "sync", channel: "demo" },
+ });
+ assert.match(textOf(none), /no editor configured/);
+});
+
+test("get_job: status, queue place and the log tail", async () => {
+ const { deps, sent } = fakeEditor((s) =>
+ s.url.includes("/api/ops/job/j-1")
+ ? {
+ status: 200,
+ body: {
+ ok: true,
+ job: {
+ id: "j-1",
+ kind: "sync",
+ channelSlug: "demo",
+ status: "queued",
+ queuedAt: Date.UTC(2026, 9, 10),
+ queue: { key: "youtube", position: 2, queued: 3, head: { id: "j-0", kind: "persist-videos", channelSlug: "other" } },
+ },
+ tail: ["line one", "line two"],
+ },
+ }
+ : { status: 404, body: { ok: false, error: "no such job" } },
+ );
+ const client = await connectWith(deps);
+ const res = await client.callTool({ name: "get_job", arguments: { job: "j-1", tail: 2 } });
+ assert.equal(new URL(sent[0].url).pathname, "/api/ops/job/j-1");
+ assert.equal(new URL(sent[0].url).searchParams.get("tail"), "2");
+ const t = textOf(res);
+ assert.match(t, /job j-1: sync on demo — \*\*queued\*\*/);
+ assert.match(t, /waiting on queue "youtube": position 2 of 3, behind persist-videos on other \(j-0\)/);
+ assert.match(t, /Still going: call get_job again later/);
+ assert.match(t, /```\nline one\nline two\n```/);
+ const gone = await client.callTool({ name: "get_job", arguments: { job: "j-9" } });
+ assert.match(textOf(gone), /the editor knows no job j-9/);
+ const bad = await client.callTool({ name: "get_job", arguments: { job: "j-1", tail: -1 } });
+ assert.match(textOf(bad), /tail is a whole number/);
+});
+
+test("channel_coverage: the query sent, and the gaps read back", async () => {
+ const { deps, sent } = fakeEditor(() => ({
+ status: 200,
+ body: {
+ ok: true,
+ slug: "vod-mirror",
+ titlePattern: "(?<year>\\d{4})-(?<month>\\d{2})-(?<day>\\d{2})",
+ held: 3,
+ dated: 2,
+ byRecordedDate: 2,
+ byUploadDate: 0,
+ undated: ["bare"],
+ first: "20230501",
+ last: "20240110",
+ byYear: { "2023": 1, "2024": 1 },
+ byMonth: {},
+ gapDays: 60,
+ gaps: [{ after: "20230501", before: "20240110", days: 254 }],
+ },
+ }));
+ const client = await connectWith(deps);
+ const res = await client.callTool({
+ name: "channel_coverage",
+ arguments: { channel: "vod-mirror", gap_days: 60, date_from: "2023-01-01", list: true },
+ });
+ const q = new URL(sent[0].url).searchParams;
+ assert.equal(new URL(sent[0].url).pathname, "/api/ops/coverage");
+ assert.equal(q.get("slug"), "vod-mirror");
+ assert.equal(q.get("gapDays"), "60");
+ assert.equal(q.get("from"), "20230101");
+ assert.equal(q.get("list"), "1");
+ const t = textOf(res);
+ assert.match(t, /3 held · 2 dated \(2 by recorded date, 0 by upload date\) · 1 undated · 2023-05-01 → 2024-01-10/);
+ assert.match(t, /the recorded date read off each title/);
+ assert.match(t, /- 2023-05-01 → 2024-01-10: 254 days with nothing held/);
+ const bad = await client.callTool({ name: "channel_coverage", arguments: { channel: "x", date_to: "Jan 2024" } });
+ assert.match(textOf(bad), /date_to is a date, YYYY-MM-DD/);
+});
+
+test("notes: list groups open notes by target; read fetches the digest; reply is the CLI command", async () => {
+ const umtool = { UMTOOL_URL: "http://umtool.test/" };
+ const { deps, sent } = fakeEditor((s) => {
+ if (s.url.includes("/api/browse/decisions")) {
+ return {
+ status: 200,
+ body: {
+ items: [
+ { kind: "open-note", project: "sites/demo-site/a-report", projectKind: "article", target: "cite c1", why: "say claimed" },
+ { kind: "cut-without-file", project: "song", target: "x", why: "y" },
+ { kind: "open-note", project: "demo/video-project", projectKind: "report-video", target: "take t2", why: "too long" },
+ ],
+ },
+ };
+ }
+ return { status: 200, body: "unused" };
+ }, umtool);
+ const client = await connectWith(deps);
+ const list = await client.callTool({ name: "notes", arguments: {} });
+ assert.equal(sent[0].url, "http://umtool.test/api/browse/decisions");
+ const t = textOf(list);
+ assert.match(t, /2 open note\(s\) on 2 target\(s\)/);
+ assert.match(t, /## sites\/demo-site\/a-report \(article\)/);
+ assert.match(t, /- \[take t2\] too long/);
+ assert.doesNotMatch(t, /song/);
+
+ assert.deepEqual(notesTarget("sites/demo-site/a-report"), { param: "article", value: "demo-site/a-report" });
+ assert.deepEqual(notesTarget("demo/video-project"), { param: "project", value: "demo/video-project" });
+ assert.equal(notesTarget("sites/../x"), null);
+
+ const digestSent: string[] = [];
+ const read = await notesTool(
+ { action: "read", target: "sites/demo-site/a-report", status: "all" },
+ {
+ env: umtool,
+ fetch: async (url) => {
+ digestSent.push(url);
+ return { status: 200, json: async () => ({}), text: async () => "# Notes on demo-site/a-report\n…" } as never;
+ },
+ },
+ );
+ const q = new URL(digestSent[0]).searchParams;
+ assert.equal(new URL(digestSent[0]).pathname, "/api/notes/context");
+ assert.equal(q.get("article"), "demo-site/a-report");
+ assert.equal(q.get("status"), "all");
+ assert.match(read.text, /^# Notes on demo-site\/a-report/);
+
+ const reply = await client.callTool({
+ name: "notes",
+ arguments: { action: "reply", note_id: "n_abc", text: "Changed it", resolve: true },
+ });
+ assert.equal(isError(reply), true);
+ assert.match(textOf(reply), /umtool notes reply n_abc "Changed it" --resolve/);
+ assert.equal(sent.length, 1, "a reply sends nothing");
+
+ const none = await notesTool({ action: "list" }, { env: {}, fetch: deps.fetch });
+ assert.match(none.text, /no umtool configured\. Set UMTOOL_URL/);
+});
diff --git a/mcp/src/archivalTools.ts b/mcp/src/archivalTools.ts
@@ -0,0 +1,479 @@
+import { describeEditorFailure, editorGet, editorPost, type EditorDeps } from "./editorOps";
+import { describeFetchError, isVideoId, REQUEST_TIMEOUT_MS } from "./fetchClip";
+
+// ─── Archival writes through the editor (release 19 A9) ───
+//
+// The fetch_clip pattern: this server asks the LOCAL editor (its URL and its
+// WORKER_TOKEN) and the editor runs the job, with its queues, pacing, holds
+// and provenance. The MCP writes nothing itself. Each tool maps to ONE
+// existing ops route and nothing else:
+//
+// get_job GET /api/ops/job/<id>?tail=N
+// enqueue POST /api/ops/{sync | download-missing | retry-bucket |
+// transcribe-bucket | fetch-posts | import-video}
+// channel_coverage GET /api/ops/coverage?slug=…
+// notes umtool (UMTOOL_URL): GET /api/browse/decisions (list),
+// GET /api/notes/context (read)
+//
+// Settings, storage and deletes are deliberately absent (operator ruling,
+// release 19): `pnpm ops` / `archilyzer` only. A notes REPLY is not written
+// from here: umtool stamps every write through its HTTP route as the
+// OPERATOR's ("there is no way to claim to be one here", umtool's
+// app/api/notes/route.ts), so an agent reply must go through `umtool notes
+// reply`, which stamps "agent" — the tool answers with that command.
+//
+// Pure apart from the injected deps; server.ts adapts.
+
+export type ToolAnswer = { text: string; isError?: boolean };
+
+const SLUG_RE = /^[a-z0-9][a-z0-9._-]*$/i;
+const trimmed = (v: unknown) => (typeof v === "string" ? v.trim() : "");
+
+// ─── get_job ───
+
+export const DEFAULT_JOB_TAIL = 40;
+export const MAX_JOB_TAIL = 500;
+
+type JobView = {
+ id: string;
+ kind?: string;
+ channelSlug?: string;
+ videoId?: string;
+ status: string;
+ queuedAt?: number;
+ startedAt?: number;
+ endedAt?: number;
+ exitCode?: number;
+ detail?: string;
+ cancelReason?: string;
+ queue?: { key: string; position: number; queued: number; head?: { id: string; kind: string; channelSlug?: string } };
+ progress?: unknown;
+};
+
+const iso = (ms?: number) => (typeof ms === "number" && ms > 0 ? new Date(ms).toISOString() : null);
+
+export function renderJob(job: JobView, tail?: string[]): string {
+ const lines: string[] = [];
+ lines.push(
+ `job ${job.id}: ${job.kind ?? "job"}${job.channelSlug ? ` on ${job.channelSlug}` : ""}${job.videoId ? `/${job.videoId}` : ""} — **${job.status}**`,
+ );
+ if (job.detail) lines.push(`for: ${job.detail}`);
+ const times = [
+ iso(job.queuedAt) ? `queued ${iso(job.queuedAt)}` : "",
+ iso(job.startedAt) ? `started ${iso(job.startedAt)}` : "",
+ iso(job.endedAt) ? `ended ${iso(job.endedAt)}` : "",
+ ].filter(Boolean);
+ if (times.length) lines.push(times.join(" · "));
+ if (typeof job.exitCode === "number") lines.push(`exit code ${job.exitCode}`);
+ if (job.cancelReason) lines.push(`cancelled because: ${job.cancelReason}`);
+ if (job.queue) {
+ const q = job.queue;
+ lines.push(
+ q.position === 0
+ ? `running at the head of queue "${q.key}" (${q.queued} waiting behind)`
+ : `waiting on queue "${q.key}": position ${q.position} of ${q.queued}` +
+ (q.head ? `, behind ${q.head.kind}${q.head.channelSlug ? ` on ${q.head.channelSlug}` : ""} (${q.head.id})` : ""),
+ );
+ }
+ if (job.progress !== undefined) lines.push(`progress: ${JSON.stringify(job.progress)}`);
+ if (job.status === "queued" || job.status === "running") {
+ lines.push("Still going: call get_job again later (a platform queue may hold a job for hours).");
+ }
+ if (tail && tail.length) {
+ lines.push("", `last ${tail.length} log line(s):`, "```", ...tail, "```");
+ }
+ return lines.join("\n");
+}
+
+export async function getJob(args: Record<string, unknown>, deps: EditorDeps): Promise<ToolAnswer> {
+ const id = trimmed(args.job);
+ if (!id || !/^[\w.-]+$/.test(id)) return { text: "get_job: job is required (the job id an enqueue or fetch_clip returned)", isError: true };
+ const rawTail = args.tail === undefined ? DEFAULT_JOB_TAIL : Number(args.tail);
+ if (!Number.isInteger(rawTail) || rawTail < 0) return { text: "get_job: tail is a whole number of lines, 0 or more", isError: true };
+ const tail = Math.min(MAX_JOB_TAIL, rawTail);
+ const a = await editorGet(deps, `/api/ops/job/${encodeURIComponent(id)}${tail > 0 ? `?tail=${tail}` : ""}`);
+ if (a.kind === "refused" && a.status === 404) return { text: `get_job: the editor knows no job ${id}`, isError: true };
+ if (a.kind !== "ok") return { text: describeEditorFailure("get_job", a), isError: true };
+ const job = a.body.job as JobView | undefined;
+ if (!job || typeof job.status !== "string") return { text: "get_job: the editor's answer carried no job", isError: true };
+ const lines = Array.isArray(a.body.tail) ? (a.body.tail as unknown[]).map(String) : undefined;
+ return { text: renderJob(job, lines) };
+}
+
+// ─── enqueue ───
+
+export const ENQUEUE_KINDS = [
+ "sync",
+ "download-missing",
+ "retry-bucket",
+ "transcribe-bucket",
+ "fetch-posts",
+ "import-video",
+] as const;
+export type EnqueueKind = (typeof ENQUEUE_KINDS)[number];
+
+// Which tool arguments each kind takes, beyond `channel`. Anything else is
+// refused by name — the ops routes refuse unknown keys too, but saying it here
+// costs no request.
+const KIND_ARGS: Record<EnqueueKind, readonly string[]> = {
+ sync: ["full"],
+ "download-missing": [],
+ "retry-bucket": ["bucket", "ids"],
+ "transcribe-bucket": ["ids"],
+ "fetch-posts": ["full", "older"],
+ "import-video": ["url"],
+};
+const ALL_ARGS = ["full", "older", "bucket", "ids", "url"];
+
+export function enqueueBody(
+ args: Record<string, unknown>,
+): { ok: true; kind: EnqueueKind; route: string; body: Record<string, unknown> } | { ok: false; error: string } {
+ const kind = trimmed(args.kind) as EnqueueKind;
+ if (!ENQUEUE_KINDS.includes(kind)) {
+ return { ok: false, error: `enqueue: kind must be one of ${ENQUEUE_KINDS.join(", ")}` };
+ }
+ const channel = trimmed(args.channel);
+ if (!channel || !SLUG_RE.test(channel)) return { ok: false, error: "enqueue: channel is required (the channel slug)" };
+ const allowed = KIND_ARGS[kind];
+ const stray = ALL_ARGS.filter((k) => args[k] !== undefined && !allowed.includes(k));
+ if (stray.length) {
+ return {
+ ok: false,
+ error: `enqueue: ${kind} does not take ${stray.join(", ")}${allowed.length ? ` (it takes ${allowed.join(", ")})` : ""}`,
+ };
+ }
+ const body: Record<string, unknown> = { slug: channel };
+ for (const k of ["full", "older"] as const) {
+ if (args[k] === undefined) continue;
+ if (typeof args[k] !== "boolean") return { ok: false, error: `enqueue: ${k} must be true or false` };
+ if (args[k]) body[k] = true;
+ }
+ if (args.ids !== undefined) {
+ const ids = args.ids;
+ if (!Array.isArray(ids) || ids.length === 0 || ids.some((i) => typeof i !== "string" || !isVideoId(i))) {
+ return { ok: false, error: "enqueue: ids must be a non-empty list of video ids" };
+ }
+ body.ids = ids;
+ }
+ if (kind === "retry-bucket") {
+ const bucket = trimmed(args.bucket);
+ if (!bucket) return { ok: false, error: "enqueue: retry-bucket needs bucket (a bucket of the channel's report, e.g. noTranscript)" };
+ body.bucket = bucket;
+ }
+ if (kind === "import-video") {
+ const url = trimmed(args.url);
+ if (!/^https?:\/\//i.test(url)) return { ok: false, error: "enqueue: import-video needs url (the video's page)" };
+ body.url = url;
+ }
+ return { ok: true, kind, route: `/api/ops/${kind}`, body };
+}
+
+export async function enqueue(args: Record<string, unknown>, deps: EditorDeps): Promise<ToolAnswer> {
+ const plan = enqueueBody(args);
+ if (!plan.ok) return { text: plan.error, isError: true };
+ const a = await editorPost(deps, plan.route, plan.body);
+ if (a.kind !== "ok") return { text: describeEditorFailure(`enqueue ${plan.kind}`, a), isError: true };
+ const ids = Array.isArray(a.body.jobIds)
+ ? (a.body.jobIds as unknown[]).map(String)
+ : typeof a.body.jobId === "string"
+ ? [a.body.jobId]
+ : [];
+ if (ids.length === 0) {
+ return { text: `enqueue ${plan.kind}: the editor accepted it and started no job (${JSON.stringify(a.body)})` };
+ }
+ return {
+ text:
+ `enqueue ${plan.kind} on ${plan.body.slug}: queued as ${ids.map((i) => `job ${i}`).join(", ")}. ` +
+ `It runs on the editor's queue at the platform's pace and may wait behind other work; ` +
+ `follow it with get_job (job: "${ids[0]}").`,
+ };
+}
+
+// ─── channel_coverage ───
+
+type Coverage = {
+ slug: string;
+ titlePattern: string | null;
+ held: number;
+ dated: number;
+ byRecordedDate: number;
+ byUploadDate: number;
+ undated: string[];
+ first: string | null;
+ last: string | null;
+ byYear: Record<string, number>;
+ byMonth: Record<string, number>;
+ gapDays: number;
+ gaps: { after: string; before: string; days: number }[];
+ videos?: { id: string; date: string; title?: string; recordedDate?: string }[];
+};
+
+const day = (d: string | null) => (d && /^\d{8}$/.test(d) ? `${d.slice(0, 4)}-${d.slice(4, 6)}-${d.slice(6, 8)}` : "—");
+const DATE_ARG = /^(\d{4})-?(\d{2})-?(\d{2})$/;
+
+export function renderCoverage(c: Coverage): string {
+ const lines: string[] = [];
+ lines.push(`# Coverage of ${c.slug} (held on the editor's disk)`);
+ lines.push(
+ `${c.held} held · ${c.dated} dated (${c.byRecordedDate} by recorded date, ${c.byUploadDate} by upload date)` +
+ `${c.undated.length ? ` · ${c.undated.length} undated` : ""} · ${day(c.first)} → ${day(c.last)}`,
+ );
+ lines.push(
+ c.titlePattern
+ ? `Dates: the recorded date read off each title (rule \`${c.titlePattern}\`), else the upload date.`
+ : "Dates: upload dates (the channel has no recorded-date title rule).",
+ );
+ const years = Object.entries(c.byYear).sort(([a], [b]) => a.localeCompare(b));
+ if (years.length) lines.push("", "## By year", ...years.map(([y, n]) => `- ${y}: ${n}`));
+ if (c.gaps.length) {
+ lines.push("", `## Gaps longer than ${c.gapDays} days (${c.gaps.length})`);
+ for (const g of c.gaps) lines.push(`- ${day(g.after)} → ${day(g.before)}: ${g.days} days with nothing held`);
+ } else {
+ lines.push("", `No gap longer than ${c.gapDays} days.`);
+ }
+ if (c.undated.length) {
+ const shown = c.undated.slice(0, 20);
+ lines.push("", `Undated (no metadata with an upload date): ${shown.join(", ")}${c.undated.length > 20 ? `, … (+${c.undated.length - 20})` : ""}`);
+ }
+ if (c.videos) {
+ lines.push("", `## Videos (${c.videos.length})`);
+ for (const v of c.videos) lines.push(`- ${day(v.date)} ${v.id}${v.title ? ` — ${v.title}` : ""}${v.recordedDate ? " (recorded)" : ""}`);
+ }
+ return lines.join("\n");
+}
+
+export async function channelCoverageTool(args: Record<string, unknown>, deps: EditorDeps): Promise<ToolAnswer> {
+ const channel = trimmed(args.channel);
+ if (!channel || !SLUG_RE.test(channel)) return { text: "channel_coverage: channel is required (the channel slug)", isError: true };
+ const q = new URLSearchParams({ slug: channel });
+ if (args.gap_days !== undefined) {
+ const n = Number(args.gap_days);
+ if (!Number.isInteger(n) || n < 1) return { text: "channel_coverage: gap_days is a whole number of days above zero", isError: true };
+ q.set("gapDays", String(n));
+ }
+ for (const [arg, key] of [["date_from", "from"], ["date_to", "to"]] as const) {
+ if (args[arg] === undefined) continue;
+ const m = DATE_ARG.exec(trimmed(args[arg]));
+ if (!m) return { text: `channel_coverage: ${arg} is a date, YYYY-MM-DD`, isError: true };
+ q.set(key, `${m[1]}${m[2]}${m[3]}`);
+ }
+ if (args.list === true) q.set("list", "1");
+ const a = await editorGet(deps, `/api/ops/coverage?${q.toString()}`);
+ if (a.kind !== "ok") return { text: describeEditorFailure("channel_coverage", a), isError: true };
+ return { text: renderCoverage(a.body as unknown as Coverage) };
+}
+
+// ─── notes (umtool) ───
+
+export const NO_UMTOOL_TEXT =
+ "notes: no umtool configured. Set UMTOOL_URL (e.g. http://localhost:3050) when " +
+ "registering the MCP server; the notes live in umtool.";
+
+export type NotesDeps = EditorDeps;
+
+function umtoolFrom(env: Record<string, string | undefined>): string | null {
+ const raw = (env.UMTOOL_URL ?? "").trim();
+ return raw ? raw.replace(/\/+$/, "") : null;
+}
+
+async function umtoolGet(
+ deps: NotesDeps,
+ route: string,
+): Promise<{ ok: true; status: number; json?: unknown; text?: string } | { ok: false; error: string }> {
+ const base = umtoolFrom(deps.env);
+ if (!base) return { ok: false, error: NO_UMTOOL_TEXT };
+ const timeoutMs = deps.requestTimeoutMs ?? REQUEST_TIMEOUT_MS;
+ try {
+ const res = await deps.fetch(`${base}${route}`, { method: "GET", signal: AbortSignal.timeout(timeoutMs) });
+ const body = await res.json().catch(() => null);
+ return { ok: true, status: res.status, json: body };
+ } catch (e) {
+ return { ok: false, error: `notes: umtool at ${base} did not answer: ${describeFetchError(e, timeoutMs)}` };
+ }
+}
+
+// A target as notes list names it: `sites/<site>/<report>` is an article,
+// anything else a report-video project id (which may hold a "/" too — the
+// prefix is what tells them apart).
+export function notesTarget(raw: string): { param: "article" | "project"; value: string } | null {
+ const t = raw.trim();
+ if (!t || /\s|\.\.|^\//.test(t)) return null;
+ const article = /^sites\/([a-z0-9][a-z0-9-]*\/[A-Za-z0-9][A-Za-z0-9._-]*)$/.exec(t);
+ if (article) return { param: "article", value: article[1] };
+ if (t.startsWith("sites/")) return null;
+ return { param: "project", value: t };
+}
+
+type Decision = { kind: string; project: string; projectKind?: string; target?: string; why?: string; href?: string };
+
+export async function notesTool(args: Record<string, unknown>, deps: NotesDeps): Promise<ToolAnswer> {
+ const action = trimmed(args.action) || "list";
+ if (action === "reply") {
+ const id = trimmed(args.note_id) || "<note id>";
+ const text = trimmed(args.text) || "<what changed>";
+ return {
+ isError: true,
+ text:
+ "notes: a reply is not written from here. umtool records every write through its HTTP " +
+ "route as the operator's, and an agent's reply must say it is an agent's — so it goes " +
+ "through umtool's CLI, from the repo checkout:\n\n" +
+ ` umtool notes reply ${id} ${JSON.stringify(text)}${args.resolve === true ? " --resolve" : ""}\n\n` +
+ "(`node umtool/bin/umtool.mjs notes reply …` when umtool is not on the PATH.) Edit the " +
+ "note's SOURCE file first (notes read shows it), regenerate, then reply.",
+ };
+ }
+ if (action === "list") {
+ const r = await umtoolGet(deps, "/api/browse/decisions");
+ if (!r.ok) return { text: r.error, isError: true };
+ if (r.status !== 200) return { text: `notes: umtool answered HTTP ${r.status}`, isError: true };
+ const items = ((r.json as { items?: Decision[] } | null)?.items ?? []).filter(
+ (d) => d.kind === "open-note" || d.kind === "unreadable-notes",
+ );
+ if (items.length === 0) return { text: "No open notes." };
+ const byProject = new Map<string, Decision[]>();
+ for (const d of items) byProject.set(d.project, [...(byProject.get(d.project) ?? []), d]);
+ const lines = [`${items.length} open note(s) on ${byProject.size} target(s):`];
+ for (const [project, ds] of byProject) {
+ lines.push("", `## ${project} (${ds[0].projectKind ?? "project"}) — read with notes action "read", target "${project}"`);
+ for (const d of ds) {
+ lines.push(d.kind === "unreadable-notes" ? `- UNREADABLE notes.json: ${d.why ?? ""}` : `- [${d.target ?? "note"}] ${d.why ?? ""}`);
+ }
+ }
+ return { text: lines.join("\n") };
+ }
+ if (action === "read") {
+ const target = notesTarget(trimmed(args.target));
+ if (!target) {
+ return { text: 'notes: read needs target as notes list names it — "sites/<site>/<report>" for an article, else a video project id', isError: true };
+ }
+ const status = trimmed(args.status) || "open";
+ if (!["open", "resolved", "all"].includes(status)) return { text: "notes: status is open, resolved or all", isError: true };
+ const base = umtoolFrom(deps.env);
+ if (!base) return { text: NO_UMTOOL_TEXT, isError: true };
+ const timeoutMs = deps.requestTimeoutMs ?? REQUEST_TIMEOUT_MS;
+ const q = new URLSearchParams({ [target.param]: target.value, status });
+ try {
+ const res = await deps.fetch(`${base}/api/notes/context?${q.toString()}`, {
+ method: "GET",
+ signal: AbortSignal.timeout(timeoutMs),
+ });
+ // The digest is text/plain; an error is JSON.
+ const raw = (res as unknown as { text?: () => Promise<string> }).text
+ ? await (res as unknown as { text: () => Promise<string> }).text()
+ : JSON.stringify(await res.json());
+ if (res.status !== 200) {
+ let error = raw;
+ try {
+ error = (JSON.parse(raw) as { error?: string }).error ?? raw;
+ } catch {
+ /* plain text */
+ }
+ return { text: `notes: umtool refused (HTTP ${res.status}): ${error}`, isError: true };
+ }
+ return { text: raw };
+ } catch (e) {
+ return { text: `notes: umtool at ${base} did not answer: ${describeFetchError(e, timeoutMs)}`, isError: true };
+ }
+ }
+ return { text: 'notes: action is "list", "read" or "reply"', isError: true };
+}
+
+// ─── The tool definitions (server.ts's TOOLS takes them in order) ───
+
+export const ARCHIVAL_TOOLS = [
+ {
+ name: "get_job",
+ description:
+ "One editor job's state — kind, channel, status, times, exit code, where it waits " +
+ "in its queue — and the last lines of its log. For the job id enqueue or fetch_clip " +
+ "returned. Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ job: { type: "string", description: "The job id." },
+ tail: {
+ type: "number",
+ description: `Log lines to include (default ${DEFAULT_JOB_TAIL}, at most ${MAX_JOB_TAIL}; 0 for none).`,
+ },
+ },
+ required: ["job"],
+ additionalProperties: false,
+ },
+ },
+ {
+ name: "enqueue",
+ description:
+ "Ask the local editor to queue archival work on a channel it archives, as the " +
+ "channel page's buttons do: sync (list and download what is new; full: true sweeps " +
+ "the whole listing), download-missing (every listed video not held), retry-bucket " +
+ "(one bucket of the channel's report, or ids within it — the way to download chosen " +
+ "videos), transcribe-bucket (the downloaded, untranscribed videos, or ids among them), " +
+ "fetch-posts (a social channel's new posts; full / older walks), import-video (one " +
+ "video by URL into the channel). The editor runs it on its own queues at each " +
+ "platform's pace and answers a job id — follow it with get_job. Only when the " +
+ "operator asks for it. Settings, storage and deletes are not reachable from here. " +
+ "Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ kind: { type: "string", enum: [...ENQUEUE_KINDS], description: "What to queue." },
+ channel: { type: "string", description: "The channel slug (the editor's)." },
+ full: { type: "boolean", description: "sync: sweep the whole listing now; fetch-posts: re-walk the timeline." },
+ older: { type: "boolean", description: "fetch-posts: walk back below the oldest archived post." },
+ bucket: { type: "string", description: "retry-bucket: the bucket name (e.g. noTranscript, partialDownloads)." },
+ ids: {
+ type: "array",
+ items: { type: "string" },
+ description: "retry-bucket / transcribe-bucket: only these video ids (each must be in the bucket).",
+ },
+ url: { type: "string", description: "import-video: the video's page URL." },
+ },
+ required: ["kind", "channel"],
+ additionalProperties: false,
+ },
+ },
+ {
+ name: "channel_coverage",
+ description:
+ "What a channel HOLDS on the editor's disk, by date: how many videos, the first and " +
+ "last day, counts per year, and every gap longer than gap_days with nothing held. " +
+ "Each video is dated by its recorded date when the channel has a recorded-date " +
+ "title rule (a VOD mirror), else by its upload date. Counts what is held, not what " +
+ "is published. Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ channel: { type: "string", description: "The channel slug (the editor's)." },
+ gap_days: { type: "number", description: "The shortest span reported as a gap (default 30)." },
+ date_from: { type: "string", description: "Only videos on or after this date (YYYY-MM-DD)." },
+ date_to: { type: "string", description: "Only videos on or before this date (YYYY-MM-DD)." },
+ list: { type: "boolean", description: "Also list every dated video (default false)." },
+ },
+ required: ["channel"],
+ additionalProperties: false,
+ },
+ },
+ {
+ name: "notes",
+ description:
+ "The operator's notes in umtool on articles and report-video projects. action " +
+ '"list": every open note, grouped by target. action "read" with target (as list names ' +
+ 'it: "sites/<site>/<report>" for an article, else a project id): each note with its anchor ' +
+ "resolved and the SOURCE file to edit (status: open, resolved or all). action " +
+ '"reply" answers with the `umtool notes reply` command to run — umtool records a ' +
+ "reply sent over HTTP as the operator's, so an agent's reply goes through its CLI. " +
+ "Needs UMTOOL_URL.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ action: { type: "string", enum: ["list", "read", "reply"], description: 'Default "list".' },
+ target: { type: "string", description: 'read: as list names it — "sites/<site>/<report>" (an article) or a project id.' },
+ status: { type: "string", enum: ["open", "resolved", "all"], description: "read: which notes (default open)." },
+ note_id: { type: "string", description: "reply: the note id (n_…)." },
+ text: { type: "string", description: "reply: what changed." },
+ resolve: { type: "boolean", description: "reply: resolve the note too." },
+ },
+ additionalProperties: false,
+ },
+ },
+] as const;
diff --git a/mcp/src/editorOps.test.ts b/mcp/src/editorOps.test.ts
@@ -0,0 +1,91 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { connectWith, fakeEditor, isError, textOf } from "./editorOps.testkit";
+import { editorGet, editorPost, editorTranscript, renderEditorTranscript, describeEditorFailure } from "./editorOps";
+
+// ─── The editor's ops routes from the MCP (release 19 A7) ───
+//
+// No network: a fake editor answers by URL, and records what was sent.
+
+test("a GET and a POST carry the token to the editor's URL; refusals and silence are said", async () => {
+ const { deps, sent } = fakeEditor((s) =>
+ s.url.endsWith("/bad") ? { status: 400, body: { ok: false, error: "no such thing" } } : { status: 200, body: { ok: true, x: 1 } },
+ );
+ const ok = await editorGet(deps, "/api/ops/job/abc");
+ assert.equal(ok.kind, "ok");
+ assert.equal(sent[0].url, "http://editor.test/api/ops/job/abc");
+ assert.equal(sent[0].headers?.authorization, "Bearer tok");
+ const posted = await editorPost(deps, "/api/ops/sync", { slug: "x" });
+ assert.equal(posted.kind, "ok");
+ assert.deepEqual(sent[1].body, { slug: "x" });
+ assert.equal(sent[1].headers?.["content-type"], "application/json");
+ const bad = await editorGet(deps, "/bad");
+ assert.equal(bad.kind, "refused");
+ assert.match(describeEditorFailure("t", bad as never), /^t: the editor refused \(HTTP 400\): no such thing$/);
+ const none = await editorGet({ env: {}, fetch: deps.fetch }, "/x");
+ assert.equal(none.kind, "no_editor");
+ assert.match(describeEditorFailure("t", none as never), /ARCHILYZER_EDITOR_URL.*WORKER_TOKEN/);
+ const gone = await editorGet(fakeEditor(() => new TypeError("fetch failed")).deps, "/x");
+ assert.equal(gone.kind, "unreachable");
+ assert.match(describeEditorFailure("t", gone as never), /did not answer: fetch failed/);
+});
+
+const CUES = [
+ { start: 1, end: 3, text: "first line" },
+ { start: 65, end: 67, text: "second line" },
+];
+
+test("editorTranscript asks GET /api/ops/transcript; a 404 is not an error, other refusals are", async () => {
+ const { deps, sent } = fakeEditor((s) =>
+ s.url.includes("id=held")
+ ? { status: 200, body: { ok: true, slug: "demo", id: "held", source: "vtt", cuesJson: "missing", cues: CUES } }
+ : { status: 404, body: { ok: false, error: "no channel holds it" } },
+ );
+ const held = await editorTranscript(deps, "held", "demo");
+ assert.equal(held.ok, true);
+ assert.equal(new URL(sent[0].url).searchParams.get("slug"), "demo");
+ const missing = await editorTranscript(deps, "nope");
+ assert.deepEqual(missing, { ok: false, error: null });
+ assert.equal(new URL(sent[1].url).searchParams.has("slug"), false);
+ assert.deepEqual(await editorTranscript(deps, "../x"), { ok: false, error: null });
+ assert.equal(sent.length, 2);
+});
+
+test("the rendered fallback says where it came from and carries no moment links", () => {
+ const md = renderEditorTranscript(
+ { slug: "demo", id: "held", source: "built", cuesJson: "missing", title: "A title", file: "transcript.en.vtt", cues: CUES },
+ { timestamps: true },
+ );
+ assert.match(md, /^> NOT IN THE ARCHIVE — read from the local editor's disk \(channel demo\)/);
+ assert.match(md, /normalized in memory/);
+ assert.match(md, /# A title/);
+ assert.match(md, /\[1:05\] second line/);
+ assert.doesNotMatch(md, /\]\(http/);
+});
+
+test("get_transcript: a video not in the archive is read off the editor's disk", async () => {
+ const { deps } = fakeEditor(() => ({
+ status: 200,
+ body: { ok: true, slug: "demo", id: "fresh01", source: "cues.json", cuesJson: "fresh", title: "Fresh", cues: CUES },
+ }));
+ const client = await connectWith(deps);
+ const res = await client.callTool({ name: "get_transcript", arguments: { video_id: "fresh01", channel: "demo" } });
+ assert.equal(isError(res), false, textOf(res));
+ assert.match(textOf(res), /NOT IN THE ARCHIVE/);
+ assert.match(textOf(res), /first line/);
+});
+
+test("get_transcript: with no editor, or one that holds nothing, it is still 'video not found'", async () => {
+ const none = await connectWith(fakeEditor(() => ({ status: 500, body: {} }), {}).deps);
+ const a = await none.callTool({ name: "get_transcript", arguments: { video_id: "x1" } });
+ assert.equal(isError(a), true);
+ assert.match(textOf(a), /^video not found: x1/);
+ const empty = await connectWith(fakeEditor(() => ({ status: 404, body: { ok: false, error: "no" } })).deps);
+ const b = await empty.callTool({ name: "get_transcript", arguments: { video_id: "x1" } });
+ assert.match(textOf(b), /^video not found: x1/);
+ // A track is the archive's: the editor is not asked for one.
+ const { deps, sent } = fakeEditor(() => ({ status: 200, body: { ok: true, cues: [] } }));
+ const c = await (await connectWith(deps)).callTool({ name: "get_transcript", arguments: { video_id: "x1", track: "en" } });
+ assert.match(textOf(c), /^video not found: x1/);
+ assert.equal(sent.length, 0);
+});
diff --git a/mcp/src/editorOps.testkit.ts b/mcp/src/editorOps.testkit.ts
@@ -0,0 +1,91 @@
+import { Client, InMemoryTransport } from "@modelcontextprotocol/client";
+import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest";
+import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
+import type { ChannelGroups, ChannelRef, ShardSource, VideoAvailability } from "./source";
+import { SourceRegistry } from "./sourceRegistry";
+import { createServer } from "./server";
+import type { FetchClipDeps, HttpInit } from "./fetchClip";
+
+// TEST KIT for the editor-backed tools (imported only by *.test.ts): a fake
+// editor that answers by URL and records what it was sent, and a server over
+// an empty archive reached through tools/call. No network, no timers.
+
+type Sent = { url: string; method: string; headers?: Record<string, string>; body?: unknown };
+
+export function fakeEditor(
+ answer: (s: Sent) => { status: number; body: unknown } | Error,
+ env: Record<string, string | undefined> = { WORKER_TOKEN: "tok", ARCHILYZER_EDITOR_URL: "http://editor.test/" },
+): { deps: Partial<FetchClipDeps> & Pick<FetchClipDeps, "env" | "fetch">; sent: Sent[] } {
+ const sent: Sent[] = [];
+ return {
+ sent,
+ deps: {
+ env,
+ fetch: async (url: string, init?: HttpInit) => {
+ const s: Sent = {
+ url,
+ method: init?.method ?? "GET",
+ headers: init?.headers,
+ body: init?.body ? JSON.parse(init.body) : undefined,
+ };
+ sent.push(s);
+ const a = answer(s);
+ if (a instanceof Error) throw a;
+ return { status: a.status, json: async () => a.body };
+ },
+ sleep: async () => {},
+ now: () => 0,
+ },
+ };
+}
+
+
+class EmptySource implements ShardSource {
+ readonly label = "local:/srv/fixture";
+ async loadAliases(): Promise<SearchAlias[]> {
+ return [];
+ }
+ async loadGroups(): Promise<ChannelGroups> {
+ return { groups: [], defaultGroupId: "default" };
+ }
+ async listChannels(): Promise<ChannelRef[]> {
+ return [{ key: "demo", slug: "demo", name: "demo" }];
+ }
+ async transcriptsManifest(): Promise<ChannelTranscriptsManifest> {
+ return { slugToPage: {} } as unknown as ChannelTranscriptsManifest;
+ }
+ async transcriptPage(): Promise<TranscriptDetail[]> {
+ return [];
+ }
+ publicOrigin(): string | null {
+ return null;
+ }
+ async subsManifest(): Promise<null> {
+ return null;
+ }
+ async subsPage(): Promise<[]> {
+ return [];
+ }
+ async postsManifest(): Promise<null> {
+ return null;
+ }
+ async postsPage(): Promise<[]> {
+ return [];
+ }
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ return new Map();
+ }
+}
+
+export async function connectWith(deps: Partial<FetchClipDeps>): Promise<Client> {
+ const server = createServer(SourceRegistry.forSource(new EmptySource()), { fetchClipDeps: deps });
+ const [ct, st] = InMemoryTransport.createLinkedPair();
+ const client = new Client({ name: "test", version: "0" }, { capabilities: {} });
+ await Promise.all([server.connect(st), client.connect(ct)]);
+ return client;
+}
+export const textOf = (res: unknown) =>
+ (res as { content: { type: string; text: string }[] }).content.map((c) => c.text).join("\n");
+export const isError = (res: unknown) => (res as { isError?: boolean }).isError === true;
+
diff --git a/mcp/src/editorOps.ts b/mcp/src/editorOps.ts
@@ -0,0 +1,154 @@
+import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown";
+import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
+import {
+ REQUEST_TIMEOUT_MS,
+ describeFetchError,
+ editorFromEnv,
+ isVideoId,
+ type FetchClipDeps,
+ type HttpInit,
+} from "./fetchClip";
+
+// ─── The editor's ops routes, for the MCP (release 19 A7, A9) ───
+//
+// The fetch_clip pattern, generalised: the MCP asks the LOCAL editor
+// (ARCHILYZER_EDITOR_URL + WORKER_TOKEN, the editor's own) and the editor
+// reads or writes. This process still writes nothing to an archive. Every
+// request is bounded (REQUEST_TIMEOUT_MS); an editor that does not answer is
+// said in words, never as a stack.
+//
+// Settings, storage and deletes are NOT reachable from here (operator ruling,
+// release 19): those stay `pnpm ops` / `archilyzer` only.
+
+export type EditorDeps = Pick<FetchClipDeps, "env" | "fetch" | "requestTimeoutMs">;
+
+export type EditorAnswer =
+ | { kind: "ok"; status: number; body: Record<string, unknown> }
+ | { kind: "refused"; status: number; error: string; body: Record<string, unknown> }
+ | { kind: "no_editor" }
+ | { kind: "unreachable"; url: string; message: string };
+
+export const NO_EDITOR_OPS_TEXT =
+ "no editor configured. Set ARCHILYZER_EDITOR_URL (e.g. http://localhost:3001) " +
+ "and WORKER_TOKEN (the editor's own WORKER_TOKEN) when registering the MCP " +
+ "server. A public-only setup has no editor to ask.";
+
+async function call(
+ deps: EditorDeps,
+ method: "GET" | "POST",
+ route: string,
+ body?: unknown,
+): Promise<EditorAnswer> {
+ const editor = editorFromEnv(deps.env);
+ if (!editor) return { kind: "no_editor" };
+ const timeoutMs = deps.requestTimeoutMs ?? REQUEST_TIMEOUT_MS;
+ const init: HttpInit = {
+ method,
+ headers: {
+ authorization: `Bearer ${editor.token}`,
+ ...(body !== undefined ? { "content-type": "application/json" } : {}),
+ },
+ ...(body !== undefined ? { body: JSON.stringify(body) } : {}),
+ signal: AbortSignal.timeout(timeoutMs),
+ };
+ let res;
+ try {
+ res = await deps.fetch(`${editor.url}${route}`, init);
+ } catch (e) {
+ return { kind: "unreachable", url: editor.url, message: describeFetchError(e, timeoutMs) };
+ }
+ let parsed: Record<string, unknown> = {};
+ try {
+ const j = await res.json();
+ if (j && typeof j === "object" && !Array.isArray(j)) parsed = j as Record<string, unknown>;
+ } catch {
+ /* a non-JSON answer is refused below with its status */
+ }
+ if (res.status >= 200 && res.status < 300 && parsed.ok !== false) {
+ return { kind: "ok", status: res.status, body: parsed };
+ }
+ const error = typeof parsed.error === "string" && parsed.error ? parsed.error : `HTTP ${res.status}`;
+ return { kind: "refused", status: res.status, error, body: parsed };
+}
+
+export const editorGet = (deps: EditorDeps, route: string) => call(deps, "GET", route);
+export const editorPost = (deps: EditorDeps, route: string, body: unknown) => call(deps, "POST", route, body);
+
+// An answer that is not `ok`, as one sentence for the agent.
+export function describeEditorFailure(tool: string, a: Exclude<EditorAnswer, { kind: "ok" }>): string {
+ switch (a.kind) {
+ case "no_editor":
+ return `${tool}: ${NO_EDITOR_OPS_TEXT}`;
+ case "unreachable":
+ return `${tool}: the editor at ${a.url} did not answer: ${a.message}`;
+ case "refused":
+ return a.status === 401 || a.status === 503
+ ? `${tool}: the editor refused the token (HTTP ${a.status}: ${a.error}) — WORKER_TOKEN must be the editor's own`
+ : `${tool}: the editor refused (HTTP ${a.status}): ${a.error}`;
+ }
+}
+
+// ─── get_transcript's fallback: cues off the editor's disk (A7) ───
+
+export type EditorTranscript = {
+ slug: string;
+ id: string;
+ source: "cues.json" | "built" | "vtt";
+ cuesJson: "fresh" | "stale" | "missing";
+ title?: string;
+ channel?: string;
+ uploadDate?: string;
+ duration?: number;
+ webpageUrl?: string;
+ description?: string;
+ file?: string;
+ cues: Cue[];
+};
+
+const SOURCE_WORDS: Record<EditorTranscript["source"], string> = {
+ "cues.json": "its transcript.cues.json",
+ built: "its raw transcript, normalized in memory (no cues.json written)",
+ vtt: "its English VTT alone (the record has no metadata)",
+};
+
+// Ask the editor for one video's cues off disk. Null when there is no editor
+// to ask, or it holds no such video — the caller's "video not found" stands.
+export async function editorTranscript(
+ deps: EditorDeps,
+ videoId: string,
+ channel?: string,
+): Promise<{ ok: true; t: EditorTranscript } | { ok: false; error: string | null }> {
+ if (!isVideoId(videoId)) return { ok: false, error: null };
+ const q = new URLSearchParams({ id: videoId });
+ if (channel && /^[a-z0-9][a-z0-9._-]*$/i.test(channel)) q.set("slug", channel);
+ const a = await editorGet(deps, `/api/ops/transcript?${q.toString()}`);
+ if (a.kind === "no_editor") return { ok: false, error: null };
+ if (a.kind === "refused" && a.status === 404) return { ok: false, error: null };
+ if (a.kind !== "ok") return { ok: false, error: describeEditorFailure("get_transcript", a) };
+ const b = a.body as unknown as EditorTranscript;
+ if (!Array.isArray(b.cues)) return { ok: false, error: "get_transcript: the editor's answer carried no cues" };
+ return { ok: true, t: b };
+}
+
+export function renderEditorTranscript(t: EditorTranscript, opts: { timestamps: boolean }): string {
+ const header = [
+ `NOT IN THE ARCHIVE — read from the local editor's disk (channel ${t.slug}): ${SOURCE_WORDS[t.source]}${t.file ? `, ${t.file}` : ""}.`,
+ "No moment links: the video is not published yet. Cite it by its source URL and time,",
+ "and expect the published text to differ once an index build normalizes it.",
+ ].join(" ");
+ const md = transcriptToMarkdown(
+ {
+ id: t.id,
+ title: t.title ?? t.id,
+ ...(t.channel ? { channel: t.channel } : {}),
+ channelSlug: t.slug,
+ ...(t.uploadDate ? { uploadDate: t.uploadDate } : {}),
+ ...(typeof t.duration === "number" ? { duration: t.duration } : {}),
+ ...(t.webpageUrl ? { webpageUrl: t.webpageUrl } : {}),
+ ...(t.description ? { description: t.description } : {}),
+ cues: t.cues,
+ },
+ { timestamps: opts.timestamps },
+ );
+ return `> ${header}\n\n${md}`;
+}
diff --git a/mcp/src/instructions.ts b/mcp/src/instructions.ts
@@ -160,6 +160,25 @@ function clipStep(ctx: PlanContext): string {
);
}
+// The editor-backed writes and reads (release 19 A9): the same rule as
+// fetch_clip — through the editor, never by hand, and only when asked.
+function archiveStep(): string {
+ return (
+ `**Archive work — through the editor, and only when I ask.** A video the ` +
+ `archive has not published yet is read off the editor's disk by ` +
+ `\`get_transcript\` (it says so; such a video has no moment link — cite its ` +
+ `source URL and time). When I ask for a channel to be synced, downloaded, ` +
+ `transcribed, its posts fetched or a video imported, call \`enqueue\` and ` +
+ `follow the job it names with \`get_job\` — a platform queue may hold it for ` +
+ `hours; report the job, do not wait it out. \`channel_coverage\` answers ` +
+ `what a channel holds by date and where its gaps are (a VOD mirror's videos ` +
+ `by the day they were recorded). The operator's notes on articles and ` +
+ `video projects: \`notes\` (list, read; a reply is \`umtool notes reply\`). ` +
+ `Settings, storage and deletes are not reachable from here — name the ` +
+ `\`pnpm ops\` command instead.`
+ );
+}
+
// The full sweep instructions.
export function buildSweepInstructions(
req: PromptRequest,
@@ -268,6 +287,8 @@ export function buildSweepInstructions(
steps.push(clipStep(ctx));
+ steps.push(archiveStep());
+
steps.push(
`**Finish.** Work to the end of the worklist, then write a summary ` +
`section: the scope swept, **how many of the N you actually read**, ` +
@@ -287,7 +308,8 @@ export function buildSweepInstructions(
`\`${reportPath}\`.\n\n` +
`You are the sweep engine — work the whole match set methodically, using ` +
`the transcript MCP tools for evidence and your own Write/Edit tools for ` +
- `the report. The MCP is read-only; never try to change the archive. The ` +
+ `the report. The MCP never changes the archive itself, and nothing here ` +
+ `asks the editor for a change unless I do. The ` +
`corpus holds video transcripts AND archived social posts.\n\n` +
`**Corpus: \`${ctx.corpus}\`.** Pass \`source: "${ctx.corpus}"\` on every ` +
`single call, including the ones your subagents make. This server has no ` +
@@ -363,6 +385,8 @@ export function buildAskInstructions(
);
steps.push(clipStep(ctx));
+
+ steps.push(archiveStep());
}
const numbered = steps.map((s, i) => `${i + 1}. ${s}`).join("\n\n");
@@ -370,7 +394,8 @@ export function buildAskInstructions(
`Answer this question from the transcript archive: **${question}**\n\n` +
`**Corpus: \`${ctx.corpus}\`.** Pass \`source: "${ctx.corpus}"\` on every ` +
`call — this server has no active source, and a call that omits it reads ` +
- `the server default. The MCP is read-only. The corpus holds video ` +
+ `the server default. The MCP never changes the archive itself; it asks ` +
+ `the local editor only when I ask it to. The corpus holds video ` +
`transcripts AND archived social posts.`;
const notes =
diff --git a/mcp/src/protocol.test.ts b/mcp/src/protocol.test.ts
@@ -45,6 +45,10 @@ const EXPECTED_TOOLS = [
"get_transcripts",
"get_video_metadata",
"fetch_clip",
+ "get_job",
+ "enqueue",
+ "channel_coverage",
+ "notes",
"list_sources",
"resolve_source",
"open_link",
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -88,6 +88,15 @@ import {
type FetchClipDeps,
type PollProgress,
} from "./fetchClip";
+import { editorTranscript, renderEditorTranscript, type EditorDeps } from "./editorOps";
+import {
+ ARCHIVAL_TOOLS,
+ channelCoverageTool,
+ enqueue,
+ getJob,
+ notesTool,
+ type ToolAnswer,
+} from "./archivalTools";
import { extractVideoId } from "yt-dlp-transcript-common/lib/videoId";
import {
decodeShareLink,
@@ -581,7 +590,10 @@ export const TOOLS: Tool[] = [
description:
"Fetch one video's full transcript as clean markdown (metadata header + " +
"timestamped captions). Provide the video id; optionally the channel to " +
- "skip the cross-channel lookup.",
+ "skip the cross-channel lookup. A video not yet in the archive (imported " +
+ "or transcribed since the last build) is read off the local editor's " +
+ "disk when one is configured (ARCHILYZER_EDITOR_URL + WORKER_TOKEN), " +
+ "marked as such, with no moment links.",
inputSchema: {
type: "object",
properties: {
@@ -881,6 +893,8 @@ export const TOOLS: Tool[] = [
additionalProperties: false,
},
},
+ // Archival writes through the editor (release 19 A9): archivalTools.ts.
+ ...(ARCHIVAL_TOOLS as unknown as Tool[]),
{
name: "list_sources",
description:
@@ -1156,6 +1170,11 @@ async function handleFetchClip(
return rendered.isError ? errorText(body) : text(body);
}
+// An archival tool's answer (archivalTools.ts) as a tool result.
+function fromAnswer(a: ToolAnswer): ToolResult {
+ return a.isError ? errorText(a.text) : text(a.text);
+}
+
// One MCP progress notification per poll while fetch_clip waits, when the
// client asked for progress (a `progressToken` in the request's _meta). A
// client whose request timeout resets on progress then keeps waiting for as
@@ -1257,7 +1276,7 @@ export function createServer(
case "enumerate_matches":
return handleEnumerateMatches(source, args);
case "get_transcript":
- return handleGetTranscript(source, args);
+ return handleGetTranscript(source, args, fetchClipDeps);
case "get_transcripts":
return handleGetTranscripts(source, args);
case "get_post":
@@ -1274,6 +1293,14 @@ export function createServer(
fetchClipDeps,
progressNotifier(ctx.mcpReq._meta?.progressToken, ctx.mcpReq.notify),
);
+ case "get_job":
+ return fromAnswer(await getJob(args, fetchClipDeps));
+ case "enqueue":
+ return fromAnswer(await enqueue(args, fetchClipDeps));
+ case "channel_coverage":
+ return fromAnswer(await channelCoverageTool(args, fetchClipDeps));
+ case "notes":
+ return fromAnswer(await notesTool(args, fetchClipDeps));
case "list_sources":
return handleListSources(registry, resolved);
case "resolve_source":
@@ -2410,15 +2437,26 @@ async function handleGetThread(
async function handleGetTranscript(
source: ShardSource,
args: Record<string, unknown>,
+ editorDeps?: EditorDeps,
): Promise<ToolResult> {
const videoId = String(args.video_id ?? "").trim();
if (!videoId) return errorText("video_id is required");
- const found = await findVideo(
- source,
- videoId,
- typeof args.channel === "string" ? args.channel : undefined,
- );
- if (!found) return errorText(`video not found: ${videoId}`);
+ const channelArg = typeof args.channel === "string" ? args.channel : undefined;
+ const found = await findVideo(source, videoId, channelArg);
+ if (!found) {
+ // NOT IN THE ARCHIVE: a video imported or transcribed since the last
+ // build. With a local editor configured, its cues are read off the
+ // editor's disk (release 19 A7) — the primary transcript only.
+ const askedTrack = typeof args.track === "string" && args.track.trim() !== "";
+ if (editorDeps && !askedTrack) {
+ const fromEditor = await editorTranscript(editorDeps, videoId, channelArg);
+ if (fromEditor.ok) {
+ return text(renderEditorTranscript(fromEditor.t, { timestamps: args.timestamps !== false }));
+ }
+ if (fromEditor.error) return errorText(`video not found in the archive: ${videoId}\n\n${fromEditor.error}`);
+ }
+ return errorText(`video not found: ${videoId}`);
+ }
const { record, ch } = found;
const link: LinkableHit = {
slug: record.slug,
diff --git a/plans/release-19.md b/plans/release-19.md
@@ -242,6 +242,103 @@ merged in before this record. Scratch files `a-*` in the job's `tmp`.
loop (a re-run after its timeout gets the job already fetching), and `pnpm ops job wait <id>` waits with no timeout,
printing the queue position.
+#### Slices A6–A9, as shipped (2026-10-10, overnight Track A)
+
+Branch `r20/a6-a9` (worktree `r20-a6-a9`, ports editor 4001, test 4011, export 4010) off `r20/integration`
+`5b690d2e`, one Opus implementer, A6–A9 in order; `r20/integration` (release 20 D1) merged in before A9. Scratch
+files `a-*` in the job's `tmp`.
+
+| commit | slice | one line |
+|---|---|---|
+| `cf0bb7a6` | A6 | `import-archive-org {items \| query}`: one job over many items, held records skipped, RESTRICTED flagged, 401/403 storms stop; `remote-listing` / `get remote-listing <slug>` for Odysee/BitChute |
+| `f83d17b1` | — | a `usage()` paragraph for the twelve actions that had none (C2's found-and-left) |
+| `890b3355` | A7 | `build-cues {slug, ids?, force?}`; `GET /api/ops/transcript` / `get transcript <id>`; the MCP's `get_transcript` falls back to the editor |
+| `1b08ccc0` | A8 | `publish {verb: "build", allowMissingMedia}`; `archilyzer publish build <id> [--allow-missing-media] [--out <dir>]` |
+| `79b62feb` | — | `r20/integration` merged in (release 20 D1's `recordedDate.ts`) |
+| `d9318dbe` | A9 | MCP `get_job`, `enqueue`, `channel_coverage`, `notes`; `GET /api/ops/coverage` / `get coverage <slug>`; instructions, README, ENVIRONMENT |
+
+#### Slice A6, as shipped — imports and remote listings
+
+- **`import-archive-org` takes three shapes** (route, `importArchiveOrgAction`): `{item, files | match}` as before;
+ `{items: ["<id>" | {item, files? | match?}, …]}` (at most 500; a bare id takes the top-level `match`, else every
+ media original, each one record); `{query, limit?}` — archive.org's advanced search (`archiveOrgSearchUrl`: `q`,
+ `fl[]=identifier,title`, `sort[]=identifier asc`, `rows` ≤ 500, one request through the polite client's chain,
+ gap and backoff), its first `limit` (100) items. Every item is validated before any job; a stray shape is refused
+ by name.
+- **One job, the politeness shared** (`runArchiveOrgBatchImport`, `controller/archiveOrgImport.ts`): the download
+ gap (`archiveOrgGapMs`) before every download after the first across the whole batch, the consecutive-failure
+ count, the rate-limit stop (the rest `notReached`; a re-run resumes), plus **an inter-item gap**
+ (`ARCHIVE_ORG_ITEM_GAP_SECONDS` = 3 s, up to half again at random) before each item's metadata request, on top of
+ the client's 2 s. An item archive.org has no record of is `missing` and passed over. `runArchiveOrgImport` (one
+ item) is the batch of one, with its old result shape plus `restricted`.
+- **Held is skipped**: `isHeldArchiveOrgRecord` = `destinationExists` (what every download path asks) **or** a
+ saved-video pointer (`isSavedVideo`) — a record whose media lives in the saved-container tier is never fetched again.
+- **Restricted** (`archiveOrgRestriction`): an item carrying `access-restricted-item`, else each file marked
+ `private` — both answer a download with 401/403. A dry run lists each file as `held`, `RESTRICTED` or
+ `would get`; an import skips restricted files. A download answered 401/403 (`isRefusalError`: the client's
+ "archive.org answered HTTP 40[13]") marks the rest of its item restricted and is not a failure streak; **three
+ such items in a row stop the job** (`refusedStorm`) and back the platform off as a rate limit does. The log ends
+ with `summary: {dryRun, items, planned, imported, held, restricted: [{identifier, files}], failed, missing,
+ notReached, stopped?}`. A dry run is refused, as an import is, while archive.org is held or cooling down (it asks
+ archive.org for metadata and searches).
+- **`remote-listing {slug}`** (`controller/remoteListing.ts`, job kind `remote-listing`: platform queue, `needsText`,
+ not ingest): an Odysee or BitChute channel's own URL, one `fetchFlatPlaylistUrls` read (the platform's paced args)
+ on `platform:<p>` — one stream with every other request there — refused while the platform is held or cooling
+ down, after the platform's import floor (`waitForPlatformGap`), noting the floor after; an
+ `EnumerationIncompleteError` (429) backs the platform off. **Nothing is written** (no roster merge, no
+ maybe-missing). The log ends with `@@remote-listing {slug, platform, url, listedAt, listed, unparsed, held,
+ notHeld: [{id, url}], heldNotListed: [id]}`; `pnpm ops get remote-listing <slug>` posts it, waits and prints that
+ JSON on stdout (the transcribe result-marker path). Any other platform is refused, naming its sync.
+
+#### Slice A7, as shipped — cues without an index
+
+- **`build-cues {slug, ids?, force?, queueKey?}`** → the digest card's `normalizeChannelAction` (now `{ids, force}`;
+ `normalizeAllTranscripts` gains `videoIds` and `force`), on the channel's queue; every id must be held — a stray is
+ refused, named, before any job.
+- **`GET /api/ops/transcript?id[&slug]`** (`controller/videoCues.ts`, `pnpm ops get transcript <id> [--slug]`): one
+ video's cues off disk, nothing written — a fresh `transcript.cues.json` (`isCuesJsonFresh`), else what normalize
+ would write (`buildNormalizedTranscript`, factored out of `normalizeTranscript`, which now calls it), else the
+ English VTT alone when there is no metadata. Without `slug` the channel holding `data/<id>/` is found; two
+ holders are a 409 naming them. Text-guarded (503 when the drive is not answering).
+- **The MCP's `get_transcript`**, for a video its archive does not hold and no `track` asked, asks that route
+ through the configured editor (`mcp/src/editorOps.ts`, the fetch_clip env and timeout) and renders the cues headed
+ `NOT IN THE ARCHIVE — read from the local editor's disk`, with no moment links. No editor, or a 404, keeps
+ "video not found".
+
+#### Slice A8, as shipped — publish edges
+
+- `POST /api/ops/publish {verb: "build", allowMissingMedia: true}` carries the flag through `TargetAsk.build` →
+ the plan step → the stage request (`--allow-missing-media`); the docker runner refuses it (its builds do not take
+ it).
+- **`pnpm ops lane` covers `publish`** — shipped in A4 (`lane` takes `publish`); nothing added.
+- **`archilyzer publish build <id> [--allow-missing-media] [--out <dir>]`** (`common/bin/publish.ts`
+ `layOutBundle`): after the build — or the no-op of a fresh one — the bundle is laid out in `<dir>` with **hard
+ links** where the filesystem allows (a build installs a new bundle and never edits one in place), else copies;
+ symlinks kept. `<dir>` is emptied first (its contents) and refused **before any build** where a local deploy's
+ destination would be (`localDestProblem`); `--out` with `all` is refused. Not a deploy: nothing recorded.
+
+#### Slice A9, as shipped — MCP archival writes
+
+- On the fetch_clip pattern, one existing route each (`mcp/src/archivalTools.ts`): **`get_job`** (`GET
+ /api/ops/job/<id>?tail=N`: status, times, exit code, queue place and head, the log tail); **`enqueue`** (`kind`:
+ `sync` (`full`), `download-missing`, `retry-bucket` (`bucket`, `ids` — the way to download chosen videos),
+ `transcribe-bucket` (`ids`), `fetch-posts` (`full`, `older`), `import-video` (`url`); a key a kind does not take
+ is refused before any request; answers the job id and how to follow it); **`channel_coverage`** (`GET
+ /api/ops/coverage`); **`notes`** (umtool via `UMTOOL_URL`: `list` from `/api/browse/decisions`, `read` from
+ `/api/notes/context`).
+- **`GET /api/ops/coverage?slug[&gapDays][&from][&to][&list]`** (`controller/channelCoverage.ts`, `pnpm ops get
+ coverage <slug>`): every held video dated off disk with `coverageDate()` — the recorded date when the channel has a
+ `recordedDate` title rule and the title yields one, else the upload date — a small `metadata.info.json` parsed
+ whole, a big one read at both ends (16 KB head for `title`, 8 KB tail for `upload_date`); `undated` named, never
+ guessed; per year and month; gaps longer than `gapDays` (30). Text-guarded; nothing written.
+- **Ruling recorded: a notes reply is not written by the MCP.** umtool's `/api/notes` stamps every write as the
+ operator's (its own rule: no way to claim to be an agent there), so `notes` `reply` answers with the `umtool
+ notes reply <id> "<text>" [--resolve]` command instead of posting.
+- Settings, storage and deletes are not reachable from the MCP (ruled). `mcp/src/instructions.ts` gains the
+ archive-work step (both plans; only when asked; report the job, do not wait it out) and both heads say the MCP
+ changes nothing itself; README tool table; `UMTOOL_URL`, `WORKER_TOKEN`, `ARCHILYZER_EDITOR_URL` name their MCP
+ readers (ENVIRONMENT.md regenerated). `channel-config`'s help names `recordedDateTitlePattern`.
+
#### Track A gates (A1–A5, on the merged tip)
- `pnpm -r --no-bail --workspace-concurrency=1 exec tsc --noEmit` — clean before every commit and on the merged tip
@@ -271,6 +368,19 @@ merged in before this record. Scratch files `a-*` in the job's `tmp`.
ending `[stage] update-index _index: Done` — the index stage, which this branch does not touch, finishing late
under load.
+**Tonight, A6–A9 (on `d9318dbe`):**
+
+- `pnpm -r --no-bail --workspace-concurrency=1 exec tsc --noEmit` — clean before every commit (A6 94 s, A7, A8, A9).
+- **common** 3672/3672 (+15 in the files this track added or extended: archiveOrgImport 7 → 12, remoteListing 4,
+ channelCoverage 3, publish-out 3; `_cli`'s publish-row case updated). **editor unit** 237/237 (+9 here:
+ import-archive-org 1, remote-listing 2, build-cues 3, publish +1, coverage 2). **test:scripts** 732: 729 pass, 3 skip
+ (`archilyzer-ops.test.mjs` 59 → 63). **mcp** 303/303 (298 → 303: editorOps 5, archivalTools 5, protocol's tool
+ list). `docs env|files|cli --check` 0.
+- **e2e** `ops-api.spec` alone, start mode (first build here 41 s, after a 1 min 37 s wait for the heavy slot):
+ **13 passed, 27.7 s**.
+- Not run (the brief): the full editor suite (the orchestrator's); an editor build of its own beyond the e2e start
+ build; a live `/ask`-style MCP session against the live editor.
+
### Track B
Branch `worktree-agent-a8b654c51bf472562` off `4cffda3f` (r18/integration with main merged), one Opus implementer,
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -15,6 +15,9 @@
// pnpm ops get tags [<tagId>]
// pnpm ops get job <id> [--tail [<lines>]]
// pnpm ops get jobs [--active | --failed] [--kind <kind>] [--slug <slug>] [--limit <n>]
+// pnpm ops get remote-listing <slug>
+// pnpm ops get transcript <videoId> [--slug <slug>]
+// pnpm ops get coverage <slug>
// pnpm ops job <cancel|drain|promote|force-release|retry> <id>... [--wait]
// pnpm ops job retry-failed | job wait <id>...
// pnpm ops list
@@ -41,6 +44,10 @@
// pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait
// pnpm ops import-video --json '{"slug":"demo-archive","url":"https://archive.org/details/example-item"}'
// pnpm ops import-archive-org --json '{"slug":"demo-archive","item":"example-item","match":"\\.mp4$"}' --wait
+// pnpm ops import-archive-org --json '{"slug":"demo-archive","query":"collection:example-collection","dryRun":true}' --wait
+// pnpm ops get remote-listing demo-odysee
+// pnpm ops build-cues --json '{"slug":"demo-yt","ids":["<videoId>"]}' --wait
+// pnpm ops get transcript <videoId> --slug demo-yt
// pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}'
// pnpm ops channel-config --json '{"slug":"x","sites":[{"siteId":"anilyzer"}]}'
// pnpm ops channel-config --json '{"slug":"x","sites":[],"excludeFromBuild":true}'
@@ -160,6 +167,31 @@ const GETTERS = {
scheduler: () => "/api/ops/scheduler",
// One channel's cleanup row: what each sweep would reclaim, what holds the rest.
cleanup: (slug) => `/api/ops/cleanup/${encodeURIComponent(slug)}`,
+ // What a channel holds, by date, and its gaps (release 19 A9): each video
+ // dated by its recorded date when the channel has a title rule, else upload.
+ coverage: (slug) => `/api/ops/coverage?slug=${encodeURIComponent(slug)}`,
+ // One video's cues off disk, with no index (release 19 A7): a fresh
+ // cues.json, else what normalize would write, else the VTT alone. --slug
+ // names the channel (else the one holding data/<id>/).
+ transcript: (id, { jobs }) => {
+ const q = new URLSearchParams({ id });
+ if (jobs.slug) q.set("slug", jobs.slug);
+ return `/api/ops/transcript?${q.toString()}`;
+ },
+};
+
+// READS THAT ARE JOBS: the answer needs a request upstream, so the editor
+// runs it on the platform's queue and the job's log carries the result. `get`
+// POSTs the action, waits, and prints the result on stdout (--wait-timeout
+// bounds the wait).
+const GET_JOBS = {
+ // An Odysee or BitChute channel's listing, diffed against what it holds
+ // (release 19 A6): {notHeld: [{id, url}], heldNotListed: [id], …}.
+ "remote-listing": (slug) => ({
+ path: "/api/ops/remote-listing",
+ body: { slug },
+ resultMarker: REMOTE_LISTING_RESULT_MARKER,
+ }),
};
// Nouns whose read takes no argument. `get channel` without a slug is a
@@ -179,6 +211,7 @@ const GET_ARG_OPTIONAL = new Set([
// The flags each noun takes beyond the shared ones; any other is refused.
const GET_FLAGS = {
+ transcript: ["slug"],
channel: ["counts"],
job: ["tail"],
jobs: ["active", "failed", "kind", "slug", "limit"],
@@ -210,9 +243,14 @@ const ACTIONS = [
// subtitles, no media; the rewrite lands in metadata.history.json.
"refresh-metadata",
"import-video",
- // Chosen media files of ONE archive.org item ({slug, item, files: [...] |
- // match: "<regex>", dryRun?}): one job, one file at a time, paced.
+ // archive.org media into a channel: one item ({slug, item, files: [...] |
+ // match: "<regex>"}), many ({items: [...]}) or a search ({query, limit?});
+ // dryRun? lists held / RESTRICTED / would-get. One job, one file at a time,
+ // paced, a pause between items.
"import-archive-org",
+ // An Odysee or BitChute channel's listing diffed against what it holds
+ // ({slug}); `get remote-listing <slug>` is the same, waited on.
+ "remote-listing",
// A channel's held videos given their media from a LOCAL archive — a
// directory, a zip read in place, a 7z ({slug, source, items?, match?,
// createRecords?, replace?, dryRun?}): one job, nothing fetched.
@@ -220,6 +258,9 @@ const ACTIONS = [
// A channel's saved containers remuxed losslessly into browser-playable
// copies, one torrent each ({slug, ids?, trackers?, root?, dryRun?}).
"prepare-playable",
+ // Write transcript.cues.json from each video's raw transcript ({slug, ids?,
+ // force?}) — what a reader with no index build wants.
+ "build-cues",
// A podcast channel's records completed from its RSS feed ({slug, dryRun?}):
// one fetch of the feed, no media.
"feed-metadata",
@@ -306,6 +347,9 @@ const ACTIONS = [
// compact JSON. The same string is TRANSCRIBE_RESULT_MARKER in
// common/controller/transcribeFile.ts.
export const TRANSCRIBE_RESULT_MARKER = "@@transcribe-result ";
+// The same for a remote listing (REMOTE_LISTING_RESULT_MARKER in
+// common/controller/remoteListing.ts).
+export const REMOTE_LISTING_RESULT_MARKER = "@@remote-listing ";
// The provenance a tag write from this CLI carries. Everything else ignores it.
function agentSource() {
@@ -429,9 +473,25 @@ export function parseArgs(argv) {
}
if (positional[0] === "get") {
const noun = positional[1];
+ if (noun && GET_JOBS[noun]) {
+ if (!positional[2]) return { error: `get ${noun}: needs an argument` };
+ if (used.size) {
+ return { error: `get ${noun} does not take ${[...used].map((f) => `--${f}`).join(", ")}` };
+ }
+ const job = GET_JOBS[noun](positional[2]);
+ return {
+ method: "POST",
+ path: job.path,
+ body: job.body,
+ resultMarker: job.resultMarker,
+ wait: true,
+ quiet,
+ waitTimeout,
+ };
+ }
if (!noun || !GETTERS[noun]) {
return {
- error: `get: unknown noun "${noun ?? ""}" — known: ${Object.keys(GETTERS).join(", ")}`,
+ error: `get: unknown noun "${noun ?? ""}" — known: ${[...Object.keys(GETTERS), ...Object.keys(GET_JOBS)].join(", ")}`,
};
}
if (!positional[2] && !GET_ARG_OPTIONAL.has(noun)) {
@@ -511,6 +571,7 @@ export function parseArgs(argv) {
...(action === "tag-videos" ? { defaultSource: agentSource() } : {}),
// A job whose log carries a RESULT, which --wait prints on stdout.
...(action === "transcribe" ? { resultMarker: TRANSCRIBE_RESULT_MARKER } : {}),
+ ...(action === "remote-listing" ? { resultMarker: REMOTE_LISTING_RESULT_MARKER } : {}),
wait,
quiet,
waitTimeout,
@@ -593,6 +654,9 @@ export function usage() {
" pnpm ops get settings [<key>]",
" pnpm ops get storage | sites | workers | auto-queue | scheduler",
" pnpm ops get cleanup <slug>",
+ " pnpm ops get remote-listing <slug> [--wait-timeout <seconds>]",
+ " pnpm ops get transcript <videoId> [--slug <slug>]",
+ " pnpm ops get coverage <slug>",
" pnpm ops list",
"",
`Actions: ${ACTIONS.join(", ")}`,
@@ -631,10 +695,35 @@ export function usage() {
" reclaim (they overlap — never add them), what holds the rest, and the",
" failed-transcriptions count. measured: false means unknown, not zero.",
"",
+ 'remote-listing lists an Odysee or BitChute channel upstream and diffs it',
+ ' against what it holds: {"slug"}. One flat-playlist read on the platform\'s',
+ " queue (refused while it is held or cooling down; a 429 backs it off),",
+ " nothing written. `get remote-listing <slug>` waits for the job and prints",
+ " {listed, held, notHeld: [{id, url}], heldNotListed: [id], ...} on stdout.",
+ "",
+ 'build-cues writes each video\'s transcript.cues.json from its raw',
+ " transcript (the caption-track rule for VTTs), as the digest card's",
+ ' Normalize button does: {"slug"}, "ids": [...] for those videos only',
+ ' (every one held), "force": true to rewrite a fresh one. A job on the',
+ " channel's queue. The file every reader without an index build wants.",
+ "",
+ "get coverage <slug> is what a channel holds by date: held, dated (by",
+ " recorded date — the channel's recordedDate title rule — else by upload",
+ " date), undated, first and last day, per year and month, and every gap",
+ " over 30 days with nothing held. Off disk; nothing written. (The MCP's",
+ " channel_coverage takes gap days, a date window and a video list.)",
+ "",
+ "get transcript <videoId> [--slug <slug>] reads one video's cues off disk",
+ " with no index: a fresh cues.json, else what build-cues would write, else",
+ " the English VTT alone — {source, cuesJson, title?, ..., cues}. Nothing is",
+ " written. Without --slug, the channel holding data/<id>/ is found.",
+ "",
'publish runs publish stages on the editor\'s publish queue, one at a time,',
' under one run id: {"verb": …}. "index" updates the index; "build" builds',
' "siteId"/"siteIds" (forced; the index first when stale; "runner":',
- ' "docker" builds every site in containers); "deploy" ships their built',
+ ' "docker" builds every site in containers; "allowMissingMedia": true lets',
+ ' a report citation with no prepared media through compose, local builds',
+ ' only); "deploy" ships their built',
' bundles (production, "preview": "<branch>", or "to": "local"; "force"',
' redeploys a bundle already shipped there); "hub" / "homepage" build',
' them, {"deploy": true} deploys after; "now" is Publish now (the stale',
@@ -702,11 +791,79 @@ export function usage() {
' free space both sides) and moves nothing — though, as on the Storage',
' panel, a channel never tiered is first tiered in place on the corpus disk.',
"",
+ 'channel-priority sets channels\' priority, as the /channels deck\'s tier',
+ ' control does: {"slugs", "tier": "normal" | "low" | "paused"} sets the',
+ ' base tier; with "operation" (sync, transcription, download, digest,',
+ ' backfill) it pins that operation\'s tier, "tier": null clearing it back',
+ ' to the base; "preset": "sync-only" | "clear" is the deck\'s two shortcuts',
+ ' and takes no tier. A manual change clears an automatic pause.',
+ "",
+ 'sync lists a channel and downloads what is new, as its Sync button does:',
+ ' {"slug"}. "full": true runs the periodic whole-listing sweep now (else',
+ ' the configured cadence decides); "queueKey" picks another queue. On the',
+ " channel's platform queue, paced like its downloads.",
+ "",
+ 'download-missing downloads every listed video the channel does not hold:',
+ ' {"slug"}. "ignoreArchive": true drops the download archive so listed',
+ ' videos it names are fetched again (recovery from a stale archive);',
+ ' "abortOnError": false carries on past a failure (default: stop at the',
+ ' first that is not one video\'s own). Paced on the platform\'s queue.',
+ "",
+ 'metadata-scan reads every listed video\'s title, date and description',
+ ' into the channel\'s scan file, fetching no media: {"slug"}. Not held by',
+ " the download pause and needs no disk floor; one request per video, so",
+ " it respects the platform's cooldown and records one when pushed back.",
+ "",
+ 'import-video imports one video by URL into a channel: {"slug", "url"}. An',
+ " archive.org URL becomes its canonical item or file (an item of several",
+ " media files is refused — see import-archive-org); a BitChute or Odysee",
+ " URL runs on that platform's queue at its pace, refused while it is held",
+ " or cooling down, and refused when already on disk; a Wayback capture is",
+ " named by what it copies.",
+ "",
+ 'feed-metadata completes a podcast channel\'s records from its RSS feed: one',
+ ' fetch of the channel\'s url, then title, date, description and duration',
+ ' into each record that lacks them: {"slug"}. "dryRun": true counts',
+ " matched, unmatched and already complete, and writes nothing.",
+ "",
+ 'refresh-report regenerates a channel\'s report (the buckets and counts its',
+ ' page shows) on the one serial report queue: {"slug"} answers the job',
+ ' queued, or the one already waiting ("started": false); {"all": true}',
+ " queues every channel and answers {queued, jobIds, skipped}.",
+ "",
+ 'relocate-back moves channels\' media back onto the corpus disk, one job',
+ ' per channel on the relocation queue: {"slugs"}. A busy channel is',
+ ' answered in "skipped"; --wait follows the jobs started.',
+ "",
+ 'evict-clips deletes cached clip windows (data/<id>/clips/) older than a',
+ ' number of days, as /storage\'s button does: {"olderThanDays", "slug"?,',
+ ' "dryRun"?}. BY AGE: nothing knows whether a umtool report still cites a',
+ " window; an evicted window is fetched again when asked for.",
+ "",
+ 'tags edits the curated-tag vocabulary through its one writer: {"op":',
+ ' "define", "tag": {...}} (the WHOLE definition: id, label, rules and',
+ ' "sites", the sites it exists on, absent = every site) or {"op":',
+ ' "remove", "tag": "<id>"}. `get tags [<id>]` reads it.',
+ "",
+ 'tag-videos pins, unpins, suppresses or unsuppresses one tag on many videos',
+ ' in ONE write: {"tag", "op": "add" | "remove" | "suppress" | "unsuppress",',
+ ' "videos": [{"slug", "id"}, ...]}. "remove" unpins only — a rule\'s hit',
+ ' survives; "suppress" rejects it. The write records who asked',
+ ' (ARCHILYZER_AGENT); a big list goes in --file.',
+ "",
+ 'keep-videos sets the "Do not clean" marker on every held video of a',
+ ' channel whose title or description matches a download-filter pattern:',
+ ' {"slug", "match", "fields"?: ["title" | "description"], "note"?,',
+ ' "dryRun"?}. A match not held is reported in "notDownloaded", never',
+ " created; a fresh channel wants metadata-scan first.",
+ "",
'channel-config changes a channel as its Configure form does: {"slug"} and',
' any of "patch" (form field names; "" clears one), "sites" (the WHOLE',
' membership set: [{"siteId", "groupId"? | "newGroupName"?}], [] = on no',
' site; an unknown site id is refused), "excludeFromBuild" and',
- ' "excludeFromCleanup" (set to the value given, not toggled).',
+ ' "excludeFromCleanup" (set to the value given, not toggled). A VOD mirror\'s',
+ ' recorded date: "patch": {"recordedDateTitlePattern": "<regex with named',
+ ' groups year, month, day>"} (CHANNEL.md, recordedDate).',
"",
'create-channel is the New channel form: {"fields": {"name", "handling":',
' "youtube"|"transcribe", "url"?, "platform"?, "sourceKind"?, "postFetcher"?,',
@@ -773,6 +930,19 @@ export function usage() {
" queue; a low disk or a rate limit stops it, and running the same body",
" again resumes — saved videos are skipped.",
"",
+ 'import-archive-org imports archive.org media into a channel, as one job on',
+ ' archive.org\'s queue: {"slug", "item", "files": [...] | "match": "<regex>"}',
+ ' for one item; {"slug", "items": ["<id>" | {"item", "files"? | "match"?},',
+ ' ...]} for many (a bare id takes "match", else every media original); or',
+ ' {"slug", "query": "<archive.org search>", "limit"?: 100} for the first',
+ ' items a search finds (at most 500). One file at a time, a jittered pause',
+ ' between files and between items; a record already held (on disk, or in',
+ ' the saved-video store) is skipped. "dryRun": true lists each file as',
+ ' held, RESTRICTED (archive.org marks it not for download: a fetch would',
+ ' answer 401/403) or would get, and fetches nothing. A 401/403 skips the',
+ ' rest of its item; three such items in a row, a 429 or three failures in',
+ ' a row stop the job. The log ends with a summary: line; a re-run resumes.',
+ "",
'attach-media copies each held video\'s file out of a LOCAL archive into the',
' saved-video store, as its source container (nothing fetched): {"slug",',
' "source"} — an absolute path to a directory, a .zip (read in place) or a',
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -6,6 +6,7 @@
import assert from "node:assert/strict";
import test from "node:test";
import {
+ REMOTE_LISTING_RESULT_MARKER,
TRANSCRIBE_RESULT_MARKER,
editorEnvFiles,
followJob,
@@ -783,3 +784,50 @@ test("the archival writes are POSTs to their routes, each named in the usage", (
assert.match(usage(), /"dryRun": true answers each channel's preview/);
assert.match(usage(), /"publish", "enabled"\?, "held"\?, "action"\?/);
});
+
+test("get remote-listing posts the job, waits, and carries the result marker", () => {
+ const r = parseArgs(["get", "remote-listing", "demo-odysee"]);
+ assert.equal(r.method, "POST");
+ assert.equal(r.path, "/api/ops/remote-listing");
+ assert.deepEqual(r.body, { slug: "demo-odysee" });
+ assert.equal(r.wait, true);
+ assert.equal(r.resultMarker, REMOTE_LISTING_RESULT_MARKER);
+ assert.equal(REMOTE_LISTING_RESULT_MARKER, "@@remote-listing ");
+ assert.match(parseArgs(["get", "remote-listing"]).error, /needs an argument/);
+ assert.match(parseArgs(["get", "remote-listing", "x", "--counts"]).error, /does not take --counts/);
+ assert.equal(parseArgs(["get", "remote-listing", "x", "--wait-timeout", "60"]).waitTimeout, 60);
+ // The action form, waited on, prints the result too.
+ const a = parseArgs(["remote-listing", "--json", '{"slug":"x"}', "--wait"]);
+ assert.equal(a.path, "/api/ops/remote-listing");
+ assert.equal(a.resultMarker, REMOTE_LISTING_RESULT_MARKER);
+});
+
+test("usage documents import-archive-org's three shapes and the remote listing", () => {
+ const u = usage();
+ assert.match(u, /import-archive-org imports archive\.org media/);
+ assert.match(u, /"query": "<archive\.org search>"/);
+ assert.match(u, /RESTRICTED/);
+ assert.match(u, /remote-listing lists an Odysee or BitChute channel/);
+});
+
+test("build-cues is an action; get transcript reads one video's cues, --slug naming the channel", () => {
+ const b = parseArgs(["build-cues", "--json", '{"slug":"x","ids":["a"]}']);
+ assert.equal(b.path, "/api/ops/build-cues");
+ assert.deepEqual(b.body, { slug: "x", ids: ["a"] });
+ const t = parseArgs(["get", "transcript", "vid1", "--slug", "demo"]);
+ assert.equal(t.method, "GET");
+ assert.equal(t.path, "/api/ops/transcript?id=vid1&slug=demo");
+ assert.equal(parseArgs(["get", "transcript", "vid1"]).path, "/api/ops/transcript?id=vid1");
+ assert.match(parseArgs(["get", "transcript"]).error, /needs an argument/);
+ assert.match(parseArgs(["get", "transcript", "v", "--counts"]).error, /does not take --counts/);
+ assert.match(usage(), /build-cues writes each video's transcript\.cues\.json/);
+});
+
+test("get coverage reads one channel's coverage", () => {
+ const r = parseArgs(["get", "coverage", "demo"]);
+ assert.equal(r.method, "GET");
+ assert.equal(r.path, "/api/ops/coverage?slug=demo");
+ assert.match(parseArgs(["get", "coverage"]).error, /needs an argument/);
+ assert.match(usage(), /get coverage <slug> is what a channel holds by date/);
+ assert.match(usage(), /recordedDateTitlePattern/);
+});