commit 2baea15e823029add6e25dc281eaaba8575c09c9
parent 6bd5722378cd6cf08623a8962f7a9582865c9269
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 10 Oct 2026 02:33:53 -0400
ops: cues without an index — build-cues, get transcript, and the MCP's get_transcript falls back to the editor (release 19 A7)
build-cues {slug, ids?, force?} writes transcript.cues.json through the
digest card's Normalize action, narrowed to held ids (a stray refused,
named) on the channel's queue. GET /api/ops/transcript?id[&slug]
(`pnpm ops get transcript <id> [--slug]`) reads one video's cues off disk
with no index and writes nothing: a fresh cues.json, else what normalize
would write (buildNormalizedTranscript, factored out of normalizeTranscript),
else the English VTT alone. The MCP's get_transcript, for a video its
archive does not hold, asks that route through the configured editor (the
fetch_clip pattern) and marks the answer as read off the editor's disk, with
no moment links.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
16 files changed, 805 insertions(+), 42 deletions(-)
diff --git a/COMMANDS.md b/COMMANDS.md
@@ -90,6 +90,7 @@ Drives a running editor over HTTP (`/api/ops/*`, the same actions its pages run)
| `remote-listing` | remote-listing lists an Odysee or BitChute channel upstream and diffs it against what it holds: {"slug"}. One flat-playlist read on the platform's queue (refused while it is held or cooling down; a 429 backs it off), nothing written. `get remote-listing <slug>` waits for the job and prints {listed, held, notHeld: \[{id, url}\], heldNotListed: \[id\], ...} on stdout. | |
| `attach-media` | attach-media copies each held video's file out of a LOCAL archive into the saved-video store, as its source container (nothing fetched): {"slug", "source"} — an absolute path to a directory, a .zip (read in place) or a .7z. The id is the folder's trailing "(<id>)", else the file's yt-dlp suffix; "items": \[{"id", "path"}\] names exact files (path inside the source), "match" narrows the folders by regex. "createRecords": true writes a record for a video the channel does not hold; "replace": true re-attaches over a saved container. "dryRun": true logs each folder's class (attach, not-held, already-attached, lost, no-media, unmatched, ambiguous) and the held videos with no media in it, and writes nothing. The log ends with a summary: line; a re-run resumes. | |
| `prepare-playable` | prepare-playable remuxes each of a channel's saved containers, losslessly (-c copy), into a browser-playable mp4 (+faststart) or webm, and makes one single-file torrent per copy, under playable/ beside the saved-video store: {"slug"}. "ids" narrows it; "trackers": \[...\] is each .torrent's announce list (none by default; the infohash does not depend on it); "root" names another playable root. A video prepared from the same source (by sha256) is skipped; "dryRun": true logs each decision. | |
+| `build-cues` | build-cues writes each video's transcript.cues.json from its raw transcript (the caption-track rule for VTTs), as the digest card's Normalize button does: {"slug"}, "ids": \[...\] for those videos only (every one held), "force": true to rewrite a fresh one. A job on the channel's queue. The file every reader without an index build wants. | `pnpm ops build-cues --json '{"slug":"demo-yt","ids":["<videoId>"]}' --wait` |
| `feed-metadata` | feed-metadata completes a podcast channel's records from its RSS feed: one fetch of the channel's url, then title, date, description and duration into each record that lacks them: {"slug"}. "dryRun": true counts matched, unmatched and already complete, and writes nothing. | `pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait` |
| `refresh-report` | refresh-report regenerates a channel's report (the buckets and counts its page shows) on the one serial report queue: {"slug"} answers the job queued, or the one already waiting ("started": false); {"all": true} queues every channel and answers {queued, jobIds, skipped}. | `pnpm ops refresh-report --json '{"all":true}'` |
| `sync` | sync lists a channel and downloads what is new, as its Sync button does: {"slug"}. "full": true runs the periodic whole-listing sweep now (else the configured cadence decides); "queueKey" picks another queue. On the channel's platform queue, paced like its downloads. | `pnpm ops sync --json '{"slug":"the-quartering"}' --wait` |
@@ -143,6 +144,7 @@ Usage: pnpm ops <action> [--json '<body>' | --file <path>] [--wait]
pnpm ops get storage | sites | workers | auto-queue | scheduler
pnpm ops get cleanup <slug>
pnpm ops get remote-listing <slug> [--wait-timeout <seconds>]
+ pnpm ops get transcript <videoId> [--slug <slug>]
pnpm ops list
--wait follows the job's log and survives a poll that fails (a busy
@@ -179,6 +181,11 @@ get cleanup <slug> is one channel's /cleanup row: what each sweep would
reclaim (they overlap — never add them), what holds the rest, and the
failed-transcriptions count. measured: false means unknown, not zero.
+get transcript <videoId> [--slug <slug>] reads one video's cues off disk
+ with no index: a fresh cues.json, else what build-cues would write, else
+ the English VTT alone — {source, cuesJson, title?, ..., cues}. Nothing is
+ written. Without --slug, the channel holding data/<id>/ is found.
+
"preview": "<branch>" on deploy-site or build-deploy makes it a Cloudflare
Pages PREVIEW instead of production: the same bundle goes to a branch
alias, https://<branch>.<project>.pages.dev, and the live site is left
diff --git a/OPERATING.md b/OPERATING.md
@@ -105,6 +105,9 @@ pnpm ops transcribe-bucket --json '{"slug":"example"}' --wait
pnpm ops transcribe --json '{"path":"/abs/clip.mp4","start":120,"end":150,"words":true}' --wait
```
+- A transcript the index has not caught up with: `pnpm ops get transcript <videoId> --slug <slug>` reads its cues
+ off disk (the MCP's `get_transcript` does the same through the editor), and
+ `pnpm ops build-cues --json '{"slug":"<slug>","ids":["<videoId>"]}' --wait` writes its `transcript.cues.json`.
- The transcription lane takes every channel's downloaded, untranscribed audio; `transcribe-bucket` runs one
channel's now.
- `transcribe` runs one local file (or a window of it) through the corpus's own engine and model; the result JSON
diff --git a/common/controller/normalizeAll.ts b/common/controller/normalizeAll.ts
@@ -24,6 +24,12 @@ export type NormalizeAllOptions = {
// says in its own header that it was modelled on this one, so this is the
// symmetry being completed rather than a new pattern.
channelSlugs?: string[];
+ // Restrict to these video ids within the channels walked (release 19 A7,
+ // `pnpm ops build-cues {slug, ids}`). An id the channel does not hold is
+ // passed over silently here; the action refuses it before any job.
+ videoIds?: string[];
+ // Rewrite cues.json even when it is fresh.
+ force?: boolean;
onLog?: (msg: string) => void;
signal?: AbortSignal;
concurrency?: number;
@@ -68,7 +74,10 @@ export async function normalizeAllTranscripts(
result.failed++;
continue;
}
- const videoIds = await readdir(dataDir).catch(() => [] as string[]);
+ const onlyIds = opts.videoIds ? new Set(opts.videoIds) : null;
+ const videoIds = (await readdir(dataDir).catch(() => [] as string[])).filter(
+ (id) => !onlyIds || onlyIds.has(id),
+ );
log(`Normalize ${ch.slug}: ${videoIds.length} videos`);
let wrote = 0;
let fresh = 0;
@@ -83,6 +92,7 @@ export async function normalizeAllTranscripts(
videoDir: path.join(dataDir, id),
channelSlug: ch.slug,
configName: ch.config.name,
+ ...(opts.force ? { force: true } : {}),
});
if (outcome.status === "wrote") wrote++;
else if (outcome.status === "fresh") fresh++;
diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts
@@ -86,9 +86,6 @@ export async function normalizeTranscript(
const picked = pickIndexTranscript(files);
if (!picked) return { status: "skipped", reason: "no-raw-transcript" };
- const metaPath = path.join(opts.videoDir, META_FILENAME);
- const transcriptPath = path.join(opts.videoDir, picked.filename);
-
// The same question every reader of cues.json asks (isCuesJsonFresh), so a
// normalize pass rewrites exactly the files they refuse.
const freshness = await isCuesJsonFresh(opts.videoDir);
@@ -97,6 +94,37 @@ export async function normalizeTranscript(
return { status: "fresh", cuesPath };
}
+ const built = await buildNormalizedTranscript(opts, files);
+ if (built.status === "skipped") return built;
+ const out = built.transcript;
+
+ // Compact, no trailing newline: transcript.cues.json's historical bytes.
+ await writeJsonAtomic(cuesPath, out, { indent: 0, newline: false });
+ opts.log?.(
+ `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${out.vttFile ?? out.transcriptFormat}, ${out.cues?.length ?? 0} cues)`,
+ );
+ return { status: "wrote", cuesPath };
+}
+
+// WHAT normalizeTranscript WOULD WRITE, built in memory and not written: the
+// summary from metadata.info.json and the cues of the transcript the index
+// would pick (the caption-track rule for VTTs). Shared by normalize and by a
+// reader that must not write (the ops transcript read, release 19 A7).
+export async function buildNormalizedTranscript(
+ opts: Pick<NormalizeOptions, "videoDir" | "channelSlug" | "configName" | "formatHint">,
+ known?: Awaited<ReturnType<typeof readVideoFiles>>,
+): Promise<
+ | { status: "built"; transcript: NormalizedTranscript }
+ | { status: "skipped"; reason: "no-raw-transcript" | "no-metadata" }
+> {
+ const files = known ?? (await readVideoFiles(opts.videoDir));
+ if (!files.hasMeta) return { status: "skipped", reason: "no-metadata" };
+ const picked = pickIndexTranscript(files);
+ if (!picked) return { status: "skipped", reason: "no-raw-transcript" };
+ const metaPath = path.join(opts.videoDir, META_FILENAME);
+ const transcriptPath = path.join(opts.videoDir, picked.filename);
+ const cuesPath = path.join(opts.videoDir, CUES_JSON_FILENAME);
+
const metaRaw = await readFile(metaPath, "utf8");
const parsedMeta = JSON.parse(metaRaw) as RawMetadata;
const summary = summarize(
@@ -135,23 +163,19 @@ export async function normalizeTranscript(
);
}
- const out: NormalizedTranscript = {
- version: CUES_FILE_VERSION,
- source: picked.kind,
- ...(vttFile !== undefined
- ? { vttFile, captionTrackRule: CAPTION_TRACK_RULE_VERSION }
- : {}),
- transcriptFormat,
- ...summary,
- cues,
+ return {
+ status: "built",
+ transcript: {
+ version: CUES_FILE_VERSION,
+ source: picked.kind,
+ ...(vttFile !== undefined
+ ? { vttFile, captionTrackRule: CAPTION_TRACK_RULE_VERSION }
+ : {}),
+ transcriptFormat,
+ ...summary,
+ cues,
+ },
};
-
- // Compact, no trailing newline: transcript.cues.json's historical bytes.
- await writeJsonAtomic(cuesPath, out, { indent: 0, newline: false });
- opts.log?.(
- `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${vttFile ?? transcriptFormat}, ${cues.length} cues)`,
- );
- return { status: "wrote", cuesPath };
}
// Read the format recorded in a prior transcript.cues.json, if any. This is the
diff --git a/common/controller/remoteListing.ts b/common/controller/remoteListing.ts
@@ -18,12 +18,14 @@
// platforms (lib/videoId.ts).
import path from "node:path";
-import { readdir } from "node:fs/promises";
import type { ChannelConfig } from "../lib/channelConfig";
import type { Paths } from "../lib/paths";
import { assertChannelTextReadable } from "../lib/channelMedia";
import { extractVideoId } from "../lib/videoId";
import { fetchFlatPlaylistUrls } from "../ytdlp/runYtdlp";
+import { heldVideoIds } from "./videoCues";
+
+export { heldVideoIds };
export const REMOTE_LISTING_PLATFORMS = ["odysee", "bitchute"] as const;
export type RemoteListingPlatform = (typeof REMOTE_LISTING_PLATFORMS)[number];
@@ -75,19 +77,6 @@ export function diffRemoteListing(
return { listed: urls.length, unparsed, held, notHeld, heldNotListed };
}
-// The channel's held ids: its data/<id>/ directories. An absent data/ is a
-// channel with nothing held; any other read error is thrown, never read as
-// "nothing held" (the text guard runs first).
-export async function heldVideoIds(dataDir: string): Promise<Set<string>> {
- try {
- const entries = await readdir(dataDir, { withFileTypes: true });
- return new Set(entries.filter((d) => d.isDirectory() && !d.name.startsWith(".")).map((d) => d.name));
- } catch (err) {
- if ((err as NodeJS.ErrnoException).code === "ENOENT") return new Set();
- throw err;
- }
-}
-
export type RemoteListingDeps = {
list?: typeof fetchFlatPlaylistUrls;
now?: () => Date;
diff --git a/common/controller/videoCues.ts b/common/controller/videoCues.ts
@@ -0,0 +1,152 @@
+// ONE VIDEO'S CUES OFF DISK, WITHOUT AN INDEX (release 19 A7).
+//
+// A video imported or transcribed since the last index build — or one whose
+// transcript.cues.json was never written (a youtube-handling channel's
+// captions are only normalized by an explicit pass) — is invisible to every
+// reader of the published shards. This answers it from the video directory,
+// and writes nothing:
+//
+// cues.json transcript.cues.json, when it is fresh (isCuesJsonFresh — the
+// question every reader of it asks)
+// built what normalize WOULD write (buildNormalizedTranscript): the
+// metadata summary and the cues of the transcript the index
+// would pick, the caption-track rule for VTTs
+// vtt no metadata.info.json: the English VTT the rule picks, cues
+// only
+//
+// The ops transcript route serves it; the MCP's get_transcript falls back to it
+// through the editor when the archive has no such video.
+
+import path from "node:path";
+import { readdir, stat } from "node:fs/promises";
+import type { Paths } from "../lib/paths";
+import type { Cue } from "../lib/vtt";
+import { assertChannelTextReadable } from "../lib/channelMedia";
+import { readEnglishVttCues, readVideoFiles, CUES_JSON_FILENAME } from "../lib/videoStatus";
+import { readChannelConfig } from "./channels";
+import {
+ buildNormalizedTranscript,
+ isCuesJsonFresh,
+ readNormalizedTranscript,
+ type NormalizedTranscript,
+} from "./normalizeTranscript";
+
+export type VideoCues = {
+ slug: string;
+ id: string;
+ source: "cues.json" | "built" | "vtt";
+ // What transcript.cues.json says about itself: absent, stale, or fresh.
+ cuesJson: "fresh" | "stale" | "missing";
+ title?: string;
+ channel?: string;
+ uploadDate?: string;
+ duration?: number;
+ webpageUrl?: string;
+ platform?: string;
+ description?: string;
+ // The file the cues came from, when they came from a raw transcript.
+ file?: string;
+ cues: Cue[];
+};
+
+export type VideoCuesError = { error: string; status: 400 | 404 | 409 | 503 };
+
+const ONE_SEGMENT = (s: string) => s !== "." && s !== ".." && !/[/\\\0]/.test(s) && s.length > 0;
+
+// The channel's held ids: its data/<id>/ directories. An absent data/ is a
+// channel with nothing held; any other read error is thrown, never read as
+// "nothing held" (the text guard runs first).
+export async function heldVideoIds(dataDir: string): Promise<Set<string>> {
+ try {
+ const entries = await readdir(dataDir, { withFileTypes: true });
+ return new Set(entries.filter((d) => d.isDirectory() && !d.name.startsWith(".")).map((d) => d.name));
+ } catch (err) {
+ if ((err as NodeJS.ErrnoException).code === "ENOENT") return new Set();
+ throw err;
+ }
+}
+
+// The channels holding data/<id>/, for a read that was not told the channel.
+export async function channelsHoldingVideo(paths: Paths, id: string): Promise<string[]> {
+ if (!ONE_SEGMENT(id)) return [];
+ const out: string[] = [];
+ const entries = await readdir(paths.channelsDir, { withFileTypes: true }).catch(() => []);
+ for (const e of entries) {
+ if (!e.isDirectory() || e.name.startsWith(".")) continue;
+ const st = await stat(path.join(paths.channelsDir, e.name, "data", id)).catch(() => null);
+ if (st?.isDirectory()) out.push(e.name);
+ }
+ return out.sort();
+}
+
+function fromNormalized(
+ slug: string,
+ id: string,
+ t: NormalizedTranscript,
+ source: VideoCues["source"],
+ cuesJson: VideoCues["cuesJson"],
+): VideoCues {
+ const r = t as NormalizedTranscript & Record<string, unknown>;
+ const str = (v: unknown) => (typeof v === "string" && v ? v : undefined);
+ return {
+ slug,
+ id,
+ source,
+ cuesJson,
+ ...(str(r.title) ? { title: str(r.title) } : {}),
+ ...(str(r.channel) ? { channel: str(r.channel) } : {}),
+ ...(str(r.uploadDate) ? { uploadDate: str(r.uploadDate) } : {}),
+ ...(typeof r.duration === "number" ? { duration: r.duration } : {}),
+ ...(str(r.webpageUrl) ? { webpageUrl: str(r.webpageUrl) } : {}),
+ ...(str(r.platform) ? { platform: str(r.platform) } : {}),
+ ...(str(r.description) ? { description: str(r.description) } : {}),
+ ...(t.vttFile ? { file: t.vttFile } : t.source === "whisper" ? { file: "transcript.json" } : {}),
+ cues: t.cues ?? [],
+ };
+}
+
+export async function readVideoCues(
+ paths: Paths,
+ opts: { slug?: string; id: string },
+): Promise<VideoCues | VideoCuesError> {
+ const id = opts.id;
+ if (!ONE_SEGMENT(id)) return { error: `"${id}" is not a video id (one path segment)`, status: 400 };
+ let slug = opts.slug;
+ if (!slug) {
+ const holders = await channelsHoldingVideo(paths, id);
+ if (holders.length === 0) return { error: `no channel holds a video "${id}"`, status: 404 };
+ if (holders.length > 1) {
+ return { error: `${holders.length} channels hold a video "${id}" (${holders.join(", ")}) — name the channel`, status: 409 };
+ }
+ slug = holders[0];
+ }
+ const config = await readChannelConfig(paths, slug);
+ if (!config) return { error: `Channel "${slug}" not found`, status: 404 };
+ try {
+ await assertChannelTextReadable(paths, slug, config);
+ } catch (err) {
+ return { error: (err as Error).message, status: 503 };
+ }
+ const videoDir = path.join(paths.channelsDir, slug, "data", id);
+ const st = await stat(videoDir).catch(() => null);
+ if (!st?.isDirectory()) return { error: `Channel "${slug}" holds no video "${id}"`, status: 404 };
+
+ const files = await readVideoFiles(videoDir);
+ const freshness = await isCuesJsonFresh(videoDir);
+ const cuesJson: VideoCues["cuesJson"] = files.entries.includes(CUES_JSON_FILENAME)
+ ? freshness.fresh
+ ? "fresh"
+ : "stale"
+ : "missing";
+ if (cuesJson === "fresh") {
+ const t = await readNormalizedTranscript(freshness.cuesPath);
+ if (t) return fromNormalized(slug, id, t, "cues.json", cuesJson);
+ }
+ const built = await buildNormalizedTranscript({ videoDir, channelSlug: slug, configName: config.name }, files);
+ if (built.status === "built") return fromNormalized(slug, id, built.transcript, "built", cuesJson);
+ if (built.reason === "no-metadata") {
+ const read = await readEnglishVttCues(videoDir, files.entries);
+ if (read) return { slug, id, source: "vtt", cuesJson, file: read.filename, cues: read.cues };
+ }
+ return { error: `${slug}/${id} has no transcript to read (no transcript.json and no English VTT)`, status: 404 };
+}
diff --git a/editor/app/api/ops/build-cues/route.test.ts b/editor/app/api/ops/build-cues/route.test.ts
@@ -0,0 +1,99 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, readFile, stat, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { setupOpsCorpus, callPost, callGet } from "../_testCorpus";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/api/ops/build-cues/route.test.ts"
+//
+// Cues without an index (release 19 A7): `build-cues` writes
+// transcript.cues.json for the named videos only, and `GET transcript` reads a
+// video's cues off disk — from a fresh cues.json, from what normalize would
+// write, or from the VTT alone — writing nothing.
+
+const corpus = await setupOpsCorpus(null);
+await corpus.writeChannelConfig("demo-yt", { url: "https://www.youtube.com/@demo" });
+await corpus.writeChannelConfig("other-yt", { url: "https://www.youtube.com/@other" });
+const DATA = path.join(corpus.transcripts, "channels", "demo-yt", "data");
+const VTT = (text: string) => `WEBVTT\n\n00:00:01.000 --> 00:00:03.000\n${text}\n\n00:00:04.000 --> 00:00:06.000\nsecond line\n`;
+async function video(dir: string, opts: { meta?: boolean; text: string }) {
+ await mkdir(dir, { recursive: true });
+ if (opts.meta) {
+ await writeFile(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({ id: path.basename(dir), title: `Title ${path.basename(dir)}`, upload_date: "20240102", duration: 6 }),
+ );
+ }
+ await writeFile(path.join(dir, "transcript.en.vtt"), VTT(opts.text));
+}
+await video(path.join(DATA, "vid00000001"), { meta: true, text: "first video" });
+await video(path.join(DATA, "vid00000002"), { meta: true, text: "second video" });
+await video(path.join(DATA, "bare0000001"), { text: "no metadata here" });
+await video(path.join(corpus.transcripts, "channels", "other-yt", "data", "vid00000002"), { meta: true, text: "a copy" });
+
+const { POST } = await import("./route");
+const { GET } = await import("../transcript/route");
+const { getRegistry } = await import("yt-dlp-transcript-common/jobs/registry");
+test.after(() => corpus.cleanup());
+const exists = (p: string) => stat(p).then(() => true, () => false);
+
+test("build-cues refusals: keys, ids, a stray id, a channel that does not exist", async () => {
+ const cases: [Record<string, unknown>, RegExp][] = [
+ [{ slug: "demo-yt", id: "x" }, /unknown key\(s\): id/],
+ [{ slug: "demo-yt", ids: [] }, /"ids" must be a non-empty array/],
+ [{ slug: "demo-yt", ids: ["../x"] }, /not a video id/],
+ [{ slug: "demo-yt", ids: ["vid00000001", "nope"] }, /1 of "ids" not held by demo-yt: nope/],
+ [{ slug: "no-such", ids: ["a"] }, /Channel "no-such" not found/],
+ [{ slug: "demo-yt", force: "yes" }, /"force" must be a boolean/],
+ ];
+ for (const [body, re] of cases) {
+ const r = await callPost(POST, body);
+ assert.equal(r.status, 400, JSON.stringify(body));
+ assert.match(r.body.error ?? "", re, JSON.stringify(body));
+ }
+});
+
+test("transcript reads cues with no index: built from the VTT + metadata, or the VTT alone; nothing written", async () => {
+ const url = (q: string) => `http://localhost/api/ops/transcript?${q}`;
+ const built = await callGet(GET, url("id=vid00000001&slug=demo-yt"));
+ assert.equal(built.status, 200, JSON.stringify(built.body));
+ assert.equal(built.body.source, "built");
+ assert.equal(built.body.cuesJson, "missing");
+ assert.equal(built.body.title, "Title vid00000001");
+ assert.equal(built.body.file, "transcript.en.vtt");
+ assert.deepEqual((built.body.cues as { text: string }[]).map((c) => c.text), ["first video", "second line"]);
+ assert.equal(await exists(path.join(DATA, "vid00000001", "transcript.cues.json")), false);
+
+ const bare = await callGet(GET, url("id=bare0000001"));
+ assert.equal(bare.status, 200, JSON.stringify(bare.body));
+ assert.equal(bare.body.slug, "demo-yt");
+ assert.equal(bare.body.source, "vtt");
+ assert.equal((bare.body.cues as unknown[]).length, 2);
+
+ const two = await callGet(GET, url("id=vid00000002"));
+ assert.equal(two.status, 409);
+ assert.match(two.body.error ?? "", /2 channels hold a video "vid00000002" \(demo-yt, other-yt\) — name the channel/);
+ assert.equal((await callGet(GET, url("id=nothing"))).status, 404);
+ assert.match((await callGet(GET, url("slug=demo-yt"))).body.error ?? "", /"id" is required/);
+ assert.match((await callGet(GET, url("id=x&track=en"))).body.error ?? "", /unknown query key\(s\): track/);
+ assert.equal((await callGet(GET, url("id=x"), {}, {})).status, 401);
+});
+
+test("build-cues writes cues.json for the named ids only; transcript then reads it as fresh", async () => {
+ const r = await callPost(POST, { slug: "demo-yt", ids: ["vid00000001"] });
+ assert.equal(r.status, 200, JSON.stringify(r.body));
+ const jobId = r.body.jobId as string;
+ for (let i = 0; i < 200; i++) {
+ const s = getRegistry().get(jobId)?.status;
+ if (s === "done" || s === "failed") break;
+ await new Promise((res) => setTimeout(res, 50));
+ }
+ assert.equal(getRegistry().get(jobId)?.status, "done");
+ const cues = JSON.parse(await readFile(path.join(DATA, "vid00000001", "transcript.cues.json"), "utf8"));
+ assert.equal(cues.cues.length, 2);
+ assert.equal(await exists(path.join(DATA, "vid00000002", "transcript.cues.json")), false);
+ const fresh = await callGet(GET, "http://localhost/api/ops/transcript?id=vid00000001&slug=demo-yt");
+ assert.equal(fresh.body.source, "cues.json");
+ assert.equal(fresh.body.cuesJson, "fresh");
+});
diff --git a/editor/app/api/ops/build-cues/route.ts b/editor/app/api/ops/build-cues/route.ts
@@ -0,0 +1,32 @@
+import { normalizeChannelAction } from "../../../channels/[slug]/normalizeActions";
+import { OpsInputError, jobResponse, ops, optBool, optString, reqSlug, reqVideoId, type OpsBody } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, ids?: string[], force?, queueKey? } -> { ok: true, jobId }
+//
+// Write each video's transcript.cues.json from its raw transcript (the
+// caption-track rule for VTTs), as the digest card's Normalize button does
+// (normalizeActions.ts, controller/normalizeAll.ts) — the file every reader
+// without an index build wants. `ids` narrows it to those videos, every one
+// held by the channel; `force` rewrites a cues.json that is already fresh. On
+// the channel's own queue.
+function optIds(body: OpsBody): string[] | undefined {
+ const v = body.ids;
+ if (v === undefined) return undefined;
+ if (!Array.isArray(v) || v.length === 0) {
+ throw new OpsInputError('"ids" must be a non-empty array of video ids');
+ }
+ return [...new Set(v.map((id, i) => reqVideoId({ [`ids[${i}]`]: id }, `ids[${i}]`)))];
+}
+
+export async function POST(request: Request) {
+ return ops(request, ["slug", "ids", "force", "queueKey"], async (body) =>
+ jobResponse(
+ await normalizeChannelAction(reqSlug(body, "slug"), optString(body, "queueKey"), {
+ ids: optIds(body),
+ force: optBool(body, "force") ?? false,
+ }),
+ ),
+ );
+}
diff --git a/editor/app/api/ops/transcript/route.ts b/editor/app/api/ops/transcript/route.ts
@@ -0,0 +1,33 @@
+import { NextResponse } from "next/server";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { isValidChannelSlug } from "yt-dlp-transcript-common/controller/channels";
+import { readVideoCues } from "yt-dlp-transcript-common/controller/videoCues";
+import { readRoute } from "../_read";
+import { opsFail } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// GET ?id=<videoId>[&slug=<channel>]
+// -> { ok, slug, id, source: "cues.json" | "built" | "vtt",
+// cuesJson: "fresh" | "stale" | "missing", title?, channel?,
+// uploadDate?, duration?, webpageUrl?, platform?, description?, file?,
+// cues: [{ start, end, text }] }
+//
+// One video's cues off disk, with no index (controller/videoCues.ts): a fresh
+// transcript.cues.json, else what normalize would write, else the English VTT
+// alone. Nothing is written. Without `slug` the channel holding data/<id>/ is
+// found (two holders → 409, name it). The MCP's get_transcript asks this when
+// the archive it reads has no such video.
+export async function GET(request: Request) {
+ return readRoute(request, ["id", "slug"], async (q) => {
+ const id = q.get("id")?.trim() ?? "";
+ if (!id) return opsFail('"id" is required');
+ const slug = q.get("slug")?.trim() || undefined;
+ if (slug !== undefined && !isValidChannelSlug(slug)) {
+ return opsFail(`"${slug}" is not a valid channel slug`);
+ }
+ const read = await readVideoCues(getPaths(), { id, ...(slug ? { slug } : {}) });
+ if ("error" in read) return opsFail(read.error, read.status);
+ return NextResponse.json({ ok: true, ...read });
+ });
+}
diff --git a/editor/app/channels/[slug]/normalizeActions.ts b/editor/app/channels/[slug]/normalizeActions.ts
@@ -15,6 +15,7 @@
// putting it anywhere else would make the fix for a stalled digest lane queue
// up BEHIND the digest lane it is meant to unblock.
+import path from "node:path";
import { safeRevalidate } from "../../lib/safeRevalidate";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import {
@@ -27,12 +28,32 @@ import {
} from "yt-dlp-transcript-common/jobs/streamCommand";
import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler";
import { normalizeChannelTranscripts } from "yt-dlp-transcript-common/controller/normalizeAll";
+import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
+import { heldVideoIds } from "yt-dlp-transcript-common/controller/videoCues";
+// `ids` narrows the pass to those videos (`pnpm ops build-cues`, release 19 A7)
+// and every one must be held — a stray is refused, named, before any job.
+// `force` rewrites cues.json even where it is fresh.
export async function normalizeChannelAction(
slug: string,
queueKey?: string,
+ opts: { ids?: string[]; force?: boolean } = {},
): Promise<StreamActionResult> {
const paths = getPaths();
+ if (opts.ids !== undefined || opts.force !== undefined) {
+ const config = await readChannelConfig(paths, slug);
+ if (!config) return { ok: false, error: `Channel "${slug}" not found` };
+ }
+ if (opts.ids !== undefined) {
+ const held = await heldVideoIds(path.join(paths.channelsDir, slug, "data"));
+ const stray = opts.ids.filter((id) => !held.has(id));
+ if (stray.length) {
+ return {
+ ok: false,
+ error: `${stray.length} of "ids" not held by ${slug}: ${stray.join(", ")}`,
+ };
+ }
+ }
return runManagedFunction({
kind: "normalize-transcripts",
queueKey: resolveQueueKey(channelQueueKey(slug), queueKey),
@@ -42,6 +63,8 @@ export async function normalizeChannelAction(
const result = await normalizeChannelTranscripts({
paths,
channelSlug: slug,
+ ...(opts.ids ? { videoIds: opts.ids } : {}),
+ ...(opts.force ? { force: true } : {}),
onLog,
signal,
});
diff --git a/mcp/src/editorOps.test.ts b/mcp/src/editorOps.test.ts
@@ -0,0 +1,91 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { connectWith, fakeEditor, isError, textOf } from "./editorOps.testkit";
+import { editorGet, editorPost, editorTranscript, renderEditorTranscript, describeEditorFailure } from "./editorOps";
+
+// ─── The editor's ops routes from the MCP (release 19 A7) ───
+//
+// No network: a fake editor answers by URL, and records what was sent.
+
+test("a GET and a POST carry the token to the editor's URL; refusals and silence are said", async () => {
+ const { deps, sent } = fakeEditor((s) =>
+ s.url.endsWith("/bad") ? { status: 400, body: { ok: false, error: "no such thing" } } : { status: 200, body: { ok: true, x: 1 } },
+ );
+ const ok = await editorGet(deps, "/api/ops/job/abc");
+ assert.equal(ok.kind, "ok");
+ assert.equal(sent[0].url, "http://editor.test/api/ops/job/abc");
+ assert.equal(sent[0].headers?.authorization, "Bearer tok");
+ const posted = await editorPost(deps, "/api/ops/sync", { slug: "x" });
+ assert.equal(posted.kind, "ok");
+ assert.deepEqual(sent[1].body, { slug: "x" });
+ assert.equal(sent[1].headers?.["content-type"], "application/json");
+ const bad = await editorGet(deps, "/bad");
+ assert.equal(bad.kind, "refused");
+ assert.match(describeEditorFailure("t", bad as never), /^t: the editor refused \(HTTP 400\): no such thing$/);
+ const none = await editorGet({ env: {}, fetch: deps.fetch }, "/x");
+ assert.equal(none.kind, "no_editor");
+ assert.match(describeEditorFailure("t", none as never), /ARCHILYZER_EDITOR_URL.*WORKER_TOKEN/);
+ const gone = await editorGet(fakeEditor(() => new TypeError("fetch failed")).deps, "/x");
+ assert.equal(gone.kind, "unreachable");
+ assert.match(describeEditorFailure("t", gone as never), /did not answer: fetch failed/);
+});
+
+const CUES = [
+ { start: 1, end: 3, text: "first line" },
+ { start: 65, end: 67, text: "second line" },
+];
+
+test("editorTranscript asks GET /api/ops/transcript; a 404 is not an error, other refusals are", async () => {
+ const { deps, sent } = fakeEditor((s) =>
+ s.url.includes("id=held")
+ ? { status: 200, body: { ok: true, slug: "demo", id: "held", source: "vtt", cuesJson: "missing", cues: CUES } }
+ : { status: 404, body: { ok: false, error: "no channel holds it" } },
+ );
+ const held = await editorTranscript(deps, "held", "demo");
+ assert.equal(held.ok, true);
+ assert.equal(new URL(sent[0].url).searchParams.get("slug"), "demo");
+ const missing = await editorTranscript(deps, "nope");
+ assert.deepEqual(missing, { ok: false, error: null });
+ assert.equal(new URL(sent[1].url).searchParams.has("slug"), false);
+ assert.deepEqual(await editorTranscript(deps, "../x"), { ok: false, error: null });
+ assert.equal(sent.length, 2);
+});
+
+test("the rendered fallback says where it came from and carries no moment links", () => {
+ const md = renderEditorTranscript(
+ { slug: "demo", id: "held", source: "built", cuesJson: "missing", title: "A title", file: "transcript.en.vtt", cues: CUES },
+ { timestamps: true },
+ );
+ assert.match(md, /^> NOT IN THE ARCHIVE — read from the local editor's disk \(channel demo\)/);
+ assert.match(md, /normalized in memory/);
+ assert.match(md, /# A title/);
+ assert.match(md, /\[1:05\] second line/);
+ assert.doesNotMatch(md, /\]\(http/);
+});
+
+test("get_transcript: a video not in the archive is read off the editor's disk", async () => {
+ const { deps } = fakeEditor(() => ({
+ status: 200,
+ body: { ok: true, slug: "demo", id: "fresh01", source: "cues.json", cuesJson: "fresh", title: "Fresh", cues: CUES },
+ }));
+ const client = await connectWith(deps);
+ const res = await client.callTool({ name: "get_transcript", arguments: { video_id: "fresh01", channel: "demo" } });
+ assert.equal(isError(res), false, textOf(res));
+ assert.match(textOf(res), /NOT IN THE ARCHIVE/);
+ assert.match(textOf(res), /first line/);
+});
+
+test("get_transcript: with no editor, or one that holds nothing, it is still 'video not found'", async () => {
+ const none = await connectWith(fakeEditor(() => ({ status: 500, body: {} }), {}).deps);
+ const a = await none.callTool({ name: "get_transcript", arguments: { video_id: "x1" } });
+ assert.equal(isError(a), true);
+ assert.match(textOf(a), /^video not found: x1/);
+ const empty = await connectWith(fakeEditor(() => ({ status: 404, body: { ok: false, error: "no" } })).deps);
+ const b = await empty.callTool({ name: "get_transcript", arguments: { video_id: "x1" } });
+ assert.match(textOf(b), /^video not found: x1/);
+ // A track is the archive's: the editor is not asked for one.
+ const { deps, sent } = fakeEditor(() => ({ status: 200, body: { ok: true, cues: [] } }));
+ const c = await (await connectWith(deps)).callTool({ name: "get_transcript", arguments: { video_id: "x1", track: "en" } });
+ assert.match(textOf(c), /^video not found: x1/);
+ assert.equal(sent.length, 0);
+});
diff --git a/mcp/src/editorOps.testkit.ts b/mcp/src/editorOps.testkit.ts
@@ -0,0 +1,91 @@
+import { Client, InMemoryTransport } from "@modelcontextprotocol/client";
+import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest";
+import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
+import type { ChannelGroups, ChannelRef, ShardSource, VideoAvailability } from "./source";
+import { SourceRegistry } from "./sourceRegistry";
+import { createServer } from "./server";
+import type { FetchClipDeps, HttpInit } from "./fetchClip";
+
+// TEST KIT for the editor-backed tools (imported only by *.test.ts): a fake
+// editor that answers by URL and records what it was sent, and a server over
+// an empty archive reached through tools/call. No network, no timers.
+
+type Sent = { url: string; method: string; headers?: Record<string, string>; body?: unknown };
+
+export function fakeEditor(
+ answer: (s: Sent) => { status: number; body: unknown } | Error,
+ env: Record<string, string | undefined> = { WORKER_TOKEN: "tok", ARCHILYZER_EDITOR_URL: "http://editor.test/" },
+): { deps: Partial<FetchClipDeps> & Pick<FetchClipDeps, "env" | "fetch">; sent: Sent[] } {
+ const sent: Sent[] = [];
+ return {
+ sent,
+ deps: {
+ env,
+ fetch: async (url: string, init?: HttpInit) => {
+ const s: Sent = {
+ url,
+ method: init?.method ?? "GET",
+ headers: init?.headers,
+ body: init?.body ? JSON.parse(init.body) : undefined,
+ };
+ sent.push(s);
+ const a = answer(s);
+ if (a instanceof Error) throw a;
+ return { status: a.status, json: async () => a.body };
+ },
+ sleep: async () => {},
+ now: () => 0,
+ },
+ };
+}
+
+
+class EmptySource implements ShardSource {
+ readonly label = "local:/srv/fixture";
+ async loadAliases(): Promise<SearchAlias[]> {
+ return [];
+ }
+ async loadGroups(): Promise<ChannelGroups> {
+ return { groups: [], defaultGroupId: "default" };
+ }
+ async listChannels(): Promise<ChannelRef[]> {
+ return [{ key: "demo", slug: "demo", name: "demo" }];
+ }
+ async transcriptsManifest(): Promise<ChannelTranscriptsManifest> {
+ return { slugToPage: {} } as unknown as ChannelTranscriptsManifest;
+ }
+ async transcriptPage(): Promise<TranscriptDetail[]> {
+ return [];
+ }
+ publicOrigin(): string | null {
+ return null;
+ }
+ async subsManifest(): Promise<null> {
+ return null;
+ }
+ async subsPage(): Promise<[]> {
+ return [];
+ }
+ async postsManifest(): Promise<null> {
+ return null;
+ }
+ async postsPage(): Promise<[]> {
+ return [];
+ }
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ return new Map();
+ }
+}
+
+export async function connectWith(deps: Partial<FetchClipDeps>): Promise<Client> {
+ const server = createServer(SourceRegistry.forSource(new EmptySource()), { fetchClipDeps: deps });
+ const [ct, st] = InMemoryTransport.createLinkedPair();
+ const client = new Client({ name: "test", version: "0" }, { capabilities: {} });
+ await Promise.all([server.connect(st), client.connect(ct)]);
+ return client;
+}
+export const textOf = (res: unknown) =>
+ (res as { content: { type: string; text: string }[] }).content.map((c) => c.text).join("\n");
+export const isError = (res: unknown) => (res as { isError?: boolean }).isError === true;
+
diff --git a/mcp/src/editorOps.ts b/mcp/src/editorOps.ts
@@ -0,0 +1,154 @@
+import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown";
+import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
+import {
+ REQUEST_TIMEOUT_MS,
+ describeFetchError,
+ editorFromEnv,
+ isVideoId,
+ type FetchClipDeps,
+ type HttpInit,
+} from "./fetchClip";
+
+// ─── The editor's ops routes, for the MCP (release 19 A7, A9) ───
+//
+// The fetch_clip pattern, generalised: the MCP asks the LOCAL editor
+// (ARCHILYZER_EDITOR_URL + WORKER_TOKEN, the editor's own) and the editor
+// reads or writes. This process still writes nothing to an archive. Every
+// request is bounded (REQUEST_TIMEOUT_MS); an editor that does not answer is
+// said in words, never as a stack.
+//
+// Settings, storage and deletes are NOT reachable from here (operator ruling,
+// release 19): those stay `pnpm ops` / `archilyzer` only.
+
+export type EditorDeps = Pick<FetchClipDeps, "env" | "fetch" | "requestTimeoutMs">;
+
+export type EditorAnswer =
+ | { kind: "ok"; status: number; body: Record<string, unknown> }
+ | { kind: "refused"; status: number; error: string; body: Record<string, unknown> }
+ | { kind: "no_editor" }
+ | { kind: "unreachable"; url: string; message: string };
+
+export const NO_EDITOR_OPS_TEXT =
+ "no editor configured. Set ARCHILYZER_EDITOR_URL (e.g. http://localhost:3001) " +
+ "and WORKER_TOKEN (the editor's own WORKER_TOKEN) when registering the MCP " +
+ "server. A public-only setup has no editor to ask.";
+
+async function call(
+ deps: EditorDeps,
+ method: "GET" | "POST",
+ route: string,
+ body?: unknown,
+): Promise<EditorAnswer> {
+ const editor = editorFromEnv(deps.env);
+ if (!editor) return { kind: "no_editor" };
+ const timeoutMs = deps.requestTimeoutMs ?? REQUEST_TIMEOUT_MS;
+ const init: HttpInit = {
+ method,
+ headers: {
+ authorization: `Bearer ${editor.token}`,
+ ...(body !== undefined ? { "content-type": "application/json" } : {}),
+ },
+ ...(body !== undefined ? { body: JSON.stringify(body) } : {}),
+ signal: AbortSignal.timeout(timeoutMs),
+ };
+ let res;
+ try {
+ res = await deps.fetch(`${editor.url}${route}`, init);
+ } catch (e) {
+ return { kind: "unreachable", url: editor.url, message: describeFetchError(e, timeoutMs) };
+ }
+ let parsed: Record<string, unknown> = {};
+ try {
+ const j = await res.json();
+ if (j && typeof j === "object" && !Array.isArray(j)) parsed = j as Record<string, unknown>;
+ } catch {
+ /* a non-JSON answer is refused below with its status */
+ }
+ if (res.status >= 200 && res.status < 300 && parsed.ok !== false) {
+ return { kind: "ok", status: res.status, body: parsed };
+ }
+ const error = typeof parsed.error === "string" && parsed.error ? parsed.error : `HTTP ${res.status}`;
+ return { kind: "refused", status: res.status, error, body: parsed };
+}
+
+export const editorGet = (deps: EditorDeps, route: string) => call(deps, "GET", route);
+export const editorPost = (deps: EditorDeps, route: string, body: unknown) => call(deps, "POST", route, body);
+
+// An answer that is not `ok`, as one sentence for the agent.
+export function describeEditorFailure(tool: string, a: Exclude<EditorAnswer, { kind: "ok" }>): string {
+ switch (a.kind) {
+ case "no_editor":
+ return `${tool}: ${NO_EDITOR_OPS_TEXT}`;
+ case "unreachable":
+ return `${tool}: the editor at ${a.url} did not answer: ${a.message}`;
+ case "refused":
+ return a.status === 401 || a.status === 503
+ ? `${tool}: the editor refused the token (HTTP ${a.status}: ${a.error}) — WORKER_TOKEN must be the editor's own`
+ : `${tool}: the editor refused (HTTP ${a.status}): ${a.error}`;
+ }
+}
+
+// ─── get_transcript's fallback: cues off the editor's disk (A7) ───
+
+export type EditorTranscript = {
+ slug: string;
+ id: string;
+ source: "cues.json" | "built" | "vtt";
+ cuesJson: "fresh" | "stale" | "missing";
+ title?: string;
+ channel?: string;
+ uploadDate?: string;
+ duration?: number;
+ webpageUrl?: string;
+ description?: string;
+ file?: string;
+ cues: Cue[];
+};
+
+const SOURCE_WORDS: Record<EditorTranscript["source"], string> = {
+ "cues.json": "its transcript.cues.json",
+ built: "its raw transcript, normalized in memory (no cues.json written)",
+ vtt: "its English VTT alone (the record has no metadata)",
+};
+
+// Ask the editor for one video's cues off disk. Null when there is no editor
+// to ask, or it holds no such video — the caller's "video not found" stands.
+export async function editorTranscript(
+ deps: EditorDeps,
+ videoId: string,
+ channel?: string,
+): Promise<{ ok: true; t: EditorTranscript } | { ok: false; error: string | null }> {
+ if (!isVideoId(videoId)) return { ok: false, error: null };
+ const q = new URLSearchParams({ id: videoId });
+ if (channel && /^[a-z0-9][a-z0-9._-]*$/i.test(channel)) q.set("slug", channel);
+ const a = await editorGet(deps, `/api/ops/transcript?${q.toString()}`);
+ if (a.kind === "no_editor") return { ok: false, error: null };
+ if (a.kind === "refused" && a.status === 404) return { ok: false, error: null };
+ if (a.kind !== "ok") return { ok: false, error: describeEditorFailure("get_transcript", a) };
+ const b = a.body as unknown as EditorTranscript;
+ if (!Array.isArray(b.cues)) return { ok: false, error: "get_transcript: the editor's answer carried no cues" };
+ return { ok: true, t: b };
+}
+
+export function renderEditorTranscript(t: EditorTranscript, opts: { timestamps: boolean }): string {
+ const header = [
+ `NOT IN THE ARCHIVE — read from the local editor's disk (channel ${t.slug}): ${SOURCE_WORDS[t.source]}${t.file ? `, ${t.file}` : ""}.`,
+ "No moment links: the video is not published yet. Cite it by its source URL and time,",
+ "and expect the published text to differ once an index build normalizes it.",
+ ].join(" ");
+ const md = transcriptToMarkdown(
+ {
+ id: t.id,
+ title: t.title ?? t.id,
+ ...(t.channel ? { channel: t.channel } : {}),
+ channelSlug: t.slug,
+ ...(t.uploadDate ? { uploadDate: t.uploadDate } : {}),
+ ...(typeof t.duration === "number" ? { duration: t.duration } : {}),
+ ...(t.webpageUrl ? { webpageUrl: t.webpageUrl } : {}),
+ ...(t.description ? { description: t.description } : {}),
+ cues: t.cues,
+ },
+ { timestamps: opts.timestamps },
+ );
+ return `> ${header}\n\n${md}`;
+}
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -88,6 +88,7 @@ import {
type FetchClipDeps,
type PollProgress,
} from "./fetchClip";
+import { editorTranscript, renderEditorTranscript, type EditorDeps } from "./editorOps";
import { extractVideoId } from "yt-dlp-transcript-common/lib/videoId";
import {
decodeShareLink,
@@ -581,7 +582,10 @@ export const TOOLS: Tool[] = [
description:
"Fetch one video's full transcript as clean markdown (metadata header + " +
"timestamped captions). Provide the video id; optionally the channel to " +
- "skip the cross-channel lookup.",
+ "skip the cross-channel lookup. A video not yet in the archive (imported " +
+ "or transcribed since the last build) is read off the local editor's " +
+ "disk when one is configured (ARCHILYZER_EDITOR_URL + WORKER_TOKEN), " +
+ "marked as such, with no moment links.",
inputSchema: {
type: "object",
properties: {
@@ -1257,7 +1261,7 @@ export function createServer(
case "enumerate_matches":
return handleEnumerateMatches(source, args);
case "get_transcript":
- return handleGetTranscript(source, args);
+ return handleGetTranscript(source, args, fetchClipDeps);
case "get_transcripts":
return handleGetTranscripts(source, args);
case "get_post":
@@ -2410,15 +2414,26 @@ async function handleGetThread(
async function handleGetTranscript(
source: ShardSource,
args: Record<string, unknown>,
+ editorDeps?: EditorDeps,
): Promise<ToolResult> {
const videoId = String(args.video_id ?? "").trim();
if (!videoId) return errorText("video_id is required");
- const found = await findVideo(
- source,
- videoId,
- typeof args.channel === "string" ? args.channel : undefined,
- );
- if (!found) return errorText(`video not found: ${videoId}`);
+ const channelArg = typeof args.channel === "string" ? args.channel : undefined;
+ const found = await findVideo(source, videoId, channelArg);
+ if (!found) {
+ // NOT IN THE ARCHIVE: a video imported or transcribed since the last
+ // build. With a local editor configured, its cues are read off the
+ // editor's disk (release 19 A7) — the primary transcript only.
+ const askedTrack = typeof args.track === "string" && args.track.trim() !== "";
+ if (editorDeps && !askedTrack) {
+ const fromEditor = await editorTranscript(editorDeps, videoId, channelArg);
+ if (fromEditor.ok) {
+ return text(renderEditorTranscript(fromEditor.t, { timestamps: args.timestamps !== false }));
+ }
+ if (fromEditor.error) return errorText(`video not found in the archive: ${videoId}\n\n${fromEditor.error}`);
+ }
+ return errorText(`video not found: ${videoId}`);
+ }
const { record, ch } = found;
const link: LinkableHit = {
slug: record.slug,
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -16,6 +16,7 @@
// pnpm ops get job <id> [--tail [<lines>]]
// pnpm ops get jobs [--active | --failed] [--kind <kind>] [--slug <slug>] [--limit <n>]
// pnpm ops get remote-listing <slug>
+// pnpm ops get transcript <videoId> [--slug <slug>]
// pnpm ops job <cancel|drain|promote|force-release|retry> <id>... [--wait]
// pnpm ops job retry-failed | job wait <id>...
// pnpm ops list
@@ -44,6 +45,8 @@
// pnpm ops import-archive-org --json '{"slug":"demo-archive","item":"example-item","match":"\\.mp4$"}' --wait
// pnpm ops import-archive-org --json '{"slug":"demo-archive","query":"collection:example-collection","dryRun":true}' --wait
// pnpm ops get remote-listing demo-odysee
+// pnpm ops build-cues --json '{"slug":"demo-yt","ids":["<videoId>"]}' --wait
+// pnpm ops get transcript <videoId> --slug demo-yt
// pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}'
// pnpm ops channel-config --json '{"slug":"x","sites":[{"siteId":"anilyzer"}]}'
// pnpm ops channel-config --json '{"slug":"x","sites":[],"excludeFromBuild":true}'
@@ -163,6 +166,14 @@ const GETTERS = {
scheduler: () => "/api/ops/scheduler",
// One channel's cleanup row: what each sweep would reclaim, what holds the rest.
cleanup: (slug) => `/api/ops/cleanup/${encodeURIComponent(slug)}`,
+ // One video's cues off disk, with no index (release 19 A7): a fresh
+ // cues.json, else what normalize would write, else the VTT alone. --slug
+ // names the channel (else the one holding data/<id>/).
+ transcript: (id, { jobs }) => {
+ const q = new URLSearchParams({ id });
+ if (jobs.slug) q.set("slug", jobs.slug);
+ return `/api/ops/transcript?${q.toString()}`;
+ },
};
// READS THAT ARE JOBS: the answer needs a request upstream, so the editor
@@ -196,6 +207,7 @@ const GET_ARG_OPTIONAL = new Set([
// The flags each noun takes beyond the shared ones; any other is refused.
const GET_FLAGS = {
+ transcript: ["slug"],
channel: ["counts"],
job: ["tail"],
jobs: ["active", "failed", "kind", "slug", "limit"],
@@ -242,6 +254,9 @@ const ACTIONS = [
// A channel's saved containers remuxed losslessly into browser-playable
// copies, one torrent each ({slug, ids?, trackers?, root?, dryRun?}).
"prepare-playable",
+ // Write transcript.cues.json from each video's raw transcript ({slug, ids?,
+ // force?}) — what a reader with no index build wants.
+ "build-cues",
// A podcast channel's records completed from its RSS feed ({slug, dryRun?}):
// one fetch of the feed, no media.
"feed-metadata",
@@ -636,6 +651,7 @@ export function usage() {
" pnpm ops get storage | sites | workers | auto-queue | scheduler",
" pnpm ops get cleanup <slug>",
" pnpm ops get remote-listing <slug> [--wait-timeout <seconds>]",
+ " pnpm ops get transcript <videoId> [--slug <slug>]",
" pnpm ops list",
"",
`Actions: ${ACTIONS.join(", ")}`,
@@ -680,6 +696,17 @@ export function usage() {
" nothing written. `get remote-listing <slug>` waits for the job and prints",
" {listed, held, notHeld: [{id, url}], heldNotListed: [id], ...} on stdout.",
"",
+ 'build-cues writes each video\'s transcript.cues.json from its raw',
+ " transcript (the caption-track rule for VTTs), as the digest card's",
+ ' Normalize button does: {"slug"}, "ids": [...] for those videos only',
+ ' (every one held), "force": true to rewrite a fresh one. A job on the',
+ " channel's queue. The file every reader without an index build wants.",
+ "",
+ "get transcript <videoId> [--slug <slug>] reads one video's cues off disk",
+ " with no index: a fresh cues.json, else what build-cues would write, else",
+ " the English VTT alone — {source, cuesJson, title?, ..., cues}. Nothing is",
+ " written. Without --slug, the channel holding data/<id>/ is found.",
+ "",
'publish runs publish stages on the editor\'s publish queue, one at a time,',
' under one run id: {"verb": …}. "index" updates the index; "build" builds',
' "siteId"/"siteIds" (forced; the index first when stale; "runner":',
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -809,3 +809,16 @@ test("usage documents import-archive-org's three shapes and the remote listing",
assert.match(u, /RESTRICTED/);
assert.match(u, /remote-listing lists an Odysee or BitChute channel/);
});
+
+test("build-cues is an action; get transcript reads one video's cues, --slug naming the channel", () => {
+ const b = parseArgs(["build-cues", "--json", '{"slug":"x","ids":["a"]}']);
+ assert.equal(b.path, "/api/ops/build-cues");
+ assert.deepEqual(b.body, { slug: "x", ids: ["a"] });
+ const t = parseArgs(["get", "transcript", "vid1", "--slug", "demo"]);
+ assert.equal(t.method, "GET");
+ assert.equal(t.path, "/api/ops/transcript?id=vid1&slug=demo");
+ assert.equal(parseArgs(["get", "transcript", "vid1"]).path, "/api/ops/transcript?id=vid1");
+ assert.match(parseArgs(["get", "transcript"]).error, /needs an argument/);
+ assert.match(parseArgs(["get", "transcript", "v", "--counts"]).error, /does not take --counts/);
+ assert.match(usage(), /build-cues writes each video's transcript\.cues\.json/);
+});