commit 56a52ba8c959611cae4f09c8b7fa5dcad44bf37f
parent b4322ab1cd3b4f67e0ed1ec7a4d4ccec5a1c2218
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 10 Oct 2026 02:39:21 -0400
Merge r20/integration (release 20 D1 recorded dates) into r20/a6-a9
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
19 files changed, 712 insertions(+), 1 deletion(-)
diff --git a/CHANNEL.md b/CHANNEL.md
@@ -46,6 +46,7 @@ Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/f
| `cookieMode` | config | Per-channel override of the global cookie mode. Absent = inherit. |
| `sleepBetweenDownloadsSeconds` | config | Per-channel override for the global pause between downloads. Absent = inherit; 0 = no sleep; floored and capped at 600. |
| `audioCheck` | config | Opt-in audio-integrity checking for sources that intermittently serve corrupt audio mid-download (e.g. Odysee "original"): the managed downloader periodically validates the in-progress `.part` file and rolls back to the last known-good snapshot on corruption. transcribe-handling only. See [`audioCheck`](#audiocheck). |
+| `recordedDate` | config | For a channel that MIRRORS another's streams (a VOD archive): how to read the date a video was RECORDED from its title, since its `upload_date` is the date of the copy. The index stores the result as the record's `recordedDate` (`YYYYMMDD`), which coverage reads before the upload date. A title it does not match, a date that is not a real day, or one after the upload date gives none. Changing the rule re-derives the channel's records on the next index build. An object without a usable `titlePattern` is dropped. See [`recordedDate`](#recordeddate). |
#### `downloadFilter`
@@ -65,3 +66,9 @@ Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/f
| `maxRollbacks` | absent | Rollbacks to the last known-good snapshot before the download is given up; clamped to [1, 20]. Absent = 5. |
| `copyTimeoutSeconds` | absent | Seconds allowed for the snapshot copy; clamped to [5, 120]. Absent = 30. |
| `resumeDuringProbe` | absent | When false (default), yt-dlp stays SIGSTOPped across each probe, so it never downloads bytes a malformed verdict would discard and force a re-fetch — minimising HTTP 429 risk. True = the legacy behaviour: resume right after the snapshot copy and probe while the download keeps running. |
+
+#### `recordedDate`
+
+| Key | Default | Description |
+|---|---|---|
+| `titlePattern` | required | Case-insensitive regex SOURCE (no delimiters, no flags) matched against each video's title, with three named groups: `year` (four digits, or two read as 20YY), `month` (1–12, or an English month name, whole or cut to three or more letters) and `day`. Trimmed; at most 200 characters, no nested quantifier (the download filter's rules). A pattern that breaks any of this drops the whole rule. |
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -173,6 +173,7 @@ import {
reapplyCuratedTags,
} from "./curatedTagsIndex";
import { tagsForSite } from "../lib/curatedTags";
+import { compileRecordedDateRule, deriveRecordedDate } from "../lib/recordedDate";
// v10: multi-site build. Shared per-channel transcript/subs pages are written
// once; per-site summaries + subs manifests are filtered selections. Bumped to
@@ -222,6 +223,13 @@ const PLATFORM_LABELS_KEY = "platformLabels";
// how many now read different text, how many had none and now do.
const CAPTION_TRACK_KEY = "captionTrackRule";
+// RECORDED DATES — per channel, the same "re-derive what the change reaches"
+// shape: `recordedDateRule:<slug>` holds the `recordedDate.titlePattern` the
+// channel's records were last derived under (absent: none). A build that finds
+// another re-processes every record of that channel, then records the
+// pattern.
+const RECORDED_DATE_RULE_KEY_PREFIX = "recordedDateRule:";
+
// A cue list's text, for "did the words change" — timing alone is not a
// different transcript.
function cueText(list: readonly Cue[]): string {
@@ -933,6 +941,41 @@ export async function buildIndex({
}
let altTrackRecords = 0;
+ // RECORDED DATES (release 20 D1, lib/recordedDate.ts): each channel's rule,
+ // compiled once, and the channels whose rule differs from the one their
+ // records were derived under (RECORDED_DATE_RULE_KEY_PREFIX) — every record
+ // of those is re-processed like a changed one. A held channel is left for
+ // the build that can read it.
+ const recordedDateRes = new Map<string, RegExp>();
+ const recordedDateRuleChanged: string[] = [];
+ for (const [slug, cfg] of channelConfigs) {
+ const re = cfg.recordedDate ? compileRecordedDateRule(cfg.recordedDate) : null;
+ if (re) recordedDateRes.set(slug, re);
+ if (held.has(slug)) continue;
+ const stored = meta.get(`${RECORDED_DATE_RULE_KEY_PREFIX}${slug}`);
+ const now = re ? cfg.recordedDate!.titlePattern : undefined;
+ if ((typeof stored === "string" ? stored : undefined) !== now) {
+ recordedDateRuleChanged.push(slug);
+ }
+ }
+ if (recordedDateRuleChanged.length > 0 && !schemaBumped) {
+ const ruled = new Set(recordedDateRuleChanged);
+ const queued = new Set(
+ [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])),
+ );
+ const perChannel = new Map<string, number>();
+ for (const s of live) {
+ if (!ruled.has(s.channelSlug)) continue;
+ const pk: PathKey = [s.channelSlug, s.videoDir];
+ if (!mtimes.get(pk)) continue;
+ perChannel.set(s.channelSlug, (perChannel.get(s.channelSlug) ?? 0) + 1);
+ if (!queued.has(pathKeyId(pk))) changed.push(s);
+ }
+ for (const [slug, n] of perChannel) {
+ log(`Recorded dates: ${slug}: its rule changed; ${n} record(s) re-derived.`);
+ }
+ }
+
const anyMutations =
added.length > 0 || changed.length > 0 || removed.length > 0;
@@ -1045,6 +1088,15 @@ export async function buildIndex({
log(`Skipping ${s.channelSlug}/${s.videoDir}: no upload_date`);
return;
}
+ // The channel's recorded-date rule, applied to the summary as it
+ // stands (a normalized one included): set when it reads a date, and
+ // never carried over from an older derivation.
+ delete summary.recordedDate;
+ const recordedRe = recordedDateRes.get(s.channelSlug);
+ if (recordedRe) {
+ const recorded = deriveRecordedDate(recordedRe, summary.title, summary.uploadDate);
+ if (recorded) summary.recordedDate = recorded;
+ }
const indexKey: IndexKey = [
summary.uploadDate,
s.channelSlug,
@@ -2529,6 +2581,12 @@ export async function buildIndex({
if (altTracksDue && held.size === 0) {
await meta.put(ALT_TRACKS_KEY, ALT_TRACKS_VERSION);
}
+ for (const slug of recordedDateRuleChanged) {
+ const rule = channelConfigs.get(slug)?.recordedDate;
+ const key = `${RECORDED_DATE_RULE_KEY_PREFIX}${slug}`;
+ if (rule && recordedDateRes.has(slug)) await meta.put(key, rule.titlePattern);
+ else await meta.remove(key);
+ }
await meta.flushed;
await root.close();
diff --git a/common/controller/buildIndexCaptionTrack.test.ts b/common/controller/buildIndexCaptionTrack.test.ts
@@ -154,3 +154,31 @@ test("an index built under the old rule is re-read once, for exactly the records
const again = await runIndex();
assert.equal(again.some((l) => /Caption track/.test(l)), false, again.join("\n"));
});
+
+// THE 2026-10-01 BUG, CLOSED (release 20 D3): a preferred track that parses to
+// no cues never hides a later one with text — in either order of the two
+// tracks the bug was about, and past `en` to a regional track. Added to the
+// index built above, so the records are read under the current rule.
+test("a preferred English track with no cues falls through to the next one with text", async () => {
+ const EMPTY_VTT = "WEBVTT\nKind: captions\nLanguage: en\n\n";
+ const ORIG_EMPTY = "OrigEmpty001"; // en-orig empty, en with text
+ const EN_EMPTY = "EnEmptyUS001"; // en empty, no en-orig, en-US with text
+ for (const [i, id] of [ORIG_EMPTY, EN_EMPTY].entries()) {
+ writeJson(path.join(dirOf(id), "metadata.info.json"), {
+ id,
+ title: `Video ${id}`,
+ upload_date: `2026070${i + 1}`,
+ duration: 30,
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ }
+ writeFileSync(path.join(dirOf(ORIG_EMPTY), "transcript.en-orig.vtt"), EMPTY_VTT);
+ writeFileSync(path.join(dirOf(ORIG_EMPTY), "transcript.en.vtt"), ROLLING);
+ writeFileSync(path.join(dirOf(EN_EMPTY), "transcript.en.vtt"), EMPTY_VTT);
+ writeFileSync(path.join(dirOf(EN_EMPTY), "transcript.en-US.vtt"), CUE_BLOCKS);
+ await runIndex();
+ assert.equal(cuesOf(ORIG_EMPTY)?.length, 3);
+ assert.equal(cuesOf(EN_EMPTY)?.length, 7);
+ assert.equal(cuesOf(EN_EMPTY)?.[0].text.startsWith("welcome back everyone"), true);
+});
diff --git a/common/controller/buildIndexRecordedDate.test.ts b/common/controller/buildIndexRecordedDate.test.ts
@@ -0,0 +1,164 @@
+// Integration: a VOD mirror's recorded dates (release 20 D1) through the REAL
+// buildIndex over a temp corpus — derived from the title by the channel's
+// `recordedDate` rule, absent for a channel without one, re-derived when the
+// rule changes and dropped when it goes. Synthetic titles.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/buildIndexRecordedDate.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-recorded-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.ARCHILYZER_INDEX_ALLOW_HELD;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { open } = await import("lmdb");
+
+const paths = getPaths();
+const MIRROR = "example-vods";
+const PLAIN = "example-plain";
+const SITE = "testsite";
+const ISO = String.raw`(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})`;
+const US = String.raw`(?<month>\d{1,2})/(?<day>\d{1,2})/(?<year>\d{2})`;
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const ROLLING = readFileSync(
+ path.join(import.meta.dirname, "..", "lib", "__fixtures__", "vtt-rolling.vtt"),
+ "utf8",
+);
+
+function config(slug: string, recordedDate?: { titlePattern: string }) {
+ writeJson(path.join(paths.channelsDir, slug, "config.json"), {
+ handling: "youtube",
+ name: slug,
+ ...(recordedDate ? { recordedDate } : {}),
+ });
+}
+
+function video(slug: string, id: string, title: string, uploadDate: string) {
+ const dir = path.join(paths.channelsDir, slug, "data", id);
+ writeJson(path.join(dir, "metadata.info.json"), {
+ id,
+ title,
+ upload_date: uploadDate,
+ duration: 30,
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ writeFileSync(path.join(dir, "transcript.en.vtt"), ROLLING);
+}
+
+function seed(): void {
+ writeFileSync(paths.settingsFile, "{}");
+ config(MIRROR, { titlePattern: ISO });
+ config(PLAIN);
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [
+ { slug: MIRROR, groupId: "default" },
+ { slug: PLAIN, groupId: "default" },
+ ],
+ });
+ video(MIRROR, "MirrorVod001", "Stream VOD 2024-03-05 [3/5/24]", "20240310");
+ video(MIRROR, "MirrorVod002", "Stream VOD with no date", "20240311");
+ video(MIRROR, "MirrorVod003", "Stream VOD — next one 2024-04-01", "20240312");
+ video(PLAIN, "PlainVideo01", "An upload about 2024-01-02", "20240115");
+}
+
+type Summary = { id: string; uploadDate: string; recordedDate?: string };
+
+function recorded(): Record<string, string | undefined> {
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ try {
+ const sums = root.openDB({ name: "sums", encoding: "msgpack" });
+ const out: Record<string, string | undefined> = {};
+ for (const { value } of sums.getRange()) {
+ const s = value as Summary;
+ out[s.id] = s.recordedDate;
+ }
+ return out;
+ } finally {
+ root.close();
+ }
+}
+
+// Every published transcript page of a channel, as text.
+function pagesOf(slug: string): string {
+ const dir = path.join(paths.exportSharedTranscriptsDir, slug);
+ return readdirSync(dir)
+ .filter((f) => f.endsWith(".json"))
+ .map((f) => readFileSync(path.join(dir, f), "utf8"))
+ .join("\n");
+}
+
+async function runIndex(): Promise<string[]> {
+ const log: string[] = [];
+ await buildIndex({ paths, onLog: (s) => log.push(s) });
+ return log;
+}
+
+test("a mirror's records carry the title's recorded date; another channel's never do", async () => {
+ seed();
+ await runIndex();
+ assert.deepEqual(recorded(), {
+ MirrorVod001: "20240305",
+ MirrorVod002: undefined, // no date in the title
+ MirrorVod003: undefined, // the title's date is after the upload
+ PlainVideo01: undefined, // no rule
+ });
+ assert.match(pagesOf(MIRROR), /"recordedDate":\s*"20240305"/);
+ assert.doesNotMatch(pagesOf(PLAIN), /recordedDate/);
+});
+
+test("a changed rule re-derives the channel's records once; a removed rule drops them", async () => {
+ config(MIRROR, { titlePattern: US });
+ const log = await runIndex();
+ assert.ok(
+ log.includes(`Recorded dates: ${MIRROR}: its rule changed; 3 record(s) re-derived.`),
+ log.join("\n"),
+ );
+ assert.equal(log.some((l) => l.includes(`Recorded dates: ${PLAIN}`)), false);
+ assert.equal(recorded().MirrorVod001, "20240305");
+
+ // Unchanged: no second pass.
+ const again = await runIndex();
+ assert.equal(again.some((l) => l.startsWith("Recorded dates:")), false, again.join("\n"));
+
+ config(MIRROR);
+ const dropped = await runIndex();
+ assert.ok(dropped.includes(`Recorded dates: ${MIRROR}: its rule changed; 3 record(s) re-derived.`));
+ assert.equal(recorded().MirrorVod001, undefined);
+ assert.doesNotMatch(pagesOf(MIRROR), /recordedDate/);
+});
diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts
@@ -7,6 +7,7 @@ import {
} from "../ytdlp/downloadFormat";
import { isCookieMode, type CookieMode } from "./cookiePolicy";
import type { FieldDocs } from "./fieldDocs";
+import { coerceRecordedDateRule, type RecordedDateRule } from "./recordedDate";
export type ChannelHandling = "youtube" | "transcribe";
@@ -146,6 +147,7 @@ export type ChannelConfig = {
cookieMode?: CookieMode;
sleepBetweenDownloadsSeconds?: number;
audioCheck?: AudioCheckConfig;
+ recordedDate?: RecordedDateRule;
};
// In the order parseChannelConfig emits (and CHANNEL.md lists) them.
@@ -204,6 +206,8 @@ export const CHANNEL_CONFIG_FIELD_DOCS: FieldDocs<ChannelConfig> = {
"Per-channel override for the global pause between downloads. Absent = inherit; 0 = no sleep; floored and capped at 600.",
audioCheck:
"Opt-in audio-integrity checking for sources that intermittently serve corrupt audio mid-download (e.g. Odysee \"original\"): the managed downloader periodically validates the in-progress `.part` file and rolls back to the last known-good snapshot on corruption. transcribe-handling only.",
+ recordedDate:
+ "For a channel that MIRRORS another's streams (a VOD archive): how to read the date a video was RECORDED from its title, since its `upload_date` is the date of the copy. The index stores the result as the record's `recordedDate` (`YYYYMMDD`), which coverage reads before the upload date. A title it does not match, a date that is not a real day, or one after the upload date gives none. Changing the rule re-derives the channel's records on the next index build. An object without a usable `titlePattern` is dropped."
};
// Every key a config.json may carry, in emission order. The unknown-key oracle:
@@ -407,6 +411,7 @@ export const CHANNEL_CONFIG_COERCIONS: {
? Math.min(Math.floor(v), CHANNEL_SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS)
: undefined,
audioCheck: coerceAudioCheck,
+ recordedDate: coerceRecordedDateRule,
};
// A channel's config.json as a ChannelConfig, or null when it is not a channel
diff --git a/common/lib/channelConfigSchema.test.ts b/common/lib/channelConfigSchema.test.ts
@@ -39,7 +39,7 @@ test("one key list: docs = coercions = schema shape, sync-state keys inside it",
assert.deepEqual(Object.keys(CHANNEL_CONFIG_COERCIONS), [...CHANNEL_CONFIG_KEYS]);
assert.deepEqual(Object.keys(channelConfigObjectSchema.shape), [...CHANNEL_CONFIG_KEYS]);
assert.deepEqual(Object.keys(CHANNEL_CONFIG_FIELD_DOCS), [...CHANNEL_CONFIG_KEYS]);
- assert.equal(CHANNEL_CONFIG_KEYS.length, 32);
+ assert.equal(CHANNEL_CONFIG_KEYS.length, 33);
assert.equal(sameKeys, true);
assert.equal(fits, true);
for (const k of CHANNEL_SYNC_STATE_KEYS) assert.ok(CHANNEL_CONFIG_KEYS.includes(k), k);
diff --git a/common/lib/channelConfigSchema.ts b/common/lib/channelConfigSchema.ts
@@ -75,6 +75,7 @@ export const channelConfigObjectSchema = z.object({
cookieMode: field("cookieMode"),
sleepBetweenDownloadsSeconds: field("sleepBetweenDownloadsSeconds"),
audioCheck: field("audioCheck"),
+ recordedDate: field("recordedDate"),
});
// Drop every own key whose value is `undefined` (the OMIT rule, above).
diff --git a/common/lib/fileSchemaDocs.ts b/common/lib/fileSchemaDocs.ts
@@ -22,6 +22,7 @@ import {
CHANNEL_SYNC_STATE_KEYS,
DOWNLOAD_FILTER_FIELD_DOCS,
} from "./channelConfig";
+import { RECORDED_DATE_RULE_FIELD_DOCS } from "./recordedDate";
import { SOCIAL_LINK_FIELD_DOCS } from "./settingsSchema";
import {
RELATED_SITE_GROUP_FIELD_DOCS,
@@ -131,6 +132,13 @@ const CHANNEL_NESTED: Partial<Record<string, KeyTable[]>> = {
defaults: (key) => (key === "enabled" ? "required" : "absent"),
},
],
+ recordedDate: [
+ {
+ path: "recordedDate",
+ docs: RECORDED_DATE_RULE_FIELD_DOCS,
+ defaults: () => "required",
+ },
+ ],
};
function channelKind(key: string): string {
diff --git a/common/lib/recordedDate.test.ts b/common/lib/recordedDate.test.ts
@@ -0,0 +1,80 @@
+// A VOD mirror's recorded date, read off a title (release 20 D1). Synthetic
+// titles throughout.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/recordedDate.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ coerceRecordedDateRule,
+ compileRecordedDateRule,
+ coverageDate,
+ deriveRecordedDate,
+} from "./recordedDate";
+import { parseChannelConfig } from "./channelConfig";
+import { channelConfigSchema } from "./channelConfigSchema";
+
+const ISO = { titlePattern: String.raw`(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})` };
+const WORDS = { titlePattern: String.raw`(?<month>[a-z]+)\.? (?<day>\d{1,2}),? (?<year>\d{4})` };
+const US = { titlePattern: String.raw`(?<month>\d{1,2})/(?<day>\d{1,2})/(?<year>\d{2,4})` };
+
+test("an ISO date in the title", () => {
+ assert.equal(deriveRecordedDate(ISO, "Example Stream VOD 2024-03-05 | part 1", "20240310"), "20240305");
+});
+
+test("month names, whole, cut and dotted, any case", () => {
+ assert.equal(deriveRecordedDate(WORDS, "VOD: March 5, 2024", "20240306"), "20240305");
+ assert.equal(deriveRecordedDate(WORDS, "vod - sept 30 2023 (full)", "20231001"), "20230930");
+ assert.equal(deriveRecordedDate(WORDS, "Stream | DEC. 1, 2022", "20221202"), "20221201");
+ assert.equal(deriveRecordedDate(WORDS, "Stream | Ma 1, 2022", "20221202"), undefined, "two letters name no month");
+ assert.equal(deriveRecordedDate(WORDS, "Stream | Smarch 1, 2022", "20221202"), undefined);
+});
+
+test("a two-digit year is 20YY", () => {
+ assert.equal(deriveRecordedDate(US, "Stream 3/5/24", "20240306"), "20240305");
+ assert.equal(deriveRecordedDate(US, "Stream 12/31/2023", "20240102"), "20231231");
+});
+
+test("not a real day, no match, or after the upload: none", () => {
+ assert.equal(deriveRecordedDate(ISO, "Stream 2023-02-30", "20230401"), undefined);
+ assert.equal(deriveRecordedDate(ISO, "Stream 2023-13-01", "20240401"), undefined);
+ assert.equal(deriveRecordedDate(ISO, "A title with no date", "20240401"), undefined);
+ // A date the title MENTIONS (the next stream) is not the day it was recorded.
+ assert.equal(deriveRecordedDate(ISO, "Next stream 2024-05-01", "20240420"), undefined);
+ // The same day as the upload is fine.
+ assert.equal(deriveRecordedDate(ISO, "Stream 2024-04-20", "20240420"), "20240420");
+ // No upload date to compare: the title's date stands.
+ assert.equal(deriveRecordedDate(ISO, "Stream 2024-04-20"), "20240420");
+});
+
+test("a rule needs a compiling pattern with all three named groups, or it is dropped", () => {
+ assert.deepEqual(coerceRecordedDateRule({ titlePattern: ` ${ISO.titlePattern} ` }), ISO);
+ assert.equal(coerceRecordedDateRule({ titlePattern: String.raw`(?<year>\d{4})-(?<month>\d{2})` }), undefined);
+ assert.equal(coerceRecordedDateRule({ titlePattern: "(?<year>[" }), undefined);
+ // The download filter's safety rules: no nested quantifier, at most 200 characters.
+ assert.equal(
+ coerceRecordedDateRule({ titlePattern: String.raw`(?<year>(\d+)+)-(?<month>\d+)-(?<day>\d+)` }),
+ undefined,
+ );
+ assert.equal(coerceRecordedDateRule({ titlePattern: `${ISO.titlePattern}${"x".repeat(200)}` }), undefined);
+ assert.equal(coerceRecordedDateRule({ titlePattern: " " }), undefined);
+ assert.equal(coerceRecordedDateRule({ titlePattern: 7 }), undefined);
+ assert.equal(coerceRecordedDateRule("(?<year>.)"), undefined);
+ assert.equal(coerceRecordedDateRule([ISO]), undefined);
+ assert.equal(compileRecordedDateRule({ titlePattern: "(?<year>[" }), null);
+ assert.ok(compileRecordedDateRule(ISO)?.flags.includes("i"));
+});
+
+test("config.json: the key round-trips through both readers, and a bad one is omitted", () => {
+ const raw = { handling: "youtube", name: "Example VODs", recordedDate: ISO };
+ assert.deepEqual(parseChannelConfig(raw)?.recordedDate, ISO);
+ assert.deepEqual(channelConfigSchema.parse(raw)?.recordedDate, ISO);
+ const bad = { handling: "youtube", recordedDate: { titlePattern: "no groups" } };
+ assert.equal("recordedDate" in (parseChannelConfig(bad) ?? {}), false);
+ assert.equal("recordedDate" in (channelConfigSchema.parse(bad) ?? {}), false);
+});
+
+test("coverage reads the recorded date before the upload date", () => {
+ assert.equal(coverageDate({ uploadDate: "20240310", recordedDate: "20240305" }), "20240305");
+ assert.equal(coverageDate({ uploadDate: "20240310" }), "20240310");
+});
diff --git a/common/lib/recordedDate.ts b/common/lib/recordedDate.ts
@@ -0,0 +1,136 @@
+// A VIDEO'S RECORDED DATE, for a channel that mirrors another's streams
+// (release 20 D1).
+//
+// A VOD-mirror channel uploads yesterday's stream today, or last year's
+// streams in a batch, so its videos' `upload_date` is the date of the COPY,
+// not of the stream it copies. Anything that asks "what was said when" —
+// coverage above all ("held videos by date, gaps") — reads the wrong day for
+// every one of them. The date the stream happened is usually in the title
+// ("Stream 2024-03-05", "VOD | March 5, 2024"), in a shape that is the same
+// for the whole channel. So the channel names that shape once, in its
+// config.json (`recordedDate.titlePattern`), and the index derives each
+// record's `recordedDate` from its title (controller/buildIndex.ts).
+//
+// THE PATTERN is a case-insensitive regex SOURCE with three NAMED groups:
+// year four digits, or two (read as 20YY)
+// month 1–12, or an English month name, whole or cut to 3+ letters
+// day 1–31
+// e.g. `(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})` or
+// `(?<month>[a-z]+)\.? (?<day>\d{1,2}),? (?<year>\d{4})`. A pattern that does
+// not compile, or lacks a group, is dropped by the config coercion — the rule
+// is then absent, never half-applied.
+//
+// A DERIVED DATE MUST BE A REAL DAY, AND NO LATER THAN THE UPLOAD: a recording
+// precedes its copy, so a title date after `upload_date` is a date the title
+// mentions (a schedule, a sequel), not the day it was recorded, and none is
+// set. A title the pattern does not match has none either; readers fall back
+// to the upload date (`coverageDate`).
+//
+// Pure, client-safe (channelConfig.ts imports the coercion).
+
+import { downloadFilterPatternProblem } from "./downloadFilters";
+
+export type RecordedDateRule = {
+ titlePattern: string;
+};
+
+export const RECORDED_DATE_RULE_FIELD_DOCS: Record<keyof RecordedDateRule, string> = {
+ titlePattern:
+ "Case-insensitive regex SOURCE (no delimiters, no flags) matched against each video's title, with three named groups: `year` (four digits, or two read as 20YY), `month` (1–12, or an English month name, whole or cut to three or more letters) and `day`. Trimmed; at most 200 characters, no nested quantifier (the download filter's rules). A pattern that breaks any of this drops the whole rule.",
+};
+
+const GROUPS = ["year", "month", "day"] as const;
+
+const MONTHS = [
+ "january",
+ "february",
+ "march",
+ "april",
+ "may",
+ "june",
+ "july",
+ "august",
+ "september",
+ "october",
+ "november",
+ "december",
+];
+
+// Why a pattern cannot be a rule, as the tail of a sentence ("… <problem>"),
+// or null. It must compile, name all three groups, and pass the download
+// filter's safety check (downloadFilterPatternProblem: at most 200 characters,
+// no nested quantifier) — it runs over every title of the channel on the
+// server's one thread, at every index build.
+export function recordedDatePatternProblem(pattern: string): string | null {
+ try {
+ new RegExp(pattern, "i");
+ } catch (e) {
+ return `is not a valid regular expression: ${(e as Error).message}`;
+ }
+ const missing = GROUPS.filter((g) => !pattern.includes(`(?<${g}>`));
+ if (missing.length > 0) {
+ return `must name the groups year, month and day (missing: ${missing.join(", ")})`;
+ }
+ return downloadFilterPatternProblem(pattern);
+}
+
+// The rule's regex, or null when the pattern has a problem (above).
+export function compileRecordedDateRule(rule: RecordedDateRule): RegExp | null {
+ if (recordedDatePatternProblem(rule.titlePattern)) return null;
+ return new RegExp(rule.titlePattern, "i");
+}
+
+export function coerceRecordedDateRule(v: unknown): RecordedDateRule | undefined {
+ if (!v || typeof v !== "object" || Array.isArray(v)) return undefined;
+ const raw = (v as Record<string, unknown>).titlePattern;
+ if (typeof raw !== "string") return undefined;
+ const rule = { titlePattern: raw.trim() };
+ if (!rule.titlePattern || !compileRecordedDateRule(rule)) return undefined;
+ return rule;
+}
+
+function monthNumber(s: string): number | null {
+ if (/^\d{1,2}$/.test(s)) {
+ const n = Number(s);
+ return n >= 1 && n <= 12 ? n : null;
+ }
+ const word = s.toLowerCase().replace(/\.$/, "");
+ if (word.length < 3) return null;
+ const i = MONTHS.findIndex((m) => m.startsWith(word));
+ return i < 0 ? null : i + 1;
+}
+
+const pad = (n: number, w = 2) => String(n).padStart(w, "0");
+
+// The recorded date (`YYYYMMDD`, the shape of `upload_date`) a rule reads off a
+// title, or undefined: no match, not a real day, or later than `uploadDate`
+// (when given).
+export function deriveRecordedDate(
+ rule: RecordedDateRule | RegExp,
+ title: string,
+ uploadDate?: string,
+): string | undefined {
+ const re = rule instanceof RegExp ? rule : compileRecordedDateRule(rule);
+ if (!re) return undefined;
+ const g = re.exec(title)?.groups;
+ if (!g || !g.year || !g.month || !g.day) return undefined;
+ if (!/^\d{2}$|^\d{4}$/.test(g.year) || !/^\d{1,2}$/.test(g.day)) return undefined;
+ const year = g.year.length === 2 ? 2000 + Number(g.year) : Number(g.year);
+ const month = monthNumber(g.month);
+ const day = Number(g.day);
+ if (month === null || day < 1) return undefined;
+ // A real day: the Date round trip refuses Feb 30 and friends.
+ const d = new Date(Date.UTC(year, month - 1, day));
+ if (d.getUTCFullYear() !== year || d.getUTCMonth() !== month - 1 || d.getUTCDate() !== day) {
+ return undefined;
+ }
+ const out = `${pad(year, 4)}${pad(month)}${pad(day)}`;
+ if (uploadDate && /^\d{8}$/.test(uploadDate) && out > uploadDate) return undefined;
+ return out;
+}
+
+// The date a video's content belongs to: its recorded date when its channel's
+// rule gave one, else its upload date. What coverage reads.
+export function coverageDate(s: { uploadDate: string; recordedDate?: string }): string {
+ return s.recordedDate ?? s.uploadDate;
+}
diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts
@@ -37,6 +37,12 @@ export type TranscriptSummary = {
// an untagged corpus's pages stay byte-identical to the ones already on disk
// and the build's sha1 skip still short-circuits.
curatedTags?: string[];
+ // The day a VOD mirror's video was RECORDED (`YYYYMMDD`, the shape of
+ // `uploadDate`), read off its title by its channel's `recordedDate` rule
+ // (lib/recordedDate.ts) when the index builds the record. OMITTED when the
+ // channel has no rule or the title gives no date, so every other channel's
+ // pages stay byte-identical. Coverage reads it first (`coverageDate`).
+ recordedDate?: string;
};
export type DisplaySummary = {
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **A channel that mirrors another's streams can date its videos by the stream, not the upload.** A new channel setting, "Recorded date from the title (regex)" on the Configure form (`recordedDate.titlePattern` in config.json; `pnpm ops channel-config` with `recordedDateTitlePattern`), names where a title carries the recording's date — a regex with the groups `year`, `month` (a number or a month name) and `day`. The index then gives each of that channel's videos a `recordedDate`, which coverage reads before the upload date; a title without a date, or with one after the upload, keeps the upload date. Changing the pattern re-dates the channel's videos at the next index build.
- **A clip window of a video whose source is saved is cut from it, not fetched.** `fetch_clip`, `POST /api/media/fetch-window` and `pnpm ops fetch-windows` now cut a window out of the video's saved container (a persisted source, a full-source fetch, or media attached from a local archive) when it covers the seconds asked for, and answer at once as a cached window — no request to the platform, so a deleted channel's held videos are clippable. The window's sidecar records `source: "saved-video"`, and the video page marks it "cut from the saved video". A batch runs such windows as their own job on `clips:saved-video`, outside every platform's queue, hold and cooldown. A saved video whose file cannot be read (its drive unplugged, the file gone) is refused with the media guard's sentence rather than fetched; one that ends before the window is fetched as before.
- **`pnpm e2e` runs against a production build, and rebuilds it when the code changed.** The editor and umtool suites now run under `next start` by default (release 19's full editor suite: 24 min, against 71 under `next dev`). Before a run starts, the build's stamp — a fingerprint of the files the build reads, uncommitted edits included — is checked against the tree, and a stale build is rebuilt first through the heavy slot under a 5 GB cap, with the reason printed (`e2e build: rebuilding editor — common changed since the build …`). The test build has its own directory (`editor/.next/e2e`, `umtool/.next-e2e-start`), so it never replaces the build a running editor or umtool serves. `E2E_MODE=dev` runs `next dev` for iterating on one spec. The export and homepage suites stay on `next dev` (they are static exports). Every suite also writes `test-results/timings.json`, and `node scripts/e2e-timings.mjs` prints each spec file's time against the branch's last run. Four slow tests no longer wait on real clocks: the first rate-limit cooldown is 20 s and the clip-window gap 2 s on the test server only (`E2E_BACKOFF_BASE_MS`, `E2E_CLIP_WINDOW_GAP_MS`).
- **An X fetch with a `limit` stops at that many posts.** "Fetch posts" with `limit` (`pnpm ops fetch-posts {"limit": 400}`) on a gallery-dl X channel read the whole history instead — a new channel walked 3,803 posts under the rate limit and held the platform queue for hours — because the cap counted media files, which a metadata-only read has almost none of. It now caps the posts themselves.
diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx
@@ -870,6 +870,14 @@ export function ChannelForm({
</span>
</label>
<Field
+ label="Recorded date from the title (regex)"
+ name="recordedDateTitlePattern"
+ state={state}
+ defaultValue={c?.recordedDate?.titlePattern ?? ""}
+ placeholder="(the upload date)"
+ hint="For a channel that mirrors another's streams, whose upload dates are the copies'. Case-insensitive regex naming three groups — year, month (a number or an English month name) and day — e.g. (?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2}). The index reads each video's recorded date from its title, and coverage uses it before the upload date. A title it does not match keeps the upload date."
+ />
+ <Field
label="Cookies from browser"
name="cookiesFromBrowser"
state={state}
diff --git a/editor/app/channels/components/channelConfigToForm.ts b/editor/app/channels/components/channelConfigToForm.ts
@@ -50,6 +50,7 @@ export const CHANNEL_FORM_VALUES = [
"downloadFilterInclude",
"downloadFilterExclude",
"downloadFilterRejectedLivestreams",
+ "recordedDateTitlePattern",
"audioCheckIntervalSeconds",
"audioCheckMaxRollbacks",
"audioCheckCopyTimeoutSeconds",
@@ -114,6 +115,7 @@ export function channelConfigToFormData(config: ChannelConfig): FormData {
"downloadFilterRejectedLivestreams",
config.downloadFilter?.rejectedLivestreams,
);
+ put(fd, "recordedDateTitlePattern", config.recordedDate?.titlePattern);
if (config.audioCheck?.enabled) {
fd.set("audioCheckEnabled", "on");
putNum(fd, "audioCheckIntervalSeconds", config.audioCheck.intervalSeconds);
diff --git a/editor/app/channels/components/parseChannelForm.test.ts b/editor/app/channels/components/parseChannelForm.test.ts
@@ -0,0 +1,63 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { parseChannelForm, CHANNEL_FORM_FIELDS } from "./parseChannelForm";
+import {
+ applyChannelFormPatch,
+ channelConfigToFormData,
+ validateChannelFormPatch,
+} from "./channelConfigToForm";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/channels/components/parseChannelForm.test.ts"
+//
+// The Configure form's recorded-date field (release 20 D1): parsed into
+// `recordedDate.titlePattern`, refused when it could not be a rule, round-tripped
+// through the ops patch, and cleared by an empty value.
+
+const ISO = String.raw`(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})`;
+
+function form(fields: Record<string, string>): FormData {
+ const fd = new FormData();
+ fd.set("name", "Example VODs");
+ fd.set("handling", "youtube");
+ for (const [k, v] of Object.entries(fields)) fd.set(k, v);
+ return fd;
+}
+
+test("the form writes recordedDate.titlePattern, trimmed", () => {
+ const { config } = parseChannelForm(form({ recordedDateTitlePattern: ` ${ISO} ` }));
+ assert.deepEqual(config.recordedDate, { titlePattern: ISO });
+ assert.equal("recordedDate" in parseChannelForm(form({})).config, false);
+ // A form save clears a stored rule the form left blank.
+ assert.ok((CHANNEL_FORM_FIELDS as readonly string[]).includes("recordedDate"));
+});
+
+test("a pattern that cannot be a rule is refused with why", () => {
+ assert.throws(
+ () => parseChannelForm(form({ recordedDateTitlePattern: String.raw`(?<year>\d{4})` })),
+ /Recorded-date pattern must name the groups year, month and day \(missing: month, day\)/,
+ );
+ assert.throws(
+ () => parseChannelForm(form({ recordedDateTitlePattern: "(?<year>[" })),
+ /Recorded-date pattern is not a valid regular expression/,
+ );
+ assert.throws(
+ () =>
+ parseChannelForm(
+ form({ recordedDateTitlePattern: String.raw`(?<year>(\d+)+)-(?<month>\d+)-(?<day>\d+)` }),
+ ),
+ /Recorded-date pattern contains a nested quantifier/,
+ );
+});
+
+test("the ops patch sets and clears it over the channel's current form", () => {
+ validateChannelFormPatch({ recordedDateTitlePattern: ISO });
+ const fd = channelConfigToFormData({ handling: "youtube", name: "Example VODs" });
+ applyChannelFormPatch(fd, { recordedDateTitlePattern: ISO });
+ const set = parseChannelForm(fd).config;
+ assert.deepEqual(set.recordedDate, { titlePattern: ISO });
+ const back = channelConfigToFormData(set);
+ assert.equal(back.get("recordedDateTitlePattern"), ISO);
+ applyChannelFormPatch(back, { recordedDateTitlePattern: "" });
+ assert.equal("recordedDate" in parseChannelForm(back).config, false);
+});
diff --git a/editor/app/channels/components/parseChannelForm.ts b/editor/app/channels/components/parseChannelForm.ts
@@ -25,6 +25,7 @@ import {
isSourceVideoQuality,
} from "yt-dlp-transcript-common/ytdlp/downloadFormat";
import { downloadFilterPatternProblem } from "yt-dlp-transcript-common/lib/downloadFilters";
+import { recordedDatePatternProblem } from "yt-dlp-transcript-common/lib/recordedDate";
import { isRejectedLivestreamMode } from "yt-dlp-transcript-common/lib/channelConfig";
export type ParsedChannelForm = {
@@ -58,6 +59,7 @@ export const CHANNEL_FORM_FIELDS = [
"cookieMode",
"audioCheck",
"downloadFilter",
+ "recordedDate",
] as const satisfies ReadonlyArray<keyof ChannelConfig>;
// Maps FormData -> ChannelConfig. Throws on invalid input so the server
@@ -280,6 +282,15 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm {
}
: undefined;
+ // A VOD mirror's recorded date (release 20 D1, lib/recordedDate.ts): a regex
+ // with named groups year/month/day, refused here — as the download filter's
+ // are — rather than saved and silently dropped by the config reader.
+ const recordedDateTitlePattern = stringOrUndef(formData, "recordedDateTitlePattern");
+ if (recordedDateTitlePattern) {
+ const problem = recordedDatePatternProblem(recordedDateTitlePattern);
+ if (problem) throw new Error(`Recorded-date pattern ${problem}`);
+ }
+
const audioCheckEnabled = formData.get("audioCheckEnabled") != null;
let audioCheck: AudioCheckConfig | undefined;
if (audioCheckEnabled) {
@@ -369,6 +380,9 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm {
if (cookieMode) config.cookieMode = cookieMode;
if (audioCheck) config.audioCheck = audioCheck;
if (downloadFilter) config.downloadFilter = downloadFilter;
+ if (recordedDateTitlePattern) {
+ config.recordedDate = { titlePattern: recordedDateTitlePattern };
+ }
return { name, slug, config };
}
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -7921,6 +7921,9 @@ on anchors elsewhere in this file:
zero cues. `maybeMissingBuild.test.ts`'s VTT is such a file, which is harmless there.
- The same rule makes a video whose English track is a *manual* caption (no inline tags) index
as 0 cues. That is rare: 0 in 3,000 sampled of the-quartering, 6 of chibi-reviews.
+ > **Superseded by `200a3105` (2026-10-06; verified 2026-10-10, release 20 D3):** `parseVtt` reads a document
+ > with no timing tag as plain cue blocks (every line of a cue is its text), so a manual caption and the
+ > livestream `en` shape parse; see "The caption-track rule" at the end of this file.
- **Measured before the fix,** on the whole-pool stats of 2026-09-28T20:40Z:
- 49,798 transcripts shown, of about 77,000 on disk;
- 24,710 records with `hasTranscript` but no `transcribedDate`;
@@ -9092,3 +9095,32 @@ The record is [`landed-2026-10.md`](landed-2026-10.md). Every anchor below was r
- **/sites** lists every site's articles (`umtool/lib/articles/sites.ts`): published ids unioned with draft
report dirs, each with its open notes, its source and its linked video. An article row's `kind` is the REPORT's
kind (`factcheck|sweep`), not a project kind.
+
+## The caption-track rule (verified 2026-10-10, release 20 D3, against `585be292`)
+
+The `en` → 0 cues bug (filed 2026-10-01: the index preferred `en` over `en-orig`, and some livestream VODs' `en`
+parsed to 0 cues, indexing the video textless) is CLOSED by `200a3105` (transcripts/en-track-fallback).
+- **One rule, by name then by content** (`common/lib/videoStatus.ts`, `CAPTION_TRACK_RULE_VERSION` 1): English
+ tracks ranked operator pin (`transcript-pin.json` → `transcript.en.vtt` first) > `en-orig` > `en` > regional
+ (`en-US`, …) > `en-en-*`, ties by name (`englishVttsByPreference`); the cues are the first track in that order
+ that parses to at least one cue (`readEnglishVttCues`); all empty → the most preferred, with no cues.
+ `resolvePrimaryVtt` is the name half only (the editor's "primary" label and change detection).
+- **Every cue reader goes through it:** the index (`buildIndex.ts` → `readEnglishVttCues`), normalize
+ (`normalizeTranscript.ts`), report compose (`composeReports.ts`), the editor's track reader
+ (`captionTracks-server.ts`); umtool's `report-to-video/cues.mjs` carries a copy that `captionTrack.test.ts` holds
+ equal. `buildIndex.ts`'s other `parseVtt` call is the non-English/live-chat subtitle tracks.
+- **`parseVtt` reads two shapes** (`common/lib/vtt.ts`), decided per document by the presence of an inline
+ `<hh:mm:ss.mmm>` tag: rolling (keep the tagged line) or cue blocks (every line, all tags stripped, entities
+ decoded). The fixture of the second is `common/lib/__fixtures__/vtt-cue-blocks.vtt`.
+- **An index built under an older rule is re-read once** for the records the rule can reach (two or more English
+ VTTs, or stored cues empty), logged per channel (`Caption track v1: <slug>: N re-read, M now read different text,
+ K had no text and now do.`), then `meta.captionTrackRule` is recorded.
+- **Tests:** `lib/captionTrack.test.ts` (order, pin, the fall-through, umtool's copy), `lib/vtt.test.ts` (both
+ shapes), `controller/buildIndexCaptionTrack.test.ts` (through the real `buildIndex`: `en-orig` beside a served
+ cue-block `en`, a lone cue-block `en`, the one-shot re-read, and — release 20 D3 — an empty `en-orig` before an
+ `en` with text and an empty `en` before an `en-US` with text).
+- **Not closed by it (found 2026-10-10, not fixed):** the stats cache (see "The stats cache key") keys a video's
+ stat on its `mtimes` record, and the caption-track pass rewrites a record's cues with an UNCHANGED `mtimes`
+ record (`buildIndex.ts`, the `mtimes.put` after processing writes the same `metaMs`/`transcriptMs`/…). So a
+ record the pass took from no text to text keeps `hasTranscript: false` and its old `cueCount` in the stats until
+ something else moves its record.
diff --git a/plans/STATE.md b/plans/STATE.md
@@ -22,6 +22,11 @@ first. Nothing of release 18 is live.
[`release-20.md`](release-20.md) (data model); Track B release 19 B1–B6 (machine safety and tooling); Track C docs
and plans (OPERATING.md and `archilyzer docs cli`, doc fixes, homepage docs, FACTS). Integration branch
`r19/integration`, branched from `4cffda3f`.
+- **The `en` → 0 cues caption bug (filed 2026-10-01) is CLOSED** by `200a3105` (en-orig first, empty tracks fall
+ through, cue-block VTTs parse, a one-shot index re-read); verified against the tree and an index fixture by release
+ 20 D3 (2026-10-10; FACTS, "The caption-track rule"). Left open: the stats cache does not see that pass's re-reads.
+- **Release 20 D2 (Twitch ids) is HELD for an operator ruling**: built and kept on `r20/twitch-ids`, reverted on
+ `r20/d2-r20`; the two options are in `release-20.md` ("Slice D2 — held").
**Previously (2026-10-06): release 18 — publishing as queueable stages — is complete on `r18/integration`** (record:
[`release-18.md`](release-18.md): slices S1 the stage contract, stamps, lock, bundles and CLI; S2 deploy hardening; S3
diff --git a/plans/release-20.md b/plans/release-20.md
@@ -35,4 +35,97 @@ spec(s) D3 names, then the full editor suite once if any editor code changed.
(Each slice adds a "### Slice <X>, as shipped" section here, before "## Rollout".)
+Track D of the overnight batch (2026-10-10), on `r20/d2-r20` off `r20/integration` (`585be292`), after release 21
+D2 on the same branch; D3 first, then D1; D2 built, then held (below). Every fixture is synthetic; nothing read the
+corpus beyond config.json key names and shapes, and a count of two Twitch channels' directories (below).
+
+### Slice D3, as shipped — the caption-track bug, closed
+
+**Verified against the tree.** The `en` → 0 cues bug (filed 2026-10-01) is closed by `200a3105`: one rule in
+`common/lib/videoStatus.ts` (`CAPTION_TRACK_RULE_VERSION` 1) — pin > `en-orig` > `en` > regional > `en-en-*` by name,
+then the first track that parses to at least one cue (`readEnglishVttCues`) — and `parseVtt` reads a document with no
+inline timing tag as plain cue blocks, the livestream `en` shape that used to parse to nothing. Every cue reader goes
+through the rule: the index, normalize, report compose, the editor's track reader, and umtool's `cues.mjs` copy (held
+equal by `captionTrack.test.ts`); `buildIndex.ts`'s other `parseVtt` call is the non-English and live-chat tracks. An
+index built under the old rule is re-read once for the records the rule reaches.
+
+**Added:** `controller/buildIndexCaptionTrack.test.ts` +1 through the real `buildIndex` — an empty `en-orig` before an
+`en` with text, and an empty `en` before an `en-US` with text (no `en-orig`), both read the track with text. FACTS
+gains "The caption-track rule" (and a superseded note on the old "a caption fixture must carry timing tags" line);
+STATE closes the bug.
+
+**Found and left (report only, as ruled): the stats cache does not see the caption-track pass.** `buildStats` keys a
+video's stat on its index `mtimes` record (`indexSignature`; `plans/stats-cache-key.md`, merged `10cefd15`, in the
+tree unchanged: schema 6, the whole record plus `indexKey`). The caption-track pass re-reads a record's cues and writes
+the SAME `mtimes` record back (no input file moved), so a record the pass took from no text to text keeps
+`hasTranscript: false` and its old `cueCount` in the stats until something else moves its record.
+
+### Slice D1, as shipped — recorded dates for VOD-mirror channels
+
+**What it does.** `common/lib/recordedDate.ts` (pure, client-safe): a channel's `recordedDate: {titlePattern}` is a
+case-insensitive regex source naming the groups `year` (four digits, or two read as 20YY), `month` (1–12, or an English
+month name, whole or cut to three or more letters) and `day`; it must compile, name all three, and pass the download
+filter's safety check (at most 200 characters, no nested quantifier), or the config reader drops the whole rule.
+`deriveRecordedDate` gives `YYYYMMDD` for a real day no later than the upload date (a later title date is one the
+title mentions, not the recording's), else nothing. `coverageDate(summary)` is `recordedDate ?? uploadDate`, the hook
+coverage reads.
+- **Config:** `ChannelConfig.recordedDate` through `CHANNEL_CONFIG_COERCIONS` / `channelConfigSchema`, with a nested
+ key table; CHANNEL.md regenerated (`archilyzer docs files`).
+- **The index** (`buildIndex.ts`): after a record's summary is settled (from metadata or a fresh `transcript.cues.json`),
+ `recordedDate` is set from the channel's compiled rule or deleted. `recordedDateRule:<slug>` in the index meta holds
+ the pattern a channel's records were derived under; a build that finds another re-processes that channel's records
+ once (`Recorded dates: <slug>: its rule changed; N record(s) re-derived.`), a held channel left for a later build.
+ No channel has a rule today, so the first build after the rollout re-processes nothing.
+- **`TranscriptSummary.recordedDate?`**, omitted when absent: it reaches a channel's published transcript pages only
+ once that channel has a rule, and no page changes until then.
+- **Setting it:** the Configure form's Advanced section gains "Recorded date from the title (regex)"
+ (`recordedDateTitlePattern`, refused with why when it could not be a rule; in `CHANNEL_FORM_FIELDS`, so clearing it
+ clears the key), and `pnpm ops channel-config {"slug":…,"patch":{"recordedDateTitlePattern":…}}` sets it through the
+ same parser.
+- **Coverage** (release 19 A9's `channel_coverage`) is not on this branch; `coverageDate` is what it reads at merge.
+
+**Which channels are mirrors** was read from config.json shapes only: no key marks one; the candidates are the
+channels whose name or URL says VODs, mirror or archive (listed in the track report). No channel's config was changed;
+the patterns are the operator's to write.
+
+### Slice D2 — one Twitch id: held, operator ruling owed
+
+Built as `433815d4` and reverted on this branch; the build is kept on `r20/twitch-ids` (orchestrator, 2026-10-10:
+it changes published URLs, which is the operator's ruling).
+
+**What is there.** A Twitch VOD's directory is the canonical id, the URL's `/videos/<n>` (`extractVideoId`);
+`summarize` takes yt-dlp's native `meta.id`, `v<n>`, so the record's id and slug say `v<n>`. Two channels carry
+Twitch VODs: `hasanabi` (180 directories) and `shondo-twitch` (37). No curated tag or site report references a `v<n>`
+id (counts 0).
+
+**Readers that miss the join today** — each opens `data/<record id>/` with the record's `v<n>`:
+- MCP `fetch_clip` → the editor's fetch-window (`data/v<n>/` holds no metadata, so no URL and no cache);
+- the evidence-clip tiers (`common/lib/evidenceClip.mjs`, `videoDirOf`) behind reports prepare and report-to-video's
+ local sources, and `publish/reportMedia.ts`'s per-record availability read;
+- umtool's local cue lookup (`report-to-video/cues.mjs`, `data/<videoId>/transcript.cues.json`).
+
+**The two options:**
+- **(a) Records take `<n>`** (what `r20/twitch-ids` does: `canonicalTwitchVideoId` in `summarize` and in
+ `readNormalizedTranscript`, a one-shot index re-key, the player still given `v<n>`). 217 public transcript URLs move
+ from `…/v<n>` to `…/<n>` (hasanabi 180, shondo-twitch 37); the hub needs tombstones for the old ones.
+- **(b) The published `v<n>` stays**, and the join normalizes on the directory side (a reader maps a Twitch record id
+ `v<n>` to `data/<n>/`). No URL moves.
+
+**Commits**
+
+| Commit | What |
+|---|---|
+| `04a075b3` | `common:` D3 — the fall-through index test; FACTS "The caption-track rule"; STATE closes the bug |
+| `96409e24` | `common:` D1 — the `recordedDate` rule, its derivation, the index pass, the form field; CHANNEL.md |
+| `433815d4` | `common:` D2 — `canonicalTwitchVideoId`, `normalizeSummaryId`, the one-shot re-key, the player (kept on `r20/twitch-ids`) |
+| `949f7e41` | `common:` revert of D2 — held for an operator ruling |
+
+**Gates.** `pnpm -r --no-bail --workspace-concurrency=1 exec tsc --noEmit` clean at each commit. New tests:
+`lib/recordedDate.test.ts` 7, `controller/buildIndexRecordedDate.test.ts` 2 (derived, absent without a rule, a date
+after the upload dropped; a changed rule re-derives once, a removed one drops the field from the pages),
+`controller/buildIndexCaptionTrack.test.ts` +1, editor `app/channels/components/parseChannelForm.test.ts`
+3 (D2's 6 left with its revert and live on `r20/twitch-ids`). The whole common suite 3,661/3,661 with D2 in
+(release 21 D2's tests included), and after the revert the touched index, normalize, caption-track and architecture tests 54/54, tsc clean; editor unit 228/228; export unit 118/118;
+mcp 293/293; `archilyzer docs files --check` and `docs env --check` clean. E2E, one run in the foreground (start of the queue after the heavy slot freed): `transcript-source.spec.ts` (D3's index spec, 4) and `ops-api.spec.ts` (13, the channel-config round trip among them) — 17 passed, 1.5 min. Release end per the plan: the full editor suite once, the orchestrator's.
+
## Rollout