commit 0f614b3a43376d07d562101aae4c83424749ccd3
parent c3890608162a0f674ecb611034f333fc35200d28
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 10 Oct 2026 01:02:30 -0400
common: release 20 D1 — recorded dates for VOD-mirror channels, read off the title by a per-channel rule
A channel's config.json gains `recordedDate: {titlePattern}` (a
case-insensitive regex with named groups year, month, day; month a number or
an English month name; checked like the download filter's patterns). The
index derives each record's `recordedDate` (YYYYMMDD) from its title — a real
day no later than the upload date, else none — and re-derives a channel's
records once when its rule changes (`recordedDateRule:<slug>` in the index
meta). `coverageDate` reads it before the upload date. The Configure form and
`ops channel-config` set it (`recordedDateTitlePattern`); CHANNEL.md
regenerated. Omitted everywhere a channel has no rule, so no page changes
until one is set.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
14 files changed, 553 insertions(+), 1 deletion(-)
diff --git a/CHANNEL.md b/CHANNEL.md
@@ -46,6 +46,7 @@ Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/f
| `cookieMode` | config | Per-channel override of the global cookie mode. Absent = inherit. |
| `sleepBetweenDownloadsSeconds` | config | Per-channel override for the global pause between downloads. Absent = inherit; 0 = no sleep; floored and capped at 600. |
| `audioCheck` | config | Opt-in audio-integrity checking for sources that intermittently serve corrupt audio mid-download (e.g. Odysee "original"): the managed downloader periodically validates the in-progress `.part` file and rolls back to the last known-good snapshot on corruption. transcribe-handling only. See [`audioCheck`](#audiocheck). |
+| `recordedDate` | config | For a channel that MIRRORS another's streams (a VOD archive): how to read the date a video was RECORDED from its title, since its `upload_date` is the date of the copy. The index stores the result as the record's `recordedDate` (`YYYYMMDD`), which coverage reads before the upload date. A title it does not match, a date that is not a real day, or one after the upload date gives none. Changing the rule re-derives the channel's records on the next index build. An object without a usable `titlePattern` is dropped. See [`recordedDate`](#recordeddate). |
#### `downloadFilter`
@@ -65,3 +66,9 @@ Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/f
| `maxRollbacks` | absent | Rollbacks to the last known-good snapshot before the download is given up; clamped to [1, 20]. Absent = 5. |
| `copyTimeoutSeconds` | absent | Seconds allowed for the snapshot copy; clamped to [5, 120]. Absent = 30. |
| `resumeDuringProbe` | absent | When false (default), yt-dlp stays SIGSTOPped across each probe, so it never downloads bytes a malformed verdict would discard and force a re-fetch — minimising HTTP 429 risk. True = the legacy behaviour: resume right after the snapshot copy and probe while the download keeps running. |
+
+#### `recordedDate`
+
+| Key | Default | Description |
+|---|---|---|
+| `titlePattern` | required | Case-insensitive regex SOURCE (no delimiters, no flags) matched against each video's title, with three named groups: `year` (four digits, or two read as 20YY), `month` (1–12, or an English month name, whole or cut to three or more letters) and `day`. Trimmed; at most 200 characters, no nested quantifier (the download filter's rules). A pattern that breaks any of this drops the whole rule. |
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -173,6 +173,7 @@ import {
reapplyCuratedTags,
} from "./curatedTagsIndex";
import { tagsForSite } from "../lib/curatedTags";
+import { compileRecordedDateRule, deriveRecordedDate } from "../lib/recordedDate";
// v10: multi-site build. Shared per-channel transcript/subs pages are written
// once; per-site summaries + subs manifests are filtered selections. Bumped to
@@ -222,6 +223,13 @@ const PLATFORM_LABELS_KEY = "platformLabels";
// how many now read different text, how many had none and now do.
const CAPTION_TRACK_KEY = "captionTrackRule";
+// RECORDED DATES — per channel, the same "re-derive what the change reaches"
+// shape: `recordedDateRule:<slug>` holds the `recordedDate.titlePattern` the
+// channel's records were last derived under (absent: none). A build that finds
+// another re-processes every record of that channel, then records the
+// pattern.
+const RECORDED_DATE_RULE_KEY_PREFIX = "recordedDateRule:";
+
// A cue list's text, for "did the words change" — timing alone is not a
// different transcript.
function cueText(list: readonly Cue[]): string {
@@ -933,6 +941,41 @@ export async function buildIndex({
}
let altTrackRecords = 0;
+ // RECORDED DATES (release 20 D1, lib/recordedDate.ts): each channel's rule,
+ // compiled once, and the channels whose rule differs from the one their
+ // records were derived under (RECORDED_DATE_RULE_KEY_PREFIX) — every record
+ // of those is re-processed like a changed one. A held channel is left for
+ // the build that can read it.
+ const recordedDateRes = new Map<string, RegExp>();
+ const recordedDateRuleChanged: string[] = [];
+ for (const [slug, cfg] of channelConfigs) {
+ const re = cfg.recordedDate ? compileRecordedDateRule(cfg.recordedDate) : null;
+ if (re) recordedDateRes.set(slug, re);
+ if (held.has(slug)) continue;
+ const stored = meta.get(`${RECORDED_DATE_RULE_KEY_PREFIX}${slug}`);
+ const now = re ? cfg.recordedDate!.titlePattern : undefined;
+ if ((typeof stored === "string" ? stored : undefined) !== now) {
+ recordedDateRuleChanged.push(slug);
+ }
+ }
+ if (recordedDateRuleChanged.length > 0 && !schemaBumped) {
+ const ruled = new Set(recordedDateRuleChanged);
+ const queued = new Set(
+ [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])),
+ );
+ const perChannel = new Map<string, number>();
+ for (const s of live) {
+ if (!ruled.has(s.channelSlug)) continue;
+ const pk: PathKey = [s.channelSlug, s.videoDir];
+ if (!mtimes.get(pk)) continue;
+ perChannel.set(s.channelSlug, (perChannel.get(s.channelSlug) ?? 0) + 1);
+ if (!queued.has(pathKeyId(pk))) changed.push(s);
+ }
+ for (const [slug, n] of perChannel) {
+ log(`Recorded dates: ${slug}: its rule changed; ${n} record(s) re-derived.`);
+ }
+ }
+
const anyMutations =
added.length > 0 || changed.length > 0 || removed.length > 0;
@@ -1045,6 +1088,15 @@ export async function buildIndex({
log(`Skipping ${s.channelSlug}/${s.videoDir}: no upload_date`);
return;
}
+ // The channel's recorded-date rule, applied to the summary as it
+ // stands (a normalized one included): set when it reads a date, and
+ // never carried over from an older derivation.
+ delete summary.recordedDate;
+ const recordedRe = recordedDateRes.get(s.channelSlug);
+ if (recordedRe) {
+ const recorded = deriveRecordedDate(recordedRe, summary.title, summary.uploadDate);
+ if (recorded) summary.recordedDate = recorded;
+ }
const indexKey: IndexKey = [
summary.uploadDate,
s.channelSlug,
@@ -2529,6 +2581,12 @@ export async function buildIndex({
if (altTracksDue && held.size === 0) {
await meta.put(ALT_TRACKS_KEY, ALT_TRACKS_VERSION);
}
+ for (const slug of recordedDateRuleChanged) {
+ const rule = channelConfigs.get(slug)?.recordedDate;
+ const key = `${RECORDED_DATE_RULE_KEY_PREFIX}${slug}`;
+ if (rule && recordedDateRes.has(slug)) await meta.put(key, rule.titlePattern);
+ else await meta.remove(key);
+ }
await meta.flushed;
await root.close();
diff --git a/common/controller/buildIndexRecordedDate.test.ts b/common/controller/buildIndexRecordedDate.test.ts
@@ -0,0 +1,164 @@
+// Integration: a VOD mirror's recorded dates (release 20 D1) through the REAL
+// buildIndex over a temp corpus — derived from the title by the channel's
+// `recordedDate` rule, absent for a channel without one, re-derived when the
+// rule changes and dropped when it goes. Synthetic titles.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/buildIndexRecordedDate.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-recorded-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.ARCHILYZER_INDEX_ALLOW_HELD;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { open } = await import("lmdb");
+
+const paths = getPaths();
+const MIRROR = "example-vods";
+const PLAIN = "example-plain";
+const SITE = "testsite";
+const ISO = String.raw`(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})`;
+const US = String.raw`(?<month>\d{1,2})/(?<day>\d{1,2})/(?<year>\d{2})`;
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const ROLLING = readFileSync(
+ path.join(import.meta.dirname, "..", "lib", "__fixtures__", "vtt-rolling.vtt"),
+ "utf8",
+);
+
+function config(slug: string, recordedDate?: { titlePattern: string }) {
+ writeJson(path.join(paths.channelsDir, slug, "config.json"), {
+ handling: "youtube",
+ name: slug,
+ ...(recordedDate ? { recordedDate } : {}),
+ });
+}
+
+function video(slug: string, id: string, title: string, uploadDate: string) {
+ const dir = path.join(paths.channelsDir, slug, "data", id);
+ writeJson(path.join(dir, "metadata.info.json"), {
+ id,
+ title,
+ upload_date: uploadDate,
+ duration: 30,
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ writeFileSync(path.join(dir, "transcript.en.vtt"), ROLLING);
+}
+
+function seed(): void {
+ writeFileSync(paths.settingsFile, "{}");
+ config(MIRROR, { titlePattern: ISO });
+ config(PLAIN);
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [
+ { slug: MIRROR, groupId: "default" },
+ { slug: PLAIN, groupId: "default" },
+ ],
+ });
+ video(MIRROR, "MirrorVod001", "Stream VOD 2024-03-05 [3/5/24]", "20240310");
+ video(MIRROR, "MirrorVod002", "Stream VOD with no date", "20240311");
+ video(MIRROR, "MirrorVod003", "Stream VOD — next one 2024-04-01", "20240312");
+ video(PLAIN, "PlainVideo01", "An upload about 2024-01-02", "20240115");
+}
+
+type Summary = { id: string; uploadDate: string; recordedDate?: string };
+
+function recorded(): Record<string, string | undefined> {
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ try {
+ const sums = root.openDB({ name: "sums", encoding: "msgpack" });
+ const out: Record<string, string | undefined> = {};
+ for (const { value } of sums.getRange()) {
+ const s = value as Summary;
+ out[s.id] = s.recordedDate;
+ }
+ return out;
+ } finally {
+ root.close();
+ }
+}
+
+// Every published transcript page of a channel, as text.
+function pagesOf(slug: string): string {
+ const dir = path.join(paths.exportSharedTranscriptsDir, slug);
+ return readdirSync(dir)
+ .filter((f) => f.endsWith(".json"))
+ .map((f) => readFileSync(path.join(dir, f), "utf8"))
+ .join("\n");
+}
+
+async function runIndex(): Promise<string[]> {
+ const log: string[] = [];
+ await buildIndex({ paths, onLog: (s) => log.push(s) });
+ return log;
+}
+
+test("a mirror's records carry the title's recorded date; another channel's never do", async () => {
+ seed();
+ await runIndex();
+ assert.deepEqual(recorded(), {
+ MirrorVod001: "20240305",
+ MirrorVod002: undefined, // no date in the title
+ MirrorVod003: undefined, // the title's date is after the upload
+ PlainVideo01: undefined, // no rule
+ });
+ assert.match(pagesOf(MIRROR), /"recordedDate":\s*"20240305"/);
+ assert.doesNotMatch(pagesOf(PLAIN), /recordedDate/);
+});
+
+test("a changed rule re-derives the channel's records once; a removed rule drops them", async () => {
+ config(MIRROR, { titlePattern: US });
+ const log = await runIndex();
+ assert.ok(
+ log.includes(`Recorded dates: ${MIRROR}: its rule changed; 3 record(s) re-derived.`),
+ log.join("\n"),
+ );
+ assert.equal(log.some((l) => l.includes(`Recorded dates: ${PLAIN}`)), false);
+ assert.equal(recorded().MirrorVod001, "20240305");
+
+ // Unchanged: no second pass.
+ const again = await runIndex();
+ assert.equal(again.some((l) => l.startsWith("Recorded dates:")), false, again.join("\n"));
+
+ config(MIRROR);
+ const dropped = await runIndex();
+ assert.ok(dropped.includes(`Recorded dates: ${MIRROR}: its rule changed; 3 record(s) re-derived.`));
+ assert.equal(recorded().MirrorVod001, undefined);
+ assert.doesNotMatch(pagesOf(MIRROR), /recordedDate/);
+});
diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts
@@ -7,6 +7,7 @@ import {
} from "../ytdlp/downloadFormat";
import { isCookieMode, type CookieMode } from "./cookiePolicy";
import type { FieldDocs } from "./fieldDocs";
+import { coerceRecordedDateRule, type RecordedDateRule } from "./recordedDate";
export type ChannelHandling = "youtube" | "transcribe";
@@ -146,6 +147,7 @@ export type ChannelConfig = {
cookieMode?: CookieMode;
sleepBetweenDownloadsSeconds?: number;
audioCheck?: AudioCheckConfig;
+ recordedDate?: RecordedDateRule;
};
// In the order parseChannelConfig emits (and CHANNEL.md lists) them.
@@ -204,6 +206,8 @@ export const CHANNEL_CONFIG_FIELD_DOCS: FieldDocs<ChannelConfig> = {
"Per-channel override for the global pause between downloads. Absent = inherit; 0 = no sleep; floored and capped at 600.",
audioCheck:
"Opt-in audio-integrity checking for sources that intermittently serve corrupt audio mid-download (e.g. Odysee \"original\"): the managed downloader periodically validates the in-progress `.part` file and rolls back to the last known-good snapshot on corruption. transcribe-handling only.",
+ recordedDate:
+ "For a channel that MIRRORS another's streams (a VOD archive): how to read the date a video was RECORDED from its title, since its `upload_date` is the date of the copy. The index stores the result as the record's `recordedDate` (`YYYYMMDD`), which coverage reads before the upload date. A title it does not match, a date that is not a real day, or one after the upload date gives none. Changing the rule re-derives the channel's records on the next index build. An object without a usable `titlePattern` is dropped."
};
// Every key a config.json may carry, in emission order. The unknown-key oracle:
@@ -407,6 +411,7 @@ export const CHANNEL_CONFIG_COERCIONS: {
? Math.min(Math.floor(v), CHANNEL_SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS)
: undefined,
audioCheck: coerceAudioCheck,
+ recordedDate: coerceRecordedDateRule,
};
// A channel's config.json as a ChannelConfig, or null when it is not a channel
diff --git a/common/lib/channelConfigSchema.test.ts b/common/lib/channelConfigSchema.test.ts
@@ -39,7 +39,7 @@ test("one key list: docs = coercions = schema shape, sync-state keys inside it",
assert.deepEqual(Object.keys(CHANNEL_CONFIG_COERCIONS), [...CHANNEL_CONFIG_KEYS]);
assert.deepEqual(Object.keys(channelConfigObjectSchema.shape), [...CHANNEL_CONFIG_KEYS]);
assert.deepEqual(Object.keys(CHANNEL_CONFIG_FIELD_DOCS), [...CHANNEL_CONFIG_KEYS]);
- assert.equal(CHANNEL_CONFIG_KEYS.length, 32);
+ assert.equal(CHANNEL_CONFIG_KEYS.length, 33);
assert.equal(sameKeys, true);
assert.equal(fits, true);
for (const k of CHANNEL_SYNC_STATE_KEYS) assert.ok(CHANNEL_CONFIG_KEYS.includes(k), k);
diff --git a/common/lib/channelConfigSchema.ts b/common/lib/channelConfigSchema.ts
@@ -75,6 +75,7 @@ export const channelConfigObjectSchema = z.object({
cookieMode: field("cookieMode"),
sleepBetweenDownloadsSeconds: field("sleepBetweenDownloadsSeconds"),
audioCheck: field("audioCheck"),
+ recordedDate: field("recordedDate"),
});
// Drop every own key whose value is `undefined` (the OMIT rule, above).
diff --git a/common/lib/fileSchemaDocs.ts b/common/lib/fileSchemaDocs.ts
@@ -22,6 +22,7 @@ import {
CHANNEL_SYNC_STATE_KEYS,
DOWNLOAD_FILTER_FIELD_DOCS,
} from "./channelConfig";
+import { RECORDED_DATE_RULE_FIELD_DOCS } from "./recordedDate";
import { SOCIAL_LINK_FIELD_DOCS } from "./settingsSchema";
import {
RELATED_SITE_GROUP_FIELD_DOCS,
@@ -131,6 +132,13 @@ const CHANNEL_NESTED: Partial<Record<string, KeyTable[]>> = {
defaults: (key) => (key === "enabled" ? "required" : "absent"),
},
],
+ recordedDate: [
+ {
+ path: "recordedDate",
+ docs: RECORDED_DATE_RULE_FIELD_DOCS,
+ defaults: () => "required",
+ },
+ ],
};
function channelKind(key: string): string {
diff --git a/common/lib/recordedDate.test.ts b/common/lib/recordedDate.test.ts
@@ -0,0 +1,80 @@
+// A VOD mirror's recorded date, read off a title (release 20 D1). Synthetic
+// titles throughout.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/recordedDate.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ coerceRecordedDateRule,
+ compileRecordedDateRule,
+ coverageDate,
+ deriveRecordedDate,
+} from "./recordedDate";
+import { parseChannelConfig } from "./channelConfig";
+import { channelConfigSchema } from "./channelConfigSchema";
+
+const ISO = { titlePattern: String.raw`(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})` };
+const WORDS = { titlePattern: String.raw`(?<month>[a-z]+)\.? (?<day>\d{1,2}),? (?<year>\d{4})` };
+const US = { titlePattern: String.raw`(?<month>\d{1,2})/(?<day>\d{1,2})/(?<year>\d{2,4})` };
+
+test("an ISO date in the title", () => {
+ assert.equal(deriveRecordedDate(ISO, "Example Stream VOD 2024-03-05 | part 1", "20240310"), "20240305");
+});
+
+test("month names, whole, cut and dotted, any case", () => {
+ assert.equal(deriveRecordedDate(WORDS, "VOD: March 5, 2024", "20240306"), "20240305");
+ assert.equal(deriveRecordedDate(WORDS, "vod - sept 30 2023 (full)", "20231001"), "20230930");
+ assert.equal(deriveRecordedDate(WORDS, "Stream | DEC. 1, 2022", "20221202"), "20221201");
+ assert.equal(deriveRecordedDate(WORDS, "Stream | Ma 1, 2022", "20221202"), undefined, "two letters name no month");
+ assert.equal(deriveRecordedDate(WORDS, "Stream | Smarch 1, 2022", "20221202"), undefined);
+});
+
+test("a two-digit year is 20YY", () => {
+ assert.equal(deriveRecordedDate(US, "Stream 3/5/24", "20240306"), "20240305");
+ assert.equal(deriveRecordedDate(US, "Stream 12/31/2023", "20240102"), "20231231");
+});
+
+test("not a real day, no match, or after the upload: none", () => {
+ assert.equal(deriveRecordedDate(ISO, "Stream 2023-02-30", "20230401"), undefined);
+ assert.equal(deriveRecordedDate(ISO, "Stream 2023-13-01", "20240401"), undefined);
+ assert.equal(deriveRecordedDate(ISO, "A title with no date", "20240401"), undefined);
+ // A date the title MENTIONS (the next stream) is not the day it was recorded.
+ assert.equal(deriveRecordedDate(ISO, "Next stream 2024-05-01", "20240420"), undefined);
+ // The same day as the upload is fine.
+ assert.equal(deriveRecordedDate(ISO, "Stream 2024-04-20", "20240420"), "20240420");
+ // No upload date to compare: the title's date stands.
+ assert.equal(deriveRecordedDate(ISO, "Stream 2024-04-20"), "20240420");
+});
+
+test("a rule needs a compiling pattern with all three named groups, or it is dropped", () => {
+ assert.deepEqual(coerceRecordedDateRule({ titlePattern: ` ${ISO.titlePattern} ` }), ISO);
+ assert.equal(coerceRecordedDateRule({ titlePattern: String.raw`(?<year>\d{4})-(?<month>\d{2})` }), undefined);
+ assert.equal(coerceRecordedDateRule({ titlePattern: "(?<year>[" }), undefined);
+ // The download filter's safety rules: no nested quantifier, at most 200 characters.
+ assert.equal(
+ coerceRecordedDateRule({ titlePattern: String.raw`(?<year>(\d+)+)-(?<month>\d+)-(?<day>\d+)` }),
+ undefined,
+ );
+ assert.equal(coerceRecordedDateRule({ titlePattern: `${ISO.titlePattern}${"x".repeat(200)}` }), undefined);
+ assert.equal(coerceRecordedDateRule({ titlePattern: " " }), undefined);
+ assert.equal(coerceRecordedDateRule({ titlePattern: 7 }), undefined);
+ assert.equal(coerceRecordedDateRule("(?<year>.)"), undefined);
+ assert.equal(coerceRecordedDateRule([ISO]), undefined);
+ assert.equal(compileRecordedDateRule({ titlePattern: "(?<year>[" }), null);
+ assert.ok(compileRecordedDateRule(ISO)?.flags.includes("i"));
+});
+
+test("config.json: the key round-trips through both readers, and a bad one is omitted", () => {
+ const raw = { handling: "youtube", name: "Example VODs", recordedDate: ISO };
+ assert.deepEqual(parseChannelConfig(raw)?.recordedDate, ISO);
+ assert.deepEqual(channelConfigSchema.parse(raw)?.recordedDate, ISO);
+ const bad = { handling: "youtube", recordedDate: { titlePattern: "no groups" } };
+ assert.equal("recordedDate" in (parseChannelConfig(bad) ?? {}), false);
+ assert.equal("recordedDate" in (channelConfigSchema.parse(bad) ?? {}), false);
+});
+
+test("coverage reads the recorded date before the upload date", () => {
+ assert.equal(coverageDate({ uploadDate: "20240310", recordedDate: "20240305" }), "20240305");
+ assert.equal(coverageDate({ uploadDate: "20240310" }), "20240310");
+});
diff --git a/common/lib/recordedDate.ts b/common/lib/recordedDate.ts
@@ -0,0 +1,136 @@
+// A VIDEO'S RECORDED DATE, for a channel that mirrors another's streams
+// (release 20 D1).
+//
+// A VOD-mirror channel uploads yesterday's stream today, or last year's
+// streams in a batch, so its videos' `upload_date` is the date of the COPY,
+// not of the stream it copies. Anything that asks "what was said when" —
+// coverage above all ("held videos by date, gaps") — reads the wrong day for
+// every one of them. The date the stream happened is usually in the title
+// ("Stream 2024-03-05", "VOD | March 5, 2024"), in a shape that is the same
+// for the whole channel. So the channel names that shape once, in its
+// config.json (`recordedDate.titlePattern`), and the index derives each
+// record's `recordedDate` from its title (controller/buildIndex.ts).
+//
+// THE PATTERN is a case-insensitive regex SOURCE with three NAMED groups:
+// year four digits, or two (read as 20YY)
+// month 1–12, or an English month name, whole or cut to 3+ letters
+// day 1–31
+// e.g. `(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})` or
+// `(?<month>[a-z]+)\.? (?<day>\d{1,2}),? (?<year>\d{4})`. A pattern that does
+// not compile, or lacks a group, is dropped by the config coercion — the rule
+// is then absent, never half-applied.
+//
+// A DERIVED DATE MUST BE A REAL DAY, AND NO LATER THAN THE UPLOAD: a recording
+// precedes its copy, so a title date after `upload_date` is a date the title
+// mentions (a schedule, a sequel), not the day it was recorded, and none is
+// set. A title the pattern does not match has none either; readers fall back
+// to the upload date (`coverageDate`).
+//
+// Pure, client-safe (channelConfig.ts imports the coercion).
+
+import { downloadFilterPatternProblem } from "./downloadFilters";
+
+export type RecordedDateRule = {
+ titlePattern: string;
+};
+
+export const RECORDED_DATE_RULE_FIELD_DOCS: Record<keyof RecordedDateRule, string> = {
+ titlePattern:
+ "Case-insensitive regex SOURCE (no delimiters, no flags) matched against each video's title, with three named groups: `year` (four digits, or two read as 20YY), `month` (1–12, or an English month name, whole or cut to three or more letters) and `day`. Trimmed; at most 200 characters, no nested quantifier (the download filter's rules). A pattern that breaks any of this drops the whole rule.",
+};
+
+const GROUPS = ["year", "month", "day"] as const;
+
+const MONTHS = [
+ "january",
+ "february",
+ "march",
+ "april",
+ "may",
+ "june",
+ "july",
+ "august",
+ "september",
+ "october",
+ "november",
+ "december",
+];
+
+// Why a pattern cannot be a rule, as the tail of a sentence ("… <problem>"),
+// or null. It must compile, name all three groups, and pass the download
+// filter's safety check (downloadFilterPatternProblem: at most 200 characters,
+// no nested quantifier) — it runs over every title of the channel on the
+// server's one thread, at every index build.
+export function recordedDatePatternProblem(pattern: string): string | null {
+ try {
+ new RegExp(pattern, "i");
+ } catch (e) {
+ return `is not a valid regular expression: ${(e as Error).message}`;
+ }
+ const missing = GROUPS.filter((g) => !pattern.includes(`(?<${g}>`));
+ if (missing.length > 0) {
+ return `must name the groups year, month and day (missing: ${missing.join(", ")})`;
+ }
+ return downloadFilterPatternProblem(pattern);
+}
+
+// The rule's regex, or null when the pattern has a problem (above).
+export function compileRecordedDateRule(rule: RecordedDateRule): RegExp | null {
+ if (recordedDatePatternProblem(rule.titlePattern)) return null;
+ return new RegExp(rule.titlePattern, "i");
+}
+
+export function coerceRecordedDateRule(v: unknown): RecordedDateRule | undefined {
+ if (!v || typeof v !== "object" || Array.isArray(v)) return undefined;
+ const raw = (v as Record<string, unknown>).titlePattern;
+ if (typeof raw !== "string") return undefined;
+ const rule = { titlePattern: raw.trim() };
+ if (!rule.titlePattern || !compileRecordedDateRule(rule)) return undefined;
+ return rule;
+}
+
+function monthNumber(s: string): number | null {
+ if (/^\d{1,2}$/.test(s)) {
+ const n = Number(s);
+ return n >= 1 && n <= 12 ? n : null;
+ }
+ const word = s.toLowerCase().replace(/\.$/, "");
+ if (word.length < 3) return null;
+ const i = MONTHS.findIndex((m) => m.startsWith(word));
+ return i < 0 ? null : i + 1;
+}
+
+const pad = (n: number, w = 2) => String(n).padStart(w, "0");
+
+// The recorded date (`YYYYMMDD`, the shape of `upload_date`) a rule reads off a
+// title, or undefined: no match, not a real day, or later than `uploadDate`
+// (when given).
+export function deriveRecordedDate(
+ rule: RecordedDateRule | RegExp,
+ title: string,
+ uploadDate?: string,
+): string | undefined {
+ const re = rule instanceof RegExp ? rule : compileRecordedDateRule(rule);
+ if (!re) return undefined;
+ const g = re.exec(title)?.groups;
+ if (!g || !g.year || !g.month || !g.day) return undefined;
+ if (!/^\d{2}$|^\d{4}$/.test(g.year) || !/^\d{1,2}$/.test(g.day)) return undefined;
+ const year = g.year.length === 2 ? 2000 + Number(g.year) : Number(g.year);
+ const month = monthNumber(g.month);
+ const day = Number(g.day);
+ if (month === null || day < 1) return undefined;
+ // A real day: the Date round trip refuses Feb 30 and friends.
+ const d = new Date(Date.UTC(year, month - 1, day));
+ if (d.getUTCFullYear() !== year || d.getUTCMonth() !== month - 1 || d.getUTCDate() !== day) {
+ return undefined;
+ }
+ const out = `${pad(year, 4)}${pad(month)}${pad(day)}`;
+ if (uploadDate && /^\d{8}$/.test(uploadDate) && out > uploadDate) return undefined;
+ return out;
+}
+
+// The date a video's content belongs to: its recorded date when its channel's
+// rule gave one, else its upload date. What coverage reads.
+export function coverageDate(s: { uploadDate: string; recordedDate?: string }): string {
+ return s.recordedDate ?? s.uploadDate;
+}
diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts
@@ -37,6 +37,12 @@ export type TranscriptSummary = {
// an untagged corpus's pages stay byte-identical to the ones already on disk
// and the build's sha1 skip still short-circuits.
curatedTags?: string[];
+ // The day a VOD mirror's video was RECORDED (`YYYYMMDD`, the shape of
+ // `uploadDate`), read off its title by its channel's `recordedDate` rule
+ // (lib/recordedDate.ts) when the index builds the record. OMITTED when the
+ // channel has no rule or the title gives no date, so every other channel's
+ // pages stay byte-identical. Coverage reads it first (`coverageDate`).
+ recordedDate?: string;
};
export type DisplaySummary = {
diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx
@@ -870,6 +870,14 @@ export function ChannelForm({
</span>
</label>
<Field
+ label="Recorded date from the title (regex)"
+ name="recordedDateTitlePattern"
+ state={state}
+ defaultValue={c?.recordedDate?.titlePattern ?? ""}
+ placeholder="(the upload date)"
+ hint="For a channel that mirrors another's streams, whose upload dates are the copies'. Case-insensitive regex naming three groups — year, month (a number or an English month name) and day — e.g. (?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2}). The index reads each video's recorded date from its title, and coverage uses it before the upload date. A title it does not match keeps the upload date."
+ />
+ <Field
label="Cookies from browser"
name="cookiesFromBrowser"
state={state}
diff --git a/editor/app/channels/components/channelConfigToForm.ts b/editor/app/channels/components/channelConfigToForm.ts
@@ -50,6 +50,7 @@ export const CHANNEL_FORM_VALUES = [
"downloadFilterInclude",
"downloadFilterExclude",
"downloadFilterRejectedLivestreams",
+ "recordedDateTitlePattern",
"audioCheckIntervalSeconds",
"audioCheckMaxRollbacks",
"audioCheckCopyTimeoutSeconds",
@@ -114,6 +115,7 @@ export function channelConfigToFormData(config: ChannelConfig): FormData {
"downloadFilterRejectedLivestreams",
config.downloadFilter?.rejectedLivestreams,
);
+ put(fd, "recordedDateTitlePattern", config.recordedDate?.titlePattern);
if (config.audioCheck?.enabled) {
fd.set("audioCheckEnabled", "on");
putNum(fd, "audioCheckIntervalSeconds", config.audioCheck.intervalSeconds);
diff --git a/editor/app/channels/components/parseChannelForm.test.ts b/editor/app/channels/components/parseChannelForm.test.ts
@@ -0,0 +1,63 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { parseChannelForm, CHANNEL_FORM_FIELDS } from "./parseChannelForm";
+import {
+ applyChannelFormPatch,
+ channelConfigToFormData,
+ validateChannelFormPatch,
+} from "./channelConfigToForm";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/channels/components/parseChannelForm.test.ts"
+//
+// The Configure form's recorded-date field (release 20 D1): parsed into
+// `recordedDate.titlePattern`, refused when it could not be a rule, round-tripped
+// through the ops patch, and cleared by an empty value.
+
+const ISO = String.raw`(?<year>\d{4})-(?<month>\d{2})-(?<day>\d{2})`;
+
+function form(fields: Record<string, string>): FormData {
+ const fd = new FormData();
+ fd.set("name", "Example VODs");
+ fd.set("handling", "youtube");
+ for (const [k, v] of Object.entries(fields)) fd.set(k, v);
+ return fd;
+}
+
+test("the form writes recordedDate.titlePattern, trimmed", () => {
+ const { config } = parseChannelForm(form({ recordedDateTitlePattern: ` ${ISO} ` }));
+ assert.deepEqual(config.recordedDate, { titlePattern: ISO });
+ assert.equal("recordedDate" in parseChannelForm(form({})).config, false);
+ // A form save clears a stored rule the form left blank.
+ assert.ok((CHANNEL_FORM_FIELDS as readonly string[]).includes("recordedDate"));
+});
+
+test("a pattern that cannot be a rule is refused with why", () => {
+ assert.throws(
+ () => parseChannelForm(form({ recordedDateTitlePattern: String.raw`(?<year>\d{4})` })),
+ /Recorded-date pattern must name the groups year, month and day \(missing: month, day\)/,
+ );
+ assert.throws(
+ () => parseChannelForm(form({ recordedDateTitlePattern: "(?<year>[" })),
+ /Recorded-date pattern is not a valid regular expression/,
+ );
+ assert.throws(
+ () =>
+ parseChannelForm(
+ form({ recordedDateTitlePattern: String.raw`(?<year>(\d+)+)-(?<month>\d+)-(?<day>\d+)` }),
+ ),
+ /Recorded-date pattern contains a nested quantifier/,
+ );
+});
+
+test("the ops patch sets and clears it over the channel's current form", () => {
+ validateChannelFormPatch({ recordedDateTitlePattern: ISO });
+ const fd = channelConfigToFormData({ handling: "youtube", name: "Example VODs" });
+ applyChannelFormPatch(fd, { recordedDateTitlePattern: ISO });
+ const set = parseChannelForm(fd).config;
+ assert.deepEqual(set.recordedDate, { titlePattern: ISO });
+ const back = channelConfigToFormData(set);
+ assert.equal(back.get("recordedDateTitlePattern"), ISO);
+ applyChannelFormPatch(back, { recordedDateTitlePattern: "" });
+ assert.equal("recordedDate" in parseChannelForm(back).config, false);
+});
diff --git a/editor/app/channels/components/parseChannelForm.ts b/editor/app/channels/components/parseChannelForm.ts
@@ -25,6 +25,7 @@ import {
isSourceVideoQuality,
} from "yt-dlp-transcript-common/ytdlp/downloadFormat";
import { downloadFilterPatternProblem } from "yt-dlp-transcript-common/lib/downloadFilters";
+import { recordedDatePatternProblem } from "yt-dlp-transcript-common/lib/recordedDate";
import { isRejectedLivestreamMode } from "yt-dlp-transcript-common/lib/channelConfig";
export type ParsedChannelForm = {
@@ -58,6 +59,7 @@ export const CHANNEL_FORM_FIELDS = [
"cookieMode",
"audioCheck",
"downloadFilter",
+ "recordedDate",
] as const satisfies ReadonlyArray<keyof ChannelConfig>;
// Maps FormData -> ChannelConfig. Throws on invalid input so the server
@@ -280,6 +282,15 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm {
}
: undefined;
+ // A VOD mirror's recorded date (release 20 D1, lib/recordedDate.ts): a regex
+ // with named groups year/month/day, refused here — as the download filter's
+ // are — rather than saved and silently dropped by the config reader.
+ const recordedDateTitlePattern = stringOrUndef(formData, "recordedDateTitlePattern");
+ if (recordedDateTitlePattern) {
+ const problem = recordedDatePatternProblem(recordedDateTitlePattern);
+ if (problem) throw new Error(`Recorded-date pattern ${problem}`);
+ }
+
const audioCheckEnabled = formData.get("audioCheckEnabled") != null;
let audioCheck: AudioCheckConfig | undefined;
if (audioCheckEnabled) {
@@ -369,6 +380,9 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm {
if (cookieMode) config.cookieMode = cookieMode;
if (audioCheck) config.audioCheck = audioCheck;
if (downloadFilter) config.downloadFilter = downloadFilter;
+ if (recordedDateTitlePattern) {
+ config.recordedDate = { titlePattern: recordedDateTitlePattern };
+ }
return { name, slug, config };
}