commit 63fb2c0083af66b91d6e667e88e98aac51472e6b
parent c600686f70bd403eab11b027ed0b07c6eacbbc66
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 6 Oct 2026 07:29:05 -0400
index: a record labelled before its platform was known is relabelled — once, from LMDB, no schema bump
- platformLabelStale(summary): the record's own page is on a platform whose
page decides the label (archive.org, BitChute) and the label says
otherwise — what a BitChute record summarized before the platform existed
carries ("youtube", no file to play)
- buildIndex: PLATFORM_LABELS_VERSION (1); the first build that sees a new
version reads every stored summary once and queues the stale ones as
changed; recorded only when no channel is held
- a normalized transcript.cues.json whose frozen summary is stale is
re-derived from the metadata, its cues kept — by the index and by report
composition
- integration test over a temp corpus through the real buildIndex
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
4 files changed, 274 insertions(+), 4 deletions(-)
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -46,6 +46,7 @@ import { parseVtt, type Cue } from "../lib/vtt";
import { parseTranscriptJson } from "../lib/whisper";
import { parseLiveChat } from "../lib/liveChat";
import {
+ platformLabelStale,
summarize,
toDisplaySummary,
type RawMetadata,
@@ -192,6 +193,18 @@ import {
// so the clearAsync() enumeration above is unchanged too.
const SCHEMA_VERSION = 13;
+// PLATFORM LABELS — also NOT a schema bump. When the app learns a platform
+// whose records may already be indexed under another label (BitChute's were
+// summarized "youtube", with no file to play), a bump would wipe the cache and
+// re-read every video directory to fix a handful. Instead, the first build
+// that sees a new PLATFORM_LABELS_VERSION reads every stored summary out of
+// LMDB once (no disk), queues each one platformLabelStale() flags as changed,
+// and records the version — only when no channel is held, so a held channel's
+// records are still looked at once its text is back. Bump this whenever
+// platformFromMetadata learns such a platform.
+const PLATFORM_LABELS_VERSION = 1;
+const PLATFORM_LABELS_KEY = "platformLabels";
+
// Per-channel post stats, persisted so per-site aggregates survive a no-op
// rebuild that doesn't re-encode the post pages. Mirrors ChannelSubsStat.
type ChannelPostsStat = {
@@ -764,6 +777,32 @@ export async function buildIndex({
);
}
+ // Summaries labelled before their platform was known (PLATFORM_LABELS_
+ // VERSION): re-derived like a changed record. A schema bump has cleared
+ // every summary, so it has none to look at.
+ const relabelDue =
+ meta.get(PLATFORM_LABELS_KEY) !== PLATFORM_LABELS_VERSION;
+ if (relabelDue && !schemaBumped) {
+ const queued = new Set(
+ [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])),
+ );
+ let relabelled = 0;
+ for (const s of live) {
+ const pk: PathKey = [s.channelSlug, s.videoDir];
+ if (queued.has(pathKeyId(pk))) continue;
+ const prev = mtimes.get(pk);
+ if (!prev) continue;
+ const sum = sums.get(prev.indexKey);
+ if (sum && platformLabelStale(sum)) {
+ changed.push(s);
+ relabelled++;
+ }
+ }
+ log(
+ `Platform labels v${PLATFORM_LABELS_VERSION}: ${relabelled} record(s) labelled before their platform was known, re-derived.`,
+ );
+ }
+
const anyMutations =
added.length > 0 || changed.length > 0 || removed.length > 0;
@@ -846,6 +885,9 @@ export async function buildIndex({
channel: s.configName ?? rest.channel,
};
cueList = cuesField;
+ // Normalized before its platform was known: the summary is
+ // re-derived from the metadata below, the cues are kept.
+ if (platformLabelStale(summary)) summary = undefined;
}
}
if (!summary) {
@@ -857,7 +899,7 @@ export async function buildIndex({
parsedMeta,
s.configName,
);
- if (s.transcriptMs !== null && s.transcriptKind) {
+ if (cueList === undefined && s.transcriptMs !== null && s.transcriptKind) {
try {
const raw = await readFile(s.transcriptPath, "utf8");
cueList =
@@ -2287,6 +2329,9 @@ export async function buildIndex({
}
for (const k of staleFpKeys) meta.remove(k);
await meta.put(INDEX_SCANNED_AT_KEY, scanStartedAt);
+ if (relabelDue && held.size === 0) {
+ await meta.put(PLATFORM_LABELS_KEY, PLATFORM_LABELS_VERSION);
+ }
await meta.flushed;
await root.close();
diff --git a/common/controller/buildIndexPlatformLabels.test.ts b/common/controller/buildIndexPlatformLabels.test.ts
@@ -0,0 +1,201 @@
+// Integration: a record labelled before its platform was known is relabelled
+// by the index — through the REAL buildIndex, over a temp corpus.
+//
+// A BitChute record downloaded before the bitchute platform existed was
+// summarized "youtube" (and with no file to play), and a transcribed one
+// carries that summary frozen in its transcript.cues.json. Its files never
+// change again, so the mtime diff alone would never look at it: the build's
+// one-off platform-labels pass (PLATFORM_LABELS_VERSION) queues it, and the
+// per-video step re-derives a stale normalized summary from the metadata.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/buildIndexPlatformLabels.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, rmSync, utimesSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-labels-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.ARCHILYZER_INDEX_ALLOW_HELD;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { open } = await import("lmdb");
+
+const paths = getPaths();
+const CHANNEL = "example-channel";
+const SITE = "testsite";
+const BC = "Zq3xVb7Kp2Lm";
+const YT = "AbC123xyz_9";
+const BC_PAGE = `https://www.bitchute.com/video/${BC}/`;
+const BC_FILE = `https://seed901.bitchute.com/AbCdEfGhIjKl/${BC}.mp4`;
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const dirOf = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id);
+
+const VTT =
+ "WEBVTT\nKind: captions\nLanguage: en\n\n" +
+ "00:00:00.000 --> 00:00:05.000 align:start position:0%\n" +
+ "First<00:00:01.000><c> caption</c><00:00:02.000><c> line.</c>\n";
+
+const CUES = [{ start: 0, end: 5, text: "Normalized line." }];
+
+function seed(): void {
+ rmSync(paths.transcriptsDir, { recursive: true, force: true });
+ rmSync(PINNED.EXPORT_INDEX_DIR, { recursive: true, force: true });
+ writeFileSync(paths.settingsFile, "{}");
+ writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
+ handling: "transcribe",
+ name: CHANNEL,
+ });
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHANNEL, groupId: "default" }],
+ });
+ writeJson(path.join(dirOf(YT), "metadata.info.json"), {
+ id: YT,
+ title: "A YouTube video",
+ upload_date: "20260601",
+ duration: 120,
+ webpage_url: `https://www.youtube.com/watch?v=${YT}`,
+ extractor_key: "Youtube",
+ });
+ writeFileSync(path.join(dirOf(YT), "transcript.en.vtt"), VTT);
+ writeJson(path.join(dirOf(BC), "metadata.info.json"), {
+ id: BC,
+ title: "A BitChute video",
+ upload_date: "20260602",
+ duration: 61,
+ webpage_url: BC_PAGE,
+ extractor: "BitChute",
+ extractor_key: "BitChute",
+ formats: [{ format_id: "0", ext: "mp4", url: BC_FILE }],
+ url: BC_FILE,
+ });
+ writeFileSync(path.join(dirOf(BC), "transcript.en.vtt"), VTT);
+ // The normalized transcript an earlier build of the app wrote: its summary
+ // says "youtube" and carries no file. Newer than everything, so fresh.
+ const cuesPath = path.join(dirOf(BC), "transcript.cues.json");
+ writeJson(cuesPath, {
+ version: 2,
+ source: "vtt",
+ slug: `${CHANNEL}/${BC}`,
+ id: BC,
+ channelSlug: CHANNEL,
+ title: "A BitChute video",
+ uploadDate: "20260602",
+ duration: 61,
+ channel: CHANNEL,
+ description: "",
+ tags: [],
+ isLivestream: false,
+ ageRestricted: false,
+ platform: "youtube",
+ webpageUrl: BC_PAGE,
+ cues: CUES,
+ });
+ const later = new Date(Date.now() + 10_000);
+ utimesSync(cuesPath, later, later);
+}
+
+type Summary = { id: string; platform: string; mediaUrl?: string };
+
+function withIndex<T>(fn: (db: (name: string) => ReturnType<ReturnType<typeof open>["openDB"]>) => T): T {
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ try {
+ return fn((name) => root.openDB({ name, encoding: "msgpack" }));
+ } finally {
+ root.close();
+ }
+}
+const summaryOf = (id: string): Summary =>
+ withIndex((db) => {
+ for (const { value } of db("sums").getRange()) {
+ if ((value as Summary).id === id) return value as Summary;
+ }
+ throw new Error(`${id} is not indexed`);
+ });
+const cuesOf = (id: string) =>
+ withIndex((db) => {
+ for (const { key, value } of db("sums").getRange()) {
+ if ((value as Summary).id === id) return db("cues").get(key) as typeof CUES;
+ }
+ return undefined;
+ });
+
+async function runIndex(): Promise<string[]> {
+ const log: string[] = [];
+ await buildIndex({ paths, onLog: (s) => log.push(s) });
+ return log;
+}
+
+test("a normalized summary frozen as youtube is re-derived as bitchute, with its file; its cues are kept", async () => {
+ seed();
+ await runIndex();
+ const bc = summaryOf(BC);
+ assert.equal(bc.platform, "bitchute");
+ assert.equal(bc.mediaUrl, BC_FILE);
+ assert.deepEqual(cuesOf(BC), CUES);
+ assert.equal(summaryOf(YT).platform, "youtube");
+});
+
+test("a summary already indexed under the old label is relabelled once, by the platform-labels pass", async () => {
+ // The index as an older build left it: the record stored as "youtube", and
+ // no platform-labels version recorded.
+ withIndex((db) => {
+ const sums = db("sums");
+ for (const { key, value } of sums.getRange()) {
+ const v = value as Summary;
+ if (v.id === BC) {
+ const stale = { ...v, platform: "youtube" } as Summary;
+ delete stale.mediaUrl;
+ sums.putSync(key, stale);
+ }
+ }
+ db("meta").removeSync("platformLabels");
+ });
+ assert.equal(summaryOf(BC).platform, "youtube");
+
+ const log = await runIndex();
+ assert.ok(
+ log.some((l) => /Platform labels v\d+: 1 record\(s\)/.test(l)),
+ log.join("\n"),
+ );
+ const bc = summaryOf(BC);
+ assert.equal(bc.platform, "bitchute");
+ assert.equal(bc.mediaUrl, BC_FILE);
+ assert.equal(summaryOf(YT).platform, "youtube");
+
+ // Recorded: the next build does not look again.
+ const again = await runIndex();
+ assert.equal(again.some((l) => /Platform labels/.test(l)), false, again.join("\n"));
+});
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -95,10 +95,29 @@ export function platformFromMetadata(meta: RawMetadata): Platform {
if (/^bitchute/i.test(key)) return "bitchute";
if (/^youtube/i.test(key)) return "youtube";
const fromPage = detectPlatform(meta.webpage_url);
- if (fromPage === "archiveorg" || fromPage === "bitchute") return fromPage;
+ if (fromPage && PAGE_DECIDES.includes(fromPage)) return fromPage;
return "youtube";
}
+// The platforms whose page decides a record's label even under an extractor
+// the app does not know (above).
+const PAGE_DECIDES: ReadonlyArray<Platform> = ["archiveorg", "bitchute"];
+
+// A SUMMARY LABELLED BEFORE ITS PLATFORM WAS KNOWN: its own page is on a
+// platform whose page decides the label, and the label says otherwise — what a
+// BitChute record summarized before the bitchute platform existed carries
+// ("youtube", and no file to play). The index re-derives such a summary from
+// the record's metadata (controller/buildIndex.ts), and so does every reader
+// of a normalized transcript.cues.json, whose summary was frozen when it was
+// written. Pure: the summary alone decides.
+export function platformLabelStale(summary: {
+ platform?: Platform;
+ webpageUrl?: string;
+}): boolean {
+ const fromPage = detectPlatform(summary.webpageUrl);
+ return fromPage !== null && PAGE_DECIDES.includes(fromPage) && fromPage !== summary.platform;
+}
+
// The broad "is this a livestream (or stream VOD/upcoming)" notion used by the
// coverage detector and the download-time duration guard, so both skip the same
// content (stream captures have unreliable metadata durations). Distinct from
diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts
@@ -66,7 +66,11 @@ import { assertChannelTextReadable } from "../lib/channelMedia";
import type { ChannelConfig } from "../lib/channelConfig";
import { readChannelConfig } from "../controller/channels";
import { isCuesJsonFresh, readNormalizedTranscript } from "../controller/normalizeTranscript";
-import { loadRawMetadataFromDir, summarize } from "../lib/transcripts-server";
+import {
+ loadRawMetadataFromDir,
+ platformLabelStale,
+ summarize,
+} from "../lib/transcripts-server";
import type { TranscriptSummary } from "../lib/transcripts";
import { parseVtt, type Cue } from "../lib/vtt";
import { parseTranscriptJson } from "../lib/whisper";
@@ -275,7 +279,8 @@ export async function readCitedRecord(
const fresh = await isCuesJsonFresh(dir);
if (fresh.fresh) {
const n = await readNormalizedTranscript(fresh.cuesPath);
- if (n) {
+ // A summary frozen before its platform was known is re-derived below.
+ if (n && !platformLabelStale(n)) {
summary = n;
cues = n.cues ?? [];
}