Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 63fb2c0083af66b91d6e667e88e98aac51472e6b
parent c600686f70bd403eab11b027ed0b07c6eacbbc66
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  6 Oct 2026 07:29:05 -0400

index: a record labelled before its platform was known is relabelled — once, from LMDB, no schema bump

- platformLabelStale(summary): the record's own page is on a platform whose
  page decides the label (archive.org, BitChute) and the label says
  otherwise — what a BitChute record summarized before the platform existed
  carries ("youtube", no file to play)
- buildIndex: PLATFORM_LABELS_VERSION (1); the first build that sees a new
  version reads every stored summary once and queues the stale ones as
  changed; recorded only when no channel is held
- a normalized transcript.cues.json whose frozen summary is stale is
  re-derived from the metadata, its cues kept — by the index and by report
  composition
- integration test over a temp corpus through the real buildIndex

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/buildIndex.ts | 47++++++++++++++++++++++++++++++++++++++++++++++-
Acommon/controller/buildIndexPlatformLabels.test.ts | 201+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/transcripts-server.ts | 21++++++++++++++++++++-
Mcommon/publish/composeReports.ts | 9+++++++--
4 files changed, 274 insertions(+), 4 deletions(-)

diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -46,6 +46,7 @@ import { parseVtt, type Cue } from "../lib/vtt"; import { parseTranscriptJson } from "../lib/whisper"; import { parseLiveChat } from "../lib/liveChat"; import { + platformLabelStale, summarize, toDisplaySummary, type RawMetadata, @@ -192,6 +193,18 @@ import { // so the clearAsync() enumeration above is unchanged too. const SCHEMA_VERSION = 13; +// PLATFORM LABELS — also NOT a schema bump. When the app learns a platform +// whose records may already be indexed under another label (BitChute's were +// summarized "youtube", with no file to play), a bump would wipe the cache and +// re-read every video directory to fix a handful. Instead, the first build +// that sees a new PLATFORM_LABELS_VERSION reads every stored summary out of +// LMDB once (no disk), queues each one platformLabelStale() flags as changed, +// and records the version — only when no channel is held, so a held channel's +// records are still looked at once its text is back. Bump this whenever +// platformFromMetadata learns such a platform. +const PLATFORM_LABELS_VERSION = 1; +const PLATFORM_LABELS_KEY = "platformLabels"; + // Per-channel post stats, persisted so per-site aggregates survive a no-op // rebuild that doesn't re-encode the post pages. Mirrors ChannelSubsStat. type ChannelPostsStat = { @@ -764,6 +777,32 @@ export async function buildIndex({ ); } + // Summaries labelled before their platform was known (PLATFORM_LABELS_ + // VERSION): re-derived like a changed record. A schema bump has cleared + // every summary, so it has none to look at. + const relabelDue = + meta.get(PLATFORM_LABELS_KEY) !== PLATFORM_LABELS_VERSION; + if (relabelDue && !schemaBumped) { + const queued = new Set( + [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])), + ); + let relabelled = 0; + for (const s of live) { + const pk: PathKey = [s.channelSlug, s.videoDir]; + if (queued.has(pathKeyId(pk))) continue; + const prev = mtimes.get(pk); + if (!prev) continue; + const sum = sums.get(prev.indexKey); + if (sum && platformLabelStale(sum)) { + changed.push(s); + relabelled++; + } + } + log( + `Platform labels v${PLATFORM_LABELS_VERSION}: ${relabelled} record(s) labelled before their platform was known, re-derived.`, + ); + } + const anyMutations = added.length > 0 || changed.length > 0 || removed.length > 0; @@ -846,6 +885,9 @@ export async function buildIndex({ channel: s.configName ?? rest.channel, }; cueList = cuesField; + // Normalized before its platform was known: the summary is + // re-derived from the metadata below, the cues are kept. + if (platformLabelStale(summary)) summary = undefined; } } if (!summary) { @@ -857,7 +899,7 @@ export async function buildIndex({ parsedMeta, s.configName, ); - if (s.transcriptMs !== null && s.transcriptKind) { + if (cueList === undefined && s.transcriptMs !== null && s.transcriptKind) { try { const raw = await readFile(s.transcriptPath, "utf8"); cueList = @@ -2287,6 +2329,9 @@ export async function buildIndex({ } for (const k of staleFpKeys) meta.remove(k); await meta.put(INDEX_SCANNED_AT_KEY, scanStartedAt); + if (relabelDue && held.size === 0) { + await meta.put(PLATFORM_LABELS_KEY, PLATFORM_LABELS_VERSION); + } await meta.flushed; await root.close(); diff --git a/common/controller/buildIndexPlatformLabels.test.ts b/common/controller/buildIndexPlatformLabels.test.ts @@ -0,0 +1,201 @@ +// Integration: a record labelled before its platform was known is relabelled +// by the index — through the REAL buildIndex, over a temp corpus. +// +// A BitChute record downloaded before the bitchute platform existed was +// summarized "youtube" (and with no file to play), and a transcribed one +// carries that summary frozen in its transcript.cues.json. Its files never +// change again, so the mtime diff alone would never look at it: the build's +// one-off platform-labels pass (PLATFORM_LABELS_VERSION) queues it, and the +// per-video step re-derives a stale normalized summary from the metadata. +// +// Run with: node_modules/.bin/tsx --test common/controller/buildIndexPlatformLabels.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync, mkdtempSync, rmSync, utimesSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-labels-")); +const PINNED: Record<string, string> = { + TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), + SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), + SITES_DIR: path.join(ROOT, "transcripts", "sites"), + SETTINGS_FILE: path.join(ROOT, "settings.json"), + EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), + EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), + EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), + EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), + EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), + CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), + SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), + CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), + ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), + ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), +}; +Object.assign(process.env, PINNED); +delete process.env.ARCHILYZER_INDEX_ALLOW_HELD; +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const { getPaths } = await import("../lib/paths"); +const { buildIndex } = await import("./buildIndex"); +const { open } = await import("lmdb"); + +const paths = getPaths(); +const CHANNEL = "example-channel"; +const SITE = "testsite"; +const BC = "Zq3xVb7Kp2Lm"; +const YT = "AbC123xyz_9"; +const BC_PAGE = `https://www.bitchute.com/video/${BC}/`; +const BC_FILE = `https://seed901.bitchute.com/AbCdEfGhIjKl/${BC}.mp4`; + +const writeJson = (file: string, value: unknown) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value, null, 2)); +}; +const dirOf = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id); + +const VTT = + "WEBVTT\nKind: captions\nLanguage: en\n\n" + + "00:00:00.000 --> 00:00:05.000 align:start position:0%\n" + + "First<00:00:01.000><c> caption</c><00:00:02.000><c> line.</c>\n"; + +const CUES = [{ start: 0, end: 5, text: "Normalized line." }]; + +function seed(): void { + rmSync(paths.transcriptsDir, { recursive: true, force: true }); + rmSync(PINNED.EXPORT_INDEX_DIR, { recursive: true, force: true }); + writeFileSync(paths.settingsFile, "{}"); + writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { + handling: "transcribe", + name: CHANNEL, + }); + writeJson(path.join(paths.sitesDir, SITE, "site.json"), { + siteId: SITE, + siteTitle: "Test Site", + siteDescription: "fixture", + headerTitle: "Test Site", + homeTagline: "", + socialLinks: [], + groups: [{ id: "default", name: "All channels", selectedByDefault: true }], + defaultGroupId: "default", + channels: [{ slug: CHANNEL, groupId: "default" }], + }); + writeJson(path.join(dirOf(YT), "metadata.info.json"), { + id: YT, + title: "A YouTube video", + upload_date: "20260601", + duration: 120, + webpage_url: `https://www.youtube.com/watch?v=${YT}`, + extractor_key: "Youtube", + }); + writeFileSync(path.join(dirOf(YT), "transcript.en.vtt"), VTT); + writeJson(path.join(dirOf(BC), "metadata.info.json"), { + id: BC, + title: "A BitChute video", + upload_date: "20260602", + duration: 61, + webpage_url: BC_PAGE, + extractor: "BitChute", + extractor_key: "BitChute", + formats: [{ format_id: "0", ext: "mp4", url: BC_FILE }], + url: BC_FILE, + }); + writeFileSync(path.join(dirOf(BC), "transcript.en.vtt"), VTT); + // The normalized transcript an earlier build of the app wrote: its summary + // says "youtube" and carries no file. Newer than everything, so fresh. + const cuesPath = path.join(dirOf(BC), "transcript.cues.json"); + writeJson(cuesPath, { + version: 2, + source: "vtt", + slug: `${CHANNEL}/${BC}`, + id: BC, + channelSlug: CHANNEL, + title: "A BitChute video", + uploadDate: "20260602", + duration: 61, + channel: CHANNEL, + description: "", + tags: [], + isLivestream: false, + ageRestricted: false, + platform: "youtube", + webpageUrl: BC_PAGE, + cues: CUES, + }); + const later = new Date(Date.now() + 10_000); + utimesSync(cuesPath, later, later); +} + +type Summary = { id: string; platform: string; mediaUrl?: string }; + +function withIndex<T>(fn: (db: (name: string) => ReturnType<ReturnType<typeof open>["openDB"]>) => T): T { + const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true }); + try { + return fn((name) => root.openDB({ name, encoding: "msgpack" })); + } finally { + root.close(); + } +} +const summaryOf = (id: string): Summary => + withIndex((db) => { + for (const { value } of db("sums").getRange()) { + if ((value as Summary).id === id) return value as Summary; + } + throw new Error(`${id} is not indexed`); + }); +const cuesOf = (id: string) => + withIndex((db) => { + for (const { key, value } of db("sums").getRange()) { + if ((value as Summary).id === id) return db("cues").get(key) as typeof CUES; + } + return undefined; + }); + +async function runIndex(): Promise<string[]> { + const log: string[] = []; + await buildIndex({ paths, onLog: (s) => log.push(s) }); + return log; +} + +test("a normalized summary frozen as youtube is re-derived as bitchute, with its file; its cues are kept", async () => { + seed(); + await runIndex(); + const bc = summaryOf(BC); + assert.equal(bc.platform, "bitchute"); + assert.equal(bc.mediaUrl, BC_FILE); + assert.deepEqual(cuesOf(BC), CUES); + assert.equal(summaryOf(YT).platform, "youtube"); +}); + +test("a summary already indexed under the old label is relabelled once, by the platform-labels pass", async () => { + // The index as an older build left it: the record stored as "youtube", and + // no platform-labels version recorded. + withIndex((db) => { + const sums = db("sums"); + for (const { key, value } of sums.getRange()) { + const v = value as Summary; + if (v.id === BC) { + const stale = { ...v, platform: "youtube" } as Summary; + delete stale.mediaUrl; + sums.putSync(key, stale); + } + } + db("meta").removeSync("platformLabels"); + }); + assert.equal(summaryOf(BC).platform, "youtube"); + + const log = await runIndex(); + assert.ok( + log.some((l) => /Platform labels v\d+: 1 record\(s\)/.test(l)), + log.join("\n"), + ); + const bc = summaryOf(BC); + assert.equal(bc.platform, "bitchute"); + assert.equal(bc.mediaUrl, BC_FILE); + assert.equal(summaryOf(YT).platform, "youtube"); + + // Recorded: the next build does not look again. + const again = await runIndex(); + assert.equal(again.some((l) => /Platform labels/.test(l)), false, again.join("\n")); +}); diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts @@ -95,10 +95,29 @@ export function platformFromMetadata(meta: RawMetadata): Platform { if (/^bitchute/i.test(key)) return "bitchute"; if (/^youtube/i.test(key)) return "youtube"; const fromPage = detectPlatform(meta.webpage_url); - if (fromPage === "archiveorg" || fromPage === "bitchute") return fromPage; + if (fromPage && PAGE_DECIDES.includes(fromPage)) return fromPage; return "youtube"; } +// The platforms whose page decides a record's label even under an extractor +// the app does not know (above). +const PAGE_DECIDES: ReadonlyArray<Platform> = ["archiveorg", "bitchute"]; + +// A SUMMARY LABELLED BEFORE ITS PLATFORM WAS KNOWN: its own page is on a +// platform whose page decides the label, and the label says otherwise — what a +// BitChute record summarized before the bitchute platform existed carries +// ("youtube", and no file to play). The index re-derives such a summary from +// the record's metadata (controller/buildIndex.ts), and so does every reader +// of a normalized transcript.cues.json, whose summary was frozen when it was +// written. Pure: the summary alone decides. +export function platformLabelStale(summary: { + platform?: Platform; + webpageUrl?: string; +}): boolean { + const fromPage = detectPlatform(summary.webpageUrl); + return fromPage !== null && PAGE_DECIDES.includes(fromPage) && fromPage !== summary.platform; +} + // The broad "is this a livestream (or stream VOD/upcoming)" notion used by the // coverage detector and the download-time duration guard, so both skip the same // content (stream captures have unreliable metadata durations). Distinct from diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts @@ -66,7 +66,11 @@ import { assertChannelTextReadable } from "../lib/channelMedia"; import type { ChannelConfig } from "../lib/channelConfig"; import { readChannelConfig } from "../controller/channels"; import { isCuesJsonFresh, readNormalizedTranscript } from "../controller/normalizeTranscript"; -import { loadRawMetadataFromDir, summarize } from "../lib/transcripts-server"; +import { + loadRawMetadataFromDir, + platformLabelStale, + summarize, +} from "../lib/transcripts-server"; import type { TranscriptSummary } from "../lib/transcripts"; import { parseVtt, type Cue } from "../lib/vtt"; import { parseTranscriptJson } from "../lib/whisper"; @@ -275,7 +279,8 @@ export async function readCitedRecord( const fresh = await isCuesJsonFresh(dir); if (fresh.fresh) { const n = await readNormalizedTranscript(fresh.cuesPath); - if (n) { + // A summary frozen before its platform was known is re-derived below. + if (n && !platformLabelStale(n)) { summary = n; cues = n.cues ?? []; }