Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 91735e8fb9a20e6522b39555f6c7901fa64e952d
parent 51a1b8f02c7dd7c5cea46020e8ef10164987d5b7
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat,  4 Jul 2026 13:22:15 -0400

Overflow oversize archives to R2; per-channel only; per-site Duplicates toggle

Archive zips over the Cloudflare Pages 25MB per-file cap used to be dropped
from the deploy entirely, so large sites served nothing downloadable. Now,
when a global R2 overflow bucket is configured (SiteSettings.archiveStorage),
compose-site stages oversize archives to export/.r2-staging/<siteId>/ and
records their public URL in the manifest; the editor deploy uploads them via
`wrangler r2 object put` before the Pages deploy. Without R2 configured the
old drop-and-flag ("Too large to host") behavior is unchanged.

- Drop the combined "whole-site" archives (per-channel only) — they duplicated
  per-channel content and were always first over the cap.
- ArchiveManifestEntry gains `url?`; hasArchives()/Downloads page prefer it.
- Per-site `Site.duplicates` opt-out (default on) + build-time hasDuplicates()
  so the Duplicates nav link hides on opt-out or when a site has no clusters.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

Diffstat:
M.gitignore | 2++
Mcommon/bin/compose-site.ts | 146++++++++++++++++++++++++++++++++++++++++++++++---------------------------------
Mcommon/lib/archiveOptions.ts | 12+++++++++---
Mcommon/lib/settings.ts | 26++++++++++++++++++++++++++
Mcommon/lib/site.ts | 7+++++++
Meditor/CHANGELOG.md | 4+++-
Meditor/app/build/buildAction.ts | 8++++++++
Meditor/app/deploy/buildDeployCore.ts | 68++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/deploy/deployAction.ts | 13++++++++++++-
Meditor/app/settings/actions.ts | 7+++++++
Meditor/app/settings/components/SettingsForm.tsx | 40++++++++++++++++++++++++++++++++++++----
Meditor/app/sites/actions.ts | 3+++
Meditor/app/sites/components/SiteForm.tsx | 22++++++++++++++++++----
Mexport/CHANGELOG.md | 5++++-
Mexport/app/components/Header.tsx | 16++++++++++------
Mexport/app/downloads/page.tsx | 65+++++++++++++++++++++++++----------------------------------------
Mexport/app/lib/archives.ts | 7++++---
Aexport/app/lib/duplicates.ts | 22++++++++++++++++++++++
18 files changed, 349 insertions(+), 124 deletions(-)

diff --git a/.gitignore b/.gitignore @@ -62,6 +62,8 @@ yarn-error.log* /export/public/hub-sites.json # per-site bulk-download archive zips + manifest (regenerate with `pnpm build`) /export/public/archives/ +# oversize archives staged for the deploy-time R2 upload +/export/.r2-staging/ /export/.export-index/ /transcripts/index.mdb/ diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts @@ -14,7 +14,7 @@ // only that site's data. Checked-in static assets in public/ are left intact. import path from "node:path"; -import { cp, mkdir, rm, readdir, access, readFile, writeFile, stat } from "node:fs/promises"; +import { cp, mkdir, rm, readdir, access, readFile, writeFile, stat, rename } from "node:fs/promises"; import { getPaths } from "../lib/paths"; import { getSite, resolveSocialLinks, resolveHubUrl, type Site } from "../lib/site"; import { getSettings } from "../lib/settings"; @@ -25,14 +25,8 @@ import { } from "../lib/duplicates"; import type { Manifest, SubsManifest } from "../lib/manifest"; import { buildSiteDescriptor } from "../lib/siteDescriptor"; -import { - archiveTranscripts, - archiveCombinedTranscripts, -} from "../controller/archiveTranscripts"; -import { - archiveLiveChat, - archiveCombinedLiveChat, -} from "../controller/archiveLiveChat"; +import { archiveTranscripts } from "../controller/archiveTranscripts"; +import { archiveLiveChat } from "../controller/archiveLiveChat"; import { ARCHIVE_MANIFEST_FILENAME, DEFAULT_ARCHIVE_MAX_BYTES, @@ -158,9 +152,21 @@ async function channelTitles( return titles; } -// Stat a produced archive, and enforce the size cap: an oversize file is removed -// from the served dir (so a capped host won't reject the whole deploy) and marked -// oversize in the manifest, which the Downloads page renders as unavailable. +// Overflow object-storage (Cloudflare R2) target for archives that exceed the +// Pages per-file cap: where to stage the file for the deploy-time upload, and the +// public URL base + key prefix the manifest points visitors at. Absent = no +// overflow storage configured, so oversize files are dropped instead. +type OverflowTarget = { + stagingDir: string; // export/.r2-staging/<siteId>/archives + publicBaseUrl: string; // e.g. https://archives.example.com (no trailing slash) + keyPrefix: string; // <siteId>/archives +}; + +// Stat a produced archive and enforce the size cap. An oversize file can't ship +// as a Pages asset, so either: (a) overflow storage is configured — move it to +// the staging dir for the deploy-time R2 upload and record its public `url`; or +// (b) it isn't — drop it and mark the entry `oversize`, which the Downloads page +// renders as unavailable. Under-cap files are served locally from /archives/. async function describeArchive( kind: ArchiveManifestEntry["kind"], scope: string, @@ -168,6 +174,7 @@ async function describeArchive( filename: string, videoCount: number, maxBytes: number, + overflow: OverflowTarget | null, channelTitle?: string, ): Promise<ArchiveManifestEntry> { const filePath = path.join(outDir, filename); @@ -177,16 +184,26 @@ async function describeArchive( } catch { bytes = 0; } + const entry: ArchiveManifestEntry = { kind, scope, filename, bytes, videoCount }; + if (channelTitle) entry.channelTitle = channelTitle; + const oversize = maxBytes > 0 && bytes > maxBytes; - if (oversize) { + if (oversize && overflow) { + // Move it out of the served tree and into the upload staging area; the deploy + // step pushes it to R2 and this URL resolves. + await mkdir(overflow.stagingDir, { recursive: true }); + await rename(filePath, path.join(overflow.stagingDir, filename)); + entry.url = `${overflow.publicBaseUrl}/${overflow.keyPrefix}/${filename}`; + console.log( + `[archives] ${filename} is ${humanBytes(bytes)} (over ${humanBytes(maxBytes)} cap) — staged for R2 upload`, + ); + } else if (oversize) { await rm(filePath, { force: true }); + entry.oversize = true; console.warn( - `[archives] ${filename} is ${humanBytes(bytes)} (over ${humanBytes(maxBytes)} cap) — not served`, + `[archives] ${filename} is ${humanBytes(bytes)} (over ${humanBytes(maxBytes)} cap) — not served (no overflow storage configured)`, ); } - const entry: ArchiveManifestEntry = { kind, scope, filename, bytes, videoCount }; - if (channelTitle) entry.channelTitle = channelTitle; - if (oversize) entry.oversize = true; return entry; } @@ -200,15 +217,37 @@ async function composeArchives( paths: ReturnType<typeof getPaths>, ): Promise<void> { const outDir = path.join(paths.exportPublicDir, "archives"); - // Always clear first so a disabled site (or one that lost the feature) never - // serves a stale bundle from a previous build. + // The deploy-time R2 upload reads staged oversize files from here. Kept OUTSIDE + // public/ so it's never deployed as Pages assets; namespaced by site so + // concurrent per-site builds don't clobber each other. + const stagingDir = path.join( + path.dirname(paths.exportPublicDir), + ".r2-staging", + site.siteId, + "archives", + ); + // Always clear both first so a disabled site (or one that lost the feature) + // never serves — or uploads — a stale bundle from a previous build. await rm(outDir, { recursive: true, force: true }); + await rm(stagingDir, { recursive: true, force: true }); if (!archivesEnabled(site)) { console.log("[archives] disabled for this build — skipping."); return; } await mkdir(outDir, { recursive: true }); + // Overflow object storage (R2) for oversize archives, if configured globally. + // Blank/absent config → oversize archives are dropped (today's behavior). + const storage = getSettings().archiveStorage; + const overflow: OverflowTarget | null = + storage && storage.bucket && storage.publicBaseUrl + ? { + stagingDir, + publicBaseUrl: storage.publicBaseUrl.replace(/\/+$/, ""), + keyPrefix: `${site.siteId}/archives`, + } + : null; + const maxBytes = archiveMaxBytes(site); const log = (m: string) => { if (m) console.log(`[archives] ${m}`); @@ -216,29 +255,18 @@ async function composeArchives( const build = { format: "zip" as const }; const common = { paths, channelSlugs: memberSlugs, outDir, build, onLog: log }; + // Per-channel archives only — the combined "whole-site" bundles were dropped + // (they duplicated the per-channel content and were always the first to blow + // past the size cap). const transcripts = await archiveTranscripts(common); - const transcriptsAll = await archiveCombinedTranscripts(common); const liveChat = await archiveLiveChat(common); - const liveChatAll = await archiveCombinedLiveChat(common); const titles = await channelTitles(paths); const bySlug = (a: { slug: string }, b: { slug: string }) => a.slug.localeCompare(b.slug); const entries: ArchiveManifestEntry[] = []; - // Transcripts: combined ("all") first, then per-channel by slug. - if (transcriptsAll.archivePath) { - entries.push( - await describeArchive( - "transcripts", - "all", - outDir, - path.basename(transcriptsAll.archivePath), - transcriptsAll.totalTranscripts, - maxBytes, - ), - ); - } + // Transcripts per channel, ordered by slug for a reproducible manifest. for (const a of [...transcripts.archives].sort(bySlug)) { entries.push( await describeArchive( @@ -248,24 +276,13 @@ async function composeArchives( path.basename(a.archivePath), a.transcriptCount, maxBytes, + overflow, titles.get(a.slug), ), ); } - // Live chat: same ordering. Channels with no chat produce nothing to list. - if (liveChatAll.archivePath) { - entries.push( - await describeArchive( - "live-chat", - "all", - outDir, - path.basename(liveChatAll.archivePath), - liveChatAll.totalLiveChats, - maxBytes, - ), - ); - } + // Live chat per channel. Channels with no chat produce nothing to list. for (const a of [...liveChat.archives].sort(bySlug)) { entries.push( await describeArchive( @@ -275,6 +292,7 @@ async function composeArchives( path.basename(a.archivePath), a.liveChatCount, maxBytes, + overflow, titles.get(a.slug), ), ); @@ -289,9 +307,10 @@ async function composeArchives( path.join(outDir, ARCHIVE_MANIFEST_FILENAME), JSON.stringify(manifest, null, 2) + "\n", ); - const served = entries.filter((e) => !e.oversize).length; + const available = entries.filter((e) => !e.oversize).length; + const staged = entries.filter((e) => e.url).length; console.log( - `[archives] wrote ${served}/${entries.length} archive(s) to public/archives.`, + `[archives] wrote ${available}/${entries.length} archive(s) (${staged} staged for R2) to public/archives.`, ); } @@ -380,26 +399,31 @@ async function main(): Promise<void> { // The detector writes one global duplicates.json over the whole channel pool; // each site only serves its own channels, so filter clusters to the site's // members (dropping out-of-site refs and clusters that fall below 2 members) - // before serving. Absent report → no file, and the page shows its empty state. + // before serving. The file is written only when the feature is enabled for this + // site (per-site `duplicates` opt-out) AND there's at least one in-scope + // cluster — so its mere presence is what hasDuplicates() keys off to show the + // nav link. No file → the page shows its empty state and the link self-hides. const dupSrc = path.join(paths.transcriptsDir, DUPLICATES_FILENAME); const dupDest = path.join(paths.exportPublicDir, DUPLICATES_FILENAME); await rm(dupDest, { force: true }); - if (await exists(dupSrc)) { + if (site.duplicates !== false && (await exists(dupSrc))) { const report = JSON.parse(await readFile(dupSrc, "utf8")) as DuplicateReport; const memberSet = new Set(memberSlugs); const clusters = report.clusters .map((c) => filterClusterToChannels(c, memberSet)) .filter((c): c is NonNullable<typeof c> => c !== null); - const filtered: DuplicateReport = { - ...report, - totals: { - ...report.totals, - clusters: clusters.length, - videosInClusters: clusters.reduce((n, c) => n + c.videoRefs.length, 0), - }, - clusters, - }; - await writeFile(dupDest, JSON.stringify(filtered)); + if (clusters.length > 0) { + const filtered: DuplicateReport = { + ...report, + totals: { + ...report.totals, + clusters: clusters.length, + videosInClusters: clusters.reduce((n, c) => n + c.videoRefs.length, 0), + }, + clusters, + }; + await writeFile(dupDest, JSON.stringify(filtered)); + } } // --- federation contract: /site.json descriptor + CORS _headers --- diff --git a/common/lib/archiveOptions.ts b/common/lib/archiveOptions.ts @@ -46,9 +46,15 @@ export type ArchiveManifestEntry = { filename: string; bytes: number; videoCount: number; - // True when the file exceeded thresholdBytes: it is NOT served (removed so a - // size-capped host like Cloudflare Pages won't reject the whole deploy) and - // the UI renders it as unavailable rather than a broken link. + // Set when the file exceeded thresholdBytes but was uploaded to overflow object + // storage (Cloudflare R2) instead of being dropped: the absolute public URL to + // download it from. When present, the UI links here and the entry is NOT + // oversize-unavailable. Absent for archives served locally from /archives/. + url?: string; + // True when the file exceeded thresholdBytes AND no overflow storage was + // configured: it is NOT served (removed so a size-capped host like Cloudflare + // Pages won't reject the whole deploy) and the UI renders it as unavailable + // rather than a broken link. Mutually exclusive with `url`. oversize?: boolean; }; diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -102,6 +102,13 @@ export type SiteSettings = { // can opt out via site.json `archives: false`, and a single build can skip via // the "Skip archive zips" build control. Opt-out: default true. buildArchives: boolean; + // Overflow object storage (Cloudflare R2) for archive zips that exceed the + // Pages per-file size cap (see Site.archiveMaxBytes). When both fields are set, + // an oversize archive is uploaded here on deploy — via `wrangler r2 object put`, + // keyed `<siteId>/archives/<file>` — instead of being dropped, and the Downloads + // page links to `<publicBaseUrl>/<key>`. Blank/absent → no overflow, so oversize + // archives stay unavailable ("Too large to host"). + archiveStorage?: { bucket: string; publicBaseUrl: string }; // Debounce preset for the global snapshot scheduler: how long it waits after // the last report-changing action before regenerating affected channel // reports. See REPORT_DEBOUNCE_PRESETS. Default "fast" (~1s, no cap). @@ -455,6 +462,7 @@ function defaults(): SiteSettings { inlineTranscribeOnFallback: false, skipLiveDownloads: true, buildArchives: true, + archiveStorage: { bucket: "", publicBaseUrl: "" }, reportDebouncePreset: DEFAULT_REPORT_DEBOUNCE_PRESET, autoRefreshIntervalSeconds: AUTO_REFRESH_INTERVAL_DEFAULT_SECONDS, syncScheduler: defaultSyncScheduler(), @@ -646,6 +654,14 @@ export function getSettings(): SiteSettings { if (typeof merged.buildArchives !== "boolean") { merged.buildArchives = true; } + { + const s = merged.archiveStorage; + merged.archiveStorage = { + bucket: s && typeof s.bucket === "string" ? s.bucket : "", + publicBaseUrl: + s && typeof s.publicBaseUrl === "string" ? s.publicBaseUrl : "", + }; + } if (!isReportDebouncePreset(merged.reportDebouncePreset)) { merged.reportDebouncePreset = DEFAULT_REPORT_DEBOUNCE_PRESET; } @@ -823,6 +839,16 @@ export async function writeSettings(next: SiteSettings): Promise<void> { inlineTranscribeOnFallback: next.inlineTranscribeOnFallback === true, skipLiveDownloads: next.skipLiveDownloads !== false, buildArchives: next.buildArchives !== false, + archiveStorage: { + bucket: + typeof next.archiveStorage?.bucket === "string" + ? next.archiveStorage.bucket.trim() + : "", + publicBaseUrl: + typeof next.archiveStorage?.publicBaseUrl === "string" + ? next.archiveStorage.publicBaseUrl.trim() + : "", + }, reportDebouncePreset: isReportDebouncePreset(next.reportDebouncePreset) ? next.reportDebouncePreset : DEFAULT_REPORT_DEBOUNCE_PRESET, diff --git a/common/lib/site.ts b/common/lib/site.ts @@ -87,6 +87,11 @@ export type Site = { // capped host (Cloudflare Pages: 25 MB) won't reject the deploy. 0 = no cap. // Absent = the global DEFAULT_ARCHIVE_MAX_BYTES / MAX_ARCHIVE_BYTES env. archiveMaxBytes?: number; + // Whether this site publishes the Duplicates page (and its Header nav link). + // Opt-OUT: undefined/true = on, only explicit `false` hides it. Even when on, + // the link/page auto-hide when the site has no in-scope duplicate clusters (the + // build simply writes no duplicates.json — see compose-site.ts / hasDuplicates). + duplicates?: boolean; // Per-site override for the hub this site belongs under (the PWA it points // visitors toward). Absent = inherit the family default SiteSettings.homepageUrl. // Resolve with resolveHubUrl(). Surfaced on /site.json so a hub can tell member @@ -259,6 +264,7 @@ export function parseSite(siteId: string, raw: unknown): Site { pwa: r.pwa === true, // Opt-out: only an explicit false disables. Absent/true stays on. archives: r.archives !== false, + duplicates: r.duplicates !== false, archiveMaxBytes: typeof r.archiveMaxBytes === "number" && Number.isFinite(r.archiveMaxBytes) && @@ -429,6 +435,7 @@ export async function writeSite( ...(site.pwa ? { pwa: true } : {}), // Persist only the non-default: archives is on unless explicitly disabled. ...(site.archives === false ? { archives: false } : {}), + ...(site.duplicates === false ? { duplicates: false } : {}), ...(typeof site.archiveMaxBytes === "number" && Number.isFinite(site.archiveMaxBytes) && site.archiveMaxBytes >= 0 diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,7 +1,9 @@ # Changelog ## [Unreleased] -- **Every site build now bundles downloadable transcript & live-chat archive zips.** The archive builders (per-channel `<slug>.zip`, combined `all-transcripts.zip` / `all-live-chat.zip`) previously only ran as standalone actions that wrote to a non-served directory; now `compose-site` generates them for the site's own channels straight into the served `public/archives/` and writes a `manifest.json` (sizes + counts) that the site's new **Downloads** page reads. **`zip` is now the default archive format** everywhere (was `tar.gz`), and the Build page's format help text tracks the selected format. Generation is **on by default with three opt-out levels**: a global **Generate archive zips on build** toggle in Settings, a per-site **Generate archive zips** toggle (plus an optional **Archive size cap (MB)**) on the site's page, and a per-build **Skip archive zips** checkbox on the Build and Build & Deploy controls (`BUILD_ARCHIVES=0`). Because a single file over ~25 MB breaks a Cloudflare Pages deploy, any archive over the cap (default 25 MB; `0` = no cap) is dropped from what's served and flagged `oversize` in the manifest so the deploy still succeeds and the Downloads page shows it as unavailable rather than a dead link. See `common/bin/compose-site.ts` (`composeArchives`), `common/controller/archive{Transcripts,LiveChat}.ts` (new `outDir` option), `common/lib/archiveOptions.ts` (default + manifest types), `common/lib/{site,settings}.ts` (opt-out flags), and `editor/app/{deploy/buildDeployCore.ts,build/buildAction.ts,deploy/components/Build{Export,Deploy}Button.tsx,sites/components/SiteForm.tsx,settings/components/SettingsForm.tsx}`. +- **Every site build now bundles downloadable per-channel transcript & live-chat archive zips.** The archive builders (per-channel `<slug>.zip` / `<slug>.live_chat.zip`) previously only ran as standalone actions that wrote to a non-served directory; now `compose-site` generates them for the site's own channels straight into the served `public/archives/` and writes a `manifest.json` (sizes + counts) that the site's new **Downloads** page reads. **`zip` is now the default archive format** everywhere (was `tar.gz`), and the Build page's format help text tracks the selected format. Generation is **on by default with three opt-out levels**: a global **Generate archive zips on build** toggle in Settings, a per-site **Generate archive zips** toggle (plus an optional **Archive size cap (MB)**) on the site's page, and a per-build **Skip archive zips** checkbox on the Build and Build & Deploy controls (`BUILD_ARCHIVES=0`). See `common/bin/compose-site.ts` (`composeArchives`), `common/controller/archive{Transcripts,LiveChat}.ts` (new `outDir` option), `common/lib/archiveOptions.ts` (default + manifest types), `common/lib/{site,settings}.ts` (opt-out flags), and `editor/app/{deploy/buildDeployCore.ts,build/buildAction.ts,deploy/components/Build{Export,Deploy}Button.tsx,sites/components/SiteForm.tsx,settings/components/SettingsForm.tsx}`. +- **Oversize archives now overflow to Cloudflare R2 instead of being dropped.** A single file over 25 MB breaks a Cloudflare Pages deploy, so a channel zip over the cap (default 25 MB; `0` = no cap) used to be removed from what's served and flagged `oversize`. Now, when **Archive overflow storage** is configured in Settings (an R2 **bucket** + its **public URL**), `compose-site` stages each oversize archive to `export/.r2-staging/<siteId>/` and records its future public URL in the manifest; the deploy step then uploads it via `wrangler r2 object put <bucket>/<siteId>/archives/<file>` (reusing the host's wrangler auth) **before** the Pages deploy, so the Downloads page links straight to R2. With no bucket configured the old drop-and-flag behavior is unchanged (uploads run only in the editor's Deploy / Build & deploy actions, not a raw `pnpm deploy`). The **combined "whole site" archives were removed** — they duplicated the per-channel content and were always the first to blow the cap. See `common/lib/settings.ts` (`archiveStorage`), `common/bin/compose-site.ts` (staging + manifest `url`), `common/lib/archiveOptions.ts` (`ArchiveManifestEntry.url`), and `editor/app/deploy/buildDeployCore.ts` (`runArchiveUploadIntoLog`) / `deploy/deployAction.ts` / `build/buildAction.ts`. +- **The Duplicates page is now a per-site toggle and hides itself when empty.** Each site's editor page gains a **Show the Duplicates page** checkbox (on by default). `compose-site` writes the site-filtered `duplicates.json` only when the toggle is on *and* there's at least one in-scope cluster, and the export Header keys its Duplicates nav link off a new `hasDuplicates()` — so the link and page disappear both when a site opts out and when it simply has no detected duplicates. See `common/lib/site.ts` (`duplicates` flag), `common/bin/compose-site.ts` (gated write), `export/app/lib/duplicates.ts` (new), `export/app/components/Header.tsx`, and `editor/app/sites/{components/SiteForm.tsx,actions.ts}`. - **You can now change a channel's slug (its id) — deliberately, from the Danger zone.** A channel's slug *is* its on-disk directory name (`transcripts/channels/<slug>/`), so it used to be fixed at creation ("Slug is fixed once a channel is created"). A new **Rename** form in the channel's Danger zone lifts that: enter a new slug and **type the current slug to confirm** (same friction as delete), and the rename is blocked while the channel has running/queued jobs (the in-memory registry keys by slug). Because the slug is a directory name, the rename does a **full migration** of every slug-keyed store so nothing silently breaks: it moves the channel dir (config, data, playlist, snapshot, shards, failed lists) **and** the saved-video store dir — rewriting each `saved-video.json` pointer's absolute `dir` so persisted source videos still resolve — then retargets every site.json membership, the sync scheduler's per-channel backoff state, and any job bookmarks. The two filesystem moves run first and roll back on failure; the metadata updates that follow are atomic and best-effort (surfaced as warnings). Renaming **changes the channel's public URL** (the old one 404s), which the form warns about. The slug grammar is also now validated on create. See `common/controller/renameChannel.ts`, `common/controller/channels.ts` (`isValidChannelSlug`), `common/lib/savedVideo-server.ts` (`rewriteSavedVideoDir`), `common/jobs/bookmarks.ts` (`renameChannelInBookmarks`), `editor/app/channels/{actions.ts,components/RenameChannelForm.tsx,[slug]/page.tsx}`, and `editor/e2e/channel-rename.spec.ts`. - **New Queue diagnostics page (`/jobs/queue`): see & force-release stuck jobs.** The job system has two sources of truth that can drift — the registry owns each job's `status`, the scheduler owns the running SLOT per queue. A cancel that never finalizes (a child that ignored SIGTERM, a crashed finalizer) leaves a job "cancelled" in the registry while the scheduler still marks its slot running, silently blocking every job behind it on that queue — and the Active Jobs page hides it (it filters to running/queued). The new **Queue** page reconciles the two: it builds from the **scheduler** as the source of truth for slots, cross-checks each against its registry record, and flags a running head as **stuck** when the record is terminal-but-holding-slot, evicted, or (softer) a live job idle past 10 minutes. It **auto-heals** the hard cases on every view/poll (frees terminal/evicted slots), shows a health strip (active queues, running, queued, **stuck**, workers), per-queue cards with the held-for duration / PID (`kill -9` hint) / last log line, and a **Force-release** button per slot (SIGKILLs the child and frees the slot unconditionally) plus a **Reap all stuck** action. Force-release is also available on any running job in Active Jobs, and Active Jobs links to Queue with a stuck-count badge. See `common/jobs/registry.ts` (`forceRelease`), `editor/app/jobs/queue/*`, `editor/app/jobs/{actions.ts,components/ForceReleaseJobButton.tsx}`, and `editor/e2e/queue.spec.ts`. - **Jobs page: real log retention + pagination (replaces the dead "Clear archived logs" button).** The old button only deleted logs absent from the in-memory registry — which, since the registry keeps the 100 newest finished jobs and sidecars preserve their real status, was almost never anything, so it did nothing. It's replaced by a **Clear logs** dropdown that prunes finished-job logs by age (older than 7 / 30 / 90 days) or all at once; running/queued jobs are never deleted. The `.jobs` directory also **self-trims on job finish** (throttled; keep newest 500, drop >30 days) so it can't grow unbounded. Job ids are now **ULIDs** (lexicographically time-sortable, timestamp decodable from the id), letting the list **paginate** — `listAllJobs` returns one page (default 50, grown by a **Load more** link) and only `stat`s/reads the sidecar for the shown page instead of every file on every load. `jobIdTime()` decodes both ULID and the legacy `<t36>-<rand>` ids, so existing on-disk logs still sort/read correctly. See `common/jobs/{ulid,listJobs,registry,streamCommand}.ts`, `editor/app/jobs/{page.tsx,actions.ts,components/ClearLogsMenu.tsx,[id]/page.tsx}`, and `editor/e2e/jobs.spec.ts`. diff --git a/editor/app/build/buildAction.ts b/editor/app/build/buildAction.ts @@ -22,6 +22,7 @@ import { } from "yt-dlp-transcript-common/jobs/streamCommand"; import { resolveOutDir, + runArchiveUploadIntoLog, runBuildPhase, runDeployIntoLog, } from "../deploy/buildDeployCore"; @@ -127,6 +128,13 @@ export async function buildAndDeployAction( throw new Error(`Build failed (exit ${buildCode}) — not deploying.`); } onLog("\n=== Deploy ===\n"); + // Push oversize archives to R2 before the Pages deploy (no-op when R2 + // isn't configured), so the published manifest URLs resolve. + const uploadCode = await runArchiveUploadIntoLog(onLog, signal, site, paths); + if (signal.aborted) return; + if (uploadCode !== 0) { + throw new Error(`Archive R2 upload failed (exit ${uploadCode}).`); + } const deployCode = await runDeployIntoLog( onLog, signal, diff --git a/editor/app/deploy/buildDeployCore.ts b/editor/app/deploy/buildDeployCore.ts @@ -5,6 +5,7 @@ // here as plain helpers and import them into the thin action wrappers. import path from "node:path"; +import { readdir } from "node:fs/promises"; import { runChildIntoLog } from "yt-dlp-transcript-common/jobs/runChild"; import type { Paths } from "yt-dlp-transcript-common/lib/paths"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; @@ -72,6 +73,73 @@ export async function runBuildPhase( }); } +// Where compose-site staged this site's oversize archives for R2 upload. Kept +// outside export/public so they never ship as Pages assets. Mirrors the path +// composeArchives writes to in common/bin/compose-site.ts. +function archiveStagingDir(siteId: string, paths: Paths): string { + return path.join( + path.dirname(paths.exportPublicDir), + ".r2-staging", + siteId, + "archives", + ); +} + +// Upload this site's staged oversize archives to the configured R2 bucket, so the +// remote URLs the served manifest points at actually resolve. Reuses the host's +// wrangler auth (process.env), same as the Pages deploy. No-op (returns 0) when +// overflow storage isn't configured or nothing was staged. Keys match what +// composeArchives wrote into the manifest: `<siteId>/archives/<file>`. Must run +// BEFORE the Pages deploy so the manifest never points at a missing object. +export async function runArchiveUploadIntoLog( + onLog: (line: string) => void, + signal: AbortSignal, + site: Site, + paths: Paths, +): Promise<number> { + const bucket = getSettings().archiveStorage?.bucket?.trim(); + if (!bucket) return 0; + + const stagingDir = archiveStagingDir(site.siteId, paths); + let files: string[]; + try { + files = (await readdir(stagingDir)).filter((f) => !f.startsWith(".")); + } catch { + // No staging dir → nothing oversize this build. + return 0; + } + if (files.length === 0) return 0; + + onLog( + `[archives] uploading ${files.length} oversize archive(s) to R2 bucket "${bucket}"…\n`, + ); + for (const file of files) { + const key = `${site.siteId}/archives/${file}`; + const code = await runChildIntoLog(onLog, signal, { + command: "pnpm", + args: [ + "dlx", + "wrangler", + "r2", + "object", + "put", + `${bucket}/${key}`, + `--file=${path.join(stagingDir, file)}`, + "--content-type=application/zip", + "--remote", + ], + cwd: paths.exportDir, + env: { ...process.env, NODE_ENV: "production" }, + }); + if (code !== 0) { + onLog(`[archives] R2 upload failed for ${key} (exit ${code}).\n`); + return code; + } + } + onLog("[archives] R2 upload complete.\n"); + return 0; +} + // Deploy a previously-built static bundle (`outDir`) to the site's Cloudflare // Pages project, streaming into `onLog`, returning the exit code. Runs on the // host with the host's Cloudflare credentials (process.env) — deploy never runs diff --git a/editor/app/deploy/deployAction.ts b/editor/app/deploy/deployAction.ts @@ -6,7 +6,11 @@ import { runManagedFunction, type StreamActionResult, } from "yt-dlp-transcript-common/jobs/streamCommand"; -import { resolveOutDir, runDeployIntoLog } from "./buildDeployCore"; +import { + resolveOutDir, + runArchiveUploadIntoLog, + runDeployIntoLog, +} from "./buildDeployCore"; const DEPLOY_QUEUE = "deploy"; @@ -29,6 +33,13 @@ export async function deployExportAction( queueKey: DEPLOY_QUEUE, paths, fn: async (onLog, signal) => { + // Push oversize archives to R2 first, so the manifest URLs the Pages + // deploy publishes resolve immediately. No-op when R2 isn't configured. + const uploadCode = await runArchiveUploadIntoLog(onLog, signal, site, paths); + if (signal.aborted) return; + if (uploadCode !== 0) { + throw new Error(`Archive R2 upload failed (exit ${uploadCode}).`); + } const code = await runDeployIntoLog( onLog, signal, diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -51,6 +51,12 @@ export async function saveSettingsAction( formData.get("inlineTranscribeOnFallback") === "on"; const skipLiveDownloads = formData.get("skipLiveDownloads") === "on"; const buildArchives = formData.get("buildArchives") === "on"; + const archiveStorage = { + bucket: String(formData.get("archiveStorageBucket") ?? "").trim(), + publicBaseUrl: String( + formData.get("archiveStoragePublicBaseUrl") ?? "", + ).trim(), + }; const reportDebouncePresetRaw = String( formData.get("reportDebouncePreset") ?? "", ).trim(); @@ -214,6 +220,7 @@ export async function saveSettingsAction( inlineTranscribeOnFallback, skipLiveDownloads, buildArchives, + archiveStorage, reportDebouncePreset, autoRefreshIntervalSeconds: autoRefreshParsed, syncScheduler, diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -175,14 +175,46 @@ export function SettingsForm({ initial, apps }: Props) { Generate downloadable archive zips on build </span> <span className="text-xs text-muted-foreground"> - Each site build produces transcript &amp; live-chat archive zips (per - channel and combined) and lists them on the site&apos;s Downloads - page. On by default. A site can opt out on its own page, and a single - build can skip them from the Build controls. + Each site build produces per-channel transcript &amp; live-chat + archive zips and lists them on the site&apos;s Downloads page. On by + default. A site can opt out on its own page, and a single build can + skip them from the Build controls. </span> </span> </label> <label className="flex flex-col gap-1 text-sm"> + <span className="font-medium">Archive overflow storage (R2 bucket)</span> + <input + type="text" + name="archiveStorageBucket" + defaultValue={initial.archiveStorage?.bucket ?? ""} + placeholder="my-archives-bucket" + className="rounded border border-border bg-card px-2 py-1 text-sm" + /> + <span className="text-xs text-muted-foreground"> + Cloudflare R2 bucket name. Archive zips larger than the Pages 25&nbsp;MB + per-file cap are uploaded here on deploy (via <code>wrangler</code>) + instead of being dropped. Leave blank to keep oversize archives + unavailable. + </span> + </label> + <label className="flex flex-col gap-1 text-sm"> + <span className="font-medium">Archive overflow public URL</span> + <input + type="text" + name="archiveStoragePublicBaseUrl" + defaultValue={initial.archiveStorage?.publicBaseUrl ?? ""} + placeholder="https://archives.example.com" + className="rounded border border-border bg-card px-2 py-1 text-sm" + /> + <span className="text-xs text-muted-foreground"> + Public base URL the bucket is served from (its <code>r2.dev</code> + subdomain or a custom domain). The Downloads page links to{" "} + <code>&lt;base&gt;/&lt;siteId&gt;/archives/&lt;file&gt;.zip</code>. + Required for overflow uploads to work. + </span> + </label> + <label className="flex flex-col gap-1 text-sm"> <span className="font-medium">Report refresh debounce</span> <select name="reportDebouncePreset" diff --git a/editor/app/sites/actions.ts b/editor/app/sites/actions.ts @@ -74,6 +74,8 @@ export async function saveSiteAction( // Archives default on: an unchecked (default-on) box yields no "archives" key // → false → persisted as the explicit opt-out. const archives = formData.get("archives") === "on"; + // Duplicates default on, same opt-out idiom as archives. + const duplicates = formData.get("duplicates") === "on"; const archiveMaxMBRaw = String(formData.get("archiveMaxMB") ?? "").trim(); let archiveMaxBytes: number | undefined; if (archiveMaxMBRaw) { @@ -178,6 +180,7 @@ export async function saveSiteAction( ...(hubUrl ? { hubUrl } : {}), ...(pwa ? { pwa: true } : {}), ...(archives ? {} : { archives: false }), + ...(duplicates ? {} : { duplicates: false }), ...(archiveMaxBytes !== undefined ? { archiveMaxBytes } : {}), ...(relatedSites.length > 0 ? { relatedSites } : {}), }; diff --git a/editor/app/sites/components/SiteForm.tsx b/editor/app/sites/components/SiteForm.tsx @@ -270,9 +270,9 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) { Generate downloadable archive zips on build </label> <p className="-mt-2 text-xs text-muted-foreground"> - On by default: each build produces transcript &amp; live-chat zips (per - channel and combined) and lists them on the site&apos;s Downloads page. - Turn off to skip generation for this site. + On by default: each build produces per-channel transcript &amp; live-chat + zips and lists them on the site&apos;s Downloads page. Turn off to skip + generation for this site. </p> <Field label="Archive size cap (MB)" @@ -283,8 +283,22 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) { ? String(Math.round(initial.archiveMaxBytes / (1024 * 1024))) : "" } - hint="Archives larger than this are not served (they'd break a size-capped host like Cloudflare Pages' 25 MB limit) and show as unavailable. Leave blank for the default 25 MB; 0 = no cap." + hint="Archives larger than this can't ship as Cloudflare Pages assets (25 MB limit). If R2 overflow storage is configured in Settings they upload there on deploy; otherwise they show as unavailable. Leave blank for the default 25 MB; 0 = no cap." /> + <label className="flex items-center gap-2 text-sm"> + <input + type="checkbox" + name="duplicates" + defaultChecked={initial.duplicates !== false} + className="accent-brand" + /> + Show the Duplicates page + </label> + <p className="-mt-2 text-xs text-muted-foreground"> + On by default: publishes the cross-channel duplicate-shorts page and its + header link. Turn off to hide it for this site. Even when on, the link and + page auto-hide when this site has no detected duplicates. + </p> <fieldset className="flex flex-col gap-3 border border-border rounded p-3"> <legend className="px-1 text-sm font-medium">Channels</legend> diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -1,7 +1,10 @@ # Changelog ## [Unreleased] -- **A new Downloads page lets you take the whole archive with you.** Reachable from the header nav and footer (shown only when a build actually produced archives), `/downloads` lists the site's transcript and live-chat bundles as `.zip` downloads — the whole site up top, then per channel — each printing its video count and file size. A bundle too large to host (over the build's size cap) is shown as unavailable with the reason instead of a broken link. The zips are regenerated on every build. +- **A new Downloads page lets you take a channel's archive with you.** Reachable from the header nav and footer (shown only when a build actually produced downloadable archives), `/downloads` lists the site's transcript and live-chat bundles as per-channel `.zip` downloads, each printing its video count and file size. The zips are regenerated on every build. +- **Big archives are no longer dropped — they overflow to object storage.** Cloudflare Pages rejects any single asset over 25 MB, so oversize channel zips (a busy channel's live-chat log can be hundreds of MB) used to show as "too large to host." When an R2 overflow bucket is configured, those archives now upload there on deploy and the Downloads page links straight to them. Without a bucket configured, the old "unavailable" behavior stands. +- **Combined "whole site" archives were removed** in favor of per-channel bundles — they duplicated the per-channel content and were always the first to blow past the size cap. +- **The Duplicates page can be turned off per site, and hides itself when empty.** Sites can opt out of the cross-channel duplicate-shorts page from their editor settings, and even when it's on, the header link and page now hide automatically when a site has no detected duplicates. - **Download a single video's transcript or live chat as a file.** The player toolbar has a new download control (`⤓`) that saves whatever you're viewing — the transcript, or the live chat — as `.txt`, `.srt`, or `.json`, generated in your browser from the already-loaded cues (no download of the full archive needed). ## [0.6.0] - 2026-07-02 diff --git a/export/app/components/Header.tsx b/export/app/components/Header.tsx @@ -11,6 +11,7 @@ import { ThemeMenu } from "yt-dlp-transcript-common/components/ThemeMenu"; import { currentSite } from "../lib/site"; import { instanceMode } from "../lib/mode"; import { hasArchives } from "../lib/archives"; +import { hasDuplicates } from "../lib/duplicates"; import SiblingSwitcher from "./SiblingSwitcher"; // The export site's masthead: a brand-colored mark + wordmark, calm sans nav, @@ -35,6 +36,7 @@ export default function Header() { // single site, not on the hub (whose "family" is the runtime shelf). const related = isSite ? resolveRelatedSites(site, listSites()) : []; const showDownloads = hasArchives(); + const showDuplicates = hasDuplicates(); return ( <header className="sticky top-0 z-20 border-b border-border bg-background/80 backdrop-blur-md"> @@ -50,12 +52,14 @@ export default function Header() { <Link href="/" className="text-foreground hover:text-brand transition-colors"> Search </Link> - <Link - href="/duplicates" - className="text-foreground hover:text-brand transition-colors" - > - Duplicates - </Link> + {showDuplicates && ( + <Link + href="/duplicates" + className="text-foreground hover:text-brand transition-colors" + > + Duplicates + </Link> + )} {showDownloads && ( <Link href="/downloads" diff --git a/export/app/downloads/page.tsx b/export/app/downloads/page.tsx @@ -24,13 +24,15 @@ function countLabel(kind: ArchiveManifestEntry["kind"], n: number): string { } // One archive rendered as a manifest line: a printed contents label on the left, -// the download action on the right. Oversize entries were not shipped, so they -// read as unavailable with the reason rather than a dead link. +// the download action on the right. An oversize entry with no overflow URL was +// not shipped, so it reads as unavailable with the reason rather than a dead +// link; everything else links to its local /archives/ file or its overflow URL. function ArchiveLine({ entry }: { entry: ArchiveManifestEntry }) { const label = entry.channelTitle ?? (entry.scope === "all" ? "Everything" : entry.scope); const kindLabel = entry.kind === "transcripts" ? "Transcripts" : "Live chat"; + const href = entry.url ?? `/archives/${entry.filename}`; - if (entry.oversize) { + if (entry.oversize && !entry.url) { return ( <div className="flex flex-wrap items-baseline justify-between gap-x-4 gap-y-1 border-l-2 border-border/60 py-3 pl-4 opacity-70"> <div className="min-w-0"> @@ -50,7 +52,7 @@ function ArchiveLine({ entry }: { entry: ArchiveManifestEntry }) { return ( <a - href={`/archives/${entry.filename}`} + href={href} download className="group flex flex-wrap items-baseline justify-between gap-x-4 gap-y-1 border-l-2 border-brand/40 py-3 pl-4 transition-colors hover:border-brand hover:bg-muted/40" > @@ -74,13 +76,11 @@ export default function DownloadsPage() { const manifest = readArchiveManifest(); const entries = manifest?.entries ?? []; - const combined = entries.filter((e) => e.scope === "all"); - const perChannel = entries.filter((e) => e.scope !== "all"); - const channelSlugs = Array.from(new Set(perChannel.map((e) => e.scope))); - const servedCount = entries.filter((e) => !e.oversize).length; - const totalBytes = entries - .filter((e) => !e.oversize) - .reduce((n, e) => n + e.bytes, 0); + // An entry is available if it ships locally (not oversize) or lives in overflow + // storage (has a url). Oversize-without-url entries are unavailable. + const available = entries.filter((e) => !e.oversize || e.url); + const servedCount = available.length; + const totalBytes = available.reduce((n, e) => n + e.bytes, 0); if (entries.length === 0) { return ( @@ -107,10 +107,10 @@ export default function DownloadsPage() { Take the whole archive with you </h1> <p className="max-w-prose text-sm text-muted-foreground"> - Every transcript and live-chat log on this site, packaged as{" "} - <code className="font-mono">.zip</code> bundles. Each archive holds the - compact cues JSON per video plus a channel manifest — grab the whole - site or a single channel. + Every transcript and live-chat log on this site, packaged per channel + as <code className="font-mono">.zip</code> bundles. Each archive holds + the compact cues JSON per video plus a channel manifest — grab a whole + channel at once. </p> <p className="font-mono text-xs text-muted-foreground/80"> {servedCount} bundle{servedCount === 1 ? "" : "s"} available ·{" "} @@ -118,31 +118,16 @@ export default function DownloadsPage() { </p> </header> - {combined.length > 0 && ( - <section className="flex flex-col gap-3"> - <h2 className="font-mono text-xs uppercase tracking-[0.14em] text-muted-foreground"> - Whole site - </h2> - <div className="flex flex-col divide-y divide-border rounded-lg border border-border bg-card/40"> - {combined.map((e) => ( - <ArchiveLine key={e.filename} entry={e} /> - ))} - </div> - </section> - )} - - {channelSlugs.length > 1 && ( - <section className="flex flex-col gap-3"> - <h2 className="font-mono text-xs uppercase tracking-[0.14em] text-muted-foreground"> - By channel - </h2> - <div className="flex flex-col divide-y divide-border rounded-lg border border-border bg-card/40"> - {perChannel.map((e) => ( - <ArchiveLine key={e.filename} entry={e} /> - ))} - </div> - </section> - )} + <section className="flex flex-col gap-3"> + <h2 className="font-mono text-xs uppercase tracking-[0.14em] text-muted-foreground"> + By channel + </h2> + <div className="flex flex-col divide-y divide-border rounded-lg border border-border bg-card/40"> + {entries.map((e) => ( + <ArchiveLine key={e.filename} entry={e} /> + ))} + </div> + </section> <p className="text-xs text-muted-foreground/70"> Archives refresh on every site build. Transcripts and live chat are diff --git a/export/app/lib/archives.ts b/export/app/lib/archives.ts @@ -22,9 +22,10 @@ export function readArchiveManifest(): ArchiveManifest | null { } } -// Whether this site has at least one actually-served (non-oversize) archive to -// link to. Drives showing the Downloads nav/footer entry. +// Whether this site has at least one downloadable archive to link to — either +// served locally (not oversize) or uploaded to overflow storage (has a `url`). +// Drives showing the Downloads nav/footer entry. export function hasArchives(): boolean { const manifest = readArchiveManifest(); - return !!manifest && manifest.entries.some((e) => !e.oversize); + return !!manifest && manifest.entries.some((e) => !e.oversize || e.url); } diff --git a/export/app/lib/duplicates.ts b/export/app/lib/duplicates.ts @@ -0,0 +1,22 @@ +import fs from "node:fs"; +import path from "node:path"; +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { + DUPLICATES_FILENAME, + type DuplicateReport, +} from "yt-dlp-transcript-common/lib/duplicates"; + +// Whether this site has a duplicate-shorts report to show. compose-site.ts writes +// public/<DUPLICATES_FILENAME> only when the per-site `duplicates` toggle is on +// AND there's at least one in-scope cluster, so its mere presence is the signal: +// no file → the Duplicates nav link hides and the page shows its empty state. +// Read synchronously — this is a static export, evaluated once per build. +export function hasDuplicates(): boolean { + try { + const file = path.join(getPaths().exportPublicDir, DUPLICATES_FILENAME); + const parsed = JSON.parse(fs.readFileSync(file, "utf8")) as DuplicateReport; + return Array.isArray(parsed?.clusters) && parsed.clusters.length > 0; + } catch { + return false; + } +}