Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 7b14346fd00ff02fcd2e09fae5f8cb15d83a233a
parent 9fed0dd7170efe759505fd5a750913eaae884421
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu, 21 May 2026 23:37:24 -0400

fix audio integrity checker and search matches

Diffstat:
Mcommon/lib/searchEval.ts | 11++++++++++-
Mcommon/ytdlp/audioCheckedDownload.ts | 25+++++++++++++++++++------
Meditor/CHANGELOG.md | 1+
Mexport/CHANGELOG.md | 1+
Mexport/e2e/query-tree.spec.ts | 43+++++++++++++++++++++++++++++++++++++++++++
5 files changed, 74 insertions(+), 7 deletions(-)

diff --git a/common/lib/searchEval.ts b/common/lib/searchEval.ts @@ -370,7 +370,16 @@ function applyCached( const slugs = new Set<string>(); for (const s of cached.slugs) slugs.add(s); const hits = new Map<string, LayerHit[]>(); - for (const [slug, h] of cached.hits) hits.set(slug, h); + // Cached hits carry the leafId from whatever session wrote them. That id + // is module-counter-scoped (searchQuery.ts) and won't survive a reload, + // so rewrite to the current leaf's id — otherwise the renderer's + // bucket-by-leafId step in TranscriptSearch drops every cached hit. + for (const [slug, list] of cached.hits) { + hits.set( + slug, + list.map((h) => ({ ...h, leafId: leaf.id })), + ); + } ctx.leafResults.set(leaf.id, { slugs, hits }); setLeafState(ctx, leaf.id, { slugCount: slugs.size, diff --git a/common/ytdlp/audioCheckedDownload.ts b/common/ytdlp/audioCheckedDownload.ts @@ -196,6 +196,7 @@ async function pathExists(p: string): Promise<boolean> { async function prepareDataTree( channelDir: string, + expectedVideoIdHint: string | null, onLog: (s: string) => void, precheck?: { paths: Paths; @@ -205,7 +206,11 @@ async function prepareDataTree( ): Promise<void> { // Drop any leftover .testing snapshots from crashed prior runs, and // promote any orphan .good back to .part so yt-dlp resume picks up where - // we left off. Scoped to <channelDir>/data/*. + // we left off. When `expectedVideoIdHint` is set, scoped to that single + // subdir under <channelDir>/data/ — sibling dirs belong to other videos + // that yt-dlp won't touch on this launch, so probing them is wasted work. + // When the hint is null (Rumble: yt-dlp's internal id only appears after + // metadata.info.json), fall back to scanning every subdir. // // When `precheck` is supplied, also probe each pre-existing .part (i.e. // one we did NOT just restore from a .good — the .good was already @@ -220,6 +225,9 @@ async function prepareDataTree( } catch { return; } + if (expectedVideoIdHint) { + videoDirs = videoDirs.filter((d) => d === expectedVideoIdHint); + } for (const sub of videoDirs) { const dir = path.join(dataDir, sub); let entries: string[] = []; @@ -475,11 +483,16 @@ export async function runAudioCheckedYtdlp( // Sticky across rollback/restart loops since the id stays the same. let resolvedVideoDir: string | null = null; - await prepareDataTree(opts.channelDir, opts.onLog, { - paths: opts.paths, - signal: opts.signal, - onCheckpoint: (rec) => checkpoints.push(rec), - }); + await prepareDataTree( + opts.channelDir, + opts.expectedVideoIdHint, + opts.onLog, + { + paths: opts.paths, + signal: opts.signal, + onCheckpoint: (rec) => checkpoints.push(rec), + }, + ); // Snapshot of subdirs that existed before yt-dlp launched. discoverPartFile // and the success-path fallback scan use this to ignore stale .part files diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -23,4 +23,5 @@ - Clicking a mobile stage badge (or a stage rail item) now auto-expands the collapsible "Channel pipeline & settings" wrapper if the user had collapsed it. Previously the hash navigation succeeded but the target section was inside a closed `<details>` and not visible. The target is also re-scrolled into view after the expansion so it lands at the correct position. - Audio-check probe now flags `.part` files with heavy mid-stream codec corruption as malformed even when ffmpeg exits 0. Previously a stream could rack up hundreds of per-packet AAC decode failures (`Error submitting packet to decoder`) while ffmpeg still exited 0, and the file was promoted to a finished download. The classifier now also treats >5 decoder-error lines in stderr as malformed. - Audio-check pipeline now scopes its `.part` discovery and final-probe to the video being downloaded, instead of scanning the whole channel `data/` tree and picking the first match. On channels with many in-progress `.part` files (cornbreadman: ~248), the orchestrator would latch onto a stale neighbour's `.part`, probe it instead of the current download, and then the success-path fallback would resolve a totally unrelated subdir's already-finalized `audio.mp3` as "the final file" — skipping transcode and letting the real `.mp4` finalize with no integrity check, only to fail noisily during the next transcode-untranscoded sweep. +- Audio-check pre-launch pass (`.testing` cleanup, `.good` → `.part` restore, malformed-`.part` probe) is now also scoped to the targeted video's subdir when the video id is known from the URL. Previously the pre-pass scanned every subdir under `<channel>/data/` and full-decoded each pre-existing `.part` through ffmpeg, so kicking off a single download on a channel with many interrupted prior attempts (cornbreadman) would stall for minutes emitting `Pre-check OK ... verdict=...` lines for ~30 sibling videos before yt-dlp even started. Rumble URLs (id not knowable until `metadata.info.json` is written) still scan all subdirs as before. - Editor's `/changelog` page no longer renders unstyled. The `@source` directive in `editor/app/globals.css` pointed at a non-existent path, so Tailwind never scanned the shared `<Changelog>` component for class usage. diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -9,6 +9,7 @@ - Per-heading copy-link buttons on the changelog page for permalinks to any section. ### Fixed +- Layered search no longer drops every hit row after a page reload. `LayerHit.leafId` is set from a session-scoped counter id and was stored verbatim in the IndexedDB layer cache; on reload `parseRoot` re-allocates leaf ids from a fresh counter, so cached hits arrived carrying stale ids and got silently dropped by the renderer's bucket-by-leaf step. The "N hits" count still summed correctly because it's computed before bucketing — matching the reported symptom. `applyCached` now rewrites each cached hit's `leafId` to the current leaf's id. - Layered search no longer drops transcript hit rows for AND-chained leaves after certain edit sequences. The per-leaf result cache was keyed without `contributeHits`, so a leaf evaluated with "Show hits in results" off would write an empty-hits cache entry that a later same-query/scope hits-on leaf would silently inherit — the result list surfaced the video but no hit timestamps. Adding `contributeHits` to the canonical hash separates the two payload shapes. Toggling regex (which already changed the hash) was the existing workaround. - Filter changes survive a refresh again. When the user had any saved profile active, `commitSearch` was writing the new state to the working snapshot but leaving the profile pointer set; hydration then preferred the unchanged profile snapshot and silently reverted the commit. Committing changes that diverge from the active profile now clears the profile pointer (the selector falls back to "(unsaved)") so the next reload reads the working snapshot. - `build:index` no longer runs out of memory on large datasets. Both the transcripts and the subs page writers now stream each entry directly to disk and hash it incrementally instead of materialising the joined page body in memory, which previously OOMed when a single live-chat track encoded to hundreds of MB. The post-processing "is this video deleted?" pass is now an in-memory LMDB scan (the availability check is cached in the mtime record at mutation time, schema bumped to 8 to invalidate the old cache) instead of ~28k sequential `availability.json` reads on every build. The `build:index` script also pre-sets `--max-old-space-size=8192` as a backstop. diff --git a/export/e2e/query-tree.spec.ts b/export/e2e/query-tree.spec.ts @@ -355,6 +355,49 @@ test.describe("composite search — query tree", () => { expect(checkedStates.filter(Boolean).length).toBe(1); }); + test("cache survives leaf-id changes across page reloads (regression)", async ({ + page, + }) => { + // Regression: LayerHits carry a session-scoped `leafId`. The layer cache + // stores hits verbatim, so a reload (which resets the id counter and + // re-allocates ids via parseRoot) used to surface stale ids — the + // renderer buckets hits by leafId, so every cached hit was silently + // dropped while the total-hits count (computed before bucketing) stayed + // correct. Build the AND via the UI so the counter advances past where + // parseRoot lands, then reload and assert hits still render. + await page.goto("/"); + const firstInput = page + .locator('input[data-testid^="leaf-query-"]') + .first(); + await firstInput.fill("alpha"); + await page.getByTestId("compact-add-layer").click(); + const inputs = page.locator('input[data-testid^="leaf-query-"]'); + await inputs.nth(1).fill("gamma"); + await page.getByTestId("search-submit").click(); + await expectResultSlugs(page, [ + TRANSCRIPT_ONLY_SLUG, + CHAT_SMALL_SLUG, + CHAT_LARGE_SLUG, + ]); + await expect( + page.locator( + `[data-result-slug="${TRANSCRIPT_ONLY_SLUG}"] [data-leaf-section]`, + ), + ).toHaveCount(2); + + await page.reload(); + await expectResultSlugs(page, [ + TRANSCRIPT_ONLY_SLUG, + CHAT_SMALL_SLUG, + CHAT_LARGE_SLUG, + ]); + await expect( + page.locator( + `[data-result-slug="${TRANSCRIPT_ONLY_SLUG}"] [data-leaf-section]`, + ), + ).toHaveCount(2, { timeout: 15_000 }); + }); + test("builder UI: + Add layer adds a second leaf and commits on Search", async ({ page, }) => {