commit c747a348ee1825ab51e2209ae16fde9c8abe57a8
parent 715d3ef8340b25cb7934063039556516a6722b80
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 5 Aug 2026 00:11:53 -0400
Cover the full sweep, and don't trust an empty listing
Three follow-ons to the sweep itself:
An empty enumeration is never trustworthy enough to act on — it is what a
transient upstream failure, an expired cookie, and a clean 101 exit all look
like from here. Acting on one would truncate the stored playlist AND flag
every video we own as missing. The sweep now leaves both alone, and declines
to stamp lastFullSweepAt so the next sync retries rather than waiting out the
cadence.
The sweep is off in the e2e default settings, the way
verifyAvailabilityBeforeClean is. Otherwise every pre-existing sync spec
would silently start exercising the sweep instead of the paged walk, since
no fixture channel has ever swept and a never-swept channel is due.
settings/actions.ts preserves both new keys when the form doesn't carry them,
rather than coercing an absent field to NaN and resetting it to the default.
The form controls land with Part B.
sync-deep.spec.ts covers the three behaviours: a due sweep replaces a stale
playlist and flags a video missing from the listing (resolving it as deleted,
under the cap); over the cap the suspects are flagged but never probed; and
the next sync stays on the paged walk until the cadence elapses.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Diffstat:
5 files changed, 291 insertions(+), 1 deletion(-)
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -1358,6 +1358,19 @@ async function syncFullSweep(opts: RunYtdlpOpts): Promise<void> {
const urls = await enumeratePlaylistUrls(opts, root);
if (opts.signal.aborted) return;
+ // An empty listing is never trustworthy enough to act on: it is what a
+ // transient upstream failure, a cookie expiry, or a clean 101 exit all look
+ // like from here. Acting on it would truncate the stored playlist AND flag
+ // every video we own as missing. Leave both alone and don't stamp the sweep,
+ // so the next sync tries again rather than waiting out the cadence.
+ if (urls.length === 0) {
+ opts.onLog(
+ `Full sweep: the channel listing came back empty — leaving the stored playlist and missing-video flags untouched, and not counting this as a sweep.\n`,
+ );
+ await touchLastSync(opts);
+ return;
+ }
+
// 2. Refresh the stored playlist. Everything downstream — "download missing",
// the snapshot's undownloadedIds — reads this file, and before the sweep
// only an explicit "store playlist" ever rewrote it.
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Sync now notices when a video disappears.** A channel's video list was being fetched three separate times for three purposes that never shared their work: "store playlist" refreshed the stored list, Sync walked the newest 50 entries at a time, and "Quick check" re-read the whole listing to find videos that had gone missing. Because Sync only ever saw the newest slice, it could never spot a deletion — and it never refreshed the stored list either, so "Download missing" and the report's not-yet-downloaded count kept working off whatever the last "store playlist" click wrote, possibly months earlier. Sync now periodically pays for **one** full read of the channel and gets all three out of it: the stored list is refreshed, videos that have left the listing are flagged, and new uploads are downloaded as before. That means **Sync all** surfaces upstream deletions across every channel on its own, where it used to take a per-channel "Quick check" click. The deep pass runs at most once a day per channel by default (it is much more expensive than a normal sync on a large channel); every sync in between stays exactly as cheap as it was. When a handful of videos are flagged — 25 or fewer by default — the same job goes on to work out which are deleted, private, or merely unlisted; past that it flags them and leaves the call to you. **What it will not do is download anything a normal sync wouldn't**: on a channel you deliberately keep only the newest few hundred of, a deep pass will not start dragging down the back catalogue. It changes what the editor *knows*, never what it *fetches*. If the channel listing comes back empty — a network blip, an expired cookie — the deep pass leaves the stored list and the missing-video flags untouched rather than concluding your whole archive vanished.
- **The editor is fast now.** Every page in the editor had a floor of about 4.4 seconds on it, and the reason was one line in the sidebar. The reclaimable-disk badge — the little "12.4 GB" pill next to Cleanup — asked for the channel list, and the function it asked was the one that counts the corpus from scratch: a `readdir` for each of the **78,350** video directories plus a digest sidecar read for each of the ~70,000 transcribed ones, **~474,559 files touched, measured at 3,985 ms**, to describe **98 videos**. Every count that walk produced was then thrown away. It sat in the root layout, so *every* document load paid it; the 5-second auto-refresh re-ran it on a timer, on every route, forever; and three widget endpoints called it on each poll. It now reads the 65 per-channel snapshots it could always have read — the same numbers, **68 ms**, a 59× improvement — and the sidebar badge itself is down to ~40 ms. Loading `/channels` went from 4.5 s to roughly a tenth of a second; the dashboard from ~10 s. The corpus-walking function still exists under a name that says what it costs (`listChannelStatsFromDisk`) for the batch jobs that genuinely need ground truth, and a test now fails the build if it ever reappears anywhere the editor renders. **The honest trade:** the video, transcript and download counts on `/channels` and the dashboard now come from each channel's last generated report rather than from disk directly, so a job that just finished can take a moment — the snapshot scheduler's ~1 second debounce — to show up. Verified against the live corpus: those three counts match a full walk **exactly** on all 65 channels. The one field that doesn't is digest coverage, which reads 0 for the 11 channels whose reports predate per-engine digest counts until their next report refresh. `/channels` now prints how old the oldest report on the page is, rather than leaving you to assume the numbers are live.
- **Auto-refresh no longer refreshes when nothing has changed.** The passive refresher called `router.refresh()` on a timer — a full server-side re-render of the entire page tree, every 5 seconds, on every route, whether or not anything had actually happened. Against the real corpus that was ~4.9 seconds of work per tick, and it held the editor's server process at roughly **22% of a CPU core, permanently, with a single idle tab open**. It now asks a new `/api/pulse` endpoint whether anything moved — a change token built from in-memory job, queue and worker state plus two file timestamps, no corpus reads at all — and re-renders only when the answer is yes. An idle page now performs **zero** re-renders where it used to perform one every five seconds; there's a test that fails if that ever regresses. Same setting, same 5-second default, same "0 disables" behaviour, and editing settings still repaints the sidebar immediately, because the settings file's timestamp is part of the token. The sidebar's job and reclaimable-disk pills now update on their own rather than requiring the whole page to re-render — which in turn let fifteen server actions stop invalidating the client's entire navigation cache to announce that a job count had changed. Also fixed while in here: opening a channel with no report used to **generate one inside the page load**, a full analysis of every video directory in that channel — minutes, on the big ones, with no progress and no way to stop it. It now shows a banner offering to run it as a normal background job, and renders the rest of the page as usual — the video list comes from disk, not the report, and the pipeline controls are exactly what you want on a channel you haven't analysed yet.
- **A page that throws no longer takes the whole editor with it, and moving between pages is instant.** There were **zero** error boundaries in the editor: anything that threw while rendering — a malformed config, a half-written snapshot — blanked the entire document, sidebar and all, with nothing to click and nothing to read. There is now a route error boundary that keeps the chrome alive, shows the error's digest so you can find it in the server log, and offers *Try again* (Next 16.2's `unstable_retry`, which actually re-fetches, rather than the older `reset`, which only clears the error state). Alongside it, a 15-second client router cache (`staleTimes`), which is what makes bouncing between two sidebar links immediate instead of a fresh server round trip each way. Loading skeletons were built and then **removed**, on measurement, for two reasons worth recording: in Next 16 a `loading.tsx` does not actually produce a fallback during a client-side navigation (the framework's own reference says so, and three different tests confirmed it), and worse, the streaming boundary it creates commits the HTTP status *before* the page decides — so every page that says "this doesn't exist" was returning **200** with 404 content inside it. Measured directly: a missing channel returned 200 with the boundaries in place and 404 without. There's now a test pinning that. Fast navigation here comes from prefetching plus the router cache.
diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts
@@ -163,6 +163,18 @@ export async function saveSettingsAction(
const n = Number.parseInt(raw, 10);
return Number.isFinite(n) ? n : null;
};
+ // For fields a given render of this form may not carry: an absent key keeps
+ // the saved value instead of coercing to NaN (which would reset it to the
+ // default). A present-but-blank value is treated as absent for the same reason.
+ const current = getSettings();
+ const intOrKeep = (key: string, fallback: number): number => {
+ const raw = formData.get(key);
+ if (raw === null) return fallback;
+ const trimmed = String(raw).trim();
+ if (!trimmed) return fallback;
+ const n = Number.parseInt(trimmed, 10);
+ return Number.isFinite(n) ? n : fallback;
+ };
const syncScheduler = {
enabled: formData.get("syncSchedulerEnabled") === "on",
defaultIntervalMinutes: intOrNaN("syncSchedulerDefaultIntervalMinutes"),
@@ -175,6 +187,18 @@ export async function saveSettingsAction(
keepLatestCheckIntervalMinutes: intOrNaN(
"syncSchedulerKeepLatestCheckIntervalMinutes",
),
+ // Full-sweep cadence + auto-confirm cap. Read from the form when present,
+ // otherwise the currently-saved value is preserved: these are 0-sentinel
+ // fields ("never"), so an unrelated save must not quietly reset a
+ // hand-tuned value to the default.
+ fullSweepIntervalMinutes: intOrKeep(
+ "syncSchedulerFullSweepIntervalMinutes",
+ current.syncScheduler.fullSweepIntervalMinutes,
+ ),
+ fullSweepConfirmMaxSuspects: intOrKeep(
+ "syncSchedulerFullSweepConfirmMaxSuspects",
+ current.syncScheduler.fullSweepConfirmMaxSuspects,
+ ),
};
let socialInput: unknown;
diff --git a/editor/e2e/fixtures/test-settings.default.json b/editor/e2e/fixtures/test-settings.default.json
@@ -3,5 +3,8 @@
"maxTranscriptPageBytes": 8388608,
"sleepBetweenDownloadsSeconds": 0,
"minFreeDiskGB": 0,
- "verifyAvailabilityBeforeClean": false
+ "verifyAvailabilityBeforeClean": false,
+ "syncScheduler": {
+ "fullSweepIntervalMinutes": 0
+ }
}
diff --git a/editor/e2e/sync-deep.spec.ts b/editor/e2e/sync-deep.spec.ts
@@ -0,0 +1,249 @@
+import { readFile, writeFile } from "node:fs/promises";
+import { test, expect } from "@playwright/test";
+import {
+ generateReport,
+ readJson,
+ resetData,
+ resolvePath,
+ writeSettings,
+} from "./helpers";
+
+// The sync FULL SWEEP: one full listing enumeration that refreshes the stored
+// `playlist`, flags videos that have left the listing into maybe-missing.json,
+// and (under a cap) resolves those suspects with the per-video availability
+// probe — all inside the ordinary `sync` job.
+//
+// The sweep is OFF in the e2e default settings (fullSweepIntervalMinutes: 0),
+// the same way verifyAvailabilityBeforeClean is, so every pre-existing sync
+// spec keeps exercising the cheap paged walk. Specs that want the sweep opt in,
+// as this one does.
+
+const CHANNEL = "availability-test";
+const FIXTURE = "availability-baseline";
+const channelRoot = `test-transcripts/channels/${CHANNEL}`;
+
+// The baseline fixture's on-disk video dirs (canonical ids).
+const ALL_IDS = [
+ "vidpublic1",
+ "vidpublic2",
+ "viddeleted1",
+ "vidprivate1",
+ "vidmembers1",
+ "vidneedsauth1",
+];
+
+type MaybeMissing = {
+ checkedAt: string;
+ freshPlaylistCount: number;
+ ids: string[];
+};
+
+type ChannelConfigFile = {
+ lastSyncedAt?: string;
+ lastFullSweepAt?: string;
+};
+
+// Controls which ids the fake yt-dlp's flat-playlist branch emits. Its cwd
+// sidecar serves the sweep's single unranged call exactly as it serves the
+// quick check's.
+async function setFreshPlaylist(ids: string[]): Promise<void> {
+ await writeFile(
+ resolvePath(`${channelRoot}/.fake-ytdlp-flat-playlist.json`),
+ JSON.stringify({ ids }),
+ );
+}
+
+// Archive every known video so the sweep's download walk finds nothing new —
+// the download set is not what this spec is about, and an archived first window
+// is also the paged walk's stopping condition.
+async function seedArchive(ids: string[]): Promise<void> {
+ await writeFile(
+ resolvePath(`${channelRoot}/archive`),
+ ids.map((id) => `youtube ${id}`).join("\n") + "\n",
+ );
+}
+
+// A stale stored playlist, standing in for the real complaint: before the
+// sweep, only an explicit "store playlist" ever rewrote this file, so it could
+// be months out of date while "download missing" kept reading it.
+async function seedStalePlaylist(): Promise<void> {
+ await writeFile(
+ resolvePath(`${channelRoot}/playlist`),
+ "https://www.youtube.com/watch?v=stale00000001\n",
+ );
+}
+
+async function enableFullSweep(confirmMaxSuspects: number): Promise<void> {
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ minFreeDiskGB: 0,
+ verifyAvailabilityBeforeClean: false,
+ syncScheduler: {
+ // The cron scheduler stays off; the sweep cadence is independent of it.
+ enabled: false,
+ fullSweepIntervalMinutes: 1440,
+ fullSweepConfirmMaxSuspects: confirmMaxSuspects,
+ },
+ });
+}
+
+async function readConfig(): Promise<ChannelConfigFile> {
+ return readJson<ChannelConfigFile>(`${channelRoot}/config.json`);
+}
+
+// The fake yt-dlp logs each spawn it serves here, so a spec can prove a
+// per-video probe did or didn't happen.
+async function readInvocations(): Promise<string> {
+ try {
+ return await readFile(
+ resolvePath(`${channelRoot}/fake-ytdlp.invocations`),
+ "utf8",
+ );
+ } catch {
+ return "";
+ }
+}
+
+async function readPlaylist(): Promise<string[]> {
+ const raw = await readFile(resolvePath(`${channelRoot}/playlist`), "utf8");
+ return raw.split("\n").filter(Boolean);
+}
+
+// Click Sync and wait for the run to land, detected by config.json's
+// lastSyncedAt advancing past what it was before the click.
+async function runSync(
+ page: import("@playwright/test").Page,
+ previousLastSyncedAt: string | undefined,
+): Promise<void> {
+ await page.goto(`/channels/${CHANNEL}`);
+ // Retried: a click landing before React hydrates fires nothing at all — the
+ // long-standing flake pattern in this suite. Sync is idempotent here.
+ for (let attempt = 0; attempt < 5; attempt++) {
+ await page
+ .getByRole("button", { name: "Sync", exact: true })
+ .click({ timeout: 5_000 })
+ .catch(() => {});
+ for (let i = 0; i < 60; i++) {
+ const cfg = await readConfig().catch(() => null);
+ if (cfg?.lastSyncedAt && cfg.lastSyncedAt !== previousLastSyncedAt) return;
+ await new Promise((r) => setTimeout(r, 250));
+ }
+ }
+ throw new Error("runSync: lastSyncedAt never advanced");
+}
+
+test("a due full sweep refreshes the playlist and flags videos that left the listing", async ({
+ page,
+}) => {
+ await resetData(FIXTURE);
+ await enableFullSweep(25);
+ await seedArchive(ALL_IDS);
+ await seedStalePlaylist();
+ // The listing no longer carries viddeleted1 — it has gone from the channel.
+ await setFreshPlaylist(ALL_IDS.filter((id) => id !== "viddeleted1"));
+
+ await generateReport(page, CHANNEL);
+ await runSync(page, undefined);
+
+ // The sweep's own log line, so a failure says which pass ran.
+ await expect(page.getByLabel("Sync output")).toContainText("Full sweep");
+
+ // 1. The deletion signal, written by the sync itself — no Quick check click.
+ const record = await readJson<MaybeMissing>(
+ `${channelRoot}/maybe-missing.json`,
+ );
+ expect(record.ids).toEqual(["viddeleted1"]);
+ expect(record.freshPlaylistCount).toBe(5);
+
+ // 2. The stale stored playlist is replaced by the fresh listing.
+ const playlist = await readPlaylist();
+ expect(playlist).toHaveLength(5);
+ expect(playlist).not.toContain(
+ "https://www.youtube.com/watch?v=stale00000001",
+ );
+ expect(playlist).toContain("https://www.youtube.com/watch?v=vidpublic1");
+
+ // 3. Under the confirm cap, the same job resolved the suspect upstream.
+ const availability = await readJson<{ availability: string }>(
+ `${channelRoot}/data/viddeleted1/availability.json`,
+ );
+ expect(availability.availability).toBe("deleted");
+
+ // One job, of kind `sync` — not a second job kind bolted alongside it.
+ // Located by data-kind: the cell renders the label "Sync", so matching row
+ // text on "sync" is a case-sensitive miss.
+ await page.goto("/jobs");
+ const rows = page.getByRole("row").filter({ hasText: CHANNEL });
+ await expect(rows).toHaveCount(1);
+ await expect(rows.first()).toHaveAttribute("data-kind", "sync");
+ await expect(rows.first()).toContainText("done");
+});
+
+test("over the confirm cap, suspects are flagged but not probed", async ({
+ page,
+}) => {
+ await resetData(FIXTURE);
+ await enableFullSweep(1);
+ await seedArchive(ALL_IDS);
+ // Two videos gone, one over the cap of 1.
+ await setFreshPlaylist(
+ ALL_IDS.filter((id) => id !== "viddeleted1" && id !== "vidprivate1"),
+ );
+
+ await generateReport(page, CHANNEL);
+ await runSync(page, undefined);
+
+ const record = await readJson<MaybeMissing>(
+ `${channelRoot}/maybe-missing.json`,
+ );
+ expect(record.ids.sort()).toEqual(["viddeleted1", "vidprivate1"].sort());
+
+ // Flagged, but no per-video probe ran — that decision is left to a human.
+ await expect(page.getByLabel("Sync output")).toContainText(
+ "over the auto-confirm cap",
+ );
+ const invocations = await readInvocations();
+ expect(invocations).not.toContain(
+ "dump-json:https://www.youtube.com/watch?v=viddeleted1",
+ );
+ expect(invocations).not.toContain(
+ "dump-json:https://www.youtube.com/watch?v=vidprivate1",
+ );
+ // So the suspects stay unresolved: every sync's availability backfill still
+ // writes a record from metadata, but nothing classified these as gone.
+ const deleted = await readJson<{ availability: string }>(
+ `${channelRoot}/data/viddeleted1/availability.json`,
+ );
+ expect(deleted.availability).not.toBe("deleted");
+});
+
+test("the next sync stays on the cheap paged walk until the cadence elapses", async ({
+ page,
+}) => {
+ await resetData(FIXTURE);
+ await enableFullSweep(25);
+ await seedArchive(ALL_IDS);
+ await setFreshPlaylist(ALL_IDS.filter((id) => id !== "viddeleted1"));
+
+ await generateReport(page, CHANNEL);
+ await runSync(page, undefined);
+
+ const afterSweep = await readConfig();
+ const sweptAt = afterSweep.lastFullSweepAt;
+ expect(typeof sweptAt).toBe("string");
+ const firstRecord = await readJson<MaybeMissing>(
+ `${channelRoot}/maybe-missing.json`,
+ );
+
+ // Second sync, immediately: the daily cadence has not elapsed, so it must not
+ // re-enumerate — no new listing read, no rewritten flags.
+ await runSync(page, afterSweep.lastSyncedAt);
+
+ const secondRecord = await readJson<MaybeMissing>(
+ `${channelRoot}/maybe-missing.json`,
+ );
+ expect(secondRecord.checkedAt).toBe(firstRecord.checkedAt);
+ expect((await readConfig()).lastFullSweepAt).toBe(sweptAt);
+});