commit 660957f7c016d06249ed1037963de685bb4317f6
parent 83c81031e6af96e975ac099aa995feb97ad0821e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 25 Sep 2026 14:17:49 -0400
Merge one-core/r8-state-share — release 8 slice S: one shared auto-queue state per process; manual-sync backoff writes through the live object
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
# Conflicts:
# editor/CHANGELOG.md
Diffstat:
8 files changed, 442 insertions(+), 46 deletions(-)
diff --git a/common/controller/autoRunner.test.ts b/common/controller/autoRunner.test.ts
@@ -11,6 +11,7 @@ import {
pickMetadataScanChannel,
priorityContextFor,
resetPriorityContextForTest,
+ sharedAutoQueueState,
} from "./autoRunner";
import {
defaultChannelPriority,
@@ -20,7 +21,11 @@ import {
type FocusSummary,
} from "../lib/channelPriority";
import { LANES } from "../lib/autoQueueTypes";
-import { emptyAutoQueueState } from "../jobs/autoQueueState";
+import {
+ emptyAutoQueueState,
+ readAutoQueueState,
+ writeAutoQueueState,
+} from "../jobs/autoQueueState";
import type {
AutoQueueGroup,
AutoQueuePolicy,
@@ -543,3 +548,104 @@ test("an injected auto-queue state changes nothing about the answer", async () =
// will hand to the next lane.
assert.deepEqual(state[LANES[0]].runtime.currentWeights, { "prio-all": 17 });
});
+
+// ONE STATE OBJECT PER PROCESS, EVEN AT BOOT (release 8, slice S).
+//
+// Boot starts all four lanes' runLoops without awaiting them, so they all ask
+// `sharedAutoQueueState` inside one read's await gap. When only the resolved
+// value was cached each got its own copy, and on 2026-09-25 a lane persisting
+// its boot-time copy of the WHOLE file erased the download lane's 13:20 backoff
+// (`fails: 13` back to 12) and its video deferral. These pin the cache to the
+// in-flight read: same object, mutations visible before any write, and a write
+// through either caller carries the other's change to disk.
+
+function freshSingletonForTest(): void {
+ globalThis.__yttAutoRunner__ = undefined;
+ globalThis.__yttAutoQueueState__ = undefined;
+}
+
+function stateTempPaths(): Paths {
+ const dir = mkdtempSync(path.join(tmpdir(), "shared-aq-state-"));
+ mkdirSync(path.join(dir, ".auto-queue"), { recursive: true });
+ const file = path.join(dir, ".auto-queue", "state.json");
+ // The boot-time content from the incident: a YouTube cooldown at fails 12.
+ const seeded = emptyAutoQueueState();
+ seeded.download.platformBackoff.youtube = { fails: 12, until: 1_000 };
+ writeFileSync(file, JSON.stringify(seeded));
+ return { autoQueueStateFile: file } as Paths;
+}
+
+test("sharedAutoQueueState: concurrent callers (the boot fan-out) get ONE object", async () => {
+ freshSingletonForTest();
+ const paths = stateTempPaths();
+ const [a, b, c, d] = await Promise.all(
+ LANES.map(() => sharedAutoQueueState(paths)),
+ );
+ assert.equal(a, b);
+ assert.equal(a, c);
+ assert.equal(a, d);
+ assert.equal(a.download.platformBackoff.youtube?.fails, 12);
+ // And a caller after the read resolved gets the same one too.
+ assert.equal(await sharedAutoQueueState(paths), a);
+ freshSingletonForTest();
+});
+
+test("sharedAutoQueueState: a mutation through one caller is visible to the other before any write", async () => {
+ freshSingletonForTest();
+ const paths = stateTempPaths();
+ const pa = sharedAutoQueueState(paths);
+ const pb = sharedAutoQueueState(paths);
+ const a = await pa;
+ a.download.videoDeferrals["_60iFE_FBPQ"] = {
+ until: 99_000,
+ channelSlug: "some-channel",
+ };
+ a.download.platformBackoff.youtube = { fails: 13, until: 2_000 };
+ const b = await pb;
+ assert.equal(b.download.platformBackoff.youtube?.fails, 13);
+ assert.ok(b.download.videoDeferrals["_60iFE_FBPQ"]);
+ freshSingletonForTest();
+});
+
+test("sharedAutoQueueState: another lane's persist carries the download lane's backoff to disk", async () => {
+ freshSingletonForTest();
+ const paths = stateTempPaths();
+ // Obtained concurrently, as boot does.
+ const [downloadLane, otherLane] = await Promise.all([
+ sharedAutoQueueState(paths),
+ sharedAutoQueueState(paths),
+ ]);
+ // 13:20 — the download lane escalates the cooldown and persists.
+ downloadLane.download.platformBackoff.youtube = { fails: 13, until: 2_000 };
+ await writeAutoQueueState(paths, downloadLane);
+ // 13:33:37 — a different lane persists ITS reference to the state.
+ otherLane.digest.runtime.currentWeights = { someLeaf: 1 };
+ await writeAutoQueueState(paths, otherLane);
+ const onDisk = await readAutoQueueState(paths);
+ assert.deepEqual(onDisk.download.platformBackoff.youtube, {
+ fails: 13,
+ until: 2_000,
+ });
+ assert.deepEqual(onDisk.digest.runtime.currentWeights, { someLeaf: 1 });
+ freshSingletonForTest();
+});
+
+test("sharedAutoQueueState: a reset singleton reads afresh; a different state file is its own object", async () => {
+ freshSingletonForTest();
+ const p1 = stateTempPaths();
+ const p2 = stateTempPaths();
+ const first = await sharedAutoQueueState(p1);
+ freshSingletonForTest();
+ const afterReset = await sharedAutoQueueState(p1);
+ assert.notEqual(afterReset, first);
+ // A read for another file in flight at the same time does not hand either
+ // caller the other file's object, and the later file owns the singleton.
+ freshSingletonForTest();
+ const [x, y] = await Promise.all([
+ sharedAutoQueueState(p1),
+ sharedAutoQueueState(p2),
+ ]);
+ assert.notEqual(x, y);
+ assert.equal(await sharedAutoQueueState(p2), y);
+ freshSingletonForTest();
+});
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts
@@ -63,7 +63,9 @@ import { LANES } from "../lib/autoQueueTypes";
import {
type AutoQueueKind,
type AutoQueueState,
+ liveAutoQueueState,
readAutoQueueState,
+ sharedAutoQueueState,
recordPick,
writeAutoQueueState,
} from "../jobs/autoQueueState";
@@ -243,10 +245,6 @@ type RunnerLive = {
type AutoRunnerSingleton = {
runners: Map<AutoQueueKind, RunnerLive>;
- // ONE persisted state object, shared by every lane's runner in this process.
- // See sharedAutoQueueState.
- state: AutoQueueState | null;
- stateFile: string | null;
};
declare global {
@@ -258,39 +256,17 @@ function getSingleton(): AutoRunnerSingleton {
if (!globalThis.__yttAutoRunner__) {
globalThis.__yttAutoRunner__ = {
runners: new Map(),
- state: null,
- stateFile: null,
};
}
return globalThis.__yttAutoRunner__;
}
-// THE PERSISTED STATE IS ONE OBJECT FOR EVERY LANE'S RUNNER, and it has to be.
-//
-// `writeAutoQueueState` serializes the WHOLE file — all four lanes — so two
-// runners each holding their own copy means every persist clobbers the other
-// lane's pick log and fairness memory with whatever that copy was read with.
-// With one slow lane that is nearly invisible (a transcription is minutes
-// apart), and it became obvious the moment two fast lanes ran together: the
-// digest lane's Recent picks panel emptied itself on every backfill dispatch,
-// and `/api/auto-queue/status` — which reads the FILE — showed zero picks for a
-// lane that was demonstrably working. `lane-runner.spec.ts`'s concurrency test
-// is what found it.
-//
-// One object mutated by both is what makes a write of it true for both. Its
-// lifetime is the auto-runner singleton's, which the e2e harness already clears
-// between specs (api/test/invalidate-cache) — so a reset corpus does not
-// inherit a previous spec's picks.
-async function sharedAutoQueueState(paths: Paths): Promise<AutoQueueState> {
- const singleton = getSingleton();
- if (singleton.state && singleton.stateFile === paths.autoQueueStateFile) {
- return singleton.state;
- }
- const state = await readAutoQueueState(paths);
- singleton.state = state;
- singleton.stateFile = paths.autoQueueStateFile;
- return state;
-}
+// THE PERSISTED STATE IS ONE OBJECT FOR EVERY LANE'S RUNNER — and for
+// `recordDownloadBackoff` too. The holder and `sharedAutoQueueState` live in
+// `jobs/autoQueueState.ts` (release 8) because `jobs/downloadBackoff.ts` has to
+// write through the same object and `jobs/` may not import `controller/`.
+// Re-exported here so every existing import keeps working.
+export { sharedAutoQueueState };
export type AutoRunnerStatus = {
kind: AutoQueueKind;
@@ -998,7 +974,12 @@ export async function computeLeafPending(
// runtime.currentWeights in place. This runs on a 3-second status poll, so
// asking the live runtime would let merely HAVING the page open skew a
// round-robin group's rotation. Deep-clone first; the clone is discarded.
- const state = shared?.state ?? (await readAutoQueueState(paths));
+ // The live shared object when a runner holds one (only cloned from, below),
+ // else the file.
+ const state =
+ shared?.state ??
+ (await liveAutoQueueState(paths)) ??
+ (await readAutoQueueState(paths));
const runtime = {
currentWeights: { ...state[kind].runtime.currentWeights },
};
diff --git a/common/jobs/autoQueueState.ts b/common/jobs/autoQueueState.ts
@@ -169,6 +169,129 @@ export async function writeAutoQueueState(
await writeJsonAtomic(paths.autoQueueStateFile, out, { mkdir: true });
}
+// ONE SHARED OBJECT PER PROCESS — the holder every lane's runner and
+// `recordDownloadBackoff` read and write through. Its own global (not the
+// auto-runner singleton, which lives in controller/ where jobs/ may not reach),
+// cleared by the e2e harness beside `__yttAutoRunner__`.
+type AutoQueueStateHolder = {
+ // The ONE persisted state object, shared by every lane's runner.
+ state: AutoQueueState | null;
+ stateFile: string | null;
+ // The READ in flight, so every caller in the same window awaits ONE read and
+ // receives ONE object.
+ loading: Promise<AutoQueueState> | null;
+ loadingFile: string | null;
+};
+
+declare global {
+ // eslint-disable-next-line no-var
+ var __yttAutoQueueState__: AutoQueueStateHolder | undefined;
+}
+
+function getHolder(): AutoQueueStateHolder {
+ if (!globalThis.__yttAutoQueueState__) {
+ globalThis.__yttAutoQueueState__ = {
+ state: null,
+ stateFile: null,
+ loading: null,
+ loadingFile: null,
+ };
+ }
+ return globalThis.__yttAutoQueueState__;
+}
+
+// THE PERSISTED STATE IS ONE OBJECT FOR EVERY LANE'S RUNNER, and it has to be.
+//
+// `writeAutoQueueState` serializes the WHOLE file — all four lanes — so two
+// runners each holding their own copy means every persist clobbers the other
+// lane's pick log and fairness memory with whatever that copy was read with.
+// With one slow lane that is nearly invisible (a transcription is minutes
+// apart), and it became obvious the moment two fast lanes ran together: the
+// digest lane's Recent picks panel emptied itself on every backfill dispatch,
+// and `/api/auto-queue/status` — which reads the FILE — showed zero picks for a
+// lane that was demonstrably working. `lane-runner.spec.ts`'s concurrency test
+// is what found it.
+//
+// One object mutated by both is what makes a write of it true for both. Its
+// lifetime is the holder's, which the e2e harness clears between specs
+// (api/test/invalidate-cache) — so a reset corpus does not inherit a previous
+// spec's picks or backoff.
+//
+// THE CACHE IS THE PROMISE, NOT THE VALUE — and until release 8 it was the
+// value, which made the paragraph above false at exactly the moment it matters.
+// Boot (`startAutoRunnersIfEnabled`) starts all four lanes' `runLoop`s without
+// awaiting them, so all four reached the `await readAutoQueueState` here before
+// any of them had set the cached state; each read its own copy and each
+// runLoop kept that private copy for its lifetime. Live, 2026-09-25: the
+// download lane persisted `fails: 13`, a later `until` and a 6 h video deferral
+// at 13:20; at 13:33:37 another lane persisted its boot-time copy of the whole
+// file (`fails: 12`, deferrals `{}`) and silently undid the pacing on disk.
+// Caching the in-flight read keyed by the state file means every caller in the
+// same window awaits one read and receives the same object. A reset (the e2e
+// harness dropping `__yttAutoQueueState__`) replaces the holder wholesale, so a
+// read still in flight resolves into the OLD holder and cannot leak across.
+//
+// It lives in jobs/ (moved from controller/autoRunner.ts in release 8) because
+// `recordDownloadBackoff` must write THROUGH it: a manual Sync's 429 cooldown
+// written straight to disk was erased by the next persist of any lane, and never
+// merged at all while the download lane was paused, stopped or at capacity.
+export async function sharedAutoQueueState(
+ paths: Paths,
+): Promise<AutoQueueState> {
+ const holder = getHolder();
+ const file = paths.autoQueueStateFile;
+ if (holder.state && holder.stateFile === file) {
+ return holder.state;
+ }
+ if (holder.loading && holder.loadingFile === file) {
+ return holder.loading;
+ }
+ const loading = readAutoQueueState(paths).then(
+ (state) => {
+ // A later call for a DIFFERENT file superseded this read; it owns the
+ // holder now, and this read's object goes only to its own awaiters.
+ if (holder.loading === loading) {
+ holder.state = state;
+ holder.stateFile = file;
+ holder.loading = null;
+ holder.loadingFile = null;
+ }
+ return state;
+ },
+ (err: unknown) => {
+ // Never cache a failure: the next caller retries the read.
+ if (holder.loading === loading) {
+ holder.loading = null;
+ holder.loadingFile = null;
+ }
+ throw err;
+ },
+ );
+ holder.loading = loading;
+ holder.loadingFile = file;
+ return loading;
+}
+
+// The shared object if something in this process already holds (or is
+// loading) it for this file, else null — WITHOUT starting a read. For writers
+// outside the runner (`recordDownloadBackoff`): with no runner live there is no
+// in-memory copy to keep in step, and the plain disk read-modify-write is right.
+export async function liveAutoQueueState(
+ paths: Paths,
+): Promise<AutoQueueState | null> {
+ const holder = getHolder();
+ const file = paths.autoQueueStateFile;
+ if (holder.state && holder.stateFile === file) return holder.state;
+ if (holder.loading && holder.loadingFile === file) {
+ try {
+ return await holder.loading;
+ } catch {
+ return null;
+ }
+ }
+ return null;
+}
+
export function recordPick(state: AutoQueueKindState, pick: AutoQueuePick): void {
state.picks.unshift(pick);
if (state.picks.length > AUTO_QUEUE_PICK_LOG_LIMIT) {
diff --git a/common/jobs/downloadBackoff.test.ts b/common/jobs/downloadBackoff.test.ts
@@ -9,6 +9,12 @@ import {
recordDownloadBackoff,
} from "./downloadBackoff";
import { BACKOFF_BASE_MS } from "./platformBackoff";
+import {
+ emptyAutoQueueState,
+ readAutoQueueState,
+ sharedAutoQueueState,
+ writeAutoQueueState,
+} from "./autoQueueState";
// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/jobs/downloadBackoff.test.ts
//
@@ -22,9 +28,12 @@ async function withTempPaths(
const paths = {
autoQueueStateFile: path.join(dir, ".auto-queue", "state.json"),
} as Paths;
+ // No runner is live unless a test makes one (sharedAutoQueueState).
+ globalThis.__yttAutoQueueState__ = undefined;
try {
await fn(paths);
} finally {
+ globalThis.__yttAutoQueueState__ = undefined;
await rm(dir, { recursive: true, force: true });
}
}
@@ -75,3 +84,70 @@ test("remaining is 0 once the window lapses", async () => {
assert.equal(await platformCooldownRemainingMs("odysee", paths), 0);
});
});
+
+// WITH A RUNNER LIVE, THE COOLDOWN GOES THROUGH THE SHARED OBJECT (release 8).
+// Before, recordDownloadBackoff wrote the file only; the runners' in-memory
+// object never saw it, so the next persist of ANY lane wrote the object over
+// the file and the manual 429 cooldown was gone — and nothing merged it back
+// while the download lane was paused, stopped or at capacity.
+
+test("with a runner live, another lane's persist keeps a manual cooldown on disk", async () => {
+ await withTempPaths(async (paths) => {
+ // A runner holds the shared object (as every lane does from boot on).
+ const shared = await sharedAutoQueueState(paths);
+ // A clicked Sync hits a 429.
+ await recordDownloadBackoff("youtube", paths);
+ assert.ok(
+ shared.download.platformBackoff.youtube,
+ "the live object carries the cooldown",
+ );
+ // The transcription lane picks something and persists the shared object.
+ shared.transcription.runtime.currentWeights = { someLeaf: 1 };
+ await writeAutoQueueState(paths, shared);
+ const onDisk = await readAutoQueueState(paths);
+ assert.equal(onDisk.download.platformBackoff.youtube?.fails, 1);
+ assert.deepEqual(onDisk.transcription.runtime.currentWeights, {
+ someLeaf: 1,
+ });
+ assert.ok((await platformCooldownRemainingMs("youtube", paths)) > 0);
+ });
+});
+
+test("with a runner live, a cooldown already on disk is merged in, not dropped", async () => {
+ await withTempPaths(async (paths) => {
+ const shared = await sharedAutoQueueState(paths);
+ // Written to the file behind the runner's back (a later `until` than the
+ // live object knows about).
+ const behind = emptyAutoQueueState();
+ const until = Date.now() + 10 * 60_000;
+ behind.download.platformBackoff.odysee = { until, fails: 4 };
+ await writeAutoQueueState(paths, behind);
+ await recordDownloadBackoff("youtube", paths);
+ assert.deepEqual(shared.download.platformBackoff.odysee, { until, fails: 4 });
+ const onDisk = await readAutoQueueState(paths);
+ assert.deepEqual(onDisk.download.platformBackoff.odysee, { until, fails: 4 });
+ assert.ok(onDisk.download.platformBackoff.youtube);
+ });
+});
+
+test("platformCooldownRemainingMs prefers the live object over the file", async () => {
+ await withTempPaths(async (paths) => {
+ const shared = await sharedAutoQueueState(paths);
+ shared.download.platformBackoff.rumble = {
+ until: Date.now() + 60_000,
+ fails: 1,
+ };
+ // Not persisted yet — the file has no entry.
+ assert.ok((await platformCooldownRemainingMs("rumble", paths)) > 0);
+ });
+});
+
+test("with no runner live, recordDownloadBackoff stays a disk round trip and loads nothing", async () => {
+ await withTempPaths(async (paths) => {
+ await recordDownloadBackoff("youtube", paths);
+ assert.equal(globalThis.__yttAutoQueueState__?.state ?? null, null);
+ assert.equal(globalThis.__yttAutoQueueState__?.loading ?? null, null);
+ const onDisk = await readAutoQueueState(paths);
+ assert.equal(onDisk.download.platformBackoff.youtube?.fails, 1);
+ });
+});
diff --git a/common/jobs/downloadBackoff.ts b/common/jobs/downloadBackoff.ts
@@ -1,38 +1,67 @@
// Shared accessors for the per-platform download rate-limit cooldown that both
// the auto-download runner and manual sync/download paths respect.
//
-// The runner already owns this state in memory and persists it to
+// The runner owns this state in memory — ONE object shared by every lane's
+// runner (`sharedAutoQueueState`) — and persists it WHOLE to
// `.auto-queue/state.json` (the "download" kind's platformBackoff). These
-// helpers let code OUTSIDE the runner — a clicked Sync, an import — observe and
-// extend the same cooldown via a fresh read-modify-write, so a 429 hit by
-// either path pauses the other. Because auto-download units now serialize on the
-// same per-platform job queue as sync (see autoRunner.ts / registry.ts), a
-// momentarily racy write here can at most delay politeness by an iteration — it
-// can no longer cause two concurrent yt-dlp processes to hammer one source.
+// helpers let code OUTSIDE the runner — a clicked Sync, a metadata scan, the
+// video page — observe and extend the same cooldown, so a 429 hit by either path
+// pauses the other.
+//
+// WRITE THROUGH THE LIVE OBJECT WHEN THERE IS ONE (release 8). A disk-only
+// read-modify-write while a runner is live is erased by the next persist of ANY
+// lane — its in-memory object never saw the cooldown — and the download lane's
+// own merge-from-disk never runs while that lane is paused, stopped or at
+// capacity, which is exactly when an operator is riding out a rate limit. With
+// no runner live there is no in-memory copy, and the disk round trip is right.
+// Per-platform one-download-at-a-time is enforced by the job queue (see
+// autoRunner.ts / registry.ts), so a lost cooldown could only ever cost
+// politeness, never concurrency.
import { getPaths, type Paths } from "../lib/paths";
-import { readAutoQueueState, writeAutoQueueState } from "./autoQueueState";
+import {
+ type AutoQueueState,
+ liveAutoQueueState,
+ readAutoQueueState,
+ writeAutoQueueState,
+} from "./autoQueueState";
import { nextBackoff, pruneExpired } from "./platformBackoff";
// Milliseconds remaining in the platform's current cooldown window, or 0 if it
-// is not cooling down. Reads the persisted "download" backoff state fresh.
+// is not cooling down. Prefers the live shared object; else reads the file.
export async function platformCooldownRemainingMs(
platform: string,
paths: Paths = getPaths(),
): Promise<number> {
- const state = await readAutoQueueState(paths);
+ const state =
+ (await liveAutoQueueState(paths)) ?? (await readAutoQueueState(paths));
const entry = state.download.platformBackoff[platform];
const now = Date.now();
return entry && entry.until > now ? entry.until - now : 0;
}
// Record a rate-limit/network failure against a platform, escalating its
-// exponential cooldown. Read-modify-write of the shared download backoff state.
+// exponential cooldown — through the live shared object when a runner holds
+// one, else as a read-modify-write of the file.
export async function recordDownloadBackoff(
platform: string,
paths: Paths = getPaths(),
): Promise<void> {
- const state = await readAutoQueueState(paths);
+ const live = await liveAutoQueueState(paths);
+ let state: AutoQueueState;
+ if (live) {
+ // Fold in whatever is on disk first (the later `until` wins, as the
+ // runner's own merge does), so a cooldown written while no runner held the
+ // object is escalated from, not forgotten.
+ const onDisk = (await readAutoQueueState(paths)).download.platformBackoff;
+ for (const [pf, e] of Object.entries(onDisk)) {
+ const cur = live.download.platformBackoff[pf];
+ if (!cur || e.until > cur.until) live.download.platformBackoff[pf] = e;
+ }
+ state = live;
+ } else {
+ state = await readAutoQueueState(paths);
+ }
const now = Date.now();
pruneExpired(state.download.platformBackoff, now);
state.download.platformBackoff[platform] = nextBackoff(
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -3,6 +3,7 @@
## [Unreleased]
- **A channel's video list shows titles, and you can search by them.** On `/channels/<slug>/videos`, each row now shows the video's title, with its id in smaller type underneath. Search matches the title or the id, ignoring case. The title comes from the transcript index for transcribed videos, from the channel's metadata scan for videos that were listed but never downloaded, and otherwise from the video's `metadata.info.json`. A video none of these name shows its id, as before. Each row's selection checkbox and link are named by the id as before, and the order is unchanged (by id).
- **An undownloaded video's page shows its title and details.** The video page used to show a bare id for any video without a `metadata.info.json`. If the channel's metadata scan has read the video, the page now shows its title, upload date and duration from the scan, marked *from the listing scan — not downloaded*. Any video page with a description, from either source, has a **Description** section, collapsed by default.
+- **A rate-limit cooldown the download runner recorded is no longer undone by another lane a few minutes later.** The four automatic runners (download, transcription, digest, backfill) are meant to share one copy of `.auto-queue/state.json`, because every save writes the whole file. At startup all four read it at the same moment, and each kept its own copy. After that, whichever runner saved next wrote its startup copy over the others' progress. On 2026-09-25 that silently put a YouTube cooldown back from 13 failures to 12 and dropped a six-hour hold on one rate-limited video, 13 minutes after the download runner had saved them. The runners now wait on one read at startup and share one copy, so a save by any lane keeps every lane's latest state. A cooldown that a manual Sync, a metadata scan or the video page records is no longer erased by a runner's save either. It used to go to the file only, so the next save of any lane wrote over it, and nothing brought it back while the download runner was paused or stopped. It is now written into the copy the runners share. Nothing changes in the file format.
- **New ops action: `pnpm ops keep-videos` marks every video of a channel whose title or description matches a pattern as "do not clean".** It sets the same marker as the video page's *Do not clean* toggle, so the clean sweep, extra-format cleanup, wrong-format removal, the superseded-subs purge and saved-video eviction all leave those videos alone. The body is `{"slug", "match", "fields"?, "note"?, "dryRun"?}`. `match` is matched the way a channel's download filter *include* is: a case-insensitive regex over title + description. `fields: ["title"]` or `["description"]` narrows it to one half, and `dryRun: true` reports without writing. Videos that already carry the marker are counted and left as they are. The marker lives in the video's folder, so a match that was never downloaded is listed under `notDownloaded` and no folder is created for it. Run `download-missing` on those ids, then run `keep-videos` again. On a new channel, run `metadata-scan` first: a video with no scanned title cannot match, and the reply counts those as `unscanned`.
- **Auto-download no longer retries the same rate-limited video over and over; it moves on to the next one.** When a download answered HTTP 429, the runner paused the whole platform for a while and then picked the same video again, because it was still first in the queue. Each retry doubled the pause, up to 30 minutes. On 2026-09-24 one YouTube Short was retried 12 times this way and kept YouTube paused all evening. A YouTube 429 comes from the subtitle fetch for one video, not from the whole site. Now a rate-limited video is also **deferred for 6 hours**: auto-download skips it, so when the pause ends the runner takes the next video. The pause still grows only when *different* videos keep hitting the limit. Deferrals are kept in `.auto-queue/state.json` beside the platform cooldowns, so a restart does not retry the video early. The log line reads `… (attempt 1). <id> deferred 6h; next video after cooldown.` A manual Sync or *download missing* ignores deferrals and still fetches the video. When every video left is deferred, the runner reports that it is idle for that reason: "every pending video was rate-limited recently and is deferred".
- **The cooldown strip on `/operations/download` also lists deferred videos.** It is now a region named *Rate-limit cooldown*, with a *Platforms in cooldown* list (unchanged) and a *Deferred videos* list. Each deferred video links to its page and shows how long it has left (`alpha/a1 — 5h 59m left`). The strip appears when either list has something in it. Times over an hour now read `5h 59m` instead of `359m 58s`.
diff --git a/editor/app/api/test/invalidate-cache/route.ts b/editor/app/api/test/invalidate-cache/route.ts
@@ -91,6 +91,12 @@ function invalidate() {
// next iteration; clearing this lets the next spec start fresh runners.
// eslint-disable-next-line @typescript-eslint/no-explicit-any
(globalThis as any).__yttAutoRunner__ = undefined;
+ // And the ONE shared auto-queue state object the runners (and
+ // recordDownloadBackoff) read and write through — it lives on its own global
+ // in jobs/autoQueueState.ts. Kept, a reset corpus would inherit the previous
+ // spec's picks, fairness weights and platform backoff.
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
+ (globalThis as any).__yttAutoQueueState__ = undefined;
// Drop the shared snapshot parse memo. It is keyed on (mtime, size), so it is
// self-invalidating in production — but resetData() rewrites the SAME fixture
// paths from the same source tree, and two specs writing an identical
diff --git a/plans/release-8.md b/plans/release-8.md
@@ -105,4 +105,78 @@ accepted.
- **Commit trailers** name `Claude Opus 5.5 (1M context)`, as the release-6 and release-7
implementers did.
+### Slice S, as shipped — one shared auto-queue state per process (2026-09-25)
+
+Branch `one-core/r8-state-share` off `main` `eff6a3a7`. **The bug, live on 2026-09-25.**
+`sharedAutoQueueState` (`common/controller/autoRunner.ts`) is meant to give ONE `AutoQueueState`
+object to all four lane runners, because `writeAutoQueueState` serializes the WHOLE
+`.auto-queue/state.json` (all four lanes). It cached only the RESOLVED value. At boot,
+`startAutoRunnersIfEnabled` awaits only each lane's *registration*: `runManagedFunction` fires
+`runLoop` without awaiting it. So all four runLoops reached `await readAutoQueueState` before any of
+them set `singleton.state`. Each got its own copy and kept it for its lifetime, since `state` is
+captured once per runLoop and every `persist()` reuses it.
+
+The live sequence ran as follows:
+- **13:10–13:11** boot. Four private copies were made, all reading `download.platformBackoff.youtube = {fails: 12, until: 13:20}` and `videoDeferrals: {}`.
+- **13:20** the download lane persisted `fails: 13`, `until: 13:52` and a 6 h deferral for `_60iFE_FBPQ`.
+- **13:29** the operator renamed `paramount-tactical-videos` to `paramount-tactical`. The priority recompile re-shaped the other lanes' trees. The rename does not touch this file.
+- **13:33:37** another lane picked work and persisted its boot-time copy, which set the file back to `fails: 12` and deferrals `{}`. The pacing was silently undone on disk.
+
+The rename is only what made another lane persist at that moment. The race is what made the write wrong.
+
+**The fix** caches the in-flight read, keyed by the state file. Every caller in the same window
+awaits one `readAutoQueueState` and gets the same object. A failed read is never cached, so the next
+caller retries. A read that a call for a different file supersedes resolves only to its own awaiters.
+The e2e harness reset (`api/test/invalidate-cache` sets `__yttAutoRunner__ = undefined`) replaces the
+singleton wholesale, so a read still in flight lands in the old singleton and cannot leak into the
+next spec. The big comment above the function is kept, and a paragraph with this incident is added.
+`sharedAutoQueueState` is now exported for the tests. There is no restructuring, no file-format
+change and no change to `writeAutoQueueState`.
+
+**The same bug in the other direction, fixed in review.** `recordDownloadBackoff`
+(`common/jobs/downloadBackoff.ts:31-43`) records a 429 hit by a manual Sync, a metadata scan or the
+video page. It was a disk-only read-modify-write while every runner held the shared object in
+memory, so the next save by any lane erased it. The download lane's merge-from-disk at the top of
+each iteration did not help while that lane was paused, stopped or at capacity, which is when an
+operator is riding out a rate limit. The holder and `sharedAutoQueueState` moved to
+`jobs/autoQueueState.ts` on their own global, `__yttAutoQueueState__`. `jobs/` may not import
+`controller/`, and `controller/` importing `jobs/` is allowed, so `architecture.test.ts` is unchanged.
+`autoRunner.ts` re-exports `sharedAutoQueueState`. A new `liveAutoQueueState(paths)` returns the live
+object, or the in-flight one, without starting a read. `recordDownloadBackoff` uses it: when a runner
+is live, it merges the disk backoff in (the later `until` wins, as the runner's merge does), escalates
+on the live object and writes through it. With no runner live it keeps the disk round trip.
+`platformCooldownRemainingMs` and the status builder's `computeLeafPending`, which only clones, prefer
+the live object. The e2e reset (`api/test/invalidate-cache`) clears the new global beside
+`__yttAutoRunner__`.
+
+| sha | what |
+|---|---|
+| `135dbebd` | `autoRunner.ts`: `AutoRunnerSingleton` gains `loading` / `loadingFile`, and `sharedAutoQueueState` caches the promise (exported, with the incident paragraph). `autoRunner.test.ts` has 4 cases: (a) four concurrent calls (the boot fan-out) resolve to the same object, and a later call gets it too; (b) a mutation through one caller (backoff `fails: 13` plus a video deferral) is visible to the other before any write; (c) the write seam: the download lane escalates the backoff and writes, then another lane mutates the digest runtime and writes *its* reference, and the file on disk carries both; (d) a reset singleton reads afresh, and two files in flight at once stay two objects. Temp dirs, no module mocks. **Cases (a)–(c) FAIL against the pre-fix behaviour** (checked by disabling the in-flight branch: 3 failed) and pass with it |
+| `755e137b` | the `[Unreleased]` bullet in `editor/CHANGELOG.md` (this record is pasted into `plans/release-8.md` by the parent at merge) |
+| `4cfc7491` | (review fix) The holder and `sharedAutoQueueState` move to `jobs/autoQueueState.ts` (`__yttAutoQueueState__`, plus `liveAutoQueueState`), and `autoRunner.ts` re-exports them. `recordDownloadBackoff` merges and writes through the live object, and `platformCooldownRemainingMs` and `computeLeafPending` prefer it. `invalidate-cache` clears the new global. `downloadBackoff.test.ts` gains 4 cases: with a runner live, another lane's persist keeps a manual cooldown on disk; a cooldown already on disk is merged in, not dropped; the remaining time prefers the live object; with no runner live, the call stays a disk round trip and starts no read. **The first three FAIL on the pre-fix `downloadBackoff.ts`** (checked: 3 failed) |
+| `89a7d6ab` | the CHANGELOG clause for the cooldown fix |
+
+**Gates.** First run, on `135dbebd`: tsc clean; common **1799/1799** = 1795 + 4; editor unit
+**72/72**; `pnpm --filter editor exec next build` ok. EDITOR e2e, `auto-queue.spec.ts
+lane-runner.spec.ts pacing.spec.ts queues.spec.ts` (`s-e2e1.log`): **38 passed, 0 failed, 4.0 min**,
+no queue wait.
+
+Re-gate after the review fix, on `4cfc7491`: tsc clean; common **1803/1803** = 1799 + 4; editor unit
+**72/72**. The first `next build` attempt failed on a Google Fonts fetch (`next/font/google` module
+not found). The retry was ok, with no code change. EDITOR e2e, the same four specs plus
+`rumble-sweep.spec.ts`, which seeds `platformBackoff` (`s-e2e2.log`): **39 passed, 0 failed, 4.0 min**, no queue wait.
+
+The export build, test:scripts and mcp were not run, because the slice touches no file under
+`export/`, `scripts/` or `mcp/`. Heavy steps started only at ≥ 3 GB available memory.
+
+**Numbers: none.** No `settings.json`, `site.json` or `config.json` key changed, and the state file's
+format is unchanged.
+
+**Found and left.**
+- **The status builder's disk read is kept as a fallback.** `computeLeafPending` reads
+ `shared?.state`, then the live object, then the file. It only clones from it, so it is not a writer.
+- **An orphaned runLoop after an e2e reset (pre-existing, e2e only).** A spec-1 runLoop can still
+ persist its old object to the same fixture file until it notices its job is gone. That is a
+ cross-spec clobber, not a two-lanes-in-one-process problem, and this slice did not introduce it.
+
## Rollout