commit 120d890843d673f3335bb4db3ca39b81b60bead8
parent 5730940be7013765f94d1cac1f4d38c947b3311f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 3 Jul 2026 00:49:00 -0400
Jobs page: ULID ids, pagination, and log retention
Replace the dead "Clear archived logs" button and bound the /jobs load.
The old button only deleted logs absent from the in-memory registry, which
(since the registry keeps the 100 newest finished jobs and sidecars preserve
their real status) was almost never anything — so it did nothing. And
listAllJobs stat'd + read the sidecar for EVERY .jobs file on every load, with
no pagination.
- ULID job ids (common/jobs/ulid.ts): lexicographically time-sortable and
time-decodable. jobIdTime() decodes both ULID and the legacy <t36>-<rand>
scheme, so existing on-disk logs still sort/read correctly.
- listAllJobs paginates: returns { entries, hasMore, total }, sorts/pages by
jobIdTime, and only stats/reads the sidecar for the shown page. New
getJobEntry() resolves a single job for the detail page.
- pruneJobLogs({ keepLast, olderThanMs, all }) replaces clearArchivedLogs;
never deletes running/queued jobs. maybePruneJobLogs auto-trims on job finish
(throttled; keep 500, drop >30d).
- UI: ClearLogsMenu dropdown (7/30/90 days, all) replaces ClearArchivedButton;
page.tsx grows ?limit= via a Load more link.
- Tests: ulid.test.ts + listJobs.test.ts (11 cases); jobs.spec.ts updated for
the new menu. Jobs e2e suite green.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Diffstat:
13 files changed, 598 insertions(+), 162 deletions(-)
diff --git a/common/jobs/listJobs.test.ts b/common/jobs/listJobs.test.ts
@@ -0,0 +1,141 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, mkdir, writeFile, readdir, rm } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import { ulid } from "./ulid";
+import {
+ listAllJobs,
+ getJobEntry,
+ pruneJobLogs,
+} from "./listJobs";
+
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/jobs/listJobs.test.ts
+//
+// listAllJobs/pruneJobLogs only touch paths.jobsDir and the (empty in a fresh
+// process) global registry, so a stub Paths with just jobsDir + seeded log/meta
+// files exercises the full disk path. Ids are ULIDs stamped at a known time so
+// jobIdTime() orders them deterministically.
+
+const BASE = 1_700_000_000_000;
+
+async function withJobs(
+ fn: (paths: Paths, seed: (i: number, t: number) => Promise<string>) => Promise<void>,
+): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "list-jobs-"));
+ const jobsDir = path.join(dir, ".jobs");
+ await mkdir(jobsDir, { recursive: true });
+ const paths = { jobsDir } as Paths;
+ const seed = async (i: number, t: number): Promise<string> => {
+ const id = ulid(t);
+ await writeFile(path.join(jobsDir, `${id}.log`), `log ${i}\n`);
+ await writeFile(
+ path.join(jobsDir, `${id}.meta.json`),
+ JSON.stringify({
+ id,
+ kind: "sync",
+ queueKey: "q",
+ status: "done",
+ queuedAt: t,
+ startedAt: t,
+ endedAt: t + 500,
+ exitCode: 0,
+ }),
+ );
+ return id;
+ };
+ try {
+ await fn(paths, seed);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+async function logCount(paths: Paths): Promise<number> {
+ return (await readdir(paths.jobsDir)).filter((n) => n.endsWith(".log")).length;
+}
+
+test("listAllJobs paginates newest-first and reports hasMore/total", async () => {
+ await withJobs(async (paths, seed) => {
+ const ids: string[] = [];
+ for (let i = 0; i < 5; i++) ids.push(await seed(i, BASE + i * 1000));
+
+ const page1 = await listAllJobs(paths, { limit: 2 });
+ assert.equal(page1.entries.length, 2);
+ assert.equal(page1.hasMore, true);
+ assert.equal(page1.total, 5);
+ // Newest first: the last two seeded ids, in reverse order.
+ assert.deepEqual(
+ page1.entries.map((e) => e.id),
+ [ids[4], ids[3]],
+ );
+
+ const all = await listAllJobs(paths, { limit: 5 });
+ assert.equal(all.hasMore, false);
+ assert.equal(all.entries.length, 5);
+ });
+});
+
+test("listAllJobs before-cursor skips newer entries", async () => {
+ await withJobs(async (paths, seed) => {
+ const ids: string[] = [];
+ for (let i = 0; i < 4; i++) ids.push(await seed(i, BASE + i * 1000));
+ // Everything strictly older than the 3rd job's time.
+ const page = await listAllJobs(paths, { limit: 10, before: BASE + 2000 });
+ assert.deepEqual(
+ page.entries.map((e) => e.id),
+ [ids[1], ids[0]],
+ );
+ });
+});
+
+test("getJobEntry resolves one job and null for unknown ids", async () => {
+ await withJobs(async (paths, seed) => {
+ const id = await seed(0, BASE);
+ const entry = await getJobEntry(paths, id);
+ assert.ok(entry);
+ assert.equal(entry.id, id);
+ assert.equal(entry.kind, "sync");
+ assert.equal(await getJobEntry(paths, "does-not-exist"), null);
+ });
+});
+
+test("pruneJobLogs keepLast drops the oldest tail", async () => {
+ await withJobs(async (paths, seed) => {
+ for (let i = 0; i < 5; i++) await seed(i, BASE + i * 1000);
+ const { deleted } = await pruneJobLogs(paths, { keepLast: 2 });
+ assert.equal(deleted, 3);
+ assert.equal(await logCount(paths), 2);
+ // Meta sidecars go too.
+ assert.equal(
+ (await readdir(paths.jobsDir)).filter((n) => n.endsWith(".meta.json")).length,
+ 2,
+ );
+ });
+});
+
+test("pruneJobLogs olderThanMs drops only aged entries", async () => {
+ await withJobs(async (paths, seed) => {
+ await seed(0, BASE); // old
+ await seed(1, BASE + 1000); // old
+ const fresh = await seed(2, BASE + 10 * 60 * 60 * 1000); // 10h newer
+ const now = BASE + 11 * 60 * 60 * 1000;
+ const { deleted } = await pruneJobLogs(paths, {
+ olderThanMs: 60 * 60 * 1000, // 1h
+ now,
+ });
+ assert.equal(deleted, 2);
+ const remaining = await readdir(paths.jobsDir);
+ assert.ok(remaining.includes(`${fresh}.log`));
+ });
+});
+
+test("pruneJobLogs all clears everything", async () => {
+ await withJobs(async (paths, seed) => {
+ for (let i = 0; i < 3; i++) await seed(i, BASE + i * 1000);
+ const { deleted } = await pruneJobLogs(paths, { all: true });
+ assert.equal(deleted, 3);
+ assert.equal(await logCount(paths), 0);
+ });
+});
diff --git a/common/jobs/listJobs.ts b/common/jobs/listJobs.ts
@@ -3,6 +3,7 @@ import { readdir, stat, readFile, rm } from "node:fs/promises";
import type { Paths } from "../lib/paths";
import { getRegistry, type JobRecord, type JobStatus } from "./registry";
import { metaPath, readJobMeta } from "./jobMeta";
+import { jobIdTime } from "./ulid";
export type JobListEntry = {
id: string;
@@ -23,125 +24,189 @@ export type JobListEntry = {
logSize: number;
};
+export type JobsPage = {
+ entries: JobListEntry[];
+ // More jobs exist beyond this page (i.e. raise `limit` to see them).
+ hasMore: boolean;
+ // Total jobs known on disk + in the registry, independent of the page size.
+ total: number;
+};
+
const TERMINAL_STATUSES: ReadonlySet<JobStatus> = new Set([
"done",
"failed",
"cancelled",
]);
-// Merge in-memory registry entries with any leftover .log files from
-// previous server lifetimes. Anything not in the registry is treated as
-// "archived" — we only know that it once ran, not whether it succeeded.
-export async function listAllJobs(paths: Paths): Promise<JobListEntry[]> {
- const registry = getRegistry();
- const live: Map<string, JobRecord> = new Map();
- for (const r of registry.list()) live.set(r.id, r);
+// Default page size for the /jobs list. The page grows this via ?limit=.
+export const DEFAULT_JOBS_LIMIT = 50;
+
+// Automatic on-finish retention (see maybePruneJobLogs): keep the newest N jobs
+// and drop anything older than the max age, whichever bites first. Generous —
+// the point is to bound the directory (including the constant refresh-report
+// churn), not to be aggressive.
+const RETENTION_KEEP_LAST = 500;
+const RETENTION_MAX_AGE_MS = 30 * 24 * 60 * 60 * 1000; // 30 days
+// The list-page is `force-dynamic`, so job finishes could fire a readdir on
+// every finalize. Throttle so a burst of jobs prunes at most once per window.
+const PRUNE_THROTTLE_MS = 60_000;
- let files: string[] = [];
+async function readLogIds(paths: Paths): Promise<Set<string>> {
try {
- files = (await readdir(paths.jobsDir)).filter((n) => n.endsWith(".log"));
+ const files = await readdir(paths.jobsDir);
+ return new Set(
+ files.filter((n) => n.endsWith(".log")).map((n) => n.replace(/\.log$/, "")),
+ );
} catch {
- files = [];
+ return new Set();
}
+}
- const out: JobListEntry[] = [];
- const seen = new Set<string>();
-
- for (const file of files) {
- const id = file.replace(/\.log$/, "");
- seen.add(id);
- const logPath = path.join(paths.jobsDir, file);
- const live_ = live.get(id);
- let logSize = 0;
- let mtime = 0;
+// Build one JobListEntry, preferring the live registry record and falling back
+// to the on-disk `.meta.json` sidecar (or the log mtime for old logs with no
+// sidecar). `hasLog` says whether an `<id>.log` file exists on disk.
+async function buildEntry(
+ paths: Paths,
+ id: string,
+ live_: JobRecord | undefined,
+ hasLog: boolean,
+): Promise<JobListEntry> {
+ const logPath = hasLog
+ ? path.join(paths.jobsDir, `${id}.log`)
+ : live_?.logPath ?? path.join(paths.jobsDir, `${id}.log`);
+ let logSize = 0;
+ let mtime = 0;
+ if (hasLog) {
try {
const s = await stat(logPath);
logSize = s.size;
mtime = s.mtimeMs;
} catch {
- continue;
- }
- if (live_) {
- out.push({
- id,
- kind: live_.kind,
- channelSlug: live_.channelSlug,
- videoId: live_.videoId,
- queueKey: live_.queueKey,
- status: live_.status,
- queuedAt: live_.queuedAt,
- startedAt: live_.startedAt,
- endedAt: live_.endedAt,
- exitCode: live_.exitCode,
- inRegistry: true,
- bookmarkable: Boolean(live_.spec),
- logPath,
- logSize,
- });
- } else {
- // Not in the registry: recover what we can from the sidecar so an evicted
- // or post-restart job still shows its kind/channel/status/duration. With
- // no sidecar (old logs), fall back to the mtime-based minimal entry. A
- // non-terminal sidecar status means the job isn't actually live (it's not
- // in the registry), so report it as "archived".
- const meta = await readJobMeta(paths, id);
- if (meta) {
- out.push({
- id,
- kind: meta.kind,
- channelSlug: meta.channelSlug,
- videoId: meta.videoId,
- queueKey: meta.queueKey,
- status: TERMINAL_STATUSES.has(meta.status) ? meta.status : "archived",
- queuedAt: meta.queuedAt,
- startedAt: meta.startedAt,
- endedAt: meta.endedAt,
- exitCode: meta.exitCode,
- inRegistry: false,
- bookmarkable: Boolean(meta.spec),
- logPath,
- logSize,
- });
- } else {
- out.push({
- id,
- status: "archived",
- queuedAt: mtime,
- startedAt: mtime,
- endedAt: mtime,
- inRegistry: false,
- bookmarkable: false,
- logPath,
- logSize,
- });
- }
+ /* log vanished between listing and stat; treat as size 0 */
}
}
- // Registry entries with no log file (shouldn't happen normally, but be
- // defensive).
+ if (live_) {
+ return {
+ id,
+ kind: live_.kind,
+ channelSlug: live_.channelSlug,
+ videoId: live_.videoId,
+ queueKey: live_.queueKey,
+ status: live_.status,
+ queuedAt: live_.queuedAt,
+ startedAt: live_.startedAt,
+ endedAt: live_.endedAt,
+ exitCode: live_.exitCode,
+ inRegistry: true,
+ bookmarkable: Boolean(live_.spec),
+ logPath,
+ logSize,
+ };
+ }
+
+ // Not in the registry: recover what we can from the sidecar so an evicted or
+ // post-restart job still shows its kind/channel/status/duration. With no
+ // sidecar (old logs), fall back to the mtime-based minimal entry. A
+ // non-terminal sidecar status means the job isn't actually live (it's not in
+ // the registry), so report it as "archived".
+ const meta = await readJobMeta(paths, id);
+ if (meta) {
+ return {
+ id,
+ kind: meta.kind,
+ channelSlug: meta.channelSlug,
+ videoId: meta.videoId,
+ queueKey: meta.queueKey,
+ status: TERMINAL_STATUSES.has(meta.status) ? meta.status : "archived",
+ queuedAt: meta.queuedAt,
+ startedAt: meta.startedAt,
+ endedAt: meta.endedAt,
+ exitCode: meta.exitCode,
+ inRegistry: false,
+ bookmarkable: Boolean(meta.spec),
+ logPath,
+ logSize,
+ };
+ }
+ return {
+ id,
+ status: "archived",
+ queuedAt: mtime,
+ startedAt: mtime,
+ endedAt: mtime,
+ inRegistry: false,
+ bookmarkable: false,
+ logPath,
+ logSize,
+ };
+}
+
+// A page of the jobs list, newest first. Merges the in-memory registry with the
+// leftover `.log` files from previous server lifetimes. Ordering and paging use
+// jobIdTime(id) — decoded from the id itself (ULID, or the legacy `<t36>-<rand>`
+// scheme) — so only the returned page is stat-ed / sidecar-read, NOT the whole
+// directory. Registry records use their authoritative queuedAt as the sort key.
+export async function listAllJobs(
+ paths: Paths,
+ opts: { limit?: number; before?: number } = {},
+): Promise<JobsPage> {
+ const registry = getRegistry();
+ const live = new Map<string, JobRecord>();
+ for (const r of registry.list()) live.set(r.id, r);
+
+ const logIds = await readLogIds(paths);
+
+ // Candidate set = every log file ∪ every registry record. Time is the sort
+ // key: registry queuedAt when known, else decoded from the id.
+ const candidates = new Map<
+ string,
+ { id: string; time: number; hasLog: boolean }
+ >();
+ for (const id of logIds) {
+ candidates.set(id, { id, time: jobIdTime(id), hasLog: true });
+ }
for (const r of live.values()) {
- if (!seen.has(r.id)) {
- out.push({
- id: r.id,
- kind: r.kind,
- channelSlug: r.channelSlug,
- videoId: r.videoId,
- queueKey: r.queueKey,
- status: r.status,
- queuedAt: r.queuedAt,
- startedAt: r.startedAt,
- endedAt: r.endedAt,
- exitCode: r.exitCode,
- inRegistry: true,
- bookmarkable: Boolean(r.spec),
- logPath: r.logPath,
- logSize: 0,
- });
- }
+ const existing = candidates.get(r.id);
+ if (existing) existing.time = r.queuedAt;
+ else candidates.set(r.id, { id: r.id, time: r.queuedAt, hasLog: false });
+ }
+
+ let sorted = Array.from(candidates.values()).sort((a, b) => b.time - a.time);
+ if (typeof opts.before === "number") {
+ const before = opts.before;
+ sorted = sorted.filter((c) => c.time < before);
}
- return out.sort((a, b) => b.queuedAt - a.queuedAt);
+ const limit = opts.limit ?? DEFAULT_JOBS_LIMIT;
+ const slice = sorted.slice(0, limit + 1);
+ const hasMore = slice.length > limit;
+ const page = hasMore ? slice.slice(0, limit) : slice;
+
+ const entries: JobListEntry[] = [];
+ for (const c of page) {
+ entries.push(await buildEntry(paths, c.id, live.get(c.id), c.hasLog));
+ }
+
+ return { entries, hasMore, total: candidates.size };
+}
+
+// Resolve a single job by id for the detail page — avoids listing every job.
+// Null when neither the registry nor an `<id>.log` file knows the id.
+export async function getJobEntry(
+ paths: Paths,
+ id: string,
+): Promise<JobListEntry | null> {
+ const live_ = getRegistry().get(id);
+ let hasLog = false;
+ try {
+ await stat(path.join(paths.jobsDir, `${id}.log`));
+ hasLog = true;
+ } catch {
+ /* no log file */
+ }
+ if (!live_ && !hasLog) return null;
+ return buildEntry(paths, id, live_, hasLog);
}
export async function readLogChunk(
@@ -160,22 +225,63 @@ export async function readLogChunk(
return { content: raw.slice(fromBytes), nextOffset: raw.length };
}
-export async function clearArchivedLogs(paths: Paths): Promise<number> {
- const live = new Set(getRegistry().list().map((r) => r.id));
- let files: string[];
- try {
- files = (await readdir(paths.jobsDir)).filter((n) => n.endsWith(".log"));
- } catch {
- return 0;
- }
+// Delete `.log` + `.meta.json` pairs by retention policy. Never touches a job
+// the registry currently reports running or queued. `all` clears every finished
+// job; otherwise `keepLast` drops the tail past the newest N and `olderThanMs`
+// drops anything older than the cutoff (union — either condition deletes).
+export async function pruneJobLogs(
+ paths: Paths,
+ opts: { keepLast?: number; olderThanMs?: number; all?: boolean; now?: number },
+): Promise<{ deleted: number }> {
+ const protectedIds = new Set(
+ getRegistry()
+ .list()
+ .filter((r) => r.status === "running" || r.status === "queued")
+ .map((r) => r.id),
+ );
+
+ const logIds = await readLogIds(paths);
+ const items = Array.from(logIds)
+ .map((id) => ({ id, t: jobIdTime(id) }))
+ .sort((a, b) => b.t - a.t);
+
+ const now = opts.now ?? Date.now();
+ const cutoff = opts.olderThanMs != null ? now - opts.olderThanMs : null;
+
let deleted = 0;
- for (const file of files) {
- const id = file.replace(/\.log$/, "");
- if (live.has(id)) continue;
- await rm(path.join(paths.jobsDir, file), { force: true });
- // Drop the sidecar too, but only count deleted .log files.
+ for (let i = 0; i < items.length; i++) {
+ const { id, t } = items[i];
+ if (protectedIds.has(id)) continue;
+
+ let doDelete = Boolean(opts.all);
+ if (!doDelete && opts.keepLast != null && i >= opts.keepLast) doDelete = true;
+ // t === 0 means an unparseable id (age unknown); leave age-based pruning to
+ // keepLast so we never delete something whose age we can't establish.
+ if (!doDelete && cutoff != null && t > 0 && t < cutoff) doDelete = true;
+ if (!doDelete) continue;
+
+ await rm(path.join(paths.jobsDir, `${id}.log`), { force: true });
await rm(metaPath(paths, id), { force: true });
deleted++;
}
- return deleted;
+ return { deleted };
+}
+
+// Throttled automatic retention, called after a job finalizes. Best-effort: a
+// prune failure must never affect the job. Bounds the `.jobs` directory so the
+// list stays fast without any manual clearing.
+let lastPruneAt = 0;
+export async function maybePruneJobLogs(paths: Paths): Promise<void> {
+ const now = Date.now();
+ if (now - lastPruneAt < PRUNE_THROTTLE_MS) return;
+ lastPruneAt = now;
+ try {
+ await pruneJobLogs(paths, {
+ keepLast: RETENTION_KEEP_LAST,
+ olderThanMs: RETENTION_MAX_AGE_MS,
+ now,
+ });
+ } catch {
+ /* best-effort */
+ }
}
diff --git a/common/jobs/registry.ts b/common/jobs/registry.ts
@@ -2,6 +2,7 @@ import type { ChildProcess } from "node:child_process";
import type { JobSpec } from "./jobSpec";
import { getScheduler } from "./scheduler";
import { getJobKind, type SchedulerTier } from "./jobKinds";
+import { ulid } from "./ulid";
export type JobStatus =
| "queued"
@@ -337,7 +338,10 @@ export function getRegistry(): JobRegistry {
}
export function newJobId(): string {
- return `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
+ // ULIDs: lexicographically time-sortable and time-decodable, so the /jobs
+ // list can paginate over `.jobs` by filename without stat-ing every file.
+ // See common/jobs/ulid.ts (jobIdTime handles both this and the legacy scheme).
+ return ulid();
}
// Which kinds honor the drain signal now lives in the job-kind metadata table
diff --git a/common/jobs/streamCommand.ts b/common/jobs/streamCommand.ts
@@ -17,6 +17,7 @@ import {
shouldRequestSnapshot,
} from "./snapshotScheduler";
import { writeJobMeta } from "./jobMeta";
+import { maybePruneJobLogs } from "./listJobs";
import type { JobSpec } from "./jobSpec";
// Mark a job's channel report dirty so the debounced scheduler regenerates the
@@ -241,6 +242,8 @@ export async function runManagedCommand(
requestSnapshotOnFinish(id, opts);
// Persist terminal state (status/endedAt/exitCode now set by finalize).
void writeJobMeta(opts.paths, record);
+ // Throttled retention so the .jobs directory stays bounded on its own.
+ void maybePruneJobLogs(opts.paths);
settle(record.status);
});
};
@@ -354,6 +357,8 @@ export async function runManagedFunction(
requestSnapshotOnFinish(id, opts);
// Persist terminal state (status/endedAt/exitCode now set by finalize).
void writeJobMeta(opts.paths, record);
+ // Throttled retention so the .jobs directory stays bounded on its own.
+ void maybePruneJobLogs(opts.paths);
settle(record.status);
});
};
diff --git a/common/jobs/ulid.test.ts b/common/jobs/ulid.test.ts
@@ -0,0 +1,37 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { ulid, jobIdTime } from "./ulid";
+
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/jobs/ulid.test.ts
+
+test("ulid is 26 chars and encodes the given time", () => {
+ const now = 1_700_000_000_000;
+ const id = ulid(now);
+ assert.equal(id.length, 26);
+ assert.equal(jobIdTime(id), now);
+});
+
+test("jobIdTime round-trips a real timestamp within a ms", () => {
+ const now = 1_735_000_123_456;
+ assert.equal(jobIdTime(ulid(now)), now);
+});
+
+test("jobIdTime decodes the legacy `<time36>-<rand>` scheme", () => {
+ const now = 1_712_000_000_000;
+ const legacy = `${now.toString(36)}-ab12cd`;
+ assert.equal(jobIdTime(legacy), now);
+});
+
+test("ulids sort lexicographically in time order", () => {
+ const a = ulid(1_000);
+ const b = ulid(2_000);
+ const c = ulid(3_000);
+ const shuffled = [c, a, b];
+ const sorted = [...shuffled].sort();
+ assert.deepEqual(sorted, [a, b, c]);
+});
+
+test("jobIdTime returns 0 for garbage", () => {
+ assert.equal(jobIdTime(""), 0);
+ assert.equal(jobIdTime("!!!"), 0);
+});
diff --git a/common/jobs/ulid.ts b/common/jobs/ulid.ts
@@ -0,0 +1,58 @@
+// Job ids are ULIDs: a 48-bit millisecond timestamp followed by 80 bits of
+// randomness, Crockford base32-encoded to a fixed 26 chars. Two properties earn
+// their keep on the /jobs screen:
+// 1. Lexicographic order == chronological order, so the on-disk `.jobs`
+// directory can be sorted and paginated by filename alone.
+// 2. The timestamp is recoverable from the id itself (jobIdTime), so listing a
+// page never has to stat/read every file to know when a job ran.
+// Hand-rolled to avoid a dependency; strict within-millisecond monotonicity is
+// not needed here (tie-break order between jobs sharing a ms is irrelevant).
+
+// Crockford base32 (no I, L, O, U).
+const ENCODING = "0123456789ABCDEFGHJKMNPQRSTVWXYZ";
+const ENCODING_LEN = ENCODING.length; // 32
+const TIME_LEN = 10; // 48 bits -> 10 base32 chars
+const RANDOM_LEN = 16; // 80 bits -> 16 base32 chars
+
+function encodeTime(now: number): string {
+ let str = "";
+ for (let i = TIME_LEN; i > 0; i--) {
+ const mod = now % ENCODING_LEN;
+ str = ENCODING[mod] + str;
+ now = Math.floor(now / ENCODING_LEN);
+ }
+ return str;
+}
+
+function encodeRandom(): string {
+ let str = "";
+ for (let i = RANDOM_LEN; i > 0; i--) {
+ str += ENCODING[Math.floor(Math.random() * ENCODING_LEN)];
+ }
+ return str;
+}
+
+export function ulid(now: number = Date.now()): string {
+ return encodeTime(now) + encodeRandom();
+}
+
+// Decode the millisecond timestamp carried by a job id. Handles both id schemes
+// so on-disk logs written before the ULID switch still sort/paginate correctly:
+// - Legacy `Date.now().toString(36)-<rand>` (contains a "-").
+// - ULID (26 chars, no "-"): the first 10 chars are the base32 timestamp.
+// Returns 0 when the id is unparseable; callers fall back to the file mtime.
+export function jobIdTime(id: string): number {
+ const dash = id.indexOf("-");
+ if (dash >= 0) {
+ const t = parseInt(id.slice(0, dash), 36);
+ return Number.isFinite(t) ? t : 0;
+ }
+ if (id.length < TIME_LEN) return 0;
+ let time = 0;
+ for (let i = 0; i < TIME_LEN; i++) {
+ const idx = ENCODING.indexOf(id[i].toUpperCase());
+ if (idx < 0) return 0;
+ time = time * ENCODING_LEN + idx;
+ }
+ return time;
+}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Jobs page: real log retention + pagination (replaces the dead "Clear archived logs" button).** The old button only deleted logs absent from the in-memory registry — which, since the registry keeps the 100 newest finished jobs and sidecars preserve their real status, was almost never anything, so it did nothing. It's replaced by a **Clear logs** dropdown that prunes finished-job logs by age (older than 7 / 30 / 90 days) or all at once; running/queued jobs are never deleted. The `.jobs` directory also **self-trims on job finish** (throttled; keep newest 500, drop >30 days) so it can't grow unbounded. Job ids are now **ULIDs** (lexicographically time-sortable, timestamp decodable from the id), letting the list **paginate** — `listAllJobs` returns one page (default 50, grown by a **Load more** link) and only `stat`s/reads the sidecar for the shown page instead of every file on every load. `jobIdTime()` decodes both ULID and the legacy `<t36>-<rand>` ids, so existing on-disk logs still sort/read correctly. See `common/jobs/{ulid,listJobs,registry,streamCommand}.ts`, `editor/app/jobs/{page.tsx,actions.ts,components/ClearLogsMenu.tsx,[id]/page.tsx}`, and `editor/e2e/jobs.spec.ts`.
- **"Move to top" button on the auto-queue policy editor.** Each reorderable rule/group in the auto-queue policy tree gains a **⤒** button beside the existing ↑/↓ swap controls that jumps the node straight to the front of its sibling list in one click (disabled on the first row, like ↑). Reordering stays local until **Save policy**, matching the swap buttons. See `editor/app/auto-queue/components/PolicyTreeEditor.tsx` and `editor/e2e/auto-queue.spec.ts`.
- **Kick VOD playback + VOD-expiry indicators.** Kick becomes a first-class platform (`Platform` union, `detectPlatform`, `platformFromMetadata` `/^kick/i`, `extractVideoId` kick branch, `defaultWebpageUrl`). Kick VODs have no iframe embed, so playback streams the HLS manifest yt-dlp resolves at download time: `summarize()` persists `manifest_url` → `hlsUrl` on the transcript summary/detail, and a new client-only `common/components/KickPlayer.tsx` plays it in a native `<video>` via the bundled **hls.js** (not react-player's file player, which loads hls.js from a CDN and would break the offline export). It exposes the same `seekTo`/`onReady`/`onProgress` handle as the YouTube player, so Kick gets full scrubbing + cue highlighting; on a fatal manifest error it falls back to an expiry notice + source link. Separately, a shared `common/lib/vodExpiry.ts` (retention: Kick 30d, Twitch 14d, tunable) drives a new `VodExpiredBadge` on search result cards for likely-deleted Kick/Twitch VODs, with a `title=` tooltip explaining each platform's retention. Cache versions bumped so stale data re-derives (`transcriptStore` `DB_VERSION` 4, `normalizeTranscript` `CUES_FILE_VERSION` 2). See `common/lib/{platform,transcripts,transcripts-server,vodExpiry,format}.ts`, `common/components/{KickPlayer,PlayerProvider,badges,TranscriptSearch}.tsx`, `common/ytdlp/runYtdlp.ts`, and `export/e2e/kick-vod.spec.ts`.
- **Clicking a site on the Sites list now opens its edit page instead of bouncing back to the list.** The sidebar site selector seeds the active site into the URL (`?site=`) on mount so the scoped server pages (Dashboard, Channels, Charts, Deploy) can read it — but it was also firing on the `/sites` CRUD pages, where a mount-time `router.replace("/sites?site=<id>")` raced and clobbered the in-flight navigation to `/sites/<id>` from a list link, dumping you back on the list. The seed is now skipped on `/sites*` routes (which never consume `?site=`), so site links navigate straight to the editor; scoped-page seeding is unchanged. See `editor/app/components/SiteScopeSelect.tsx`.
diff --git a/editor/app/jobs/[id]/page.tsx b/editor/app/jobs/[id]/page.tsx
@@ -2,7 +2,7 @@ import type { Metadata } from "next";
import Link from "next/link";
import path from "node:path";
import { notFound } from "next/navigation";
-import { listAllJobs } from "yt-dlp-transcript-common/jobs/listJobs";
+import { getJobEntry } from "yt-dlp-transcript-common/jobs/listJobs";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import { JobLogTail } from "./components/JobLogTail";
import { BookmarkJobButton } from "../components/BookmarkJobButton";
@@ -16,8 +16,7 @@ export async function generateMetadata({
params: Promise<{ id: string }>;
}): Promise<Metadata> {
const { id } = await params;
- const jobs = await listAllJobs(getPaths());
- const job = jobs.find((j) => j.id === id);
+ const job = await getJobEntry(getPaths(), id);
if (!job) return { title: `${id} — Job` };
const scope = job.channelSlug
? job.videoId
@@ -35,8 +34,7 @@ export default async function JobDetailPage({
params: Promise<{ id: string }>;
}) {
const { id } = await params;
- const jobs = await listAllJobs(getPaths());
- const job = jobs.find((j) => j.id === id);
+ const job = await getJobEntry(getPaths(), id);
if (!job) notFound();
return (
<div className="flex flex-col gap-4">
diff --git a/editor/app/jobs/actions.ts b/editor/app/jobs/actions.ts
@@ -2,7 +2,7 @@
import { revalidatePath } from "next/cache";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { clearArchivedLogs } from "yt-dlp-transcript-common/jobs/listJobs";
+import { pruneJobLogs } from "yt-dlp-transcript-common/jobs/listJobs";
import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
import { readJobMeta } from "yt-dlp-transcript-common/jobs/jobMeta";
import type { JobSpec } from "yt-dlp-transcript-common/jobs/jobSpec";
@@ -109,8 +109,20 @@ export async function retryAllFailedAction(): Promise<{ count: number }> {
return { count };
}
-export async function clearArchivedAction(): Promise<{ deleted: number }> {
- const deleted = await clearArchivedLogs(getPaths());
+// Retention scopes offered by the ClearLogsMenu. "all" clears every finished
+// job's log; the day-scopes clear anything older than that. Running/queued jobs
+// are never deleted (see pruneJobLogs).
+const DAY_MS = 24 * 60 * 60 * 1000;
+export type ClearLogsScope = "7d" | "30d" | "90d" | "all";
+
+export async function clearFinishedLogsAction(
+ scope: ClearLogsScope,
+): Promise<{ deleted: number }> {
+ const opts =
+ scope === "all"
+ ? { all: true }
+ : { olderThanMs: { "7d": 7, "30d": 30, "90d": 90 }[scope] * DAY_MS };
+ const { deleted } = await pruneJobLogs(getPaths(), opts);
revalidatePath("/jobs");
return { deleted };
}
diff --git a/editor/app/jobs/components/ClearArchivedButton.tsx b/editor/app/jobs/components/ClearArchivedButton.tsx
@@ -1,28 +0,0 @@
-"use client";
-
-import { useState } from "react";
-import { useRouter } from "next/navigation";
-import { clearArchivedAction } from "../actions";
-
-export function ClearArchivedButton() {
- const [busy, setBusy] = useState(false);
- const router = useRouter();
- return (
- <button
- type="button"
- onClick={async () => {
- setBusy(true);
- try {
- await clearArchivedAction();
- router.refresh();
- } finally {
- setBusy(false);
- }
- }}
- disabled={busy}
- className="px-3 py-2 rounded-md border border-border text-sm hover:bg-muted disabled:opacity-50"
- >
- {busy ? "Clearing…" : "Clear archived logs"}
- </button>
- );
-}
diff --git a/editor/app/jobs/components/ClearLogsMenu.tsx b/editor/app/jobs/components/ClearLogsMenu.tsx
@@ -0,0 +1,62 @@
+"use client";
+
+import { useState, useTransition } from "react";
+import { useRouter } from "next/navigation";
+import { ChevronDown } from "lucide-react";
+import {
+ DropdownMenu,
+ DropdownMenuContent,
+ DropdownMenuItem,
+ DropdownMenuSeparator,
+ DropdownMenuTrigger,
+} from "yt-dlp-transcript-common/components/ui/dropdown-menu";
+import { clearFinishedLogsAction, type ClearLogsScope } from "../actions";
+
+const OPTIONS: { scope: ClearLogsScope; label: string }[] = [
+ { scope: "7d", label: "Older than 7 days" },
+ { scope: "30d", label: "Older than 30 days" },
+ { scope: "90d", label: "Older than 90 days" },
+];
+
+// Replaces the old (now no-op) "Clear archived logs" button. Prunes finished
+// job logs by age or all at once; running/queued jobs are never touched (see
+// pruneJobLogs). The directory also self-trims on job finish, so this is the
+// manual escape hatch.
+export function ClearLogsMenu() {
+ const router = useRouter();
+ const [pending, startTransition] = useTransition();
+ const [open, setOpen] = useState(false);
+
+ function clear(scope: ClearLogsScope): void {
+ startTransition(async () => {
+ await clearFinishedLogsAction(scope);
+ router.refresh();
+ });
+ }
+
+ return (
+ <DropdownMenu open={open} onOpenChange={setOpen}>
+ <DropdownMenuTrigger
+ disabled={pending}
+ className="inline-flex items-center gap-1 px-3 py-2 rounded-md border border-border text-sm hover:bg-muted disabled:opacity-50 outline-none focus-visible:ring-2 focus-visible:ring-ring"
+ >
+ {pending ? "Clearing…" : "Clear logs"}
+ <ChevronDown className="h-4 w-4" aria-hidden="true" />
+ </DropdownMenuTrigger>
+ <DropdownMenuContent align="end" className="min-w-48">
+ {OPTIONS.map((o) => (
+ <DropdownMenuItem key={o.scope} onSelect={() => clear(o.scope)}>
+ {o.label}
+ </DropdownMenuItem>
+ ))}
+ <DropdownMenuSeparator />
+ <DropdownMenuItem
+ variant="destructive"
+ onSelect={() => clear("all")}
+ >
+ All finished logs
+ </DropdownMenuItem>
+ </DropdownMenuContent>
+ </DropdownMenu>
+ );
+}
diff --git a/editor/app/jobs/page.tsx b/editor/app/jobs/page.tsx
@@ -1,7 +1,12 @@
import type { Metadata } from "next";
-import { listAllJobs } from "yt-dlp-transcript-common/jobs/listJobs";
+import Link from "next/link";
+import {
+ listAllJobs,
+ DEFAULT_JOBS_LIMIT,
+} from "yt-dlp-transcript-common/jobs/listJobs";
+import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { ClearArchivedButton } from "./components/ClearArchivedButton";
+import { ClearLogsMenu } from "./components/ClearLogsMenu";
import { JobsTable } from "./components/JobsTable";
import { RetryAllFailedButton } from "./components/RetryAllFailedButton";
import { BookmarksMenu } from "./components/BookmarksMenu";
@@ -11,30 +16,61 @@ export const dynamic = "force-dynamic";
export const metadata: Metadata = { title: "Jobs" };
-export default async function JobsPage() {
- const [jobs, { bookmarks, missingSlugs }] = await Promise.all([
- listAllJobs(getPaths()),
- loadBookmarksView(),
- ]);
- const hasRetryableFailed = jobs.some(
- (j) => j.status === "failed" && j.bookmarkable,
- );
+// Clamp so a hand-edited ?limit= can't ask the server to hydrate an unbounded
+// page. "Load more" grows the limit a page at a time.
+const MAX_JOBS_LIMIT = 2000;
+
+function parseLimit(raw: string | string[] | undefined): number {
+ const n = Number(Array.isArray(raw) ? raw[0] : raw);
+ if (!Number.isFinite(n) || n <= 0) return DEFAULT_JOBS_LIMIT;
+ return Math.min(Math.floor(n), MAX_JOBS_LIMIT);
+}
+
+export default async function JobsPage({
+ searchParams,
+}: {
+ searchParams: Promise<{ limit?: string | string[] }>;
+}) {
+ const limit = parseLimit((await searchParams).limit);
+ const [{ entries, hasMore, total }, { bookmarks, missingSlugs }] =
+ await Promise.all([listAllJobs(getPaths(), { limit }), loadBookmarksView()]);
+ // Base the Retry-all affordance on the registry (what retryAllFailedAction
+ // actually acts on), not just the loaded page.
+ const hasRetryableFailed = getRegistry()
+ .list()
+ .some((j) => j.status === "failed" && Boolean(j.spec));
return (
<div className="flex flex-col gap-4">
<div className="flex items-center justify-between">
<h1 className="text-2xl font-semibold">Jobs</h1>
<div className="flex items-center gap-2">
{hasRetryableFailed && <RetryAllFailedButton />}
- <ClearArchivedButton />
+ <ClearLogsMenu />
</div>
</div>
<BookmarksMenu bookmarks={bookmarks} missingSlugs={missingSlugs} />
- {jobs.length === 0 ? (
+ {total === 0 ? (
<p className="text-sm text-muted-foreground border border-dashed border-border rounded p-4">
No jobs have run yet.
</p>
) : (
- <JobsTable jobs={jobs} />
+ <>
+ <JobsTable jobs={entries} />
+ {hasMore && (
+ <div className="flex items-center justify-center gap-3 text-sm">
+ <span className="text-muted-foreground">
+ Showing {entries.length} of {total}
+ </span>
+ <Link
+ href={`/jobs?limit=${limit + DEFAULT_JOBS_LIMIT}`}
+ scroll={false}
+ className="px-3 py-2 rounded-md border border-border hover:bg-muted"
+ >
+ Load more
+ </Link>
+ </div>
+ )}
+ </>
)}
</div>
);
diff --git a/editor/e2e/jobs.spec.ts b/editor/e2e/jobs.spec.ts
@@ -5,7 +5,11 @@ test("empty state when no jobs have run", async ({ page }) => {
await resetData("empty");
// Clear any stray .jobs/*.log from other tests in this lifetime.
await page.goto("/jobs");
- await page.getByRole("button", { name: "Clear archived logs" }).click();
+ await page.getByRole("button", { name: "Clear logs" }).click();
+ await page.getByRole("menuitem", { name: "All finished logs" }).click();
+ // The menu clears via a transition + router.refresh(); wait for it to land,
+ // then confirm it persists on a fresh load.
+ await expect(page.getByText("No jobs have run yet.")).toBeVisible();
await page.goto("/jobs");
await expect(page.getByText("No jobs have run yet.")).toBeVisible();
});