Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 120d890843d673f3335bb4db3ca39b81b60bead8
parent 5730940be7013765f94d1cac1f4d38c947b3311f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri,  3 Jul 2026 00:49:00 -0400

Jobs page: ULID ids, pagination, and log retention

Replace the dead "Clear archived logs" button and bound the /jobs load.

The old button only deleted logs absent from the in-memory registry, which
(since the registry keeps the 100 newest finished jobs and sidecars preserve
their real status) was almost never anything — so it did nothing. And
listAllJobs stat'd + read the sidecar for EVERY .jobs file on every load, with
no pagination.

- ULID job ids (common/jobs/ulid.ts): lexicographically time-sortable and
  time-decodable. jobIdTime() decodes both ULID and the legacy <t36>-<rand>
  scheme, so existing on-disk logs still sort/read correctly.
- listAllJobs paginates: returns { entries, hasMore, total }, sorts/pages by
  jobIdTime, and only stats/reads the sidecar for the shown page. New
  getJobEntry() resolves a single job for the detail page.
- pruneJobLogs({ keepLast, olderThanMs, all }) replaces clearArchivedLogs;
  never deletes running/queued jobs. maybePruneJobLogs auto-trims on job finish
  (throttled; keep 500, drop >30d).
- UI: ClearLogsMenu dropdown (7/30/90 days, all) replaces ClearArchivedButton;
  page.tsx grows ?limit= via a Load more link.
- Tests: ulid.test.ts + listJobs.test.ts (11 cases); jobs.spec.ts updated for
  the new menu. Jobs e2e suite green.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/jobs/listJobs.test.ts | 141+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/jobs/listJobs.ts | 328++++++++++++++++++++++++++++++++++++++++++++++++++++---------------------------
Mcommon/jobs/registry.ts | 6+++++-
Mcommon/jobs/streamCommand.ts | 5+++++
Acommon/jobs/ulid.test.ts | 37+++++++++++++++++++++++++++++++++++++
Acommon/jobs/ulid.ts | 58++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/CHANGELOG.md | 1+
Meditor/app/jobs/[id]/page.tsx | 8+++-----
Meditor/app/jobs/actions.ts | 18+++++++++++++++---
Deditor/app/jobs/components/ClearArchivedButton.tsx | 28----------------------------
Aeditor/app/jobs/components/ClearLogsMenu.tsx | 62++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/jobs/page.tsx | 62+++++++++++++++++++++++++++++++++++++++++++++++++-------------
Meditor/e2e/jobs.spec.ts | 6+++++-
13 files changed, 598 insertions(+), 162 deletions(-)

diff --git a/common/jobs/listJobs.test.ts b/common/jobs/listJobs.test.ts @@ -0,0 +1,141 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, mkdir, writeFile, readdir, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import { ulid } from "./ulid"; +import { + listAllJobs, + getJobEntry, + pruneJobLogs, +} from "./listJobs"; + +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/jobs/listJobs.test.ts +// +// listAllJobs/pruneJobLogs only touch paths.jobsDir and the (empty in a fresh +// process) global registry, so a stub Paths with just jobsDir + seeded log/meta +// files exercises the full disk path. Ids are ULIDs stamped at a known time so +// jobIdTime() orders them deterministically. + +const BASE = 1_700_000_000_000; + +async function withJobs( + fn: (paths: Paths, seed: (i: number, t: number) => Promise<string>) => Promise<void>, +): Promise<void> { + const dir = await mkdtemp(path.join(tmpdir(), "list-jobs-")); + const jobsDir = path.join(dir, ".jobs"); + await mkdir(jobsDir, { recursive: true }); + const paths = { jobsDir } as Paths; + const seed = async (i: number, t: number): Promise<string> => { + const id = ulid(t); + await writeFile(path.join(jobsDir, `${id}.log`), `log ${i}\n`); + await writeFile( + path.join(jobsDir, `${id}.meta.json`), + JSON.stringify({ + id, + kind: "sync", + queueKey: "q", + status: "done", + queuedAt: t, + startedAt: t, + endedAt: t + 500, + exitCode: 0, + }), + ); + return id; + }; + try { + await fn(paths, seed); + } finally { + await rm(dir, { recursive: true, force: true }); + } +} + +async function logCount(paths: Paths): Promise<number> { + return (await readdir(paths.jobsDir)).filter((n) => n.endsWith(".log")).length; +} + +test("listAllJobs paginates newest-first and reports hasMore/total", async () => { + await withJobs(async (paths, seed) => { + const ids: string[] = []; + for (let i = 0; i < 5; i++) ids.push(await seed(i, BASE + i * 1000)); + + const page1 = await listAllJobs(paths, { limit: 2 }); + assert.equal(page1.entries.length, 2); + assert.equal(page1.hasMore, true); + assert.equal(page1.total, 5); + // Newest first: the last two seeded ids, in reverse order. + assert.deepEqual( + page1.entries.map((e) => e.id), + [ids[4], ids[3]], + ); + + const all = await listAllJobs(paths, { limit: 5 }); + assert.equal(all.hasMore, false); + assert.equal(all.entries.length, 5); + }); +}); + +test("listAllJobs before-cursor skips newer entries", async () => { + await withJobs(async (paths, seed) => { + const ids: string[] = []; + for (let i = 0; i < 4; i++) ids.push(await seed(i, BASE + i * 1000)); + // Everything strictly older than the 3rd job's time. + const page = await listAllJobs(paths, { limit: 10, before: BASE + 2000 }); + assert.deepEqual( + page.entries.map((e) => e.id), + [ids[1], ids[0]], + ); + }); +}); + +test("getJobEntry resolves one job and null for unknown ids", async () => { + await withJobs(async (paths, seed) => { + const id = await seed(0, BASE); + const entry = await getJobEntry(paths, id); + assert.ok(entry); + assert.equal(entry.id, id); + assert.equal(entry.kind, "sync"); + assert.equal(await getJobEntry(paths, "does-not-exist"), null); + }); +}); + +test("pruneJobLogs keepLast drops the oldest tail", async () => { + await withJobs(async (paths, seed) => { + for (let i = 0; i < 5; i++) await seed(i, BASE + i * 1000); + const { deleted } = await pruneJobLogs(paths, { keepLast: 2 }); + assert.equal(deleted, 3); + assert.equal(await logCount(paths), 2); + // Meta sidecars go too. + assert.equal( + (await readdir(paths.jobsDir)).filter((n) => n.endsWith(".meta.json")).length, + 2, + ); + }); +}); + +test("pruneJobLogs olderThanMs drops only aged entries", async () => { + await withJobs(async (paths, seed) => { + await seed(0, BASE); // old + await seed(1, BASE + 1000); // old + const fresh = await seed(2, BASE + 10 * 60 * 60 * 1000); // 10h newer + const now = BASE + 11 * 60 * 60 * 1000; + const { deleted } = await pruneJobLogs(paths, { + olderThanMs: 60 * 60 * 1000, // 1h + now, + }); + assert.equal(deleted, 2); + const remaining = await readdir(paths.jobsDir); + assert.ok(remaining.includes(`${fresh}.log`)); + }); +}); + +test("pruneJobLogs all clears everything", async () => { + await withJobs(async (paths, seed) => { + for (let i = 0; i < 3; i++) await seed(i, BASE + i * 1000); + const { deleted } = await pruneJobLogs(paths, { all: true }); + assert.equal(deleted, 3); + assert.equal(await logCount(paths), 0); + }); +}); diff --git a/common/jobs/listJobs.ts b/common/jobs/listJobs.ts @@ -3,6 +3,7 @@ import { readdir, stat, readFile, rm } from "node:fs/promises"; import type { Paths } from "../lib/paths"; import { getRegistry, type JobRecord, type JobStatus } from "./registry"; import { metaPath, readJobMeta } from "./jobMeta"; +import { jobIdTime } from "./ulid"; export type JobListEntry = { id: string; @@ -23,125 +24,189 @@ export type JobListEntry = { logSize: number; }; +export type JobsPage = { + entries: JobListEntry[]; + // More jobs exist beyond this page (i.e. raise `limit` to see them). + hasMore: boolean; + // Total jobs known on disk + in the registry, independent of the page size. + total: number; +}; + const TERMINAL_STATUSES: ReadonlySet<JobStatus> = new Set([ "done", "failed", "cancelled", ]); -// Merge in-memory registry entries with any leftover .log files from -// previous server lifetimes. Anything not in the registry is treated as -// "archived" — we only know that it once ran, not whether it succeeded. -export async function listAllJobs(paths: Paths): Promise<JobListEntry[]> { - const registry = getRegistry(); - const live: Map<string, JobRecord> = new Map(); - for (const r of registry.list()) live.set(r.id, r); +// Default page size for the /jobs list. The page grows this via ?limit=. +export const DEFAULT_JOBS_LIMIT = 50; + +// Automatic on-finish retention (see maybePruneJobLogs): keep the newest N jobs +// and drop anything older than the max age, whichever bites first. Generous — +// the point is to bound the directory (including the constant refresh-report +// churn), not to be aggressive. +const RETENTION_KEEP_LAST = 500; +const RETENTION_MAX_AGE_MS = 30 * 24 * 60 * 60 * 1000; // 30 days +// The list-page is `force-dynamic`, so job finishes could fire a readdir on +// every finalize. Throttle so a burst of jobs prunes at most once per window. +const PRUNE_THROTTLE_MS = 60_000; - let files: string[] = []; +async function readLogIds(paths: Paths): Promise<Set<string>> { try { - files = (await readdir(paths.jobsDir)).filter((n) => n.endsWith(".log")); + const files = await readdir(paths.jobsDir); + return new Set( + files.filter((n) => n.endsWith(".log")).map((n) => n.replace(/\.log$/, "")), + ); } catch { - files = []; + return new Set(); } +} - const out: JobListEntry[] = []; - const seen = new Set<string>(); - - for (const file of files) { - const id = file.replace(/\.log$/, ""); - seen.add(id); - const logPath = path.join(paths.jobsDir, file); - const live_ = live.get(id); - let logSize = 0; - let mtime = 0; +// Build one JobListEntry, preferring the live registry record and falling back +// to the on-disk `.meta.json` sidecar (or the log mtime for old logs with no +// sidecar). `hasLog` says whether an `<id>.log` file exists on disk. +async function buildEntry( + paths: Paths, + id: string, + live_: JobRecord | undefined, + hasLog: boolean, +): Promise<JobListEntry> { + const logPath = hasLog + ? path.join(paths.jobsDir, `${id}.log`) + : live_?.logPath ?? path.join(paths.jobsDir, `${id}.log`); + let logSize = 0; + let mtime = 0; + if (hasLog) { try { const s = await stat(logPath); logSize = s.size; mtime = s.mtimeMs; } catch { - continue; - } - if (live_) { - out.push({ - id, - kind: live_.kind, - channelSlug: live_.channelSlug, - videoId: live_.videoId, - queueKey: live_.queueKey, - status: live_.status, - queuedAt: live_.queuedAt, - startedAt: live_.startedAt, - endedAt: live_.endedAt, - exitCode: live_.exitCode, - inRegistry: true, - bookmarkable: Boolean(live_.spec), - logPath, - logSize, - }); - } else { - // Not in the registry: recover what we can from the sidecar so an evicted - // or post-restart job still shows its kind/channel/status/duration. With - // no sidecar (old logs), fall back to the mtime-based minimal entry. A - // non-terminal sidecar status means the job isn't actually live (it's not - // in the registry), so report it as "archived". - const meta = await readJobMeta(paths, id); - if (meta) { - out.push({ - id, - kind: meta.kind, - channelSlug: meta.channelSlug, - videoId: meta.videoId, - queueKey: meta.queueKey, - status: TERMINAL_STATUSES.has(meta.status) ? meta.status : "archived", - queuedAt: meta.queuedAt, - startedAt: meta.startedAt, - endedAt: meta.endedAt, - exitCode: meta.exitCode, - inRegistry: false, - bookmarkable: Boolean(meta.spec), - logPath, - logSize, - }); - } else { - out.push({ - id, - status: "archived", - queuedAt: mtime, - startedAt: mtime, - endedAt: mtime, - inRegistry: false, - bookmarkable: false, - logPath, - logSize, - }); - } + /* log vanished between listing and stat; treat as size 0 */ } } - // Registry entries with no log file (shouldn't happen normally, but be - // defensive). + if (live_) { + return { + id, + kind: live_.kind, + channelSlug: live_.channelSlug, + videoId: live_.videoId, + queueKey: live_.queueKey, + status: live_.status, + queuedAt: live_.queuedAt, + startedAt: live_.startedAt, + endedAt: live_.endedAt, + exitCode: live_.exitCode, + inRegistry: true, + bookmarkable: Boolean(live_.spec), + logPath, + logSize, + }; + } + + // Not in the registry: recover what we can from the sidecar so an evicted or + // post-restart job still shows its kind/channel/status/duration. With no + // sidecar (old logs), fall back to the mtime-based minimal entry. A + // non-terminal sidecar status means the job isn't actually live (it's not in + // the registry), so report it as "archived". + const meta = await readJobMeta(paths, id); + if (meta) { + return { + id, + kind: meta.kind, + channelSlug: meta.channelSlug, + videoId: meta.videoId, + queueKey: meta.queueKey, + status: TERMINAL_STATUSES.has(meta.status) ? meta.status : "archived", + queuedAt: meta.queuedAt, + startedAt: meta.startedAt, + endedAt: meta.endedAt, + exitCode: meta.exitCode, + inRegistry: false, + bookmarkable: Boolean(meta.spec), + logPath, + logSize, + }; + } + return { + id, + status: "archived", + queuedAt: mtime, + startedAt: mtime, + endedAt: mtime, + inRegistry: false, + bookmarkable: false, + logPath, + logSize, + }; +} + +// A page of the jobs list, newest first. Merges the in-memory registry with the +// leftover `.log` files from previous server lifetimes. Ordering and paging use +// jobIdTime(id) — decoded from the id itself (ULID, or the legacy `<t36>-<rand>` +// scheme) — so only the returned page is stat-ed / sidecar-read, NOT the whole +// directory. Registry records use their authoritative queuedAt as the sort key. +export async function listAllJobs( + paths: Paths, + opts: { limit?: number; before?: number } = {}, +): Promise<JobsPage> { + const registry = getRegistry(); + const live = new Map<string, JobRecord>(); + for (const r of registry.list()) live.set(r.id, r); + + const logIds = await readLogIds(paths); + + // Candidate set = every log file ∪ every registry record. Time is the sort + // key: registry queuedAt when known, else decoded from the id. + const candidates = new Map< + string, + { id: string; time: number; hasLog: boolean } + >(); + for (const id of logIds) { + candidates.set(id, { id, time: jobIdTime(id), hasLog: true }); + } for (const r of live.values()) { - if (!seen.has(r.id)) { - out.push({ - id: r.id, - kind: r.kind, - channelSlug: r.channelSlug, - videoId: r.videoId, - queueKey: r.queueKey, - status: r.status, - queuedAt: r.queuedAt, - startedAt: r.startedAt, - endedAt: r.endedAt, - exitCode: r.exitCode, - inRegistry: true, - bookmarkable: Boolean(r.spec), - logPath: r.logPath, - logSize: 0, - }); - } + const existing = candidates.get(r.id); + if (existing) existing.time = r.queuedAt; + else candidates.set(r.id, { id: r.id, time: r.queuedAt, hasLog: false }); + } + + let sorted = Array.from(candidates.values()).sort((a, b) => b.time - a.time); + if (typeof opts.before === "number") { + const before = opts.before; + sorted = sorted.filter((c) => c.time < before); } - return out.sort((a, b) => b.queuedAt - a.queuedAt); + const limit = opts.limit ?? DEFAULT_JOBS_LIMIT; + const slice = sorted.slice(0, limit + 1); + const hasMore = slice.length > limit; + const page = hasMore ? slice.slice(0, limit) : slice; + + const entries: JobListEntry[] = []; + for (const c of page) { + entries.push(await buildEntry(paths, c.id, live.get(c.id), c.hasLog)); + } + + return { entries, hasMore, total: candidates.size }; +} + +// Resolve a single job by id for the detail page — avoids listing every job. +// Null when neither the registry nor an `<id>.log` file knows the id. +export async function getJobEntry( + paths: Paths, + id: string, +): Promise<JobListEntry | null> { + const live_ = getRegistry().get(id); + let hasLog = false; + try { + await stat(path.join(paths.jobsDir, `${id}.log`)); + hasLog = true; + } catch { + /* no log file */ + } + if (!live_ && !hasLog) return null; + return buildEntry(paths, id, live_, hasLog); } export async function readLogChunk( @@ -160,22 +225,63 @@ export async function readLogChunk( return { content: raw.slice(fromBytes), nextOffset: raw.length }; } -export async function clearArchivedLogs(paths: Paths): Promise<number> { - const live = new Set(getRegistry().list().map((r) => r.id)); - let files: string[]; - try { - files = (await readdir(paths.jobsDir)).filter((n) => n.endsWith(".log")); - } catch { - return 0; - } +// Delete `.log` + `.meta.json` pairs by retention policy. Never touches a job +// the registry currently reports running or queued. `all` clears every finished +// job; otherwise `keepLast` drops the tail past the newest N and `olderThanMs` +// drops anything older than the cutoff (union — either condition deletes). +export async function pruneJobLogs( + paths: Paths, + opts: { keepLast?: number; olderThanMs?: number; all?: boolean; now?: number }, +): Promise<{ deleted: number }> { + const protectedIds = new Set( + getRegistry() + .list() + .filter((r) => r.status === "running" || r.status === "queued") + .map((r) => r.id), + ); + + const logIds = await readLogIds(paths); + const items = Array.from(logIds) + .map((id) => ({ id, t: jobIdTime(id) })) + .sort((a, b) => b.t - a.t); + + const now = opts.now ?? Date.now(); + const cutoff = opts.olderThanMs != null ? now - opts.olderThanMs : null; + let deleted = 0; - for (const file of files) { - const id = file.replace(/\.log$/, ""); - if (live.has(id)) continue; - await rm(path.join(paths.jobsDir, file), { force: true }); - // Drop the sidecar too, but only count deleted .log files. + for (let i = 0; i < items.length; i++) { + const { id, t } = items[i]; + if (protectedIds.has(id)) continue; + + let doDelete = Boolean(opts.all); + if (!doDelete && opts.keepLast != null && i >= opts.keepLast) doDelete = true; + // t === 0 means an unparseable id (age unknown); leave age-based pruning to + // keepLast so we never delete something whose age we can't establish. + if (!doDelete && cutoff != null && t > 0 && t < cutoff) doDelete = true; + if (!doDelete) continue; + + await rm(path.join(paths.jobsDir, `${id}.log`), { force: true }); await rm(metaPath(paths, id), { force: true }); deleted++; } - return deleted; + return { deleted }; +} + +// Throttled automatic retention, called after a job finalizes. Best-effort: a +// prune failure must never affect the job. Bounds the `.jobs` directory so the +// list stays fast without any manual clearing. +let lastPruneAt = 0; +export async function maybePruneJobLogs(paths: Paths): Promise<void> { + const now = Date.now(); + if (now - lastPruneAt < PRUNE_THROTTLE_MS) return; + lastPruneAt = now; + try { + await pruneJobLogs(paths, { + keepLast: RETENTION_KEEP_LAST, + olderThanMs: RETENTION_MAX_AGE_MS, + now, + }); + } catch { + /* best-effort */ + } } diff --git a/common/jobs/registry.ts b/common/jobs/registry.ts @@ -2,6 +2,7 @@ import type { ChildProcess } from "node:child_process"; import type { JobSpec } from "./jobSpec"; import { getScheduler } from "./scheduler"; import { getJobKind, type SchedulerTier } from "./jobKinds"; +import { ulid } from "./ulid"; export type JobStatus = | "queued" @@ -337,7 +338,10 @@ export function getRegistry(): JobRegistry { } export function newJobId(): string { - return `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`; + // ULIDs: lexicographically time-sortable and time-decodable, so the /jobs + // list can paginate over `.jobs` by filename without stat-ing every file. + // See common/jobs/ulid.ts (jobIdTime handles both this and the legacy scheme). + return ulid(); } // Which kinds honor the drain signal now lives in the job-kind metadata table diff --git a/common/jobs/streamCommand.ts b/common/jobs/streamCommand.ts @@ -17,6 +17,7 @@ import { shouldRequestSnapshot, } from "./snapshotScheduler"; import { writeJobMeta } from "./jobMeta"; +import { maybePruneJobLogs } from "./listJobs"; import type { JobSpec } from "./jobSpec"; // Mark a job's channel report dirty so the debounced scheduler regenerates the @@ -241,6 +242,8 @@ export async function runManagedCommand( requestSnapshotOnFinish(id, opts); // Persist terminal state (status/endedAt/exitCode now set by finalize). void writeJobMeta(opts.paths, record); + // Throttled retention so the .jobs directory stays bounded on its own. + void maybePruneJobLogs(opts.paths); settle(record.status); }); }; @@ -354,6 +357,8 @@ export async function runManagedFunction( requestSnapshotOnFinish(id, opts); // Persist terminal state (status/endedAt/exitCode now set by finalize). void writeJobMeta(opts.paths, record); + // Throttled retention so the .jobs directory stays bounded on its own. + void maybePruneJobLogs(opts.paths); settle(record.status); }); }; diff --git a/common/jobs/ulid.test.ts b/common/jobs/ulid.test.ts @@ -0,0 +1,37 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { ulid, jobIdTime } from "./ulid"; + +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/jobs/ulid.test.ts + +test("ulid is 26 chars and encodes the given time", () => { + const now = 1_700_000_000_000; + const id = ulid(now); + assert.equal(id.length, 26); + assert.equal(jobIdTime(id), now); +}); + +test("jobIdTime round-trips a real timestamp within a ms", () => { + const now = 1_735_000_123_456; + assert.equal(jobIdTime(ulid(now)), now); +}); + +test("jobIdTime decodes the legacy `<time36>-<rand>` scheme", () => { + const now = 1_712_000_000_000; + const legacy = `${now.toString(36)}-ab12cd`; + assert.equal(jobIdTime(legacy), now); +}); + +test("ulids sort lexicographically in time order", () => { + const a = ulid(1_000); + const b = ulid(2_000); + const c = ulid(3_000); + const shuffled = [c, a, b]; + const sorted = [...shuffled].sort(); + assert.deepEqual(sorted, [a, b, c]); +}); + +test("jobIdTime returns 0 for garbage", () => { + assert.equal(jobIdTime(""), 0); + assert.equal(jobIdTime("!!!"), 0); +}); diff --git a/common/jobs/ulid.ts b/common/jobs/ulid.ts @@ -0,0 +1,58 @@ +// Job ids are ULIDs: a 48-bit millisecond timestamp followed by 80 bits of +// randomness, Crockford base32-encoded to a fixed 26 chars. Two properties earn +// their keep on the /jobs screen: +// 1. Lexicographic order == chronological order, so the on-disk `.jobs` +// directory can be sorted and paginated by filename alone. +// 2. The timestamp is recoverable from the id itself (jobIdTime), so listing a +// page never has to stat/read every file to know when a job ran. +// Hand-rolled to avoid a dependency; strict within-millisecond monotonicity is +// not needed here (tie-break order between jobs sharing a ms is irrelevant). + +// Crockford base32 (no I, L, O, U). +const ENCODING = "0123456789ABCDEFGHJKMNPQRSTVWXYZ"; +const ENCODING_LEN = ENCODING.length; // 32 +const TIME_LEN = 10; // 48 bits -> 10 base32 chars +const RANDOM_LEN = 16; // 80 bits -> 16 base32 chars + +function encodeTime(now: number): string { + let str = ""; + for (let i = TIME_LEN; i > 0; i--) { + const mod = now % ENCODING_LEN; + str = ENCODING[mod] + str; + now = Math.floor(now / ENCODING_LEN); + } + return str; +} + +function encodeRandom(): string { + let str = ""; + for (let i = RANDOM_LEN; i > 0; i--) { + str += ENCODING[Math.floor(Math.random() * ENCODING_LEN)]; + } + return str; +} + +export function ulid(now: number = Date.now()): string { + return encodeTime(now) + encodeRandom(); +} + +// Decode the millisecond timestamp carried by a job id. Handles both id schemes +// so on-disk logs written before the ULID switch still sort/paginate correctly: +// - Legacy `Date.now().toString(36)-<rand>` (contains a "-"). +// - ULID (26 chars, no "-"): the first 10 chars are the base32 timestamp. +// Returns 0 when the id is unparseable; callers fall back to the file mtime. +export function jobIdTime(id: string): number { + const dash = id.indexOf("-"); + if (dash >= 0) { + const t = parseInt(id.slice(0, dash), 36); + return Number.isFinite(t) ? t : 0; + } + if (id.length < TIME_LEN) return 0; + let time = 0; + for (let i = 0; i < TIME_LEN; i++) { + const idx = ENCODING.indexOf(id[i].toUpperCase()); + if (idx < 0) return 0; + time = time * ENCODING_LEN + idx; + } + return time; +} diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Jobs page: real log retention + pagination (replaces the dead "Clear archived logs" button).** The old button only deleted logs absent from the in-memory registry — which, since the registry keeps the 100 newest finished jobs and sidecars preserve their real status, was almost never anything, so it did nothing. It's replaced by a **Clear logs** dropdown that prunes finished-job logs by age (older than 7 / 30 / 90 days) or all at once; running/queued jobs are never deleted. The `.jobs` directory also **self-trims on job finish** (throttled; keep newest 500, drop >30 days) so it can't grow unbounded. Job ids are now **ULIDs** (lexicographically time-sortable, timestamp decodable from the id), letting the list **paginate** — `listAllJobs` returns one page (default 50, grown by a **Load more** link) and only `stat`s/reads the sidecar for the shown page instead of every file on every load. `jobIdTime()` decodes both ULID and the legacy `<t36>-<rand>` ids, so existing on-disk logs still sort/read correctly. See `common/jobs/{ulid,listJobs,registry,streamCommand}.ts`, `editor/app/jobs/{page.tsx,actions.ts,components/ClearLogsMenu.tsx,[id]/page.tsx}`, and `editor/e2e/jobs.spec.ts`. - **"Move to top" button on the auto-queue policy editor.** Each reorderable rule/group in the auto-queue policy tree gains a **⤒** button beside the existing ↑/↓ swap controls that jumps the node straight to the front of its sibling list in one click (disabled on the first row, like ↑). Reordering stays local until **Save policy**, matching the swap buttons. See `editor/app/auto-queue/components/PolicyTreeEditor.tsx` and `editor/e2e/auto-queue.spec.ts`. - **Kick VOD playback + VOD-expiry indicators.** Kick becomes a first-class platform (`Platform` union, `detectPlatform`, `platformFromMetadata` `/^kick/i`, `extractVideoId` kick branch, `defaultWebpageUrl`). Kick VODs have no iframe embed, so playback streams the HLS manifest yt-dlp resolves at download time: `summarize()` persists `manifest_url` → `hlsUrl` on the transcript summary/detail, and a new client-only `common/components/KickPlayer.tsx` plays it in a native `<video>` via the bundled **hls.js** (not react-player's file player, which loads hls.js from a CDN and would break the offline export). It exposes the same `seekTo`/`onReady`/`onProgress` handle as the YouTube player, so Kick gets full scrubbing + cue highlighting; on a fatal manifest error it falls back to an expiry notice + source link. Separately, a shared `common/lib/vodExpiry.ts` (retention: Kick 30d, Twitch 14d, tunable) drives a new `VodExpiredBadge` on search result cards for likely-deleted Kick/Twitch VODs, with a `title=` tooltip explaining each platform's retention. Cache versions bumped so stale data re-derives (`transcriptStore` `DB_VERSION` 4, `normalizeTranscript` `CUES_FILE_VERSION` 2). See `common/lib/{platform,transcripts,transcripts-server,vodExpiry,format}.ts`, `common/components/{KickPlayer,PlayerProvider,badges,TranscriptSearch}.tsx`, `common/ytdlp/runYtdlp.ts`, and `export/e2e/kick-vod.spec.ts`. - **Clicking a site on the Sites list now opens its edit page instead of bouncing back to the list.** The sidebar site selector seeds the active site into the URL (`?site=`) on mount so the scoped server pages (Dashboard, Channels, Charts, Deploy) can read it — but it was also firing on the `/sites` CRUD pages, where a mount-time `router.replace("/sites?site=<id>")` raced and clobbered the in-flight navigation to `/sites/<id>` from a list link, dumping you back on the list. The seed is now skipped on `/sites*` routes (which never consume `?site=`), so site links navigate straight to the editor; scoped-page seeding is unchanged. See `editor/app/components/SiteScopeSelect.tsx`. diff --git a/editor/app/jobs/[id]/page.tsx b/editor/app/jobs/[id]/page.tsx @@ -2,7 +2,7 @@ import type { Metadata } from "next"; import Link from "next/link"; import path from "node:path"; import { notFound } from "next/navigation"; -import { listAllJobs } from "yt-dlp-transcript-common/jobs/listJobs"; +import { getJobEntry } from "yt-dlp-transcript-common/jobs/listJobs"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { JobLogTail } from "./components/JobLogTail"; import { BookmarkJobButton } from "../components/BookmarkJobButton"; @@ -16,8 +16,7 @@ export async function generateMetadata({ params: Promise<{ id: string }>; }): Promise<Metadata> { const { id } = await params; - const jobs = await listAllJobs(getPaths()); - const job = jobs.find((j) => j.id === id); + const job = await getJobEntry(getPaths(), id); if (!job) return { title: `${id} — Job` }; const scope = job.channelSlug ? job.videoId @@ -35,8 +34,7 @@ export default async function JobDetailPage({ params: Promise<{ id: string }>; }) { const { id } = await params; - const jobs = await listAllJobs(getPaths()); - const job = jobs.find((j) => j.id === id); + const job = await getJobEntry(getPaths(), id); if (!job) notFound(); return ( <div className="flex flex-col gap-4"> diff --git a/editor/app/jobs/actions.ts b/editor/app/jobs/actions.ts @@ -2,7 +2,7 @@ import { revalidatePath } from "next/cache"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { clearArchivedLogs } from "yt-dlp-transcript-common/jobs/listJobs"; +import { pruneJobLogs } from "yt-dlp-transcript-common/jobs/listJobs"; import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { readJobMeta } from "yt-dlp-transcript-common/jobs/jobMeta"; import type { JobSpec } from "yt-dlp-transcript-common/jobs/jobSpec"; @@ -109,8 +109,20 @@ export async function retryAllFailedAction(): Promise<{ count: number }> { return { count }; } -export async function clearArchivedAction(): Promise<{ deleted: number }> { - const deleted = await clearArchivedLogs(getPaths()); +// Retention scopes offered by the ClearLogsMenu. "all" clears every finished +// job's log; the day-scopes clear anything older than that. Running/queued jobs +// are never deleted (see pruneJobLogs). +const DAY_MS = 24 * 60 * 60 * 1000; +export type ClearLogsScope = "7d" | "30d" | "90d" | "all"; + +export async function clearFinishedLogsAction( + scope: ClearLogsScope, +): Promise<{ deleted: number }> { + const opts = + scope === "all" + ? { all: true } + : { olderThanMs: { "7d": 7, "30d": 30, "90d": 90 }[scope] * DAY_MS }; + const { deleted } = await pruneJobLogs(getPaths(), opts); revalidatePath("/jobs"); return { deleted }; } diff --git a/editor/app/jobs/components/ClearArchivedButton.tsx b/editor/app/jobs/components/ClearArchivedButton.tsx @@ -1,28 +0,0 @@ -"use client"; - -import { useState } from "react"; -import { useRouter } from "next/navigation"; -import { clearArchivedAction } from "../actions"; - -export function ClearArchivedButton() { - const [busy, setBusy] = useState(false); - const router = useRouter(); - return ( - <button - type="button" - onClick={async () => { - setBusy(true); - try { - await clearArchivedAction(); - router.refresh(); - } finally { - setBusy(false); - } - }} - disabled={busy} - className="px-3 py-2 rounded-md border border-border text-sm hover:bg-muted disabled:opacity-50" - > - {busy ? "Clearing…" : "Clear archived logs"} - </button> - ); -} diff --git a/editor/app/jobs/components/ClearLogsMenu.tsx b/editor/app/jobs/components/ClearLogsMenu.tsx @@ -0,0 +1,62 @@ +"use client"; + +import { useState, useTransition } from "react"; +import { useRouter } from "next/navigation"; +import { ChevronDown } from "lucide-react"; +import { + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuSeparator, + DropdownMenuTrigger, +} from "yt-dlp-transcript-common/components/ui/dropdown-menu"; +import { clearFinishedLogsAction, type ClearLogsScope } from "../actions"; + +const OPTIONS: { scope: ClearLogsScope; label: string }[] = [ + { scope: "7d", label: "Older than 7 days" }, + { scope: "30d", label: "Older than 30 days" }, + { scope: "90d", label: "Older than 90 days" }, +]; + +// Replaces the old (now no-op) "Clear archived logs" button. Prunes finished +// job logs by age or all at once; running/queued jobs are never touched (see +// pruneJobLogs). The directory also self-trims on job finish, so this is the +// manual escape hatch. +export function ClearLogsMenu() { + const router = useRouter(); + const [pending, startTransition] = useTransition(); + const [open, setOpen] = useState(false); + + function clear(scope: ClearLogsScope): void { + startTransition(async () => { + await clearFinishedLogsAction(scope); + router.refresh(); + }); + } + + return ( + <DropdownMenu open={open} onOpenChange={setOpen}> + <DropdownMenuTrigger + disabled={pending} + className="inline-flex items-center gap-1 px-3 py-2 rounded-md border border-border text-sm hover:bg-muted disabled:opacity-50 outline-none focus-visible:ring-2 focus-visible:ring-ring" + > + {pending ? "Clearing…" : "Clear logs"} + <ChevronDown className="h-4 w-4" aria-hidden="true" /> + </DropdownMenuTrigger> + <DropdownMenuContent align="end" className="min-w-48"> + {OPTIONS.map((o) => ( + <DropdownMenuItem key={o.scope} onSelect={() => clear(o.scope)}> + {o.label} + </DropdownMenuItem> + ))} + <DropdownMenuSeparator /> + <DropdownMenuItem + variant="destructive" + onSelect={() => clear("all")} + > + All finished logs + </DropdownMenuItem> + </DropdownMenuContent> + </DropdownMenu> + ); +} diff --git a/editor/app/jobs/page.tsx b/editor/app/jobs/page.tsx @@ -1,7 +1,12 @@ import type { Metadata } from "next"; -import { listAllJobs } from "yt-dlp-transcript-common/jobs/listJobs"; +import Link from "next/link"; +import { + listAllJobs, + DEFAULT_JOBS_LIMIT, +} from "yt-dlp-transcript-common/jobs/listJobs"; +import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { ClearArchivedButton } from "./components/ClearArchivedButton"; +import { ClearLogsMenu } from "./components/ClearLogsMenu"; import { JobsTable } from "./components/JobsTable"; import { RetryAllFailedButton } from "./components/RetryAllFailedButton"; import { BookmarksMenu } from "./components/BookmarksMenu"; @@ -11,30 +16,61 @@ export const dynamic = "force-dynamic"; export const metadata: Metadata = { title: "Jobs" }; -export default async function JobsPage() { - const [jobs, { bookmarks, missingSlugs }] = await Promise.all([ - listAllJobs(getPaths()), - loadBookmarksView(), - ]); - const hasRetryableFailed = jobs.some( - (j) => j.status === "failed" && j.bookmarkable, - ); +// Clamp so a hand-edited ?limit= can't ask the server to hydrate an unbounded +// page. "Load more" grows the limit a page at a time. +const MAX_JOBS_LIMIT = 2000; + +function parseLimit(raw: string | string[] | undefined): number { + const n = Number(Array.isArray(raw) ? raw[0] : raw); + if (!Number.isFinite(n) || n <= 0) return DEFAULT_JOBS_LIMIT; + return Math.min(Math.floor(n), MAX_JOBS_LIMIT); +} + +export default async function JobsPage({ + searchParams, +}: { + searchParams: Promise<{ limit?: string | string[] }>; +}) { + const limit = parseLimit((await searchParams).limit); + const [{ entries, hasMore, total }, { bookmarks, missingSlugs }] = + await Promise.all([listAllJobs(getPaths(), { limit }), loadBookmarksView()]); + // Base the Retry-all affordance on the registry (what retryAllFailedAction + // actually acts on), not just the loaded page. + const hasRetryableFailed = getRegistry() + .list() + .some((j) => j.status === "failed" && Boolean(j.spec)); return ( <div className="flex flex-col gap-4"> <div className="flex items-center justify-between"> <h1 className="text-2xl font-semibold">Jobs</h1> <div className="flex items-center gap-2"> {hasRetryableFailed && <RetryAllFailedButton />} - <ClearArchivedButton /> + <ClearLogsMenu /> </div> </div> <BookmarksMenu bookmarks={bookmarks} missingSlugs={missingSlugs} /> - {jobs.length === 0 ? ( + {total === 0 ? ( <p className="text-sm text-muted-foreground border border-dashed border-border rounded p-4"> No jobs have run yet. </p> ) : ( - <JobsTable jobs={jobs} /> + <> + <JobsTable jobs={entries} /> + {hasMore && ( + <div className="flex items-center justify-center gap-3 text-sm"> + <span className="text-muted-foreground"> + Showing {entries.length} of {total} + </span> + <Link + href={`/jobs?limit=${limit + DEFAULT_JOBS_LIMIT}`} + scroll={false} + className="px-3 py-2 rounded-md border border-border hover:bg-muted" + > + Load more + </Link> + </div> + )} + </> )} </div> ); diff --git a/editor/e2e/jobs.spec.ts b/editor/e2e/jobs.spec.ts @@ -5,7 +5,11 @@ test("empty state when no jobs have run", async ({ page }) => { await resetData("empty"); // Clear any stray .jobs/*.log from other tests in this lifetime. await page.goto("/jobs"); - await page.getByRole("button", { name: "Clear archived logs" }).click(); + await page.getByRole("button", { name: "Clear logs" }).click(); + await page.getByRole("menuitem", { name: "All finished logs" }).click(); + // The menu clears via a transition + router.refresh(); wait for it to land, + // then confirm it persists on a fresh load. + await expect(page.getByText("No jobs have run yet.")).toBeVisible(); await page.goto("/jobs"); await expect(page.getByText("No jobs have run yet.")).toBeVisible(); });