commit 44787dc1489f30ba075958844f095a9a046573e3
parent 5ebf4015874adcb45426c9f31ee3c20903afa6c1
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 8 Oct 2026 22:26:10 -0400
umtool S0: notes store, SITES_DIR, corpus-notes predicate, source discovery
- lib/paths.mjs: SITES_DIR (read-only root) and isCorpusNotesFile, the one
corpus path umtool may write (sites/<site>/reports/<id>/notes.json)
- lib/annotations: shape (pure contract + ops), anchor (TextQuote re-anchoring),
store (cross-process lockfile, mtime token, never overwrite unparseable),
targets (article / video-project -> notes file + source)
- lib/articles/sources.mjs: draft + generator discovery for a report
- common listReportDirs; editor's report list uses it
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
14 files changed, 1553 insertions(+), 23 deletions(-)
diff --git a/common/publish/reportMedia.ts b/common/publish/reportMedia.ts
@@ -42,6 +42,7 @@
// a channel whose media tier is not reachable (lib/channelMedia.ts) is
// reported as unreachable, with the reason, rather than as missing.
+import type { Dirent } from "node:fs";
import { readdir, rm } from "node:fs/promises";
import path from "node:path";
import { readJsonFile, writeJsonAtomic } from "../lib/jsonFile-server";
@@ -55,7 +56,7 @@ import { readChannelConfig } from "../controller/channels";
import { momentKey, momentOf, momentProblem, type Moment } from "../lib/citations/moments";
import type { CitationPad } from "../lib/citations/schema";
import { parseReport } from "../lib/report/validate";
-import type { Report } from "../lib/report/schema";
+import { isReportId, type Report } from "../lib/report/schema";
import { isPermanentlyGone } from "../lib/availability";
import { loadAvailability } from "../lib/availability-server";
import { MAX_CLIP_WINDOW_SECONDS } from "../lib/clipWindow";
@@ -93,6 +94,23 @@ export function siteReportFile(paths: Paths, siteId: string, reportId: string):
return path.join(siteReportDir(paths, siteId, reportId), "report.json");
}
+// Every report directory of a site, published or draft: a directory under
+// `sites/<siteId>/reports/` whose name is a report id. Anything else there (a
+// stray file, a bad name) is not a report. Sorted. Whether one is PUBLISHED is
+// the site's `reports` list (site.json), not anything on disk.
+export async function listReportDirs(paths: Paths, siteId: string): Promise<string[]> {
+ let entries: Dirent[];
+ try {
+ entries = await readdir(path.join(siteDir(paths, siteId), "reports"), { withFileTypes: true });
+ } catch {
+ return [];
+ }
+ return entries
+ .filter((e) => e.isDirectory() && isReportId(e.name))
+ .map((e) => e.name)
+ .sort((a, b) => a.localeCompare(b));
+}
+
export type ReportMediaEntry =
| EvidenceMedia
| {
diff --git a/editor/app/sites/lib/reportListServer.ts b/editor/app/sites/lib/reportListServer.ts
@@ -1,12 +1,8 @@
import "server-only";
-import type { Dirent } from "node:fs";
-import { readdir } from "node:fs/promises";
-import path from "node:path";
import type { Paths } from "yt-dlp-transcript-common/lib/paths";
import { readJsonFile } from "yt-dlp-transcript-common/lib/jsonFile-server";
-import { isReportId } from "yt-dlp-transcript-common/lib/report/schema";
-import { siteDir, type Site } from "yt-dlp-transcript-common/lib/site";
-import { siteReportFile } from "yt-dlp-transcript-common/publish/reportMedia";
+import type { Site } from "yt-dlp-transcript-common/lib/site";
+import { listReportDirs, siteReportFile } from "yt-dlp-transcript-common/publish/reportMedia";
import {
publishableReportExports,
readReportExportManifest,
@@ -18,24 +14,11 @@ import { listAllJobs, type JobListEntry } from "yt-dlp-transcript-common/jobs/li
import { readJobMeta } from "yt-dlp-transcript-common/jobs/jobMeta";
import { reportRows, type ReportFileRead, type ReportRow } from "./reportList";
-// The report directories under `sites/<siteId>/reports/`: a directory whose
-// name is a report id. Anything else there (a stray file, a bad name) is not a
-// report and is not listed.
-async function reportDirIds(paths: Paths, siteId: string): Promise<string[]> {
- let entries: Dirent[];
- try {
- entries = await readdir(path.join(siteDir(paths, siteId), "reports"), { withFileTypes: true });
- } catch {
- return [];
- }
- return entries.filter((e) => e.isDirectory() && isReportId(e.name)).map((e) => e.name);
-}
-
// Every report of the site, published (in order) then drafts, each read and
// validated.
export async function readSiteReportRows(paths: Paths, site: Site): Promise<ReportRow[]> {
const published = site.reports ?? [];
- const dirIds = await reportDirIds(paths, site.siteId);
+ const dirIds = await listReportDirs(paths, site.siteId);
const ids = [...new Set([...published, ...dirIds])];
const reads = new Map<string, ReportFileRead>(
await Promise.all(
diff --git a/package.json b/package.json
@@ -22,7 +22,7 @@
"e2e": "node scripts/worktree.mjs run -- pnpm --filter editor run e2e",
"wt": "node scripts/worktree.mjs",
"e2e:sharded": "node scripts/run-sharded-e2e.mjs",
- "test:scripts": "node --test scripts/*.test.mjs umtool/report-to-video/*.test.mjs umtool/lib/report/*.test.mjs",
+ "test:scripts": "node --test scripts/*.test.mjs umtool/report-to-video/*.test.mjs umtool/lib/report/*.test.mjs umtool/lib/annotations/*.test.mjs umtool/lib/articles/*.test.mjs",
"lint": "pnpm --filter export run lint",
"ops": "node scripts/archilyzer-ops.mjs"
},
diff --git a/umtool/lib/annotations/anchor.mjs b/umtool/lib/annotations/anchor.mjs
@@ -0,0 +1,176 @@
+// Finding a text anchor again in text that may have changed.
+//
+// A note on an article is anchored by its QUOTE plus 32 characters of context
+// either side (the W3C TextQuoteSelector), never by an offset: report.json is
+// regenerated from a draft, and an offset into the old text points at nothing
+// after the agent edits the paragraph above it.
+//
+// PURE and client-safe: no node imports. The article page runs it against the
+// rendered DOM text of a section; `umtool notes` runs it against the section's
+// plain text; the unit test runs it against both.
+
+export const CONTEXT = 32;
+
+/** Common-suffix length of `a` and `b` (how much of the prefix still precedes). */
+function suffixMatch(a, b) {
+ let n = 0;
+ while (n < a.length && n < b.length && a[a.length - 1 - n] === b[b.length - 1 - n]) n += 1;
+ return n;
+}
+/** Common-prefix length of `a` and `b` (how much of the suffix still follows). */
+function prefixMatch(a, b) {
+ let n = 0;
+ while (n < a.length && n < b.length && a[n] === b[n]) n += 1;
+ return n;
+}
+
+function allIndexes(hay, needle) {
+ const out = [];
+ if (!needle) return out;
+ for (let i = hay.indexOf(needle); i !== -1; i = hay.indexOf(needle, i + 1)) out.push(i);
+ return out;
+}
+
+/** Of several hits, the one whose surroundings best match prefix/suffix. Ties: the first. */
+function best(hay, hits, len, prefix, suffix) {
+ let top = hits[0];
+ let topScore = -1;
+ for (const i of hits) {
+ const score =
+ suffixMatch(hay.slice(Math.max(0, i - prefix.length), i), prefix) +
+ prefixMatch(hay.slice(i + len, i + len + suffix.length), suffix);
+ if (score > topScore) {
+ top = i;
+ topScore = score;
+ }
+ }
+ return top;
+}
+
+/**
+ * Whitespace collapsed (and typographic quotes/dashes folded), with a map from
+ * each normalised index back to the original one. `lower` also lowercases.
+ */
+export function normalise(text, { lower = false } = {}) {
+ const chars = [];
+ const map = [];
+ let space = false;
+ for (let i = 0; i < text.length; i += 1) {
+ let c = text[i];
+ if (/\s/.test(c)) {
+ if (space || chars.length === 0) continue;
+ space = true;
+ chars.push(" ");
+ map.push(i);
+ continue;
+ }
+ space = false;
+ if (c === "‘" || c === "’") c = "'";
+ else if (c === "“" || c === "”") c = '"';
+ else if (c === "–" || c === "—") c = "-";
+ else if (c === "…") c = ".";
+ if (lower) c = c.toLowerCase();
+ chars.push(c);
+ map.push(i);
+ }
+ if (chars[chars.length - 1] === " ") {
+ chars.pop();
+ map.pop();
+ }
+ map.push(text.length);
+ return { text: chars.join(""), map };
+}
+
+const norm = (s, lower) => normalise(s ?? "", { lower }).text;
+
+/**
+ * Locate a text anchor. `{ found: true, start, end, how }` with offsets into
+ * `text`, or `{ found: false }` -- an ORPHANED note, still shown, pinned to its
+ * section. `how`: "exact", "normalised" (whitespace/quotes/case differ), or
+ * "context" (the quote itself was edited, but what came before and after it is
+ * still there, close together).
+ *
+ * @param {string} text
+ * @param {{ quote: string, prefix?: string, suffix?: string }} anchor
+ */
+export function locateQuote(text, anchor) {
+ const quote = anchor?.quote ?? "";
+ const prefix = anchor?.prefix ?? "";
+ const suffix = anchor?.suffix ?? "";
+ if (!text || !quote) return { found: false };
+
+ const exact = allIndexes(text, quote);
+ if (exact.length) {
+ const i = best(text, exact, quote.length, prefix, suffix);
+ return { found: true, start: i, end: i + quote.length, how: "exact" };
+ }
+
+ for (const lower of [false, true]) {
+ const n = normalise(text, { lower });
+ const q = norm(quote, lower);
+ if (!q) continue;
+ const hits = allIndexes(n.text, q);
+ if (hits.length) {
+ const i = best(n.text, hits, q.length, norm(prefix, lower), norm(suffix, lower));
+ return { found: true, start: n.map[i], end: n.map[i + q.length - 1] + 1, how: "normalised" };
+ }
+ }
+
+ // The quote was rewritten. If the context on BOTH sides survives, close
+ // together, the span between them is where it was.
+ const n = normalise(text, { lower: true });
+ const p = norm(prefix, true);
+ const s = norm(suffix, true);
+ if (p.length >= 8 && s.length >= 8) {
+ const q = norm(quote, true);
+ for (const pi of allIndexes(n.text, p)) {
+ const from = pi + p.length;
+ const si = n.text.indexOf(s, from);
+ if (si === -1) continue;
+ const span = si - from;
+ if (span <= 0 || span > Math.max(q.length * 2, q.length + 80)) continue;
+ // Trim the separator spaces the normalised text keeps around the span.
+ let a = from;
+ let b = si;
+ while (a < b && n.text[a] === " ") a += 1;
+ while (b > a && n.text[b - 1] === " ") b -= 1;
+ if (a >= b) continue;
+ return { found: true, start: n.map[a], end: n.map[b - 1] + 1, how: "context" };
+ }
+ }
+ return { found: false };
+}
+
+/**
+ * The anchor for a selection `[start, end)` of `text`: the quote (trimmed of
+ * surrounding whitespace) and up to CONTEXT characters either side.
+ *
+ * @param {string} text
+ * @param {number} start
+ * @param {number} end
+ * @param {number} [context]
+ */
+export function quoteAnchor(text, start, end, context = CONTEXT) {
+ let a = Math.max(0, Math.min(start, end));
+ let b = Math.min(text.length, Math.max(start, end));
+ while (a < b && /\s/.test(text[a])) a += 1;
+ while (b > a && /\s/.test(text[b - 1])) b -= 1;
+ return {
+ quote: text.slice(a, b),
+ prefix: text.slice(Math.max(0, a - context), a),
+ suffix: text.slice(b, b + context),
+ };
+}
+
+/** The sentence of `text` around `[start, end)`, for a reader with no page open. */
+export function sentenceAround(text, start, end, max = 400) {
+ const before = text.slice(0, start);
+ const after = text.slice(end);
+ const boundary = /[.?!]\s+|\n/g;
+ let from = 0;
+ for (let m = boundary.exec(before); m; m = boundary.exec(before)) from = m.index + m[0].length;
+ const m = after.search(/[.?!](\s|$)|\n/);
+ const to = m === -1 ? text.length : end + m + (after[m] === "\n" ? 0 : 1);
+ const out = text.slice(from, to).replace(/\s+/g, " ").trim();
+ return out.length > max ? `${out.slice(0, max - 1)}…` : out;
+}
diff --git a/umtool/lib/annotations/anchor.test.mjs b/umtool/lib/annotations/anchor.test.mjs
@@ -0,0 +1,65 @@
+// Re-anchoring a quote in text that changed.
+//
+// Run with: pnpm test:scripts
+import assert from "node:assert/strict";
+import test from "node:test";
+import { locateQuote, normalise, quoteAnchor, sentenceAround } from "./anchor.mjs";
+
+const P1 = "She said the vote was rigged in 2020. Nobody checked the claim at the time.";
+const P2 = "Two years later she said the vote was rigged again, on a different show.";
+const TEXT = `${P1}\n\n${P2}`;
+
+test("an exact quote is found, and the context picks between two copies", () => {
+ const second = TEXT.indexOf("the vote was rigged", P1.length);
+ const a = quoteAnchor(TEXT, second, second + "the vote was rigged".length);
+ assert.equal(a.quote, "the vote was rigged");
+ const r = locateQuote(TEXT, a);
+ assert.deepEqual([r.found, r.start, r.how], [true, second, "exact"]);
+ const first = locateQuote(TEXT, quoteAnchor(TEXT, TEXT.indexOf("the vote"), TEXT.indexOf("the vote") + 19));
+ assert.equal(first.start, TEXT.indexOf("the vote"));
+});
+
+test("a paragraph edited ABOVE the quote does not move the note off it", () => {
+ const start = TEXT.indexOf("on a different show");
+ const a = quoteAnchor(TEXT, start, start + "on a different show".length);
+ const edited = `A new opening paragraph the agent added.\n\n${P1.replace("Nobody checked", "No outlet checked")}\n\n${P2}`;
+ const r = locateQuote(edited, a);
+ assert.equal(r.found, true);
+ assert.equal(edited.slice(r.start, r.end), "on a different show");
+});
+
+test("whitespace, typographic quotes and case still match (normalised)", () => {
+ const a = { quote: "it's the\nclaim", prefix: "", suffix: "" };
+ const t = "And then: It’s the claim, again.";
+ const r = locateQuote(t, a);
+ assert.equal(r.found, true);
+ assert.equal(r.how, "normalised");
+ assert.equal(t.slice(r.start, r.end), "It’s the claim");
+});
+
+test("a rewritten quote is found by its surviving context; a vanished one is orphaned", () => {
+ const start = TEXT.indexOf("Nobody checked the claim");
+ const a = quoteAnchor(TEXT, start, start + "Nobody checked the claim".length);
+ const rewritten = TEXT.replace("Nobody checked the claim", "No outlet verified it");
+ const r = locateQuote(rewritten, a);
+ assert.equal(r.found, true);
+ assert.equal(r.how, "context");
+ assert.equal(rewritten.slice(r.start, r.end), "No outlet verified it");
+
+ const gone = locateQuote("A completely different article.", a);
+ assert.deepEqual(gone, { found: false });
+ assert.deepEqual(locateQuote("", a), { found: false });
+});
+
+test("normalise maps back to the original offsets", () => {
+ const n = normalise(" a \n\n b—c ");
+ assert.equal(n.text, "a b-c");
+ assert.deepEqual(n.map.slice(0, 5), [2, 3, 7, 8, 9]);
+});
+
+test("sentenceAround gives the sentence holding the quote", () => {
+ const s = TEXT.indexOf("Nobody checked");
+ assert.equal(sentenceAround(TEXT, s, s + 6), "Nobody checked the claim at the time.");
+ const f = TEXT.indexOf("Two years");
+ assert.equal(sentenceAround(TEXT, f, f + 3), P2);
+});
diff --git a/umtool/lib/annotations/shape.mjs b/umtool/lib/annotations/shape.mjs
@@ -0,0 +1,302 @@
+// notes.json: its constants, its validation and the edits made to it -- PURE
+// (no node imports), so the store (./store.mjs, node), the CLI and a client
+// component all hold the same rules. ./types.ts gives them types.
+//
+// { "format": "umtool-notes", "version": 1,
+// "subject": { kind: "article", site, report } | { kind: "video-project", project },
+// "source": { draft?, generator?, manifest?, how? }, which file to EDIT
+// "notes": [ { id, status, author, text, at, updatedAt, anchor, replies,
+// resolvedAt?, resolvedBy? } ] }
+//
+// docs/notes.md is the prose.
+
+export const NOTES_FORMAT = "umtool-notes";
+export const NOTES_VERSION = 1;
+export const NOTE_TEXT_LIMIT = 8000;
+export const NOTE_STATUSES = ["open", "resolved", "wontfix"];
+export const NOTE_AUTHORS = ["operator", "agent"];
+export const ANCHOR_KINDS = ["text", "cite", "section", "whole", "moment", "entry", "take", "edit"];
+export const REPORT_BLOCKS = ["title", "subtitle", "summary", "method"];
+
+const ID = /^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/;
+const TAKE_ID = /^[a-z0-9][a-z0-9-]{0,63}$/;
+const NOTE_ID = /^n_[a-z0-9]{4,32}$/;
+const isStr = (v) => typeof v === "string";
+const short = (v, max) => isStr(v) && v.length <= max;
+
+export const isNoteId = (v) => isStr(v) && NOTE_ID.test(v);
+
+/** A fresh note id: time then randomness, base36. */
+export function newNoteId(now = Date.now()) {
+ return `n_${now.toString(36)}${Math.random().toString(36).slice(2, 6).padEnd(4, "0")}`;
+}
+
+/** A moment's rel path: relative, no `..`, no empty segment, no backslash. */
+function relFile(v) {
+ return short(v, 512) && v.length > 0 && !v.startsWith("/") && !/[\\\0]/.test(v) && !v.split("/").some((s) => s === ".." || s === "" || s === ".");
+}
+
+/** A JSON value small enough to keep: an edit's from/to. */
+function smallJson(v) {
+ if (v === undefined) return true;
+ try {
+ return JSON.stringify(v).length <= 8000;
+ } catch {
+ return false;
+ }
+}
+
+const RESOLVED_STR = ["entry", "title", "quote", "channel", "video", "url"];
+
+/**
+ * An anchor as stored, or `{ error }`. Extra keys are dropped; a text anchor's
+ * prefix/suffix default to "".
+ *
+ * @param {unknown} raw
+ * @returns {{ anchor: Record<string, unknown> } | { error: string }}
+ */
+export function validateAnchor(raw) {
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) return { error: "anchor is not an object" };
+ const a = /** @type {Record<string, unknown>} */ (raw);
+ switch (a.kind) {
+ case "whole":
+ return { anchor: { kind: "whole" } };
+ case "section":
+ if (!short(a.section, 128) || !ID.test(a.section)) return { error: "section anchor needs a section id" };
+ return { anchor: { kind: "section", section: a.section } };
+ case "text": {
+ if (!short(a.section, 128) || !ID.test(a.section)) return { error: "text anchor needs a section id" };
+ if (!short(a.quote, 4000) || !a.quote.trim()) return { error: "text anchor needs a quote" };
+ if (a.prefix !== undefined && !short(a.prefix, 256)) return { error: "prefix is not a short string" };
+ if (a.suffix !== undefined && !short(a.suffix, 256)) return { error: "suffix is not a short string" };
+ return { anchor: { kind: "text", section: a.section, quote: a.quote, prefix: a.prefix ?? "", suffix: a.suffix ?? "" } };
+ }
+ case "cite":
+ if (!short(a.cite, 128) || !ID.test(a.cite)) return { error: "cite anchor needs a citation id" };
+ return { anchor: { kind: "cite", cite: a.cite } };
+ case "entry":
+ if (!short(a.entry, 128) || !ID.test(a.entry)) return { error: "entry anchor needs an entry id" };
+ return { anchor: { kind: "entry", entry: a.entry } };
+ case "take":
+ if (!isStr(a.take) || !TAKE_ID.test(a.take)) return { error: "take anchor needs a take id" };
+ return { anchor: { kind: "take", take: a.take } };
+ case "moment": {
+ if (!relFile(a.file)) return { error: "moment anchor needs a relative file" };
+ const t = Number(a.t);
+ if (!Number.isFinite(t) || t < 0) return { error: "moment anchor needs t ≥ 0" };
+ const out = { kind: "moment", file: a.file, t: Number(t.toFixed(2)) };
+ if (a.take !== undefined) {
+ if (!isStr(a.take) || !TAKE_ID.test(a.take)) return { error: "moment take is not a take id" };
+ out.take = a.take;
+ }
+ if (a.entry !== undefined) {
+ if (!short(a.entry, 128) || !ID.test(a.entry)) return { error: "moment entry is not an entry id" };
+ out.entry = a.entry;
+ }
+ if (a.resolved !== undefined) {
+ if (!a.resolved || typeof a.resolved !== "object" || Array.isArray(a.resolved)) return { error: "resolved is not an object" };
+ const r = /** @type {Record<string, unknown>} */ (a.resolved);
+ const res = {};
+ for (const k of RESOLVED_STR) if (short(r[k], 2000)) res[k] = r[k];
+ if (typeof r.sourceT === "number" && Number.isFinite(r.sourceT)) res.sourceT = Number(r.sourceT.toFixed(2));
+ if (r.approx === true) res.approx = true;
+ out.resolved = res;
+ }
+ return { anchor: out };
+ }
+ case "edit": {
+ if (!short(a.field, 128) || !a.field) return { error: "edit anchor needs a field" };
+ if (a.entry !== undefined && (!short(a.entry, 128) || !ID.test(a.entry))) return { error: "edit entry is not an entry id" };
+ if (!smallJson(a.from) || !smallJson(a.to)) return { error: "edit from/to too large" };
+ const out = { kind: "edit", field: a.field, from: a.from ?? null, to: a.to ?? null };
+ if (a.entry !== undefined) out.entry = a.entry;
+ return { anchor: out };
+ }
+ default:
+ return { error: `anchor kind must be one of ${ANCHOR_KINDS.join(", ")}` };
+ }
+}
+
+/** @returns {{ subject: Record<string, string> } | { error: string }} */
+export function validateSubject(raw) {
+ if (!raw || typeof raw !== "object") return { error: "subject is not an object" };
+ const s = /** @type {Record<string, unknown>} */ (raw);
+ if (s.kind === "article" && isStr(s.site) && TAKE_ID.test(s.site) && isStr(s.report) && TAKE_ID.test(s.report)) {
+ return { subject: { kind: "article", site: s.site, report: s.report } };
+ }
+ if (s.kind === "video-project" && short(s.project, 512) && s.project.length > 0) {
+ return { subject: { kind: "video-project", project: s.project } };
+ }
+ return { error: "subject must be an article {site, report} or a video-project {project}" };
+}
+
+function cleanSource(raw) {
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) return undefined;
+ const out = {};
+ for (const k of ["draft", "generator", "manifest", "how"]) if (short(raw[k], 1000) && raw[k]) out[k] = raw[k];
+ return Object.keys(out).length ? out : undefined;
+}
+
+function cleanText(v) {
+ if (!isStr(v)) throw new NoteError("text must be a string");
+ const t = v.replace(/\s+$/, "");
+ if (!t.trim()) throw new NoteError("text is empty");
+ if (t.length > NOTE_TEXT_LIMIT) throw new NoteError(`text is over ${NOTE_TEXT_LIMIT} characters`);
+ return t;
+}
+
+/** A refused edit: the caller's fault, a 400. */
+export class NoteError extends Error {
+ constructor(message) {
+ super(message);
+ this.name = "NoteError";
+ }
+}
+
+/**
+ * A parsed notes.json, checked. `{ doc }` or `{ error }`. A note that is not
+ * the contract's shape is an error for the whole file -- the file is never
+ * "repaired" by dropping it, because the next write would erase it.
+ */
+export function parseNotesDoc(raw) {
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) return { error: "not an object" };
+ if (raw.format !== NOTES_FORMAT) return { error: `format is not ${NOTES_FORMAT}` };
+ if (raw.version !== NOTES_VERSION) return { error: `version ${raw.version} is not ${NOTES_VERSION}` };
+ const subject = validateSubject(raw.subject);
+ if ("error" in subject) return subject;
+ if (!Array.isArray(raw.notes)) return { error: "notes is not a list" };
+ const notes = [];
+ for (const [i, n] of raw.notes.entries()) {
+ const where = `notes[${i}]`;
+ if (!n || typeof n !== "object") return { error: `${where} is not an object` };
+ if (!isNoteId(n.id)) return { error: `${where}.id is not a note id` };
+ if (!NOTE_STATUSES.includes(n.status)) return { error: `${where}.status` };
+ if (!NOTE_AUTHORS.includes(n.author)) return { error: `${where}.author` };
+ if (!isStr(n.text) || !isStr(n.at) || !isStr(n.updatedAt)) return { error: `${where} text/at/updatedAt` };
+ const a = validateAnchor(n.anchor);
+ if ("error" in a) return { error: `${where}.anchor: ${a.error}` };
+ if (!Array.isArray(n.replies)) return { error: `${where}.replies is not a list` };
+ const replies = [];
+ for (const r of n.replies) {
+ if (!r || !NOTE_AUTHORS.includes(r.author) || !isStr(r.text) || !isStr(r.at)) return { error: `${where}.replies` };
+ replies.push({ author: r.author, text: r.text, at: r.at });
+ }
+ const note = { id: n.id, status: n.status, author: n.author, text: n.text, at: n.at, updatedAt: n.updatedAt, anchor: a.anchor, replies };
+ if (isStr(n.resolvedAt)) note.resolvedAt = n.resolvedAt;
+ if (NOTE_AUTHORS.includes(n.resolvedBy)) note.resolvedBy = n.resolvedBy;
+ notes.push(note);
+ }
+ const doc = { format: NOTES_FORMAT, version: NOTES_VERSION, subject: subject.subject };
+ const source = cleanSource(raw.source);
+ if (source) doc.source = source;
+ doc.notes = notes;
+ return { doc };
+}
+
+export function emptyDoc(subject, source) {
+ const doc = { format: NOTES_FORMAT, version: NOTES_VERSION, subject };
+ const s = cleanSource(source);
+ if (s) doc.source = s;
+ doc.notes = [];
+ return doc;
+}
+
+function find(doc, id) {
+ const note = doc.notes.find((n) => n.id === id);
+ if (!note) throw new NoteError(`no note ${id}`);
+ return note;
+}
+
+function setStatus(note, status, by, now) {
+ if (!NOTE_STATUSES.includes(status)) throw new NoteError(`status must be one of ${NOTE_STATUSES.join(", ")}`);
+ note.status = status;
+ note.updatedAt = now;
+ if (status === "open") {
+ delete note.resolvedAt;
+ delete note.resolvedBy;
+ } else {
+ note.resolvedAt = now;
+ note.resolvedBy = by;
+ }
+}
+
+/**
+ * Apply one op to a doc, in place. Returns the note touched (null on delete).
+ * `by` is who is writing: the app stamps "operator", the CLI "agent".
+ *
+ * { op: "add", text, anchor } a new open note
+ * { op: "edit", id, text?, anchor? } rewrite it (its author only)
+ * { op: "status", id, status } open | resolved | wontfix
+ * { op: "reply", id, text, resolve? } a threaded reply, optionally resolving
+ * { op: "delete", id } remove the note
+ * { op: "delete-reply", id, index } remove one reply
+ * { op: "source", source } correct which file to edit
+ *
+ * @param {Record<string, any>} doc
+ * @param {Record<string, any>} op
+ * @param {"operator" | "agent"} by
+ * @param {string} [now]
+ */
+export function applyOp(doc, op, by, now = new Date().toISOString()) {
+ if (!NOTE_AUTHORS.includes(by)) throw new NoteError("author must be operator or agent");
+ switch (op?.op) {
+ case "add": {
+ const a = validateAnchor(op.anchor);
+ if ("error" in a) throw new NoteError(a.error);
+ let id = newNoteId();
+ while (doc.notes.some((n) => n.id === id)) id = newNoteId();
+ const note = { id, status: "open", author: by, text: cleanText(op.text), at: now, updatedAt: now, anchor: a.anchor, replies: [] };
+ doc.notes.push(note);
+ return note;
+ }
+ case "edit": {
+ const note = find(doc, op.id);
+ if (note.author !== by) throw new NoteError(`only the ${note.author} edits this note; reply instead`);
+ if (op.text !== undefined) note.text = cleanText(op.text);
+ if (op.anchor !== undefined) {
+ const a = validateAnchor(op.anchor);
+ if ("error" in a) throw new NoteError(a.error);
+ note.anchor = a.anchor;
+ }
+ note.updatedAt = now;
+ return note;
+ }
+ case "status": {
+ const note = find(doc, op.id);
+ setStatus(note, op.status, by, now);
+ return note;
+ }
+ case "reply": {
+ const note = find(doc, op.id);
+ note.replies.push({ author: by, text: cleanText(op.text), at: now });
+ note.updatedAt = now;
+ if (op.resolve) setStatus(note, "resolved", by, now);
+ return note;
+ }
+ case "delete": {
+ const i = doc.notes.findIndex((n) => n.id === op.id);
+ if (i === -1) throw new NoteError(`no note ${op.id}`);
+ doc.notes.splice(i, 1);
+ return null;
+ }
+ case "delete-reply": {
+ const note = find(doc, op.id);
+ const i = Number(op.index);
+ if (!Number.isInteger(i) || i < 0 || i >= note.replies.length) throw new NoteError("no such reply");
+ if (note.replies[i].author !== by) throw new NoteError(`only the ${note.replies[i].author} deletes that reply`);
+ note.replies.splice(i, 1);
+ note.updatedAt = now;
+ return note;
+ }
+ case "source": {
+ const s = cleanSource(op.source);
+ if (s) doc.source = s;
+ else delete doc.source;
+ return null;
+ }
+ default:
+ throw new NoteError("op must be add, edit, status, reply, delete, delete-reply or source");
+ }
+}
+
+export const openNotes = (doc) => (doc ? doc.notes.filter((n) => n.status === "open") : []);
diff --git a/umtool/lib/annotations/store.mjs b/umtool/lib/annotations/store.mjs
@@ -0,0 +1,155 @@
+// notes.json on disk: read, lock, apply one op, write. Shared by the app's
+// /api/notes and `umtool notes`, so the operator's page and an agent's CLI go
+// through ONE writer with one set of rules (./shape.mjs).
+//
+// Three writers can race on one file -- the page, an agent's `umtool notes
+// reply`, a second agent -- and they are different PROCESSES, so the in-process
+// queues lib/state.ts and lib/report/manifest.mjs use are not enough here:
+//
+// * a LOCKFILE (`notes.json.lock`, created O_EXCL) serialises
+// read-modify-write across processes; one left by a dead writer is stale
+// after 30 s and is taken over;
+// * the write is tmp + rename, so a reader never sees half a file;
+// * every write from the page carries the TOKEN it read (the file's mtime in
+// ns, or "absent"); a stale one is a 409, never a silent overwrite of a
+// reply an agent wrote in between;
+// * a notes.json that exists and does not parse is NEVER overwritten (the
+// takes.mjs rule) -- writing over it would erase every note in it;
+// * the last note deleted deletes the file: an empty notes.json says nothing.
+//
+// WHERE a write may land is the caller's to decide BEFORE it gets here
+// (lib/annotations/targets.mjs: isCorpusNotesFile for an article, a project
+// directory for a video) -- `writeOp` takes the file it is given.
+import { open, readFile, rename, stat, unlink, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { NoteError, applyOp, emptyDoc, parseNotesDoc } from "./shape.mjs";
+
+export { NoteError };
+export const LOCK_STALE_MS = 30_000;
+const LOCK_WAIT_MS = 10_000;
+
+/** A write whose token no longer matches the file: someone wrote in between. */
+export class StaleNotes extends Error {
+ constructor(expected, got) {
+ super(`the notes changed since you read them (${got} vs ${expected})`);
+ this.name = "StaleNotes";
+ this.status = 409;
+ }
+}
+
+/** A notes.json that exists and is not one: refused, never overwritten. */
+export class NotesUnreadable extends Error {
+ constructor(file, why) {
+ super(`${path.basename(file)} does not parse (${why}); not overwriting it`);
+ this.name = "NotesUnreadable";
+ this.status = 409;
+ }
+}
+
+/** The file's identity for a write guard: mtime in ns plus size, or "absent". */
+export async function notesToken(file) {
+ const st = await stat(/* turbopackIgnore: true */ file, { bigint: true }).catch(() => null);
+ return st ? `${st.mtimeNs}-${st.size}` : "absent";
+}
+
+/**
+ * `{ doc, token }` -- doc null when there is no file. `{ error }` too when the
+ * file is there and is not a notes doc (the page shows it; writes refuse).
+ *
+ * @param {string} file
+ */
+export async function readNotes(file) {
+ const token = await notesToken(file);
+ let text;
+ try {
+ text = await readFile(/* turbopackIgnore: true */ file, "utf8");
+ } catch (err) {
+ if (/** @type {NodeJS.ErrnoException} */ (err).code === "ENOENT") return { doc: null, token: "absent" };
+ return { doc: null, token, error: String(/** @type {Error} */ (err).message ?? err) };
+ }
+ let raw;
+ try {
+ raw = JSON.parse(text);
+ } catch (err) {
+ return { doc: null, token, error: `not JSON: ${/** @type {Error} */ (err).message}` };
+ }
+ const r = parseNotesDoc(raw);
+ if ("error" in r) return { doc: null, token, error: r.error };
+ return { doc: r.doc, token };
+}
+
+const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
+
+/**
+ * Run `fn` holding `<file>.lock`. A lock older than LOCK_STALE_MS is a dead
+ * writer's and is removed; otherwise wait (up to 10 s) and retry.
+ *
+ * @template T
+ * @param {string} file
+ * @param {() => Promise<T>} fn
+ * @param {{ waitMs?: number, staleMs?: number }} [opts]
+ * @returns {Promise<T>}
+ */
+export async function withNotesLock(file, fn, { waitMs = LOCK_WAIT_MS, staleMs = LOCK_STALE_MS } = {}) {
+ const lock = `${file}.lock`;
+ const deadline = Date.now() + waitMs;
+ let delay = 15;
+ for (;;) {
+ try {
+ const h = await open(/* turbopackIgnore: true */ lock, "wx");
+ await h.writeFile(`${process.pid} ${new Date().toISOString()}\n`).catch(() => {});
+ await h.close();
+ break;
+ } catch (err) {
+ if (/** @type {NodeJS.ErrnoException} */ (err).code !== "EEXIST") throw err;
+ const st = await stat(/* turbopackIgnore: true */ lock).catch(() => null);
+ if (st && Date.now() - st.mtimeMs > staleMs) {
+ await unlink(/* turbopackIgnore: true */ lock).catch(() => {});
+ continue;
+ }
+ if (Date.now() > deadline) throw new Error(`${path.basename(lock)} is held; try again`);
+ await sleep(delay);
+ delay = Math.min(delay * 2, 250);
+ }
+ }
+ try {
+ return await fn();
+ } finally {
+ await unlink(/* turbopackIgnore: true */ lock).catch(() => {});
+ }
+}
+
+/**
+ * Apply one op to the notes at `file` and write it back. `init` is the
+ * subject (and source) a NEW file starts with; an existing file keeps its own
+ * subject, and its `source` unless the op is `source`. `token` (when given)
+ * must match the file as it is now, or StaleNotes. `by` is "operator" (the
+ * app) or "agent" (the CLI).
+ *
+ * Returns the doc as written (null when the file was deleted), the new token,
+ * and the note the op touched.
+ *
+ * @param {string} file
+ * @param {{ subject: Record<string, string>, source?: Record<string, string> | null }} init
+ * @param {Record<string, any>} op
+ * @param {{ by: "operator" | "agent", token?: string | null }} opts
+ */
+export async function writeOp(file, init, op, { by, token = null }) {
+ return withNotesLock(file, async () => {
+ const current = await readNotes(file);
+ if (current.error) throw new NotesUnreadable(file, current.error);
+ if (token !== null && token !== undefined && token !== current.token) throw new StaleNotes(token, current.token);
+ const doc = current.doc ?? emptyDoc(init.subject, init.source ?? undefined);
+ const note = applyOp(doc, op, by);
+ if (doc.notes.length === 0) {
+ await unlink(/* turbopackIgnore: true */ file).catch((err) => {
+ if (err.code !== "ENOENT") throw err;
+ });
+ return { doc: null, token: "absent", note };
+ }
+ const tmp = `${file}.tmp-${process.pid}-${Math.random().toString(36).slice(2, 8)}`;
+ await writeFile(/* turbopackIgnore: true */ tmp, JSON.stringify(doc, null, 2) + "\n", "utf8");
+ await rename(/* turbopackIgnore: true */ tmp, file);
+ return { doc, token: await notesToken(file), note };
+ });
+}
diff --git a/umtool/lib/annotations/store.test.mjs b/umtool/lib/annotations/store.test.mjs
@@ -0,0 +1,206 @@
+// notes.json: the store, the corpus write predicate and the targets.
+//
+// Run with: pnpm test:scripts
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, readFile, rm, stat, symlink, utimes, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import test from "node:test";
+import { corpusNotesFile, isCorpusNotesFile } from "../paths.mjs";
+import { NoteError, applyOp, emptyDoc, parseNotesDoc, validateAnchor } from "./shape.mjs";
+import { NotesUnreadable, StaleNotes, readNotes, withNotesLock, writeOp } from "./store.mjs";
+import { articleTarget, listNotesFiles, writeNote } from "./targets.mjs";
+
+const SUBJECT = { kind: "article", site: "s1", report: "r1" };
+
+async function tmp() {
+ return mkdtemp(path.join(tmpdir(), "umtool-notes-"));
+}
+
+test("add, reply, resolve, reopen, delete: round-trip, and the last delete removes the file", async () => {
+ const dir = await tmp();
+ const file = path.join(dir, "notes.json");
+ const a = await writeOp(file, { subject: SUBJECT, source: { draft: "~/d.json" } }, { op: "add", text: "fix this", anchor: { kind: "whole" } }, { by: "operator" });
+ assert.equal(a.doc.notes.length, 1);
+ assert.equal(a.doc.source.draft, "~/d.json");
+ const id = a.note.id;
+ assert.match(id, /^n_[a-z0-9]+$/);
+
+ const back = await readNotes(file);
+ assert.deepEqual(back.doc, a.doc);
+ assert.equal(back.token, a.token);
+
+ const r = await writeOp(file, { subject: SUBJECT }, { op: "reply", id, text: "done in drafts/x.json", resolve: true }, { by: "agent", token: a.token });
+ assert.equal(r.note.status, "resolved");
+ assert.equal(r.note.resolvedBy, "agent");
+ assert.equal(r.note.replies[0].author, "agent");
+
+ const o = await writeOp(file, { subject: SUBJECT }, { op: "status", id, status: "open" }, { by: "operator" });
+ assert.equal(o.note.status, "open");
+ assert.equal(o.note.resolvedAt, undefined);
+
+ const d = await writeOp(file, { subject: SUBJECT }, { op: "delete", id }, { by: "operator" });
+ assert.equal(d.doc, null);
+ await assert.rejects(stat(file), /ENOENT/);
+ await rm(dir, { recursive: true });
+});
+
+test("a stale token is a 409 and the file is untouched", async () => {
+ const dir = await tmp();
+ const file = path.join(dir, "notes.json");
+ const a = await writeOp(file, { subject: SUBJECT }, { op: "add", text: "one", anchor: { kind: "whole" } }, { by: "operator", token: "absent" });
+ await writeOp(file, { subject: SUBJECT }, { op: "add", text: "two (agent)", anchor: { kind: "whole" } }, { by: "agent" });
+ const before = await readFile(file, "utf8");
+ await assert.rejects(
+ writeOp(file, { subject: SUBJECT }, { op: "add", text: "three", anchor: { kind: "whole" } }, { by: "operator", token: a.token }),
+ (err) => err instanceof StaleNotes && err.status === 409,
+ );
+ assert.equal(await readFile(file, "utf8"), before);
+ // "absent" against a file that exists is stale too
+ await assert.rejects(
+ writeOp(file, { subject: SUBJECT }, { op: "add", text: "x", anchor: { kind: "whole" } }, { by: "operator", token: "absent" }),
+ StaleNotes,
+ );
+ await rm(dir, { recursive: true });
+});
+
+test("an unparseable notes.json is never overwritten", async () => {
+ const dir = await tmp();
+ const file = path.join(dir, "notes.json");
+ await writeFile(file, "{ not json");
+ assert.match((await readNotes(file)).error, /not JSON/);
+ await assert.rejects(writeOp(file, { subject: SUBJECT }, { op: "add", text: "x", anchor: { kind: "whole" } }, { by: "operator" }), NotesUnreadable);
+ assert.equal(await readFile(file, "utf8"), "{ not json");
+ // a parseable file with a bad note is refused the same way
+ await writeFile(file, JSON.stringify({ format: "umtool-notes", version: 1, subject: SUBJECT, notes: [{ id: "bad" }] }));
+ await assert.rejects(writeOp(file, { subject: SUBJECT }, { op: "add", text: "x", anchor: { kind: "whole" } }, { by: "operator" }), NotesUnreadable);
+ await rm(dir, { recursive: true });
+});
+
+test("lock contention: parallel writers all land; a held lock waits; a stale lock is taken over", async () => {
+ const dir = await tmp();
+ const file = path.join(dir, "notes.json");
+ await Promise.all(
+ Array.from({ length: 12 }, (_, i) =>
+ writeOp(file, { subject: SUBJECT }, { op: "add", text: `n${i}`, anchor: { kind: "whole" } }, { by: i % 2 ? "agent" : "operator" }),
+ ),
+ );
+ assert.equal((await readNotes(file)).doc.notes.length, 12);
+
+ // Another process holds it (a fresh lock file): we wait, then give up.
+ await writeFile(`${file}.lock`, "999999 now\n");
+ await assert.rejects(withNotesLock(file, async () => 1, { waitMs: 120 }), /held/);
+ // The same lock, 31 s old: a dead writer's, taken over.
+ const old = new Date(Date.now() - 31_000);
+ await utimes(`${file}.lock`, old, old);
+ assert.equal(await withNotesLock(file, async () => 2, { waitMs: 120 }), 2);
+ await assert.rejects(stat(`${file}.lock`), /ENOENT/);
+ await rm(dir, { recursive: true });
+});
+
+test("ops refuse what the contract does not allow", () => {
+ const doc = emptyDoc(SUBJECT);
+ assert.throws(() => applyOp(doc, { op: "add", text: " ", anchor: { kind: "whole" } }, "operator"), NoteError);
+ assert.throws(() => applyOp(doc, { op: "add", text: "x", anchor: { kind: "nope" } }, "operator"), NoteError);
+ assert.throws(() => applyOp(doc, { op: "add", text: "x".repeat(8001), anchor: { kind: "whole" } }, "operator"), NoteError);
+ const n = applyOp(doc, { op: "add", text: "x", anchor: { kind: "whole" } }, "operator");
+ assert.throws(() => applyOp(doc, { op: "edit", id: n.id, text: "agent rewrites it" }, "agent"), /reply instead/);
+ assert.throws(() => applyOp(doc, { op: "status", id: n.id, status: "done" }, "agent"), NoteError);
+ assert.throws(() => applyOp(doc, { op: "reply", id: "n_missing0", text: "x" }, "agent"), /no note/);
+ assert.throws(() => applyOp(doc, { op: "frobnicate" }, "agent"), NoteError);
+ assert.throws(() => applyOp(doc, { op: "add", text: "x", anchor: { kind: "whole" } }, "someone"), NoteError);
+ // source: set, then cleared
+ applyOp(doc, { op: "source", source: { draft: "~/x.json", junk: 1 } }, "agent");
+ assert.deepEqual(doc.source, { draft: "~/x.json" });
+ assert.ok(parseNotesDoc(JSON.parse(JSON.stringify(doc))).doc);
+});
+
+test("anchors: every kind validates; bad ones refuse", () => {
+ const ok = [
+ { kind: "whole" },
+ { kind: "section", section: "s-2" },
+ { kind: "text", section: "summary", quote: "the claim", prefix: "before ", suffix: " after" },
+ { kind: "cite", cite: "c12" },
+ { kind: "moment", file: "takes/deck/preview.mp4", t: 12.345, take: "deck", entry: "e3", resolved: { title: "T", sourceT: 81.234, approx: true, bogus: 1 } },
+ { kind: "entry", entry: "clip-4" },
+ { kind: "take", take: "cold-open" },
+ { kind: "edit", entry: "e3", field: "quote", from: "a", to: "b" },
+ ];
+ for (const a of ok) assert.ok("anchor" in validateAnchor(a), JSON.stringify(a));
+ assert.equal(validateAnchor(ok[4]).anchor.t, 12.35);
+ assert.deepEqual(validateAnchor(ok[4]).anchor.resolved, { title: "T", sourceT: 81.23, approx: true });
+ assert.equal(validateAnchor({ kind: "text", section: "s", quote: "q" }).anchor.prefix, "");
+ const bad = [
+ null,
+ { kind: "text", section: "s" },
+ { kind: "moment", file: "../x.mp4", t: 1 },
+ { kind: "moment", file: "/abs.mp4", t: 1 },
+ { kind: "moment", file: "a.mp4", t: -1 },
+ { kind: "take", take: "Bad Id" },
+ { kind: "section", section: "has space" },
+ { kind: "edit", field: "" },
+ ];
+ for (const a of bad) assert.ok("error" in validateAnchor(a), JSON.stringify(a));
+});
+
+async function sitesFixture() {
+ const root = await tmp();
+ const sites = path.join(root, "sites");
+ await mkdir(path.join(sites, "s1", "reports", "r1"), { recursive: true });
+ await writeFile(path.join(sites, "s1", "site.json"), "{}");
+ await writeFile(path.join(sites, "s1", "reports", "r1", "report.json"), "{}");
+ const outside = path.join(root, "outside", "r2");
+ await mkdir(outside, { recursive: true });
+ await symlink(outside, path.join(sites, "s1", "reports", "r2"));
+ return { root, sites };
+}
+
+test("isCorpusNotesFile: exactly sites/<site>/reports/<id>/notes.json, and nothing else", async () => {
+ const { root, sites } = await sitesFixture();
+ const opt = { sitesDir: sites };
+ const good = path.join(sites, "s1", "reports", "r1", "notes.json");
+ assert.equal(corpusNotesFile("s1", "r1", opt), good);
+ assert.equal(await isCorpusNotesFile(good, opt), true);
+ // traversal, spelled several ways
+ assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "reports", "r1", "..", "r1", "notes.json").replace(/\/r1\/notes/, "/../r1/r1/notes"), opt), false);
+ assert.equal(await isCorpusNotesFile(`${sites}/s1/reports/../reports/r1/notes.json`, opt), false);
+ assert.equal(await isCorpusNotesFile(`${sites}/s1/reports//r1/notes.json`, opt), false);
+ assert.equal(await isCorpusNotesFile("s1/reports/r1/notes.json", opt), false);
+ // wrong name, wrong depth, wrong middle segment, bad ids
+ assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "reports", "r1", "report.json"), opt), false);
+ assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "reports", "r1", "x", "notes.json"), opt), false);
+ assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "stills", "r1", "notes.json"), opt), false);
+ assert.equal(await isCorpusNotesFile(path.join(sites, "S1", "reports", "r1", "notes.json"), opt), false);
+ assert.equal(corpusNotesFile("s1", "../r1", opt), null);
+ // a report dir that is a symlink out of the site
+ assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "reports", "r2", "notes.json"), opt), false);
+ // a report that does not exist: a note never creates its directory
+ assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "reports", "r9", "notes.json"), opt), false);
+ // a notes.json that is itself a symlink
+ await writeFile(path.join(root, "elsewhere.json"), "{}");
+ await symlink(path.join(root, "elsewhere.json"), good);
+ assert.equal(await isCorpusNotesFile(good, opt), false);
+ await rm(root, { recursive: true });
+});
+
+test("articleTarget: writes land beside report.json with the discovered source; refusals are typed", async () => {
+ const { root, sites } = await sitesFixture();
+ const reports = path.join(root, "reports");
+ await mkdir(path.join(reports, "ws", "polemics", "drafts"), { recursive: true });
+ await writeFile(path.join(reports, "ws", "polemics", "drafts", "r1.json"), JSON.stringify({ id: "r1" }));
+ await writeFile(path.join(reports, "ws", "polemics", "make-site.py"), 'OUT = "sites/s1/reports"\nfor d in drafts: pass\n');
+ const opt = { sitesDir: sites, reportsRoot: reports };
+
+ const t = await articleTarget("s1/r1", opt);
+ const w = await writeNote(t, { op: "add", text: "x", anchor: { kind: "whole" } }, { by: "operator" });
+ assert.equal(w.doc.source.draft.endsWith(path.join("ws", "polemics", "drafts", "r1.json")), true);
+ assert.equal(w.doc.source.generator.endsWith("make-site.py"), true);
+ const list = await listNotesFiles(opt);
+ assert.deepEqual(list.map((l) => [l.kind, l.id, l.doc.notes.length]), [["article", "s1/r1", 1]]);
+
+ await assert.rejects(articleTarget("s1/r9", opt), (e) => e.status === 404);
+ await assert.rejects(articleTarget("s1/r2", opt), (e) => e.status === 403);
+ await assert.rejects(articleTarget("s1/../r1", opt), (e) => e.status === 400);
+ await assert.rejects(articleTarget("s1", opt), (e) => e.status === 400);
+ await rm(root, { recursive: true });
+});
diff --git a/umtool/lib/annotations/targets.mjs b/umtool/lib/annotations/targets.mjs
@@ -0,0 +1,172 @@
+// What a note is ON, and therefore which notes.json it lives in -- decided
+// here, once, for the app's /api/notes and for `umtool notes`.
+//
+// article `SITES_DIR/<site>/reports/<report>/notes.json`, beside
+// report.json. The generators that write a report dir
+// overwrite only report.json, video.mp4 and poster.jpg, so it
+// survives a regenerate; the compose stage and the report
+// history never read it (common/publish tests hold that), so
+// it is never published. The ONE corpus file umtool writes,
+// and only through isCorpusNotesFile.
+// video-project `<project>/notes.json`, beside video.manifest.json. A
+// report-video project under REPORTS_ROOT, found by the same
+// walk `umtool ls` uses.
+import { readdir, readFile, realpath, stat } from "node:fs/promises";
+import path from "node:path";
+import { NOTES_FILENAME, REPORTS_ROOT, SEGMENT_RE, SITES_DIR, corpusNotesFile, inside, isCorpusNotesFile } from "../paths.mjs";
+import { projectRefs, resolveProject } from "../projects/core.mjs";
+import { sourceFor, tildify } from "../articles/sources.mjs";
+import { readNotes, writeOp } from "./store.mjs";
+
+const exists = (p) => stat(/* turbopackIgnore: true */ p).then(() => true, () => false);
+
+/** A refused target: the caller's fault (400/404). */
+export class TargetError extends Error {
+ constructor(message, status = 400) {
+ super(message);
+ this.name = "TargetError";
+ this.status = status;
+ }
+}
+
+/**
+ * An article target, checked: the site and report ids, the report directory
+ * on disk, and the notes path through isCorpusNotesFile.
+ *
+ * @param {string} spec `<site>/<report>`
+ * @param {{ sitesDir?: string, reportsRoot?: string }} [opts]
+ */
+export async function articleTarget(spec, { sitesDir = SITES_DIR, reportsRoot = REPORTS_ROOT } = {}) {
+ const [site, report, ...rest] = String(spec ?? "").split("/");
+ if (rest.length || !SEGMENT_RE.test(site ?? "") || !SEGMENT_RE.test(report ?? "")) {
+ throw new TargetError(`not an article: ${JSON.stringify(spec)} (want <site>/<report>)`);
+ }
+ const file = corpusNotesFile(site, report, { sitesDir });
+ if (!file || !(await exists(path.dirname(/* turbopackIgnore: true */ file)))) throw new TargetError(`no report ${site}/${report}`, 404);
+ if (!(await isCorpusNotesFile(file, { sitesDir }))) throw new TargetError(`refusing to write ${file}`, 403);
+ return {
+ kind: "article",
+ id: `${site}/${report}`,
+ file,
+ subject: { kind: "article", site, report },
+ // Lazy: the scan reads every workspace's generators, and a read of an
+ // existing file never needs it.
+ source: async () => sourceFor(site, report, { reportsRoot }),
+ };
+}
+
+/**
+ * A video-project target: a report-video project under REPORTS_ROOT, by id,
+ * unique name or directory.
+ *
+ * @param {string} spec
+ * @param {{ reportsRoot?: string }} [opts]
+ */
+export async function projectTarget(spec, { reportsRoot = REPORTS_ROOT } = {}) {
+ const r = await resolveProject(String(spec ?? ""), reportsRoot);
+ if (r.ambiguous) throw new TargetError(`${spec} names ${r.ambiguous.length} projects: ${r.ambiguous.map((p) => p.id).join(", ")}`);
+ const p = r.project;
+ if (!p) throw new TargetError(`no project ${spec}`, 404);
+ if (p.kind !== "report-video") throw new TargetError(`${p.id} is a ${p.kind} project; notes are for report videos`);
+ const [realRoot, realDir] = await Promise.all([realpath(/* turbopackIgnore: true */ reportsRoot).catch(() => null), realpath(/* turbopackIgnore: true */ p.dir).catch(() => null)]);
+ if (!realRoot || !realDir || !inside(realRoot, realDir) || realRoot === realDir) {
+ throw new TargetError(`${p.id} is not under the reports root`, 403);
+ }
+ const file = path.join(/* turbopackIgnore: true */ p.dir, NOTES_FILENAME);
+ return {
+ kind: "video-project",
+ id: p.id,
+ dir: p.dir,
+ file,
+ subject: { kind: "video-project", project: p.id },
+ source: async () => projectSource(p.dir, reportsRoot),
+ };
+}
+
+/**
+ * A report-video project's source: its manifest, and -- when the manifest is
+ * generated -- the generator, resolved against the project's workspace (the
+ * first directory under REPORTS_ROOT) when that file exists.
+ */
+export async function projectSource(dir, reportsRoot = REPORTS_ROOT) {
+ const manifest = path.join(/* turbopackIgnore: true */ dir, "video.manifest.json");
+ const out = { manifest: tildify(manifest) };
+ let generatedBy = null;
+ try {
+ const m = JSON.parse(await readFile(/* turbopackIgnore: true */ manifest, "utf8"));
+ if (typeof m.generatedBy === "string" && m.generatedBy.trim()) generatedBy = m.generatedBy.trim();
+ } catch {
+ // no manifest, or not JSON: the manifest path is still the place to look
+ }
+ if (!generatedBy) {
+ out.how = "hand-edited manifest";
+ return out;
+ }
+ const rel = path.relative(/* turbopackIgnore: true */ reportsRoot, dir);
+ const ws = rel && !rel.startsWith("..") ? path.join(/* turbopackIgnore: true */ reportsRoot, rel.split(path.sep)[0]) : null;
+ const candidates = [ws && path.join(/* turbopackIgnore: true */ ws, generatedBy), path.join(/* turbopackIgnore: true */ dir, generatedBy)].filter(Boolean);
+ let gen = null;
+ for (const c of candidates) {
+ if (await exists(c)) {
+ gen = c;
+ break;
+ }
+ }
+ out.generator = gen ? tildify(gen) : generatedBy;
+ out.how = `manifest is generated by ${generatedBy}; edit its inputs, then regenerate`;
+ return out;
+}
+
+/**
+ * Resolve what the CLI was handed: `<site>/<report>` when that report exists,
+ * else a project.
+ */
+export async function resolveTarget(spec, opts = {}) {
+ const parts = String(spec ?? "").split("/");
+ if (parts.length === 2 && parts.every((s) => SEGMENT_RE.test(s))) {
+ const dir = path.join(/* turbopackIgnore: true */ opts.sitesDir ?? SITES_DIR, parts[0], "reports", parts[1]);
+ if (await exists(dir)) return articleTarget(spec, opts);
+ }
+ return projectTarget(spec, opts);
+}
+
+/**
+ * Every notes.json there is: articles under SITES_DIR, projects under
+ * REPORTS_ROOT. `{ target, file, doc, error? }` each; sorted by id.
+ *
+ * @param {{ sitesDir?: string, reportsRoot?: string }} [opts]
+ */
+export async function listNotesFiles({ sitesDir = SITES_DIR, reportsRoot = REPORTS_ROOT } = {}) {
+ const out = [];
+ for (const site of await readdir(/* turbopackIgnore: true */ sitesDir).catch(() => [])) {
+ if (!SEGMENT_RE.test(site)) continue;
+ for (const report of await readdir(/* turbopackIgnore: true */ path.join(/* turbopackIgnore: true */ sitesDir, site, "reports")).catch(() => [])) {
+ if (!SEGMENT_RE.test(report)) continue;
+ const file = path.join(/* turbopackIgnore: true */ sitesDir, site, "reports", report, NOTES_FILENAME);
+ if (!(await exists(file))) continue;
+ out.push({ kind: "article", id: `${site}/${report}`, file, ...(await readNotes(file)) });
+ }
+ }
+ for (const p of await projectRefs(reportsRoot)) {
+ if (p.kind !== "report-video") continue;
+ const file = path.join(/* turbopackIgnore: true */ p.dir, NOTES_FILENAME);
+ if (!(await exists(file))) continue;
+ out.push({ kind: "video-project", id: p.id, file, ...(await readNotes(file)) });
+ }
+ return out.sort((a, b) => a.kind.localeCompare(b.kind) || a.id.localeCompare(b.id));
+}
+
+/**
+ * One write to a target's notes. A file that does not exist yet starts with
+ * the target's subject and its discovered source (which the agent may correct
+ * later with a `source` op); an existing file keeps both.
+ *
+ * @param {{ file: string, subject: Record<string, string>, source: () => Promise<Record<string, string> | null> }} target
+ * @param {Record<string, any>} op
+ * @param {{ by: "operator" | "agent", token?: string | null }} opts
+ */
+export async function writeNote(target, op, opts) {
+ const now = await readNotes(target.file);
+ const source = now.doc || now.error ? undefined : ((await target.source()) ?? undefined);
+ return writeOp(target.file, { subject: target.subject, source }, op, opts);
+}
diff --git a/umtool/lib/annotations/types.ts b/umtool/lib/annotations/types.ts
@@ -0,0 +1,98 @@
+// The notes shapes, with NO server imports (the lib/note-types.ts rule): a
+// client component may take a value from here without dragging node:fs into
+// the browser bundle. The store that reads and writes them is ./store.mjs,
+// shared by the app and `umtool notes`; docs/notes.md is the prose.
+
+// The values are ./shape.mjs's (pure, shared with the store and the CLI);
+// this file adds the types.
+export {
+ ANCHOR_KINDS,
+ NOTE_AUTHORS,
+ NOTE_STATUSES,
+ NOTE_TEXT_LIMIT,
+ NOTES_FORMAT,
+ NOTES_VERSION,
+ REPORT_BLOCKS,
+} from "./shape.mjs";
+export { CONTEXT as ANCHOR_CONTEXT } from "./anchor.mjs";
+
+export type NoteStatus = "open" | "resolved" | "wontfix";
+export type NoteAuthor = "operator" | "agent";
+
+/** What a moment resolved to when it was written, for the agent reading it back. */
+export type MomentResolved = {
+ entry?: string;
+ title?: string;
+ quote?: string;
+ channel?: string;
+ video?: string;
+ sourceT?: number;
+ url?: string;
+ /** The schedule did not match this file exactly (a preview, not out/). */
+ approx?: boolean;
+};
+
+export type Anchor =
+ | { kind: "text"; section: string; quote: string; prefix: string; suffix: string }
+ | { kind: "cite"; cite: string }
+ | { kind: "section"; section: string }
+ | { kind: "whole" }
+ | { kind: "moment"; file: string; t: number; take?: string; entry?: string; resolved?: MomentResolved }
+ | { kind: "entry"; entry: string }
+ | { kind: "take"; take: string }
+ | { kind: "edit"; entry?: string; field: string; from: unknown; to: unknown };
+
+export type AnchorKind = Anchor["kind"];
+
+export type NoteReply = { author: NoteAuthor; text: string; at: string };
+
+export type Note = {
+ id: string;
+ status: NoteStatus;
+ author: NoteAuthor;
+ text: string;
+ at: string;
+ updatedAt: string;
+ anchor: Anchor;
+ replies: NoteReply[];
+ resolvedAt?: string;
+ resolvedBy?: NoteAuthor;
+};
+
+export type NotesSubject =
+ | { kind: "article"; site: string; report: string }
+ | { kind: "video-project"; project: string };
+
+/** Which file an agent should edit to act on a note. Filled by umtool, correctable. */
+export type NotesSource = { draft?: string; generator?: string; manifest?: string; how?: string };
+
+export type NotesDoc = {
+ format: "umtool-notes";
+ version: 1;
+ subject: NotesSubject;
+ source?: NotesSource;
+ notes: Note[];
+};
+
+/** What GET /api/notes returns. `token` goes back on every write (409 when stale). */
+export type NotesRead = {
+ subject: NotesSubject;
+ file: string;
+ token: string;
+ doc: NotesDoc | null;
+ source: NotesSource | null;
+ error?: string;
+};
+
+/** One write. The server stamps author "operator" on everything the UI sends. */
+export type NoteOp =
+ | { op: "add"; text: string; anchor: Anchor }
+ | { op: "edit"; id: string; text?: string; anchor?: Anchor }
+ | { op: "status"; id: string; status: NoteStatus }
+ | { op: "reply"; id: string; text: string; resolve?: boolean }
+ | { op: "delete"; id: string }
+ | { op: "delete-reply"; id: string; index: number }
+ | { op: "source"; source: NotesSource };
+
+export const openCount = (doc: NotesDoc | null | undefined): number =>
+ doc ? doc.notes.filter((n) => n.status === "open").length : 0;
diff --git a/umtool/lib/articles/sources.mjs b/umtool/lib/articles/sources.mjs
@@ -0,0 +1,219 @@
+// Which file an agent should EDIT to change an article.
+//
+// A report.json under transcripts/sites/ is generated: a workspace under
+// ~/reports keeps the draft (`<ws>/polemics/drafts/<slug>.json`, the source of
+// truth) and a generator script that writes report.json from it
+// (`<ws>/polemics/make-site.py`, `<ws>/site/polemics.py`, ...). A note that
+// says "fix this sentence" is useless to an agent that edits report.json -- the
+// next generator run puts the old sentence back. So every notes.json carries a
+// `source` block naming the draft and the generator, found here.
+//
+// The match is a heuristic, written down with its reason (`how`), and the
+// agent may correct it (`umtool notes source`):
+//
+// draft a drafts/*.json whose `id`, or file name, is the report id --
+// allowing for a `polemic-` prefix on either side (candalyzer's
+// polemic-israel is drafts/israel.json with id polemic-israel;
+// jeralyzer-private's `blame` is drafts/blame.json with id
+// polemic-blame). Several matches: the one whose workspace has a
+// generator naming the site wins; still several, none is chosen.
+// generator a *.py / *.mts under <ws>/polemics or <ws>/site that names the
+// site (or the report id), preferring one that names the report
+// id itself, then one that reads the drafts. Backups
+// (`make-report.pre-2026-10-05.py`: a second dot) are skipped.
+//
+// Cheap: one readdir per workspace and one read per generator, cached for 30 s
+// like the project walk.
+import { readdir, readFile, stat } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { REPORTS_ROOT } from "../paths.mjs";
+
+const CACHE_MS = 30_000;
+const GEN_DIRS = ["polemics", "site"];
+const GEN_EXT = /^[^.]+\.(py|mts|mjs|sh)$/;
+const MAX_GEN_BYTES = 2 * 1024 * 1024;
+
+/** `~/…` for a path under the home directory: what a note shows an agent. */
+export function tildify(abs) {
+ const home = os.homedir();
+ return abs === home || abs.startsWith(home + path.sep) ? `~${abs.slice(home.length)}` : abs;
+}
+
+/** The ids a report or draft may go by: itself, without `polemic-`, with it. */
+export function idKeys(id) {
+ const bare = id.replace(/^polemic-/, "");
+ return new Set([id, bare, `polemic-${bare}`]);
+}
+
+/** Does `text` name `id` as a whole token (so `jasolyzer` is not `jasolyzer-private`)? */
+export function names(text, id) {
+ const esc = id.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
+ return new RegExp(`(?<![A-Za-z0-9_-])${esc}(?![A-Za-z0-9_-])`).test(text);
+}
+
+const isDir = (p) => stat(/* turbopackIgnore: true */ p).then((s) => s.isDirectory(), () => false);
+
+/** @type {Map<string, { at: number, value: Promise<any[]> }>} */
+const cache = new Map();
+
+/**
+ * Every workspace under `reportsRoot` that has drafts or a generator dir:
+ * `{ dir, name, drafts: [{ file, slug, id }], generators: [{ file, text }] }`.
+ *
+ * @param {string} [reportsRoot]
+ */
+export function scanWorkspaces(reportsRoot = REPORTS_ROOT) {
+ const hit = cache.get(reportsRoot);
+ if (hit && Date.now() - hit.at < CACHE_MS) return hit.value;
+ const value = scan(reportsRoot);
+ cache.set(reportsRoot, { at: Date.now(), value });
+ value.catch(() => cache.delete(reportsRoot));
+ return value;
+}
+
+/** Forget the scan (a test that writes a fixture, then reads it). */
+export function clearSourcesCache() {
+ cache.clear();
+}
+
+async function scan(reportsRoot) {
+ const entries = await readdir(/* turbopackIgnore: true */ reportsRoot, { withFileTypes: true }).catch(() => []);
+ const out = [];
+ for (const e of entries) {
+ if (e.name.startsWith(".") || e.name === "data") continue;
+ const dir = path.join(/* turbopackIgnore: true */ reportsRoot, e.name);
+ if (!(e.isDirectory() || (e.isSymbolicLink() && (await isDir(dir))))) continue;
+ const drafts = [];
+ const draftsDir = path.join(/* turbopackIgnore: true */ dir, "polemics", "drafts");
+ for (const f of await readdir(/* turbopackIgnore: true */ draftsDir).catch(() => [])) {
+ if (!f.endsWith(".json")) continue;
+ const file = path.join(/* turbopackIgnore: true */ draftsDir, f);
+ let id = null;
+ try {
+ const j = JSON.parse(await readFile(/* turbopackIgnore: true */ file, "utf8"));
+ if (j && typeof j.id === "string") id = j.id;
+ } catch {
+ // an unreadable draft still matches by its file name
+ }
+ drafts.push({ file, slug: f.slice(0, -5), id });
+ }
+ const generators = [];
+ for (const g of GEN_DIRS) {
+ const gdir = path.join(/* turbopackIgnore: true */ dir, g);
+ for (const f of await readdir(/* turbopackIgnore: true */ gdir).catch(() => [])) {
+ if (!GEN_EXT.test(f)) continue;
+ const file = path.join(/* turbopackIgnore: true */ gdir, f);
+ const st = await stat(/* turbopackIgnore: true */ file).catch(() => null);
+ if (!st?.isFile() || st.size > MAX_GEN_BYTES) continue;
+ generators.push({ file, text: await readFile(/* turbopackIgnore: true */ file, "utf8").catch(() => "") });
+ }
+ }
+ if (drafts.length || generators.length) {
+ drafts.sort((a, b) => a.file.localeCompare(b.file));
+ generators.sort((a, b) => a.file.localeCompare(b.file));
+ out.push({ dir, name: e.name, drafts, generators });
+ }
+ }
+ return out.sort((a, b) => a.name.localeCompare(b.name));
+}
+
+/** Is `draft` this report's? */
+function draftMatches(draft, keys) {
+ return (draft.id !== null && keys.has(draft.id)) || keys.has(draft.slug) || keys.has(`polemic-${draft.slug}`);
+}
+
+/**
+ * The source of truth for one report, or null when no workspace claims it.
+ *
+ * @param {string} siteId
+ * @param {string} reportId
+ * @param {{ reportsRoot?: string }} [opts]
+ * @returns {Promise<{ draft?: string, generator?: string, how: string, workspace?: string } | null>}
+ */
+export async function sourceFor(siteId, reportId, { reportsRoot = REPORTS_ROOT } = {}) {
+ const workspaces = await scanWorkspaces(reportsRoot);
+ const keys = idKeys(reportId);
+ const namesSite = (ws) => ws.generators.some((g) => names(g.text, siteId));
+
+ const cands = [];
+ for (const ws of workspaces) {
+ for (const d of ws.drafts) {
+ if (!draftMatches(d, keys)) continue;
+ const score = (namesSite(ws) ? 4 : 0) + (d.id === reportId ? 2 : 0) + (d.slug === reportId.replace(/^polemic-/, "") ? 1 : 0);
+ cands.push({ ws, d, score });
+ }
+ }
+ cands.sort((a, b) => b.score - a.score);
+ const top = cands[0];
+ const unique = top && (cands.length === 1 || cands[1].score < top.score);
+ const draft = unique ? top : null;
+
+ const pool = draft ? [draft.ws] : workspaces;
+ let gen = null;
+ let genScore = 0;
+ for (const ws of pool) {
+ for (const g of ws.generators) {
+ const site = names(g.text, siteId);
+ const report = names(g.text, reportId);
+ if (!site && !report) continue;
+ const base = path.basename(g.file);
+ const score =
+ (report ? 3 : 0) + (site ? 2 : 0) + (draft && /\bdrafts\b/.test(g.text) ? 1 : 0) + (/^(make-site|polemics)\./.test(base) ? 0.5 : 0);
+ if (score > genScore) {
+ gen = { ws, g };
+ genScore = score;
+ }
+ }
+ }
+ // Without a draft, a generator that only names the SITE is every report's
+ // generator and says nothing about this one; keep it only if it names the id.
+ // A bare id (`deleted`, `poker`) is also an English word, so naming it is
+ // only evidence when the generator names the site too.
+ if (!draft && gen && !(names(gen.g.text, reportId) && (reportId.includes("-") || names(gen.g.text, siteId)))) gen = null;
+ if (!draft && !gen) {
+ if (cands.length > 1) {
+ return { how: `several drafts match ${reportId}: ${cands.map((c) => tildify(c.d.file)).join(", ")}; none chosen` };
+ }
+ return null;
+ }
+
+ const how = [];
+ if (draft) {
+ const by = draft.d.id === reportId ? `id ${reportId}` : draft.d.id && keys.has(draft.d.id) ? `id ${draft.d.id}` : `file name ${draft.d.slug}`;
+ how.push(`draft matched by ${by}`);
+ if (cands.length > 1) how.push(`preferred over ${cands.length - 1} other`);
+ }
+ if (gen) {
+ const named = [siteId, reportId].filter((id) => names(gen.g.text, id));
+ how.push(`generator names ${named.join(" and ")}`);
+ }
+ const out = { how: how.join("; ") };
+ if (draft) out.draft = tildify(draft.d.file);
+ if (gen) out.generator = tildify(gen.g.file);
+ out.workspace = tildify((draft?.ws ?? gen?.ws).dir);
+ return out;
+}
+
+/**
+ * The workspace directories a site's articles come from: every workspace that
+ * holds a matched draft for one of `reportIds`, or a generator that names the
+ * site. Absolute paths, sorted.
+ *
+ * @param {string} siteId
+ * @param {string[]} reportIds
+ * @param {{ reportsRoot?: string }} [opts]
+ */
+export async function siteWorkspaces(siteId, reportIds, { reportsRoot = REPORTS_ROOT } = {}) {
+ const workspaces = await scanWorkspaces(reportsRoot);
+ const dirs = new Set();
+ for (const ws of workspaces) {
+ if (ws.generators.some((g) => names(g.text, siteId))) dirs.add(ws.dir);
+ }
+ for (const id of reportIds) {
+ const s = await sourceFor(siteId, id, { reportsRoot });
+ const ws = s?.draft ? workspaces.find((w) => w.drafts.some((d) => tildify(d.file) === s.draft)) : null;
+ if (ws) dirs.add(ws.dir);
+ }
+ return [...dirs].sort();
+}
diff --git a/umtool/lib/articles/sources.test.mjs b/umtool/lib/articles/sources.test.mjs
@@ -0,0 +1,69 @@
+// Finding an article's draft and generator, on the three workspace layouts
+// the live private sites use.
+//
+// Run with: pnpm test:scripts
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import test from "node:test";
+import { clearSourcesCache, names, siteWorkspaces, sourceFor } from "./sources.mjs";
+
+async function put(file, text) {
+ await mkdir(path.dirname(file), { recursive: true });
+ await writeFile(file, text);
+}
+
+async function fixture() {
+ const root = await mkdtemp(path.join(tmpdir(), "umtool-sources-"));
+ // candalyzer: drafts/<bare>.json with id polemic-<bare>; generator site/polemics.py;
+ // a hand-written fact-check whose generator names it; a backup generator.
+ await put(path.join(root, "candace/polemics/drafts/israel.json"), JSON.stringify({ id: "polemic-israel" }));
+ await put(path.join(root, "candace/site/polemics.py"), 'OUT = ".../sites/candalyzer/reports"\nfor f in DRAFTS.glob("drafts/*.json"): rid = f"polemic-{slug}"\n');
+ await put(path.join(root, "candace/site/make-report.py"), 'SITE = "candalyzer"\nREPORT = "deconstruction-fact-check"\n');
+ await put(path.join(root, "candace/site/make-report.pre-2026-10-05.py"), 'REPORT = "polemic-israel" # candalyzer\n');
+ // jasolyzer-private: drafts/<bare>.json, generator polemics/make-site.py
+ await put(path.join(root, "pirate/polemics/drafts/skg.json"), JSON.stringify({ id: "polemic-skg" }));
+ await put(path.join(root, "pirate/polemics/make-site.py"), 'SITE = "jasolyzer-private"\nrid = f"polemic-{slug}" # from drafts\n');
+ // jeralyzer-private: the report id is BARE (`blame`), the draft id is polemic-blame
+ await put(path.join(root, "quartering/polemics/drafts/blame.json"), JSON.stringify({ id: "polemic-blame" }));
+ await put(path.join(root, "quartering/polemics/make-site.py"), 'SITE = "jeralyzer-private"\nrid = slug # drafts\n');
+ // a decoy: a public site's workspace with a same-named draft
+ await put(path.join(root, "decoy/polemics/drafts/blame.json"), JSON.stringify({ id: "blame" }));
+ await put(path.join(root, "decoy/polemics/make-site.py"), 'SITE = "jeralyzer"\n');
+ clearSourcesCache();
+ return root;
+}
+
+test("each layout finds its draft and generator", async () => {
+ const root = await fixture();
+ const o = { reportsRoot: root };
+ const isr = await sourceFor("candalyzer", "polemic-israel", o);
+ assert.equal(isr.draft, path.join(root, "candace/polemics/drafts/israel.json"));
+ assert.equal(isr.generator, path.join(root, "candace/site/polemics.py"));
+ assert.match(isr.how, /draft matched by id polemic-israel/);
+
+ const fc = await sourceFor("candalyzer", "deconstruction-fact-check", o);
+ assert.equal(fc.draft, undefined);
+ assert.equal(fc.generator, path.join(root, "candace/site/make-report.py"));
+
+ const skg = await sourceFor("jasolyzer-private", "polemic-skg", o);
+ assert.equal(skg.draft, path.join(root, "pirate/polemics/drafts/skg.json"));
+ assert.equal(skg.generator, path.join(root, "pirate/polemics/make-site.py"));
+
+ // The decoy's draft id is exactly `blame`, but its workspace never names the site.
+ const blame = await sourceFor("jeralyzer-private", "blame", o);
+ assert.equal(blame.draft, path.join(root, "quartering/polemics/drafts/blame.json"));
+ assert.equal(blame.generator, path.join(root, "quartering/polemics/make-site.py"));
+ assert.match(blame.how, /preferred over 1 other/);
+
+ assert.equal(await sourceFor("candalyzer", "no-such-report", o), null);
+ assert.deepEqual(await siteWorkspaces("jasolyzer-private", ["polemic-skg"], o), [path.join(root, "pirate")]);
+ await rm(root, { recursive: true });
+});
+
+test("names() matches whole ids only", () => {
+ assert.equal(names('"jasolyzer-private"', "jasolyzer"), false);
+ assert.equal(names("sites/jasolyzer/reports", "jasolyzer"), true);
+ assert.equal(names("polemic-blame", "blame"), false);
+});
diff --git a/umtool/lib/paths.mjs b/umtool/lib/paths.mjs
@@ -5,6 +5,7 @@
// `umtool ls` and the page it is supposed to describe. lib/paths.ts re-exports
// everything here with types; nothing computes a root twice.
import { existsSync } from "node:fs";
+import { lstat, realpath } from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { SONG_DATA, SONG_REPORTS } from "../song/paths.mjs";
@@ -208,12 +209,74 @@ export const CHANNELS_DIR = path.resolve(
: path.join(/* turbopackIgnore: true */ REPO_ROOT, "transcripts", "channels")),
);
+// The SITES -- `transcripts/sites/<site>/`, each a site.json and its reports
+// (`reports/<id>/report.json`, `video.mp4`, `poster.jpg`). Resolved the way
+// CHANNELS_DIR is, and the way common/lib/paths.ts resolves its sitesDir, so
+// one `SITES_DIR` confines both. Readable only; the ONE file in it umtool may
+// write is a report's notes.json, and only through isCorpusNotesFile below.
+export const SITES_DIR = path.resolve(
+ /* turbopackIgnore: true */
+ process.env.SITES_DIR ??
+ (process.env.TRANSCRIPTS_DIR
+ ? path.join(/* turbopackIgnore: true */ process.env.TRANSCRIPTS_DIR, "sites")
+ : path.join(/* turbopackIgnore: true */ REPO_ROOT, "transcripts", "sites")),
+);
+
export const READ_ROOTS = dedupe(
process.env.MIX_ROOTS
? process.env.MIX_ROOTS.split(":").filter(Boolean)
- : [SONG_REPORTS, REPORTS_ROOT, SONG_DATA, SONG_SCRATCH, CHANNELS_DIR, MEDIA_ROOT],
+ : [SONG_REPORTS, REPORTS_ROOT, SONG_DATA, SONG_SCRATCH, CHANNELS_DIR, MEDIA_ROOT, SITES_DIR],
);
+/** A site id or a report id: one lowercase url-safe segment (common/lib/report/schema.ts REPORT_ID_RE). */
+export const SEGMENT_RE = /^[a-z0-9][a-z0-9-]{0,63}$/;
+export const NOTES_FILENAME = "notes.json";
+
+/**
+ * The ONE corpus path umtool may write: `SITES_DIR/<site>/reports/<id>/notes.json`.
+ *
+ * Lexically: exactly four segments under SITES_DIR, the second `reports`, the
+ * site and report ids in the report-id grammar, the file named notes.json, and
+ * no `..` or doubled separator anywhere. Then on disk: the report directory's
+ * REAL path must be the lexical one under the real SITES_DIR -- a report dir
+ * that is a symlink out of its site (or a site dir that is one) is refused --
+ * and it must already exist: a note never creates a report directory. A
+ * notes.json that exists and is not a plain file (a symlink) is refused too.
+ *
+ * WRITE_ROOTS is untouched: nothing else in the corpus becomes writable.
+ *
+ * @param {string} abs
+ * @param {{ sitesDir?: string }} [opts]
+ * @returns {Promise<boolean>}
+ */
+export async function isCorpusNotesFile(abs, { sitesDir = SITES_DIR } = {}) {
+ if (typeof abs !== "string" || !path.isAbsolute(abs) || abs.includes("\0")) return false;
+ if (path.resolve(/* turbopackIgnore: true */ abs) !== abs) return false;
+ const root = path.resolve(/* turbopackIgnore: true */ sitesDir);
+ const rel = path.relative(/* turbopackIgnore: true */ root, abs);
+ if (!rel || rel.startsWith("..") || path.isAbsolute(rel)) return false;
+ const parts = rel.split(path.sep);
+ if (parts.length !== 4) return false;
+ const [site, reports, report, name] = parts;
+ if (!SEGMENT_RE.test(site) || reports !== "reports" || !SEGMENT_RE.test(report) || name !== NOTES_FILENAME) {
+ return false;
+ }
+ const [realRoot, realDir] = await Promise.all([
+ realpath(/* turbopackIgnore: true */ root).catch(() => null),
+ realpath(/* turbopackIgnore: true */ path.dirname(/* turbopackIgnore: true */ abs)).catch(() => null),
+ ]);
+ if (!realRoot || !realDir) return false;
+ if (realDir !== path.join(/* turbopackIgnore: true */ realRoot, site, "reports", report)) return false;
+ const st = await lstat(/* turbopackIgnore: true */ abs).catch(() => null);
+ return !st || st.isFile();
+}
+
+/** `SITES_DIR/<site>/reports/<report>/notes.json`, or null for a bad id. */
+export function corpusNotesFile(site, report, { sitesDir = SITES_DIR } = {}) {
+ if (!SEGMENT_RE.test(String(site)) || !SEGMENT_RE.test(String(report))) return null;
+ return path.join(/* turbopackIgnore: true */ path.resolve(/* turbopackIgnore: true */ sitesDir), site, "reports", report, NOTES_FILENAME);
+}
+
export const WRITE_ROOTS = dedupe(
process.env.MIX_WRITE_ROOTS
? process.env.MIX_WRITE_ROOTS.split(":").filter(Boolean)
diff --git a/umtool/lib/paths.ts b/umtool/lib/paths.ts
@@ -20,7 +20,9 @@ import path from "node:path";
// ---------------------------------------------------------------------------
export {
CACHE_DIR,
+ CHANNELS_DIR,
cacheFile,
+ corpusNotesFile,
INDEX_DIR,
MEDIA_ROOT,
MEDIA_ROOTS,
@@ -31,8 +33,10 @@ export {
SONG_DATA,
SONG_REPORTS,
SONG_SCRATCH,
+ SITES_DIR,
WRITE_ROOTS,
inside,
+ isCorpusNotesFile,
labelFor,
mediaMirror,
resolveInRoots,