Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 44787dc1489f30ba075958844f095a9a046573e3
parent 5ebf4015874adcb45426c9f31ee3c20903afa6c1
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu,  8 Oct 2026 22:26:10 -0400

umtool S0: notes store, SITES_DIR, corpus-notes predicate, source discovery

- lib/paths.mjs: SITES_DIR (read-only root) and isCorpusNotesFile, the one
  corpus path umtool may write (sites/<site>/reports/<id>/notes.json)
- lib/annotations: shape (pure contract + ops), anchor (TextQuote re-anchoring),
  store (cross-process lockfile, mtime token, never overwrite unparseable),
  targets (article / video-project -> notes file + source)
- lib/articles/sources.mjs: draft + generator discovery for a report
- common listReportDirs; editor's report list uses it

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/publish/reportMedia.ts | 20+++++++++++++++++++-
Meditor/app/sites/lib/reportListServer.ts | 23+++--------------------
Mpackage.json | 2+-
Aumtool/lib/annotations/anchor.mjs | 176+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/annotations/anchor.test.mjs | 65+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/annotations/shape.mjs | 302++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/annotations/store.mjs | 155+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/annotations/store.test.mjs | 206+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/annotations/targets.mjs | 172+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/annotations/types.ts | 98+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/articles/sources.mjs | 219+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/articles/sources.test.mjs | 69+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/lib/paths.mjs | 65++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mumtool/lib/paths.ts | 4++++
14 files changed, 1553 insertions(+), 23 deletions(-)

diff --git a/common/publish/reportMedia.ts b/common/publish/reportMedia.ts @@ -42,6 +42,7 @@ // a channel whose media tier is not reachable (lib/channelMedia.ts) is // reported as unreachable, with the reason, rather than as missing. +import type { Dirent } from "node:fs"; import { readdir, rm } from "node:fs/promises"; import path from "node:path"; import { readJsonFile, writeJsonAtomic } from "../lib/jsonFile-server"; @@ -55,7 +56,7 @@ import { readChannelConfig } from "../controller/channels"; import { momentKey, momentOf, momentProblem, type Moment } from "../lib/citations/moments"; import type { CitationPad } from "../lib/citations/schema"; import { parseReport } from "../lib/report/validate"; -import type { Report } from "../lib/report/schema"; +import { isReportId, type Report } from "../lib/report/schema"; import { isPermanentlyGone } from "../lib/availability"; import { loadAvailability } from "../lib/availability-server"; import { MAX_CLIP_WINDOW_SECONDS } from "../lib/clipWindow"; @@ -93,6 +94,23 @@ export function siteReportFile(paths: Paths, siteId: string, reportId: string): return path.join(siteReportDir(paths, siteId, reportId), "report.json"); } +// Every report directory of a site, published or draft: a directory under +// `sites/<siteId>/reports/` whose name is a report id. Anything else there (a +// stray file, a bad name) is not a report. Sorted. Whether one is PUBLISHED is +// the site's `reports` list (site.json), not anything on disk. +export async function listReportDirs(paths: Paths, siteId: string): Promise<string[]> { + let entries: Dirent[]; + try { + entries = await readdir(path.join(siteDir(paths, siteId), "reports"), { withFileTypes: true }); + } catch { + return []; + } + return entries + .filter((e) => e.isDirectory() && isReportId(e.name)) + .map((e) => e.name) + .sort((a, b) => a.localeCompare(b)); +} + export type ReportMediaEntry = | EvidenceMedia | { diff --git a/editor/app/sites/lib/reportListServer.ts b/editor/app/sites/lib/reportListServer.ts @@ -1,12 +1,8 @@ import "server-only"; -import type { Dirent } from "node:fs"; -import { readdir } from "node:fs/promises"; -import path from "node:path"; import type { Paths } from "yt-dlp-transcript-common/lib/paths"; import { readJsonFile } from "yt-dlp-transcript-common/lib/jsonFile-server"; -import { isReportId } from "yt-dlp-transcript-common/lib/report/schema"; -import { siteDir, type Site } from "yt-dlp-transcript-common/lib/site"; -import { siteReportFile } from "yt-dlp-transcript-common/publish/reportMedia"; +import type { Site } from "yt-dlp-transcript-common/lib/site"; +import { listReportDirs, siteReportFile } from "yt-dlp-transcript-common/publish/reportMedia"; import { publishableReportExports, readReportExportManifest, @@ -18,24 +14,11 @@ import { listAllJobs, type JobListEntry } from "yt-dlp-transcript-common/jobs/li import { readJobMeta } from "yt-dlp-transcript-common/jobs/jobMeta"; import { reportRows, type ReportFileRead, type ReportRow } from "./reportList"; -// The report directories under `sites/<siteId>/reports/`: a directory whose -// name is a report id. Anything else there (a stray file, a bad name) is not a -// report and is not listed. -async function reportDirIds(paths: Paths, siteId: string): Promise<string[]> { - let entries: Dirent[]; - try { - entries = await readdir(path.join(siteDir(paths, siteId), "reports"), { withFileTypes: true }); - } catch { - return []; - } - return entries.filter((e) => e.isDirectory() && isReportId(e.name)).map((e) => e.name); -} - // Every report of the site, published (in order) then drafts, each read and // validated. export async function readSiteReportRows(paths: Paths, site: Site): Promise<ReportRow[]> { const published = site.reports ?? []; - const dirIds = await reportDirIds(paths, site.siteId); + const dirIds = await listReportDirs(paths, site.siteId); const ids = [...new Set([...published, ...dirIds])]; const reads = new Map<string, ReportFileRead>( await Promise.all( diff --git a/package.json b/package.json @@ -22,7 +22,7 @@ "e2e": "node scripts/worktree.mjs run -- pnpm --filter editor run e2e", "wt": "node scripts/worktree.mjs", "e2e:sharded": "node scripts/run-sharded-e2e.mjs", - "test:scripts": "node --test scripts/*.test.mjs umtool/report-to-video/*.test.mjs umtool/lib/report/*.test.mjs", + "test:scripts": "node --test scripts/*.test.mjs umtool/report-to-video/*.test.mjs umtool/lib/report/*.test.mjs umtool/lib/annotations/*.test.mjs umtool/lib/articles/*.test.mjs", "lint": "pnpm --filter export run lint", "ops": "node scripts/archilyzer-ops.mjs" }, diff --git a/umtool/lib/annotations/anchor.mjs b/umtool/lib/annotations/anchor.mjs @@ -0,0 +1,176 @@ +// Finding a text anchor again in text that may have changed. +// +// A note on an article is anchored by its QUOTE plus 32 characters of context +// either side (the W3C TextQuoteSelector), never by an offset: report.json is +// regenerated from a draft, and an offset into the old text points at nothing +// after the agent edits the paragraph above it. +// +// PURE and client-safe: no node imports. The article page runs it against the +// rendered DOM text of a section; `umtool notes` runs it against the section's +// plain text; the unit test runs it against both. + +export const CONTEXT = 32; + +/** Common-suffix length of `a` and `b` (how much of the prefix still precedes). */ +function suffixMatch(a, b) { + let n = 0; + while (n < a.length && n < b.length && a[a.length - 1 - n] === b[b.length - 1 - n]) n += 1; + return n; +} +/** Common-prefix length of `a` and `b` (how much of the suffix still follows). */ +function prefixMatch(a, b) { + let n = 0; + while (n < a.length && n < b.length && a[n] === b[n]) n += 1; + return n; +} + +function allIndexes(hay, needle) { + const out = []; + if (!needle) return out; + for (let i = hay.indexOf(needle); i !== -1; i = hay.indexOf(needle, i + 1)) out.push(i); + return out; +} + +/** Of several hits, the one whose surroundings best match prefix/suffix. Ties: the first. */ +function best(hay, hits, len, prefix, suffix) { + let top = hits[0]; + let topScore = -1; + for (const i of hits) { + const score = + suffixMatch(hay.slice(Math.max(0, i - prefix.length), i), prefix) + + prefixMatch(hay.slice(i + len, i + len + suffix.length), suffix); + if (score > topScore) { + top = i; + topScore = score; + } + } + return top; +} + +/** + * Whitespace collapsed (and typographic quotes/dashes folded), with a map from + * each normalised index back to the original one. `lower` also lowercases. + */ +export function normalise(text, { lower = false } = {}) { + const chars = []; + const map = []; + let space = false; + for (let i = 0; i < text.length; i += 1) { + let c = text[i]; + if (/\s/.test(c)) { + if (space || chars.length === 0) continue; + space = true; + chars.push(" "); + map.push(i); + continue; + } + space = false; + if (c === "‘" || c === "’") c = "'"; + else if (c === "“" || c === "”") c = '"'; + else if (c === "–" || c === "—") c = "-"; + else if (c === "…") c = "."; + if (lower) c = c.toLowerCase(); + chars.push(c); + map.push(i); + } + if (chars[chars.length - 1] === " ") { + chars.pop(); + map.pop(); + } + map.push(text.length); + return { text: chars.join(""), map }; +} + +const norm = (s, lower) => normalise(s ?? "", { lower }).text; + +/** + * Locate a text anchor. `{ found: true, start, end, how }` with offsets into + * `text`, or `{ found: false }` -- an ORPHANED note, still shown, pinned to its + * section. `how`: "exact", "normalised" (whitespace/quotes/case differ), or + * "context" (the quote itself was edited, but what came before and after it is + * still there, close together). + * + * @param {string} text + * @param {{ quote: string, prefix?: string, suffix?: string }} anchor + */ +export function locateQuote(text, anchor) { + const quote = anchor?.quote ?? ""; + const prefix = anchor?.prefix ?? ""; + const suffix = anchor?.suffix ?? ""; + if (!text || !quote) return { found: false }; + + const exact = allIndexes(text, quote); + if (exact.length) { + const i = best(text, exact, quote.length, prefix, suffix); + return { found: true, start: i, end: i + quote.length, how: "exact" }; + } + + for (const lower of [false, true]) { + const n = normalise(text, { lower }); + const q = norm(quote, lower); + if (!q) continue; + const hits = allIndexes(n.text, q); + if (hits.length) { + const i = best(n.text, hits, q.length, norm(prefix, lower), norm(suffix, lower)); + return { found: true, start: n.map[i], end: n.map[i + q.length - 1] + 1, how: "normalised" }; + } + } + + // The quote was rewritten. If the context on BOTH sides survives, close + // together, the span between them is where it was. + const n = normalise(text, { lower: true }); + const p = norm(prefix, true); + const s = norm(suffix, true); + if (p.length >= 8 && s.length >= 8) { + const q = norm(quote, true); + for (const pi of allIndexes(n.text, p)) { + const from = pi + p.length; + const si = n.text.indexOf(s, from); + if (si === -1) continue; + const span = si - from; + if (span <= 0 || span > Math.max(q.length * 2, q.length + 80)) continue; + // Trim the separator spaces the normalised text keeps around the span. + let a = from; + let b = si; + while (a < b && n.text[a] === " ") a += 1; + while (b > a && n.text[b - 1] === " ") b -= 1; + if (a >= b) continue; + return { found: true, start: n.map[a], end: n.map[b - 1] + 1, how: "context" }; + } + } + return { found: false }; +} + +/** + * The anchor for a selection `[start, end)` of `text`: the quote (trimmed of + * surrounding whitespace) and up to CONTEXT characters either side. + * + * @param {string} text + * @param {number} start + * @param {number} end + * @param {number} [context] + */ +export function quoteAnchor(text, start, end, context = CONTEXT) { + let a = Math.max(0, Math.min(start, end)); + let b = Math.min(text.length, Math.max(start, end)); + while (a < b && /\s/.test(text[a])) a += 1; + while (b > a && /\s/.test(text[b - 1])) b -= 1; + return { + quote: text.slice(a, b), + prefix: text.slice(Math.max(0, a - context), a), + suffix: text.slice(b, b + context), + }; +} + +/** The sentence of `text` around `[start, end)`, for a reader with no page open. */ +export function sentenceAround(text, start, end, max = 400) { + const before = text.slice(0, start); + const after = text.slice(end); + const boundary = /[.?!]\s+|\n/g; + let from = 0; + for (let m = boundary.exec(before); m; m = boundary.exec(before)) from = m.index + m[0].length; + const m = after.search(/[.?!](\s|$)|\n/); + const to = m === -1 ? text.length : end + m + (after[m] === "\n" ? 0 : 1); + const out = text.slice(from, to).replace(/\s+/g, " ").trim(); + return out.length > max ? `${out.slice(0, max - 1)}…` : out; +} diff --git a/umtool/lib/annotations/anchor.test.mjs b/umtool/lib/annotations/anchor.test.mjs @@ -0,0 +1,65 @@ +// Re-anchoring a quote in text that changed. +// +// Run with: pnpm test:scripts +import assert from "node:assert/strict"; +import test from "node:test"; +import { locateQuote, normalise, quoteAnchor, sentenceAround } from "./anchor.mjs"; + +const P1 = "She said the vote was rigged in 2020. Nobody checked the claim at the time."; +const P2 = "Two years later she said the vote was rigged again, on a different show."; +const TEXT = `${P1}\n\n${P2}`; + +test("an exact quote is found, and the context picks between two copies", () => { + const second = TEXT.indexOf("the vote was rigged", P1.length); + const a = quoteAnchor(TEXT, second, second + "the vote was rigged".length); + assert.equal(a.quote, "the vote was rigged"); + const r = locateQuote(TEXT, a); + assert.deepEqual([r.found, r.start, r.how], [true, second, "exact"]); + const first = locateQuote(TEXT, quoteAnchor(TEXT, TEXT.indexOf("the vote"), TEXT.indexOf("the vote") + 19)); + assert.equal(first.start, TEXT.indexOf("the vote")); +}); + +test("a paragraph edited ABOVE the quote does not move the note off it", () => { + const start = TEXT.indexOf("on a different show"); + const a = quoteAnchor(TEXT, start, start + "on a different show".length); + const edited = `A new opening paragraph the agent added.\n\n${P1.replace("Nobody checked", "No outlet checked")}\n\n${P2}`; + const r = locateQuote(edited, a); + assert.equal(r.found, true); + assert.equal(edited.slice(r.start, r.end), "on a different show"); +}); + +test("whitespace, typographic quotes and case still match (normalised)", () => { + const a = { quote: "it's the\nclaim", prefix: "", suffix: "" }; + const t = "And then: It’s the claim, again."; + const r = locateQuote(t, a); + assert.equal(r.found, true); + assert.equal(r.how, "normalised"); + assert.equal(t.slice(r.start, r.end), "It’s the claim"); +}); + +test("a rewritten quote is found by its surviving context; a vanished one is orphaned", () => { + const start = TEXT.indexOf("Nobody checked the claim"); + const a = quoteAnchor(TEXT, start, start + "Nobody checked the claim".length); + const rewritten = TEXT.replace("Nobody checked the claim", "No outlet verified it"); + const r = locateQuote(rewritten, a); + assert.equal(r.found, true); + assert.equal(r.how, "context"); + assert.equal(rewritten.slice(r.start, r.end), "No outlet verified it"); + + const gone = locateQuote("A completely different article.", a); + assert.deepEqual(gone, { found: false }); + assert.deepEqual(locateQuote("", a), { found: false }); +}); + +test("normalise maps back to the original offsets", () => { + const n = normalise(" a \n\n b—c "); + assert.equal(n.text, "a b-c"); + assert.deepEqual(n.map.slice(0, 5), [2, 3, 7, 8, 9]); +}); + +test("sentenceAround gives the sentence holding the quote", () => { + const s = TEXT.indexOf("Nobody checked"); + assert.equal(sentenceAround(TEXT, s, s + 6), "Nobody checked the claim at the time."); + const f = TEXT.indexOf("Two years"); + assert.equal(sentenceAround(TEXT, f, f + 3), P2); +}); diff --git a/umtool/lib/annotations/shape.mjs b/umtool/lib/annotations/shape.mjs @@ -0,0 +1,302 @@ +// notes.json: its constants, its validation and the edits made to it -- PURE +// (no node imports), so the store (./store.mjs, node), the CLI and a client +// component all hold the same rules. ./types.ts gives them types. +// +// { "format": "umtool-notes", "version": 1, +// "subject": { kind: "article", site, report } | { kind: "video-project", project }, +// "source": { draft?, generator?, manifest?, how? }, which file to EDIT +// "notes": [ { id, status, author, text, at, updatedAt, anchor, replies, +// resolvedAt?, resolvedBy? } ] } +// +// docs/notes.md is the prose. + +export const NOTES_FORMAT = "umtool-notes"; +export const NOTES_VERSION = 1; +export const NOTE_TEXT_LIMIT = 8000; +export const NOTE_STATUSES = ["open", "resolved", "wontfix"]; +export const NOTE_AUTHORS = ["operator", "agent"]; +export const ANCHOR_KINDS = ["text", "cite", "section", "whole", "moment", "entry", "take", "edit"]; +export const REPORT_BLOCKS = ["title", "subtitle", "summary", "method"]; + +const ID = /^[A-Za-z0-9][A-Za-z0-9._:-]{0,127}$/; +const TAKE_ID = /^[a-z0-9][a-z0-9-]{0,63}$/; +const NOTE_ID = /^n_[a-z0-9]{4,32}$/; +const isStr = (v) => typeof v === "string"; +const short = (v, max) => isStr(v) && v.length <= max; + +export const isNoteId = (v) => isStr(v) && NOTE_ID.test(v); + +/** A fresh note id: time then randomness, base36. */ +export function newNoteId(now = Date.now()) { + return `n_${now.toString(36)}${Math.random().toString(36).slice(2, 6).padEnd(4, "0")}`; +} + +/** A moment's rel path: relative, no `..`, no empty segment, no backslash. */ +function relFile(v) { + return short(v, 512) && v.length > 0 && !v.startsWith("/") && !/[\\\0]/.test(v) && !v.split("/").some((s) => s === ".." || s === "" || s === "."); +} + +/** A JSON value small enough to keep: an edit's from/to. */ +function smallJson(v) { + if (v === undefined) return true; + try { + return JSON.stringify(v).length <= 8000; + } catch { + return false; + } +} + +const RESOLVED_STR = ["entry", "title", "quote", "channel", "video", "url"]; + +/** + * An anchor as stored, or `{ error }`. Extra keys are dropped; a text anchor's + * prefix/suffix default to "". + * + * @param {unknown} raw + * @returns {{ anchor: Record<string, unknown> } | { error: string }} + */ +export function validateAnchor(raw) { + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return { error: "anchor is not an object" }; + const a = /** @type {Record<string, unknown>} */ (raw); + switch (a.kind) { + case "whole": + return { anchor: { kind: "whole" } }; + case "section": + if (!short(a.section, 128) || !ID.test(a.section)) return { error: "section anchor needs a section id" }; + return { anchor: { kind: "section", section: a.section } }; + case "text": { + if (!short(a.section, 128) || !ID.test(a.section)) return { error: "text anchor needs a section id" }; + if (!short(a.quote, 4000) || !a.quote.trim()) return { error: "text anchor needs a quote" }; + if (a.prefix !== undefined && !short(a.prefix, 256)) return { error: "prefix is not a short string" }; + if (a.suffix !== undefined && !short(a.suffix, 256)) return { error: "suffix is not a short string" }; + return { anchor: { kind: "text", section: a.section, quote: a.quote, prefix: a.prefix ?? "", suffix: a.suffix ?? "" } }; + } + case "cite": + if (!short(a.cite, 128) || !ID.test(a.cite)) return { error: "cite anchor needs a citation id" }; + return { anchor: { kind: "cite", cite: a.cite } }; + case "entry": + if (!short(a.entry, 128) || !ID.test(a.entry)) return { error: "entry anchor needs an entry id" }; + return { anchor: { kind: "entry", entry: a.entry } }; + case "take": + if (!isStr(a.take) || !TAKE_ID.test(a.take)) return { error: "take anchor needs a take id" }; + return { anchor: { kind: "take", take: a.take } }; + case "moment": { + if (!relFile(a.file)) return { error: "moment anchor needs a relative file" }; + const t = Number(a.t); + if (!Number.isFinite(t) || t < 0) return { error: "moment anchor needs t ≥ 0" }; + const out = { kind: "moment", file: a.file, t: Number(t.toFixed(2)) }; + if (a.take !== undefined) { + if (!isStr(a.take) || !TAKE_ID.test(a.take)) return { error: "moment take is not a take id" }; + out.take = a.take; + } + if (a.entry !== undefined) { + if (!short(a.entry, 128) || !ID.test(a.entry)) return { error: "moment entry is not an entry id" }; + out.entry = a.entry; + } + if (a.resolved !== undefined) { + if (!a.resolved || typeof a.resolved !== "object" || Array.isArray(a.resolved)) return { error: "resolved is not an object" }; + const r = /** @type {Record<string, unknown>} */ (a.resolved); + const res = {}; + for (const k of RESOLVED_STR) if (short(r[k], 2000)) res[k] = r[k]; + if (typeof r.sourceT === "number" && Number.isFinite(r.sourceT)) res.sourceT = Number(r.sourceT.toFixed(2)); + if (r.approx === true) res.approx = true; + out.resolved = res; + } + return { anchor: out }; + } + case "edit": { + if (!short(a.field, 128) || !a.field) return { error: "edit anchor needs a field" }; + if (a.entry !== undefined && (!short(a.entry, 128) || !ID.test(a.entry))) return { error: "edit entry is not an entry id" }; + if (!smallJson(a.from) || !smallJson(a.to)) return { error: "edit from/to too large" }; + const out = { kind: "edit", field: a.field, from: a.from ?? null, to: a.to ?? null }; + if (a.entry !== undefined) out.entry = a.entry; + return { anchor: out }; + } + default: + return { error: `anchor kind must be one of ${ANCHOR_KINDS.join(", ")}` }; + } +} + +/** @returns {{ subject: Record<string, string> } | { error: string }} */ +export function validateSubject(raw) { + if (!raw || typeof raw !== "object") return { error: "subject is not an object" }; + const s = /** @type {Record<string, unknown>} */ (raw); + if (s.kind === "article" && isStr(s.site) && TAKE_ID.test(s.site) && isStr(s.report) && TAKE_ID.test(s.report)) { + return { subject: { kind: "article", site: s.site, report: s.report } }; + } + if (s.kind === "video-project" && short(s.project, 512) && s.project.length > 0) { + return { subject: { kind: "video-project", project: s.project } }; + } + return { error: "subject must be an article {site, report} or a video-project {project}" }; +} + +function cleanSource(raw) { + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return undefined; + const out = {}; + for (const k of ["draft", "generator", "manifest", "how"]) if (short(raw[k], 1000) && raw[k]) out[k] = raw[k]; + return Object.keys(out).length ? out : undefined; +} + +function cleanText(v) { + if (!isStr(v)) throw new NoteError("text must be a string"); + const t = v.replace(/\s+$/, ""); + if (!t.trim()) throw new NoteError("text is empty"); + if (t.length > NOTE_TEXT_LIMIT) throw new NoteError(`text is over ${NOTE_TEXT_LIMIT} characters`); + return t; +} + +/** A refused edit: the caller's fault, a 400. */ +export class NoteError extends Error { + constructor(message) { + super(message); + this.name = "NoteError"; + } +} + +/** + * A parsed notes.json, checked. `{ doc }` or `{ error }`. A note that is not + * the contract's shape is an error for the whole file -- the file is never + * "repaired" by dropping it, because the next write would erase it. + */ +export function parseNotesDoc(raw) { + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return { error: "not an object" }; + if (raw.format !== NOTES_FORMAT) return { error: `format is not ${NOTES_FORMAT}` }; + if (raw.version !== NOTES_VERSION) return { error: `version ${raw.version} is not ${NOTES_VERSION}` }; + const subject = validateSubject(raw.subject); + if ("error" in subject) return subject; + if (!Array.isArray(raw.notes)) return { error: "notes is not a list" }; + const notes = []; + for (const [i, n] of raw.notes.entries()) { + const where = `notes[${i}]`; + if (!n || typeof n !== "object") return { error: `${where} is not an object` }; + if (!isNoteId(n.id)) return { error: `${where}.id is not a note id` }; + if (!NOTE_STATUSES.includes(n.status)) return { error: `${where}.status` }; + if (!NOTE_AUTHORS.includes(n.author)) return { error: `${where}.author` }; + if (!isStr(n.text) || !isStr(n.at) || !isStr(n.updatedAt)) return { error: `${where} text/at/updatedAt` }; + const a = validateAnchor(n.anchor); + if ("error" in a) return { error: `${where}.anchor: ${a.error}` }; + if (!Array.isArray(n.replies)) return { error: `${where}.replies is not a list` }; + const replies = []; + for (const r of n.replies) { + if (!r || !NOTE_AUTHORS.includes(r.author) || !isStr(r.text) || !isStr(r.at)) return { error: `${where}.replies` }; + replies.push({ author: r.author, text: r.text, at: r.at }); + } + const note = { id: n.id, status: n.status, author: n.author, text: n.text, at: n.at, updatedAt: n.updatedAt, anchor: a.anchor, replies }; + if (isStr(n.resolvedAt)) note.resolvedAt = n.resolvedAt; + if (NOTE_AUTHORS.includes(n.resolvedBy)) note.resolvedBy = n.resolvedBy; + notes.push(note); + } + const doc = { format: NOTES_FORMAT, version: NOTES_VERSION, subject: subject.subject }; + const source = cleanSource(raw.source); + if (source) doc.source = source; + doc.notes = notes; + return { doc }; +} + +export function emptyDoc(subject, source) { + const doc = { format: NOTES_FORMAT, version: NOTES_VERSION, subject }; + const s = cleanSource(source); + if (s) doc.source = s; + doc.notes = []; + return doc; +} + +function find(doc, id) { + const note = doc.notes.find((n) => n.id === id); + if (!note) throw new NoteError(`no note ${id}`); + return note; +} + +function setStatus(note, status, by, now) { + if (!NOTE_STATUSES.includes(status)) throw new NoteError(`status must be one of ${NOTE_STATUSES.join(", ")}`); + note.status = status; + note.updatedAt = now; + if (status === "open") { + delete note.resolvedAt; + delete note.resolvedBy; + } else { + note.resolvedAt = now; + note.resolvedBy = by; + } +} + +/** + * Apply one op to a doc, in place. Returns the note touched (null on delete). + * `by` is who is writing: the app stamps "operator", the CLI "agent". + * + * { op: "add", text, anchor } a new open note + * { op: "edit", id, text?, anchor? } rewrite it (its author only) + * { op: "status", id, status } open | resolved | wontfix + * { op: "reply", id, text, resolve? } a threaded reply, optionally resolving + * { op: "delete", id } remove the note + * { op: "delete-reply", id, index } remove one reply + * { op: "source", source } correct which file to edit + * + * @param {Record<string, any>} doc + * @param {Record<string, any>} op + * @param {"operator" | "agent"} by + * @param {string} [now] + */ +export function applyOp(doc, op, by, now = new Date().toISOString()) { + if (!NOTE_AUTHORS.includes(by)) throw new NoteError("author must be operator or agent"); + switch (op?.op) { + case "add": { + const a = validateAnchor(op.anchor); + if ("error" in a) throw new NoteError(a.error); + let id = newNoteId(); + while (doc.notes.some((n) => n.id === id)) id = newNoteId(); + const note = { id, status: "open", author: by, text: cleanText(op.text), at: now, updatedAt: now, anchor: a.anchor, replies: [] }; + doc.notes.push(note); + return note; + } + case "edit": { + const note = find(doc, op.id); + if (note.author !== by) throw new NoteError(`only the ${note.author} edits this note; reply instead`); + if (op.text !== undefined) note.text = cleanText(op.text); + if (op.anchor !== undefined) { + const a = validateAnchor(op.anchor); + if ("error" in a) throw new NoteError(a.error); + note.anchor = a.anchor; + } + note.updatedAt = now; + return note; + } + case "status": { + const note = find(doc, op.id); + setStatus(note, op.status, by, now); + return note; + } + case "reply": { + const note = find(doc, op.id); + note.replies.push({ author: by, text: cleanText(op.text), at: now }); + note.updatedAt = now; + if (op.resolve) setStatus(note, "resolved", by, now); + return note; + } + case "delete": { + const i = doc.notes.findIndex((n) => n.id === op.id); + if (i === -1) throw new NoteError(`no note ${op.id}`); + doc.notes.splice(i, 1); + return null; + } + case "delete-reply": { + const note = find(doc, op.id); + const i = Number(op.index); + if (!Number.isInteger(i) || i < 0 || i >= note.replies.length) throw new NoteError("no such reply"); + if (note.replies[i].author !== by) throw new NoteError(`only the ${note.replies[i].author} deletes that reply`); + note.replies.splice(i, 1); + note.updatedAt = now; + return note; + } + case "source": { + const s = cleanSource(op.source); + if (s) doc.source = s; + else delete doc.source; + return null; + } + default: + throw new NoteError("op must be add, edit, status, reply, delete, delete-reply or source"); + } +} + +export const openNotes = (doc) => (doc ? doc.notes.filter((n) => n.status === "open") : []); diff --git a/umtool/lib/annotations/store.mjs b/umtool/lib/annotations/store.mjs @@ -0,0 +1,155 @@ +// notes.json on disk: read, lock, apply one op, write. Shared by the app's +// /api/notes and `umtool notes`, so the operator's page and an agent's CLI go +// through ONE writer with one set of rules (./shape.mjs). +// +// Three writers can race on one file -- the page, an agent's `umtool notes +// reply`, a second agent -- and they are different PROCESSES, so the in-process +// queues lib/state.ts and lib/report/manifest.mjs use are not enough here: +// +// * a LOCKFILE (`notes.json.lock`, created O_EXCL) serialises +// read-modify-write across processes; one left by a dead writer is stale +// after 30 s and is taken over; +// * the write is tmp + rename, so a reader never sees half a file; +// * every write from the page carries the TOKEN it read (the file's mtime in +// ns, or "absent"); a stale one is a 409, never a silent overwrite of a +// reply an agent wrote in between; +// * a notes.json that exists and does not parse is NEVER overwritten (the +// takes.mjs rule) -- writing over it would erase every note in it; +// * the last note deleted deletes the file: an empty notes.json says nothing. +// +// WHERE a write may land is the caller's to decide BEFORE it gets here +// (lib/annotations/targets.mjs: isCorpusNotesFile for an article, a project +// directory for a video) -- `writeOp` takes the file it is given. +import { open, readFile, rename, stat, unlink, writeFile } from "node:fs/promises"; +import path from "node:path"; +import { NoteError, applyOp, emptyDoc, parseNotesDoc } from "./shape.mjs"; + +export { NoteError }; +export const LOCK_STALE_MS = 30_000; +const LOCK_WAIT_MS = 10_000; + +/** A write whose token no longer matches the file: someone wrote in between. */ +export class StaleNotes extends Error { + constructor(expected, got) { + super(`the notes changed since you read them (${got} vs ${expected})`); + this.name = "StaleNotes"; + this.status = 409; + } +} + +/** A notes.json that exists and is not one: refused, never overwritten. */ +export class NotesUnreadable extends Error { + constructor(file, why) { + super(`${path.basename(file)} does not parse (${why}); not overwriting it`); + this.name = "NotesUnreadable"; + this.status = 409; + } +} + +/** The file's identity for a write guard: mtime in ns plus size, or "absent". */ +export async function notesToken(file) { + const st = await stat(/* turbopackIgnore: true */ file, { bigint: true }).catch(() => null); + return st ? `${st.mtimeNs}-${st.size}` : "absent"; +} + +/** + * `{ doc, token }` -- doc null when there is no file. `{ error }` too when the + * file is there and is not a notes doc (the page shows it; writes refuse). + * + * @param {string} file + */ +export async function readNotes(file) { + const token = await notesToken(file); + let text; + try { + text = await readFile(/* turbopackIgnore: true */ file, "utf8"); + } catch (err) { + if (/** @type {NodeJS.ErrnoException} */ (err).code === "ENOENT") return { doc: null, token: "absent" }; + return { doc: null, token, error: String(/** @type {Error} */ (err).message ?? err) }; + } + let raw; + try { + raw = JSON.parse(text); + } catch (err) { + return { doc: null, token, error: `not JSON: ${/** @type {Error} */ (err).message}` }; + } + const r = parseNotesDoc(raw); + if ("error" in r) return { doc: null, token, error: r.error }; + return { doc: r.doc, token }; +} + +const sleep = (ms) => new Promise((r) => setTimeout(r, ms)); + +/** + * Run `fn` holding `<file>.lock`. A lock older than LOCK_STALE_MS is a dead + * writer's and is removed; otherwise wait (up to 10 s) and retry. + * + * @template T + * @param {string} file + * @param {() => Promise<T>} fn + * @param {{ waitMs?: number, staleMs?: number }} [opts] + * @returns {Promise<T>} + */ +export async function withNotesLock(file, fn, { waitMs = LOCK_WAIT_MS, staleMs = LOCK_STALE_MS } = {}) { + const lock = `${file}.lock`; + const deadline = Date.now() + waitMs; + let delay = 15; + for (;;) { + try { + const h = await open(/* turbopackIgnore: true */ lock, "wx"); + await h.writeFile(`${process.pid} ${new Date().toISOString()}\n`).catch(() => {}); + await h.close(); + break; + } catch (err) { + if (/** @type {NodeJS.ErrnoException} */ (err).code !== "EEXIST") throw err; + const st = await stat(/* turbopackIgnore: true */ lock).catch(() => null); + if (st && Date.now() - st.mtimeMs > staleMs) { + await unlink(/* turbopackIgnore: true */ lock).catch(() => {}); + continue; + } + if (Date.now() > deadline) throw new Error(`${path.basename(lock)} is held; try again`); + await sleep(delay); + delay = Math.min(delay * 2, 250); + } + } + try { + return await fn(); + } finally { + await unlink(/* turbopackIgnore: true */ lock).catch(() => {}); + } +} + +/** + * Apply one op to the notes at `file` and write it back. `init` is the + * subject (and source) a NEW file starts with; an existing file keeps its own + * subject, and its `source` unless the op is `source`. `token` (when given) + * must match the file as it is now, or StaleNotes. `by` is "operator" (the + * app) or "agent" (the CLI). + * + * Returns the doc as written (null when the file was deleted), the new token, + * and the note the op touched. + * + * @param {string} file + * @param {{ subject: Record<string, string>, source?: Record<string, string> | null }} init + * @param {Record<string, any>} op + * @param {{ by: "operator" | "agent", token?: string | null }} opts + */ +export async function writeOp(file, init, op, { by, token = null }) { + return withNotesLock(file, async () => { + const current = await readNotes(file); + if (current.error) throw new NotesUnreadable(file, current.error); + if (token !== null && token !== undefined && token !== current.token) throw new StaleNotes(token, current.token); + const doc = current.doc ?? emptyDoc(init.subject, init.source ?? undefined); + const note = applyOp(doc, op, by); + if (doc.notes.length === 0) { + await unlink(/* turbopackIgnore: true */ file).catch((err) => { + if (err.code !== "ENOENT") throw err; + }); + return { doc: null, token: "absent", note }; + } + const tmp = `${file}.tmp-${process.pid}-${Math.random().toString(36).slice(2, 8)}`; + await writeFile(/* turbopackIgnore: true */ tmp, JSON.stringify(doc, null, 2) + "\n", "utf8"); + await rename(/* turbopackIgnore: true */ tmp, file); + return { doc, token: await notesToken(file), note }; + }); +} diff --git a/umtool/lib/annotations/store.test.mjs b/umtool/lib/annotations/store.test.mjs @@ -0,0 +1,206 @@ +// notes.json: the store, the corpus write predicate and the targets. +// +// Run with: pnpm test:scripts +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, readFile, rm, stat, symlink, utimes, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import test from "node:test"; +import { corpusNotesFile, isCorpusNotesFile } from "../paths.mjs"; +import { NoteError, applyOp, emptyDoc, parseNotesDoc, validateAnchor } from "./shape.mjs"; +import { NotesUnreadable, StaleNotes, readNotes, withNotesLock, writeOp } from "./store.mjs"; +import { articleTarget, listNotesFiles, writeNote } from "./targets.mjs"; + +const SUBJECT = { kind: "article", site: "s1", report: "r1" }; + +async function tmp() { + return mkdtemp(path.join(tmpdir(), "umtool-notes-")); +} + +test("add, reply, resolve, reopen, delete: round-trip, and the last delete removes the file", async () => { + const dir = await tmp(); + const file = path.join(dir, "notes.json"); + const a = await writeOp(file, { subject: SUBJECT, source: { draft: "~/d.json" } }, { op: "add", text: "fix this", anchor: { kind: "whole" } }, { by: "operator" }); + assert.equal(a.doc.notes.length, 1); + assert.equal(a.doc.source.draft, "~/d.json"); + const id = a.note.id; + assert.match(id, /^n_[a-z0-9]+$/); + + const back = await readNotes(file); + assert.deepEqual(back.doc, a.doc); + assert.equal(back.token, a.token); + + const r = await writeOp(file, { subject: SUBJECT }, { op: "reply", id, text: "done in drafts/x.json", resolve: true }, { by: "agent", token: a.token }); + assert.equal(r.note.status, "resolved"); + assert.equal(r.note.resolvedBy, "agent"); + assert.equal(r.note.replies[0].author, "agent"); + + const o = await writeOp(file, { subject: SUBJECT }, { op: "status", id, status: "open" }, { by: "operator" }); + assert.equal(o.note.status, "open"); + assert.equal(o.note.resolvedAt, undefined); + + const d = await writeOp(file, { subject: SUBJECT }, { op: "delete", id }, { by: "operator" }); + assert.equal(d.doc, null); + await assert.rejects(stat(file), /ENOENT/); + await rm(dir, { recursive: true }); +}); + +test("a stale token is a 409 and the file is untouched", async () => { + const dir = await tmp(); + const file = path.join(dir, "notes.json"); + const a = await writeOp(file, { subject: SUBJECT }, { op: "add", text: "one", anchor: { kind: "whole" } }, { by: "operator", token: "absent" }); + await writeOp(file, { subject: SUBJECT }, { op: "add", text: "two (agent)", anchor: { kind: "whole" } }, { by: "agent" }); + const before = await readFile(file, "utf8"); + await assert.rejects( + writeOp(file, { subject: SUBJECT }, { op: "add", text: "three", anchor: { kind: "whole" } }, { by: "operator", token: a.token }), + (err) => err instanceof StaleNotes && err.status === 409, + ); + assert.equal(await readFile(file, "utf8"), before); + // "absent" against a file that exists is stale too + await assert.rejects( + writeOp(file, { subject: SUBJECT }, { op: "add", text: "x", anchor: { kind: "whole" } }, { by: "operator", token: "absent" }), + StaleNotes, + ); + await rm(dir, { recursive: true }); +}); + +test("an unparseable notes.json is never overwritten", async () => { + const dir = await tmp(); + const file = path.join(dir, "notes.json"); + await writeFile(file, "{ not json"); + assert.match((await readNotes(file)).error, /not JSON/); + await assert.rejects(writeOp(file, { subject: SUBJECT }, { op: "add", text: "x", anchor: { kind: "whole" } }, { by: "operator" }), NotesUnreadable); + assert.equal(await readFile(file, "utf8"), "{ not json"); + // a parseable file with a bad note is refused the same way + await writeFile(file, JSON.stringify({ format: "umtool-notes", version: 1, subject: SUBJECT, notes: [{ id: "bad" }] })); + await assert.rejects(writeOp(file, { subject: SUBJECT }, { op: "add", text: "x", anchor: { kind: "whole" } }, { by: "operator" }), NotesUnreadable); + await rm(dir, { recursive: true }); +}); + +test("lock contention: parallel writers all land; a held lock waits; a stale lock is taken over", async () => { + const dir = await tmp(); + const file = path.join(dir, "notes.json"); + await Promise.all( + Array.from({ length: 12 }, (_, i) => + writeOp(file, { subject: SUBJECT }, { op: "add", text: `n${i}`, anchor: { kind: "whole" } }, { by: i % 2 ? "agent" : "operator" }), + ), + ); + assert.equal((await readNotes(file)).doc.notes.length, 12); + + // Another process holds it (a fresh lock file): we wait, then give up. + await writeFile(`${file}.lock`, "999999 now\n"); + await assert.rejects(withNotesLock(file, async () => 1, { waitMs: 120 }), /held/); + // The same lock, 31 s old: a dead writer's, taken over. + const old = new Date(Date.now() - 31_000); + await utimes(`${file}.lock`, old, old); + assert.equal(await withNotesLock(file, async () => 2, { waitMs: 120 }), 2); + await assert.rejects(stat(`${file}.lock`), /ENOENT/); + await rm(dir, { recursive: true }); +}); + +test("ops refuse what the contract does not allow", () => { + const doc = emptyDoc(SUBJECT); + assert.throws(() => applyOp(doc, { op: "add", text: " ", anchor: { kind: "whole" } }, "operator"), NoteError); + assert.throws(() => applyOp(doc, { op: "add", text: "x", anchor: { kind: "nope" } }, "operator"), NoteError); + assert.throws(() => applyOp(doc, { op: "add", text: "x".repeat(8001), anchor: { kind: "whole" } }, "operator"), NoteError); + const n = applyOp(doc, { op: "add", text: "x", anchor: { kind: "whole" } }, "operator"); + assert.throws(() => applyOp(doc, { op: "edit", id: n.id, text: "agent rewrites it" }, "agent"), /reply instead/); + assert.throws(() => applyOp(doc, { op: "status", id: n.id, status: "done" }, "agent"), NoteError); + assert.throws(() => applyOp(doc, { op: "reply", id: "n_missing0", text: "x" }, "agent"), /no note/); + assert.throws(() => applyOp(doc, { op: "frobnicate" }, "agent"), NoteError); + assert.throws(() => applyOp(doc, { op: "add", text: "x", anchor: { kind: "whole" } }, "someone"), NoteError); + // source: set, then cleared + applyOp(doc, { op: "source", source: { draft: "~/x.json", junk: 1 } }, "agent"); + assert.deepEqual(doc.source, { draft: "~/x.json" }); + assert.ok(parseNotesDoc(JSON.parse(JSON.stringify(doc))).doc); +}); + +test("anchors: every kind validates; bad ones refuse", () => { + const ok = [ + { kind: "whole" }, + { kind: "section", section: "s-2" }, + { kind: "text", section: "summary", quote: "the claim", prefix: "before ", suffix: " after" }, + { kind: "cite", cite: "c12" }, + { kind: "moment", file: "takes/deck/preview.mp4", t: 12.345, take: "deck", entry: "e3", resolved: { title: "T", sourceT: 81.234, approx: true, bogus: 1 } }, + { kind: "entry", entry: "clip-4" }, + { kind: "take", take: "cold-open" }, + { kind: "edit", entry: "e3", field: "quote", from: "a", to: "b" }, + ]; + for (const a of ok) assert.ok("anchor" in validateAnchor(a), JSON.stringify(a)); + assert.equal(validateAnchor(ok[4]).anchor.t, 12.35); + assert.deepEqual(validateAnchor(ok[4]).anchor.resolved, { title: "T", sourceT: 81.23, approx: true }); + assert.equal(validateAnchor({ kind: "text", section: "s", quote: "q" }).anchor.prefix, ""); + const bad = [ + null, + { kind: "text", section: "s" }, + { kind: "moment", file: "../x.mp4", t: 1 }, + { kind: "moment", file: "/abs.mp4", t: 1 }, + { kind: "moment", file: "a.mp4", t: -1 }, + { kind: "take", take: "Bad Id" }, + { kind: "section", section: "has space" }, + { kind: "edit", field: "" }, + ]; + for (const a of bad) assert.ok("error" in validateAnchor(a), JSON.stringify(a)); +}); + +async function sitesFixture() { + const root = await tmp(); + const sites = path.join(root, "sites"); + await mkdir(path.join(sites, "s1", "reports", "r1"), { recursive: true }); + await writeFile(path.join(sites, "s1", "site.json"), "{}"); + await writeFile(path.join(sites, "s1", "reports", "r1", "report.json"), "{}"); + const outside = path.join(root, "outside", "r2"); + await mkdir(outside, { recursive: true }); + await symlink(outside, path.join(sites, "s1", "reports", "r2")); + return { root, sites }; +} + +test("isCorpusNotesFile: exactly sites/<site>/reports/<id>/notes.json, and nothing else", async () => { + const { root, sites } = await sitesFixture(); + const opt = { sitesDir: sites }; + const good = path.join(sites, "s1", "reports", "r1", "notes.json"); + assert.equal(corpusNotesFile("s1", "r1", opt), good); + assert.equal(await isCorpusNotesFile(good, opt), true); + // traversal, spelled several ways + assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "reports", "r1", "..", "r1", "notes.json").replace(/\/r1\/notes/, "/../r1/r1/notes"), opt), false); + assert.equal(await isCorpusNotesFile(`${sites}/s1/reports/../reports/r1/notes.json`, opt), false); + assert.equal(await isCorpusNotesFile(`${sites}/s1/reports//r1/notes.json`, opt), false); + assert.equal(await isCorpusNotesFile("s1/reports/r1/notes.json", opt), false); + // wrong name, wrong depth, wrong middle segment, bad ids + assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "reports", "r1", "report.json"), opt), false); + assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "reports", "r1", "x", "notes.json"), opt), false); + assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "stills", "r1", "notes.json"), opt), false); + assert.equal(await isCorpusNotesFile(path.join(sites, "S1", "reports", "r1", "notes.json"), opt), false); + assert.equal(corpusNotesFile("s1", "../r1", opt), null); + // a report dir that is a symlink out of the site + assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "reports", "r2", "notes.json"), opt), false); + // a report that does not exist: a note never creates its directory + assert.equal(await isCorpusNotesFile(path.join(sites, "s1", "reports", "r9", "notes.json"), opt), false); + // a notes.json that is itself a symlink + await writeFile(path.join(root, "elsewhere.json"), "{}"); + await symlink(path.join(root, "elsewhere.json"), good); + assert.equal(await isCorpusNotesFile(good, opt), false); + await rm(root, { recursive: true }); +}); + +test("articleTarget: writes land beside report.json with the discovered source; refusals are typed", async () => { + const { root, sites } = await sitesFixture(); + const reports = path.join(root, "reports"); + await mkdir(path.join(reports, "ws", "polemics", "drafts"), { recursive: true }); + await writeFile(path.join(reports, "ws", "polemics", "drafts", "r1.json"), JSON.stringify({ id: "r1" })); + await writeFile(path.join(reports, "ws", "polemics", "make-site.py"), 'OUT = "sites/s1/reports"\nfor d in drafts: pass\n'); + const opt = { sitesDir: sites, reportsRoot: reports }; + + const t = await articleTarget("s1/r1", opt); + const w = await writeNote(t, { op: "add", text: "x", anchor: { kind: "whole" } }, { by: "operator" }); + assert.equal(w.doc.source.draft.endsWith(path.join("ws", "polemics", "drafts", "r1.json")), true); + assert.equal(w.doc.source.generator.endsWith("make-site.py"), true); + const list = await listNotesFiles(opt); + assert.deepEqual(list.map((l) => [l.kind, l.id, l.doc.notes.length]), [["article", "s1/r1", 1]]); + + await assert.rejects(articleTarget("s1/r9", opt), (e) => e.status === 404); + await assert.rejects(articleTarget("s1/r2", opt), (e) => e.status === 403); + await assert.rejects(articleTarget("s1/../r1", opt), (e) => e.status === 400); + await assert.rejects(articleTarget("s1", opt), (e) => e.status === 400); + await rm(root, { recursive: true }); +}); diff --git a/umtool/lib/annotations/targets.mjs b/umtool/lib/annotations/targets.mjs @@ -0,0 +1,172 @@ +// What a note is ON, and therefore which notes.json it lives in -- decided +// here, once, for the app's /api/notes and for `umtool notes`. +// +// article `SITES_DIR/<site>/reports/<report>/notes.json`, beside +// report.json. The generators that write a report dir +// overwrite only report.json, video.mp4 and poster.jpg, so it +// survives a regenerate; the compose stage and the report +// history never read it (common/publish tests hold that), so +// it is never published. The ONE corpus file umtool writes, +// and only through isCorpusNotesFile. +// video-project `<project>/notes.json`, beside video.manifest.json. A +// report-video project under REPORTS_ROOT, found by the same +// walk `umtool ls` uses. +import { readdir, readFile, realpath, stat } from "node:fs/promises"; +import path from "node:path"; +import { NOTES_FILENAME, REPORTS_ROOT, SEGMENT_RE, SITES_DIR, corpusNotesFile, inside, isCorpusNotesFile } from "../paths.mjs"; +import { projectRefs, resolveProject } from "../projects/core.mjs"; +import { sourceFor, tildify } from "../articles/sources.mjs"; +import { readNotes, writeOp } from "./store.mjs"; + +const exists = (p) => stat(/* turbopackIgnore: true */ p).then(() => true, () => false); + +/** A refused target: the caller's fault (400/404). */ +export class TargetError extends Error { + constructor(message, status = 400) { + super(message); + this.name = "TargetError"; + this.status = status; + } +} + +/** + * An article target, checked: the site and report ids, the report directory + * on disk, and the notes path through isCorpusNotesFile. + * + * @param {string} spec `<site>/<report>` + * @param {{ sitesDir?: string, reportsRoot?: string }} [opts] + */ +export async function articleTarget(spec, { sitesDir = SITES_DIR, reportsRoot = REPORTS_ROOT } = {}) { + const [site, report, ...rest] = String(spec ?? "").split("/"); + if (rest.length || !SEGMENT_RE.test(site ?? "") || !SEGMENT_RE.test(report ?? "")) { + throw new TargetError(`not an article: ${JSON.stringify(spec)} (want <site>/<report>)`); + } + const file = corpusNotesFile(site, report, { sitesDir }); + if (!file || !(await exists(path.dirname(/* turbopackIgnore: true */ file)))) throw new TargetError(`no report ${site}/${report}`, 404); + if (!(await isCorpusNotesFile(file, { sitesDir }))) throw new TargetError(`refusing to write ${file}`, 403); + return { + kind: "article", + id: `${site}/${report}`, + file, + subject: { kind: "article", site, report }, + // Lazy: the scan reads every workspace's generators, and a read of an + // existing file never needs it. + source: async () => sourceFor(site, report, { reportsRoot }), + }; +} + +/** + * A video-project target: a report-video project under REPORTS_ROOT, by id, + * unique name or directory. + * + * @param {string} spec + * @param {{ reportsRoot?: string }} [opts] + */ +export async function projectTarget(spec, { reportsRoot = REPORTS_ROOT } = {}) { + const r = await resolveProject(String(spec ?? ""), reportsRoot); + if (r.ambiguous) throw new TargetError(`${spec} names ${r.ambiguous.length} projects: ${r.ambiguous.map((p) => p.id).join(", ")}`); + const p = r.project; + if (!p) throw new TargetError(`no project ${spec}`, 404); + if (p.kind !== "report-video") throw new TargetError(`${p.id} is a ${p.kind} project; notes are for report videos`); + const [realRoot, realDir] = await Promise.all([realpath(/* turbopackIgnore: true */ reportsRoot).catch(() => null), realpath(/* turbopackIgnore: true */ p.dir).catch(() => null)]); + if (!realRoot || !realDir || !inside(realRoot, realDir) || realRoot === realDir) { + throw new TargetError(`${p.id} is not under the reports root`, 403); + } + const file = path.join(/* turbopackIgnore: true */ p.dir, NOTES_FILENAME); + return { + kind: "video-project", + id: p.id, + dir: p.dir, + file, + subject: { kind: "video-project", project: p.id }, + source: async () => projectSource(p.dir, reportsRoot), + }; +} + +/** + * A report-video project's source: its manifest, and -- when the manifest is + * generated -- the generator, resolved against the project's workspace (the + * first directory under REPORTS_ROOT) when that file exists. + */ +export async function projectSource(dir, reportsRoot = REPORTS_ROOT) { + const manifest = path.join(/* turbopackIgnore: true */ dir, "video.manifest.json"); + const out = { manifest: tildify(manifest) }; + let generatedBy = null; + try { + const m = JSON.parse(await readFile(/* turbopackIgnore: true */ manifest, "utf8")); + if (typeof m.generatedBy === "string" && m.generatedBy.trim()) generatedBy = m.generatedBy.trim(); + } catch { + // no manifest, or not JSON: the manifest path is still the place to look + } + if (!generatedBy) { + out.how = "hand-edited manifest"; + return out; + } + const rel = path.relative(/* turbopackIgnore: true */ reportsRoot, dir); + const ws = rel && !rel.startsWith("..") ? path.join(/* turbopackIgnore: true */ reportsRoot, rel.split(path.sep)[0]) : null; + const candidates = [ws && path.join(/* turbopackIgnore: true */ ws, generatedBy), path.join(/* turbopackIgnore: true */ dir, generatedBy)].filter(Boolean); + let gen = null; + for (const c of candidates) { + if (await exists(c)) { + gen = c; + break; + } + } + out.generator = gen ? tildify(gen) : generatedBy; + out.how = `manifest is generated by ${generatedBy}; edit its inputs, then regenerate`; + return out; +} + +/** + * Resolve what the CLI was handed: `<site>/<report>` when that report exists, + * else a project. + */ +export async function resolveTarget(spec, opts = {}) { + const parts = String(spec ?? "").split("/"); + if (parts.length === 2 && parts.every((s) => SEGMENT_RE.test(s))) { + const dir = path.join(/* turbopackIgnore: true */ opts.sitesDir ?? SITES_DIR, parts[0], "reports", parts[1]); + if (await exists(dir)) return articleTarget(spec, opts); + } + return projectTarget(spec, opts); +} + +/** + * Every notes.json there is: articles under SITES_DIR, projects under + * REPORTS_ROOT. `{ target, file, doc, error? }` each; sorted by id. + * + * @param {{ sitesDir?: string, reportsRoot?: string }} [opts] + */ +export async function listNotesFiles({ sitesDir = SITES_DIR, reportsRoot = REPORTS_ROOT } = {}) { + const out = []; + for (const site of await readdir(/* turbopackIgnore: true */ sitesDir).catch(() => [])) { + if (!SEGMENT_RE.test(site)) continue; + for (const report of await readdir(/* turbopackIgnore: true */ path.join(/* turbopackIgnore: true */ sitesDir, site, "reports")).catch(() => [])) { + if (!SEGMENT_RE.test(report)) continue; + const file = path.join(/* turbopackIgnore: true */ sitesDir, site, "reports", report, NOTES_FILENAME); + if (!(await exists(file))) continue; + out.push({ kind: "article", id: `${site}/${report}`, file, ...(await readNotes(file)) }); + } + } + for (const p of await projectRefs(reportsRoot)) { + if (p.kind !== "report-video") continue; + const file = path.join(/* turbopackIgnore: true */ p.dir, NOTES_FILENAME); + if (!(await exists(file))) continue; + out.push({ kind: "video-project", id: p.id, file, ...(await readNotes(file)) }); + } + return out.sort((a, b) => a.kind.localeCompare(b.kind) || a.id.localeCompare(b.id)); +} + +/** + * One write to a target's notes. A file that does not exist yet starts with + * the target's subject and its discovered source (which the agent may correct + * later with a `source` op); an existing file keeps both. + * + * @param {{ file: string, subject: Record<string, string>, source: () => Promise<Record<string, string> | null> }} target + * @param {Record<string, any>} op + * @param {{ by: "operator" | "agent", token?: string | null }} opts + */ +export async function writeNote(target, op, opts) { + const now = await readNotes(target.file); + const source = now.doc || now.error ? undefined : ((await target.source()) ?? undefined); + return writeOp(target.file, { subject: target.subject, source }, op, opts); +} diff --git a/umtool/lib/annotations/types.ts b/umtool/lib/annotations/types.ts @@ -0,0 +1,98 @@ +// The notes shapes, with NO server imports (the lib/note-types.ts rule): a +// client component may take a value from here without dragging node:fs into +// the browser bundle. The store that reads and writes them is ./store.mjs, +// shared by the app and `umtool notes`; docs/notes.md is the prose. + +// The values are ./shape.mjs's (pure, shared with the store and the CLI); +// this file adds the types. +export { + ANCHOR_KINDS, + NOTE_AUTHORS, + NOTE_STATUSES, + NOTE_TEXT_LIMIT, + NOTES_FORMAT, + NOTES_VERSION, + REPORT_BLOCKS, +} from "./shape.mjs"; +export { CONTEXT as ANCHOR_CONTEXT } from "./anchor.mjs"; + +export type NoteStatus = "open" | "resolved" | "wontfix"; +export type NoteAuthor = "operator" | "agent"; + +/** What a moment resolved to when it was written, for the agent reading it back. */ +export type MomentResolved = { + entry?: string; + title?: string; + quote?: string; + channel?: string; + video?: string; + sourceT?: number; + url?: string; + /** The schedule did not match this file exactly (a preview, not out/). */ + approx?: boolean; +}; + +export type Anchor = + | { kind: "text"; section: string; quote: string; prefix: string; suffix: string } + | { kind: "cite"; cite: string } + | { kind: "section"; section: string } + | { kind: "whole" } + | { kind: "moment"; file: string; t: number; take?: string; entry?: string; resolved?: MomentResolved } + | { kind: "entry"; entry: string } + | { kind: "take"; take: string } + | { kind: "edit"; entry?: string; field: string; from: unknown; to: unknown }; + +export type AnchorKind = Anchor["kind"]; + +export type NoteReply = { author: NoteAuthor; text: string; at: string }; + +export type Note = { + id: string; + status: NoteStatus; + author: NoteAuthor; + text: string; + at: string; + updatedAt: string; + anchor: Anchor; + replies: NoteReply[]; + resolvedAt?: string; + resolvedBy?: NoteAuthor; +}; + +export type NotesSubject = + | { kind: "article"; site: string; report: string } + | { kind: "video-project"; project: string }; + +/** Which file an agent should edit to act on a note. Filled by umtool, correctable. */ +export type NotesSource = { draft?: string; generator?: string; manifest?: string; how?: string }; + +export type NotesDoc = { + format: "umtool-notes"; + version: 1; + subject: NotesSubject; + source?: NotesSource; + notes: Note[]; +}; + +/** What GET /api/notes returns. `token` goes back on every write (409 when stale). */ +export type NotesRead = { + subject: NotesSubject; + file: string; + token: string; + doc: NotesDoc | null; + source: NotesSource | null; + error?: string; +}; + +/** One write. The server stamps author "operator" on everything the UI sends. */ +export type NoteOp = + | { op: "add"; text: string; anchor: Anchor } + | { op: "edit"; id: string; text?: string; anchor?: Anchor } + | { op: "status"; id: string; status: NoteStatus } + | { op: "reply"; id: string; text: string; resolve?: boolean } + | { op: "delete"; id: string } + | { op: "delete-reply"; id: string; index: number } + | { op: "source"; source: NotesSource }; + +export const openCount = (doc: NotesDoc | null | undefined): number => + doc ? doc.notes.filter((n) => n.status === "open").length : 0; diff --git a/umtool/lib/articles/sources.mjs b/umtool/lib/articles/sources.mjs @@ -0,0 +1,219 @@ +// Which file an agent should EDIT to change an article. +// +// A report.json under transcripts/sites/ is generated: a workspace under +// ~/reports keeps the draft (`<ws>/polemics/drafts/<slug>.json`, the source of +// truth) and a generator script that writes report.json from it +// (`<ws>/polemics/make-site.py`, `<ws>/site/polemics.py`, ...). A note that +// says "fix this sentence" is useless to an agent that edits report.json -- the +// next generator run puts the old sentence back. So every notes.json carries a +// `source` block naming the draft and the generator, found here. +// +// The match is a heuristic, written down with its reason (`how`), and the +// agent may correct it (`umtool notes source`): +// +// draft a drafts/*.json whose `id`, or file name, is the report id -- +// allowing for a `polemic-` prefix on either side (candalyzer's +// polemic-israel is drafts/israel.json with id polemic-israel; +// jeralyzer-private's `blame` is drafts/blame.json with id +// polemic-blame). Several matches: the one whose workspace has a +// generator naming the site wins; still several, none is chosen. +// generator a *.py / *.mts under <ws>/polemics or <ws>/site that names the +// site (or the report id), preferring one that names the report +// id itself, then one that reads the drafts. Backups +// (`make-report.pre-2026-10-05.py`: a second dot) are skipped. +// +// Cheap: one readdir per workspace and one read per generator, cached for 30 s +// like the project walk. +import { readdir, readFile, stat } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { REPORTS_ROOT } from "../paths.mjs"; + +const CACHE_MS = 30_000; +const GEN_DIRS = ["polemics", "site"]; +const GEN_EXT = /^[^.]+\.(py|mts|mjs|sh)$/; +const MAX_GEN_BYTES = 2 * 1024 * 1024; + +/** `~/…` for a path under the home directory: what a note shows an agent. */ +export function tildify(abs) { + const home = os.homedir(); + return abs === home || abs.startsWith(home + path.sep) ? `~${abs.slice(home.length)}` : abs; +} + +/** The ids a report or draft may go by: itself, without `polemic-`, with it. */ +export function idKeys(id) { + const bare = id.replace(/^polemic-/, ""); + return new Set([id, bare, `polemic-${bare}`]); +} + +/** Does `text` name `id` as a whole token (so `jasolyzer` is not `jasolyzer-private`)? */ +export function names(text, id) { + const esc = id.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); + return new RegExp(`(?<![A-Za-z0-9_-])${esc}(?![A-Za-z0-9_-])`).test(text); +} + +const isDir = (p) => stat(/* turbopackIgnore: true */ p).then((s) => s.isDirectory(), () => false); + +/** @type {Map<string, { at: number, value: Promise<any[]> }>} */ +const cache = new Map(); + +/** + * Every workspace under `reportsRoot` that has drafts or a generator dir: + * `{ dir, name, drafts: [{ file, slug, id }], generators: [{ file, text }] }`. + * + * @param {string} [reportsRoot] + */ +export function scanWorkspaces(reportsRoot = REPORTS_ROOT) { + const hit = cache.get(reportsRoot); + if (hit && Date.now() - hit.at < CACHE_MS) return hit.value; + const value = scan(reportsRoot); + cache.set(reportsRoot, { at: Date.now(), value }); + value.catch(() => cache.delete(reportsRoot)); + return value; +} + +/** Forget the scan (a test that writes a fixture, then reads it). */ +export function clearSourcesCache() { + cache.clear(); +} + +async function scan(reportsRoot) { + const entries = await readdir(/* turbopackIgnore: true */ reportsRoot, { withFileTypes: true }).catch(() => []); + const out = []; + for (const e of entries) { + if (e.name.startsWith(".") || e.name === "data") continue; + const dir = path.join(/* turbopackIgnore: true */ reportsRoot, e.name); + if (!(e.isDirectory() || (e.isSymbolicLink() && (await isDir(dir))))) continue; + const drafts = []; + const draftsDir = path.join(/* turbopackIgnore: true */ dir, "polemics", "drafts"); + for (const f of await readdir(/* turbopackIgnore: true */ draftsDir).catch(() => [])) { + if (!f.endsWith(".json")) continue; + const file = path.join(/* turbopackIgnore: true */ draftsDir, f); + let id = null; + try { + const j = JSON.parse(await readFile(/* turbopackIgnore: true */ file, "utf8")); + if (j && typeof j.id === "string") id = j.id; + } catch { + // an unreadable draft still matches by its file name + } + drafts.push({ file, slug: f.slice(0, -5), id }); + } + const generators = []; + for (const g of GEN_DIRS) { + const gdir = path.join(/* turbopackIgnore: true */ dir, g); + for (const f of await readdir(/* turbopackIgnore: true */ gdir).catch(() => [])) { + if (!GEN_EXT.test(f)) continue; + const file = path.join(/* turbopackIgnore: true */ gdir, f); + const st = await stat(/* turbopackIgnore: true */ file).catch(() => null); + if (!st?.isFile() || st.size > MAX_GEN_BYTES) continue; + generators.push({ file, text: await readFile(/* turbopackIgnore: true */ file, "utf8").catch(() => "") }); + } + } + if (drafts.length || generators.length) { + drafts.sort((a, b) => a.file.localeCompare(b.file)); + generators.sort((a, b) => a.file.localeCompare(b.file)); + out.push({ dir, name: e.name, drafts, generators }); + } + } + return out.sort((a, b) => a.name.localeCompare(b.name)); +} + +/** Is `draft` this report's? */ +function draftMatches(draft, keys) { + return (draft.id !== null && keys.has(draft.id)) || keys.has(draft.slug) || keys.has(`polemic-${draft.slug}`); +} + +/** + * The source of truth for one report, or null when no workspace claims it. + * + * @param {string} siteId + * @param {string} reportId + * @param {{ reportsRoot?: string }} [opts] + * @returns {Promise<{ draft?: string, generator?: string, how: string, workspace?: string } | null>} + */ +export async function sourceFor(siteId, reportId, { reportsRoot = REPORTS_ROOT } = {}) { + const workspaces = await scanWorkspaces(reportsRoot); + const keys = idKeys(reportId); + const namesSite = (ws) => ws.generators.some((g) => names(g.text, siteId)); + + const cands = []; + for (const ws of workspaces) { + for (const d of ws.drafts) { + if (!draftMatches(d, keys)) continue; + const score = (namesSite(ws) ? 4 : 0) + (d.id === reportId ? 2 : 0) + (d.slug === reportId.replace(/^polemic-/, "") ? 1 : 0); + cands.push({ ws, d, score }); + } + } + cands.sort((a, b) => b.score - a.score); + const top = cands[0]; + const unique = top && (cands.length === 1 || cands[1].score < top.score); + const draft = unique ? top : null; + + const pool = draft ? [draft.ws] : workspaces; + let gen = null; + let genScore = 0; + for (const ws of pool) { + for (const g of ws.generators) { + const site = names(g.text, siteId); + const report = names(g.text, reportId); + if (!site && !report) continue; + const base = path.basename(g.file); + const score = + (report ? 3 : 0) + (site ? 2 : 0) + (draft && /\bdrafts\b/.test(g.text) ? 1 : 0) + (/^(make-site|polemics)\./.test(base) ? 0.5 : 0); + if (score > genScore) { + gen = { ws, g }; + genScore = score; + } + } + } + // Without a draft, a generator that only names the SITE is every report's + // generator and says nothing about this one; keep it only if it names the id. + // A bare id (`deleted`, `poker`) is also an English word, so naming it is + // only evidence when the generator names the site too. + if (!draft && gen && !(names(gen.g.text, reportId) && (reportId.includes("-") || names(gen.g.text, siteId)))) gen = null; + if (!draft && !gen) { + if (cands.length > 1) { + return { how: `several drafts match ${reportId}: ${cands.map((c) => tildify(c.d.file)).join(", ")}; none chosen` }; + } + return null; + } + + const how = []; + if (draft) { + const by = draft.d.id === reportId ? `id ${reportId}` : draft.d.id && keys.has(draft.d.id) ? `id ${draft.d.id}` : `file name ${draft.d.slug}`; + how.push(`draft matched by ${by}`); + if (cands.length > 1) how.push(`preferred over ${cands.length - 1} other`); + } + if (gen) { + const named = [siteId, reportId].filter((id) => names(gen.g.text, id)); + how.push(`generator names ${named.join(" and ")}`); + } + const out = { how: how.join("; ") }; + if (draft) out.draft = tildify(draft.d.file); + if (gen) out.generator = tildify(gen.g.file); + out.workspace = tildify((draft?.ws ?? gen?.ws).dir); + return out; +} + +/** + * The workspace directories a site's articles come from: every workspace that + * holds a matched draft for one of `reportIds`, or a generator that names the + * site. Absolute paths, sorted. + * + * @param {string} siteId + * @param {string[]} reportIds + * @param {{ reportsRoot?: string }} [opts] + */ +export async function siteWorkspaces(siteId, reportIds, { reportsRoot = REPORTS_ROOT } = {}) { + const workspaces = await scanWorkspaces(reportsRoot); + const dirs = new Set(); + for (const ws of workspaces) { + if (ws.generators.some((g) => names(g.text, siteId))) dirs.add(ws.dir); + } + for (const id of reportIds) { + const s = await sourceFor(siteId, id, { reportsRoot }); + const ws = s?.draft ? workspaces.find((w) => w.drafts.some((d) => tildify(d.file) === s.draft)) : null; + if (ws) dirs.add(ws.dir); + } + return [...dirs].sort(); +} diff --git a/umtool/lib/articles/sources.test.mjs b/umtool/lib/articles/sources.test.mjs @@ -0,0 +1,69 @@ +// Finding an article's draft and generator, on the three workspace layouts +// the live private sites use. +// +// Run with: pnpm test:scripts +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import test from "node:test"; +import { clearSourcesCache, names, siteWorkspaces, sourceFor } from "./sources.mjs"; + +async function put(file, text) { + await mkdir(path.dirname(file), { recursive: true }); + await writeFile(file, text); +} + +async function fixture() { + const root = await mkdtemp(path.join(tmpdir(), "umtool-sources-")); + // candalyzer: drafts/<bare>.json with id polemic-<bare>; generator site/polemics.py; + // a hand-written fact-check whose generator names it; a backup generator. + await put(path.join(root, "candace/polemics/drafts/israel.json"), JSON.stringify({ id: "polemic-israel" })); + await put(path.join(root, "candace/site/polemics.py"), 'OUT = ".../sites/candalyzer/reports"\nfor f in DRAFTS.glob("drafts/*.json"): rid = f"polemic-{slug}"\n'); + await put(path.join(root, "candace/site/make-report.py"), 'SITE = "candalyzer"\nREPORT = "deconstruction-fact-check"\n'); + await put(path.join(root, "candace/site/make-report.pre-2026-10-05.py"), 'REPORT = "polemic-israel" # candalyzer\n'); + // jasolyzer-private: drafts/<bare>.json, generator polemics/make-site.py + await put(path.join(root, "pirate/polemics/drafts/skg.json"), JSON.stringify({ id: "polemic-skg" })); + await put(path.join(root, "pirate/polemics/make-site.py"), 'SITE = "jasolyzer-private"\nrid = f"polemic-{slug}" # from drafts\n'); + // jeralyzer-private: the report id is BARE (`blame`), the draft id is polemic-blame + await put(path.join(root, "quartering/polemics/drafts/blame.json"), JSON.stringify({ id: "polemic-blame" })); + await put(path.join(root, "quartering/polemics/make-site.py"), 'SITE = "jeralyzer-private"\nrid = slug # drafts\n'); + // a decoy: a public site's workspace with a same-named draft + await put(path.join(root, "decoy/polemics/drafts/blame.json"), JSON.stringify({ id: "blame" })); + await put(path.join(root, "decoy/polemics/make-site.py"), 'SITE = "jeralyzer"\n'); + clearSourcesCache(); + return root; +} + +test("each layout finds its draft and generator", async () => { + const root = await fixture(); + const o = { reportsRoot: root }; + const isr = await sourceFor("candalyzer", "polemic-israel", o); + assert.equal(isr.draft, path.join(root, "candace/polemics/drafts/israel.json")); + assert.equal(isr.generator, path.join(root, "candace/site/polemics.py")); + assert.match(isr.how, /draft matched by id polemic-israel/); + + const fc = await sourceFor("candalyzer", "deconstruction-fact-check", o); + assert.equal(fc.draft, undefined); + assert.equal(fc.generator, path.join(root, "candace/site/make-report.py")); + + const skg = await sourceFor("jasolyzer-private", "polemic-skg", o); + assert.equal(skg.draft, path.join(root, "pirate/polemics/drafts/skg.json")); + assert.equal(skg.generator, path.join(root, "pirate/polemics/make-site.py")); + + // The decoy's draft id is exactly `blame`, but its workspace never names the site. + const blame = await sourceFor("jeralyzer-private", "blame", o); + assert.equal(blame.draft, path.join(root, "quartering/polemics/drafts/blame.json")); + assert.equal(blame.generator, path.join(root, "quartering/polemics/make-site.py")); + assert.match(blame.how, /preferred over 1 other/); + + assert.equal(await sourceFor("candalyzer", "no-such-report", o), null); + assert.deepEqual(await siteWorkspaces("jasolyzer-private", ["polemic-skg"], o), [path.join(root, "pirate")]); + await rm(root, { recursive: true }); +}); + +test("names() matches whole ids only", () => { + assert.equal(names('"jasolyzer-private"', "jasolyzer"), false); + assert.equal(names("sites/jasolyzer/reports", "jasolyzer"), true); + assert.equal(names("polemic-blame", "blame"), false); +}); diff --git a/umtool/lib/paths.mjs b/umtool/lib/paths.mjs @@ -5,6 +5,7 @@ // `umtool ls` and the page it is supposed to describe. lib/paths.ts re-exports // everything here with types; nothing computes a root twice. import { existsSync } from "node:fs"; +import { lstat, realpath } from "node:fs/promises"; import os from "node:os"; import path from "node:path"; import { SONG_DATA, SONG_REPORTS } from "../song/paths.mjs"; @@ -208,12 +209,74 @@ export const CHANNELS_DIR = path.resolve( : path.join(/* turbopackIgnore: true */ REPO_ROOT, "transcripts", "channels")), ); +// The SITES -- `transcripts/sites/<site>/`, each a site.json and its reports +// (`reports/<id>/report.json`, `video.mp4`, `poster.jpg`). Resolved the way +// CHANNELS_DIR is, and the way common/lib/paths.ts resolves its sitesDir, so +// one `SITES_DIR` confines both. Readable only; the ONE file in it umtool may +// write is a report's notes.json, and only through isCorpusNotesFile below. +export const SITES_DIR = path.resolve( + /* turbopackIgnore: true */ + process.env.SITES_DIR ?? + (process.env.TRANSCRIPTS_DIR + ? path.join(/* turbopackIgnore: true */ process.env.TRANSCRIPTS_DIR, "sites") + : path.join(/* turbopackIgnore: true */ REPO_ROOT, "transcripts", "sites")), +); + export const READ_ROOTS = dedupe( process.env.MIX_ROOTS ? process.env.MIX_ROOTS.split(":").filter(Boolean) - : [SONG_REPORTS, REPORTS_ROOT, SONG_DATA, SONG_SCRATCH, CHANNELS_DIR, MEDIA_ROOT], + : [SONG_REPORTS, REPORTS_ROOT, SONG_DATA, SONG_SCRATCH, CHANNELS_DIR, MEDIA_ROOT, SITES_DIR], ); +/** A site id or a report id: one lowercase url-safe segment (common/lib/report/schema.ts REPORT_ID_RE). */ +export const SEGMENT_RE = /^[a-z0-9][a-z0-9-]{0,63}$/; +export const NOTES_FILENAME = "notes.json"; + +/** + * The ONE corpus path umtool may write: `SITES_DIR/<site>/reports/<id>/notes.json`. + * + * Lexically: exactly four segments under SITES_DIR, the second `reports`, the + * site and report ids in the report-id grammar, the file named notes.json, and + * no `..` or doubled separator anywhere. Then on disk: the report directory's + * REAL path must be the lexical one under the real SITES_DIR -- a report dir + * that is a symlink out of its site (or a site dir that is one) is refused -- + * and it must already exist: a note never creates a report directory. A + * notes.json that exists and is not a plain file (a symlink) is refused too. + * + * WRITE_ROOTS is untouched: nothing else in the corpus becomes writable. + * + * @param {string} abs + * @param {{ sitesDir?: string }} [opts] + * @returns {Promise<boolean>} + */ +export async function isCorpusNotesFile(abs, { sitesDir = SITES_DIR } = {}) { + if (typeof abs !== "string" || !path.isAbsolute(abs) || abs.includes("\0")) return false; + if (path.resolve(/* turbopackIgnore: true */ abs) !== abs) return false; + const root = path.resolve(/* turbopackIgnore: true */ sitesDir); + const rel = path.relative(/* turbopackIgnore: true */ root, abs); + if (!rel || rel.startsWith("..") || path.isAbsolute(rel)) return false; + const parts = rel.split(path.sep); + if (parts.length !== 4) return false; + const [site, reports, report, name] = parts; + if (!SEGMENT_RE.test(site) || reports !== "reports" || !SEGMENT_RE.test(report) || name !== NOTES_FILENAME) { + return false; + } + const [realRoot, realDir] = await Promise.all([ + realpath(/* turbopackIgnore: true */ root).catch(() => null), + realpath(/* turbopackIgnore: true */ path.dirname(/* turbopackIgnore: true */ abs)).catch(() => null), + ]); + if (!realRoot || !realDir) return false; + if (realDir !== path.join(/* turbopackIgnore: true */ realRoot, site, "reports", report)) return false; + const st = await lstat(/* turbopackIgnore: true */ abs).catch(() => null); + return !st || st.isFile(); +} + +/** `SITES_DIR/<site>/reports/<report>/notes.json`, or null for a bad id. */ +export function corpusNotesFile(site, report, { sitesDir = SITES_DIR } = {}) { + if (!SEGMENT_RE.test(String(site)) || !SEGMENT_RE.test(String(report))) return null; + return path.join(/* turbopackIgnore: true */ path.resolve(/* turbopackIgnore: true */ sitesDir), site, "reports", report, NOTES_FILENAME); +} + export const WRITE_ROOTS = dedupe( process.env.MIX_WRITE_ROOTS ? process.env.MIX_WRITE_ROOTS.split(":").filter(Boolean) diff --git a/umtool/lib/paths.ts b/umtool/lib/paths.ts @@ -20,7 +20,9 @@ import path from "node:path"; // --------------------------------------------------------------------------- export { CACHE_DIR, + CHANNELS_DIR, cacheFile, + corpusNotesFile, INDEX_DIR, MEDIA_ROOT, MEDIA_ROOTS, @@ -31,8 +33,10 @@ export { SONG_DATA, SONG_REPORTS, SONG_SCRATCH, + SITES_DIR, WRITE_ROOTS, inside, + isCorpusNotesFile, labelFor, mediaMirror, resolveInRoots,