// Which file an agent should EDIT to change an article. // // A report.json under transcripts/sites/ is generated: a workspace under // ~/reports keeps the draft (`/polemics/drafts/.json`, the source of // truth) and a generator script that writes report.json from it // (`/polemics/make-site.py`, `/site/polemics.py`, ...). A note that // says "fix this sentence" is useless to an agent that edits report.json -- the // next generator run puts the old sentence back. So every notes.json carries a // `source` block naming the draft and the generator, found here. // // The match is a heuristic, written down with its reason (`how`), and the // agent may correct it (`umtool notes source`): // // draft a drafts/*.json whose `id`, or file name, is the report id -- // allowing for a `polemic-` prefix on either side (candalyzer's // polemic-israel is drafts/israel.json with id polemic-israel; // jeralyzer-private's `blame` is drafts/blame.json with id // polemic-blame). Several matches: the one whose workspace has a // generator naming the site wins; still several, none is chosen. // generator a *.py / *.mts under /polemics or /site that names the // site (or the report id), preferring one that names the report // id itself, then one that reads the drafts. Backups // (`make-report.pre-2026-10-05.py`: a second dot) are skipped. // // Cheap: one readdir per workspace and one read per generator, cached for 30 s // like the project walk. import { readdir, readFile, stat } from "node:fs/promises"; import os from "node:os"; import path from "node:path"; import { REPORTS_ROOT } from "../paths.mjs"; const CACHE_MS = 30_000; const GEN_DIRS = ["polemics", "site"]; const GEN_EXT = /^[^.]+\.(py|mts|mjs|sh)$/; const MAX_GEN_BYTES = 2 * 1024 * 1024; /** `~/…` for a path under the home directory: what a note shows an agent. */ export function tildify(abs) { const home = os.homedir(); return abs === home || abs.startsWith(home + path.sep) ? `~${abs.slice(home.length)}` : abs; } /** The ids a report or draft may go by: itself, without `polemic-`, with it. */ export function idKeys(id) { const bare = id.replace(/^polemic-/, ""); return new Set([id, bare, `polemic-${bare}`]); } /** Does `text` name `id` as a whole token (so `jasolyzer` is not `jasolyzer-private`)? */ export function names(text, id) { const esc = id.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); return new RegExp(`(? stat(/* turbopackIgnore: true */ p).then((s) => s.isDirectory(), () => false); /** @type {Map }>} */ const cache = new Map(); /** * Every workspace under `reportsRoot` that has drafts or a generator dir: * `{ dir, name, drafts: [{ file, slug, id }], generators: [{ file, text }] }`. * * @param {string} [reportsRoot] */ export function scanWorkspaces(reportsRoot = REPORTS_ROOT) { const hit = cache.get(reportsRoot); if (hit && Date.now() - hit.at < CACHE_MS) return hit.value; const value = scan(reportsRoot); cache.set(reportsRoot, { at: Date.now(), value }); value.catch(() => cache.delete(reportsRoot)); return value; } /** Forget the scan (a test that writes a fixture, then reads it). */ export function clearSourcesCache() { cache.clear(); } async function scan(reportsRoot) { const entries = await readdir(/* turbopackIgnore: true */ reportsRoot, { withFileTypes: true }).catch(() => []); const out = []; for (const e of entries) { if (e.name.startsWith(".") || e.name === "data") continue; const dir = path.join(/* turbopackIgnore: true */ reportsRoot, e.name); if (!(e.isDirectory() || (e.isSymbolicLink() && (await isDir(dir))))) continue; const drafts = []; const draftsDir = path.join(/* turbopackIgnore: true */ dir, "polemics", "drafts"); for (const f of await readdir(/* turbopackIgnore: true */ draftsDir).catch(() => [])) { if (!f.endsWith(".json")) continue; const file = path.join(/* turbopackIgnore: true */ draftsDir, f); let id = null; try { const j = JSON.parse(await readFile(/* turbopackIgnore: true */ file, "utf8")); if (j && typeof j.id === "string") id = j.id; } catch { // an unreadable draft still matches by its file name } drafts.push({ file, slug: f.slice(0, -5), id }); } const generators = []; for (const g of GEN_DIRS) { const gdir = path.join(/* turbopackIgnore: true */ dir, g); for (const f of await readdir(/* turbopackIgnore: true */ gdir).catch(() => [])) { if (!GEN_EXT.test(f)) continue; const file = path.join(/* turbopackIgnore: true */ gdir, f); const st = await stat(/* turbopackIgnore: true */ file).catch(() => null); if (!st?.isFile() || st.size > MAX_GEN_BYTES) continue; generators.push({ file, text: await readFile(/* turbopackIgnore: true */ file, "utf8").catch(() => "") }); } } if (drafts.length || generators.length) { drafts.sort((a, b) => a.file.localeCompare(b.file)); generators.sort((a, b) => a.file.localeCompare(b.file)); out.push({ dir, name: e.name, drafts, generators }); } } return out.sort((a, b) => a.name.localeCompare(b.name)); } /** Is `draft` this report's? */ function draftMatches(draft, keys) { return (draft.id !== null && keys.has(draft.id)) || keys.has(draft.slug) || keys.has(`polemic-${draft.slug}`); } /** * The source of truth for one report, or null when no workspace claims it. * * @param {string} siteId * @param {string} reportId * @param {{ reportsRoot?: string }} [opts] * @returns {Promise<{ draft?: string, generator?: string, how: string, workspace?: string } | null>} */ export async function sourceFor(siteId, reportId, { reportsRoot = REPORTS_ROOT } = {}) { const workspaces = await scanWorkspaces(reportsRoot); const keys = idKeys(reportId); const namesSite = (ws) => ws.generators.some((g) => names(g.text, siteId)); const cands = []; for (const ws of workspaces) { for (const d of ws.drafts) { if (!draftMatches(d, keys)) continue; const score = (namesSite(ws) ? 4 : 0) + (d.id === reportId ? 2 : 0) + (d.slug === reportId.replace(/^polemic-/, "") ? 1 : 0); cands.push({ ws, d, score }); } } cands.sort((a, b) => b.score - a.score); const top = cands[0]; const unique = top && (cands.length === 1 || cands[1].score < top.score); const draft = unique ? top : null; const pool = draft ? [draft.ws] : workspaces; let gen = null; let genScore = 0; for (const ws of pool) { for (const g of ws.generators) { const site = names(g.text, siteId); const report = names(g.text, reportId); if (!site && !report) continue; const base = path.basename(g.file); const score = (report ? 3 : 0) + (site ? 2 : 0) + (draft && /\bdrafts\b/.test(g.text) ? 1 : 0) + (/^(make-site|polemics)\./.test(base) ? 0.5 : 0); if (score > genScore) { gen = { ws, g }; genScore = score; } } } // Without a draft, a generator that only names the SITE is every report's // generator and says nothing about this one; keep it only if it names the id. // A bare id (`deleted`, `poker`) is also an English word, so naming it is // only evidence when the generator names the site too. if (!draft && gen && !(names(gen.g.text, reportId) && (reportId.includes("-") || names(gen.g.text, siteId)))) gen = null; if (!draft && !gen) { if (cands.length > 1) { return { how: `several drafts match ${reportId}: ${cands.map((c) => tildify(c.d.file)).join(", ")}; none chosen` }; } return null; } const how = []; if (draft) { const by = draft.d.id === reportId ? `id ${reportId}` : draft.d.id && keys.has(draft.d.id) ? `id ${draft.d.id}` : `file name ${draft.d.slug}`; how.push(`draft matched by ${by}`); if (cands.length > 1) how.push(`preferred over ${cands.length - 1} other`); } if (gen) { const named = [siteId, reportId].filter((id) => names(gen.g.text, id)); how.push(`generator names ${named.join(" and ")}`); } const out = { how: how.join("; ") }; if (draft) out.draft = tildify(draft.d.file); if (gen) out.generator = tildify(gen.g.file); out.workspace = tildify((draft?.ws ?? gen?.ws).dir); return out; } /** * The workspace directories a site's articles come from: every workspace that * holds a matched draft for one of `reportIds`, or a generator that names the * site. Absolute paths, sorted. * * @param {string} siteId * @param {string[]} reportIds * @param {{ reportsRoot?: string }} [opts] */ export async function siteWorkspaces(siteId, reportIds, { reportsRoot = REPORTS_ROOT } = {}) { const workspaces = await scanWorkspaces(reportsRoot); const dirs = new Set(); for (const ws of workspaces) { if (ws.generators.some((g) => names(g.text, siteId))) dirs.add(ws.dir); } for (const id of reportIds) { const s = await sourceFor(siteId, id, { reportsRoot }); const ws = s?.draft ? workspaces.find((w) => w.drafts.some((d) => tildify(d.file) === s.draft)) : null; if (ws) dirs.add(ws.dir); } return [...dirs].sort(); }