Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit c287d7dd7112680482363cc556c44eefa1888de7
parent 5a717dd1901e8277854225d48f61c207305890e7
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu,  8 Oct 2026 22:42:37 -0400

umtool: /sites -- every site's articles, site pages, media and workspace files

- /sites lists every site (private first) and every article, published or
  draft, with its citations, open notes, poster, linked video project and
  source draft; link filters site/status/notes=open
- /sites/<site>: articles, report videos, linked projects with take tallies,
  the workspace files (md rendered, JSON folded, HTML sandboxed)
- lib/articles: sites (common's listSites/listReportDirs/parseReport),
  links (manifest `article` key, else a unique slug in the draft's workspace),
  workspace, evidence (prepared clip > window > saved > fetch_clip line),
  article (page view with umtool's own record resolver)
- read-only routes: /api/sites/{media,evidence,workspace}
- open notes are open decisions (lib/decisions.ts noteDecisions)
- nav: sites
- e2e fixture: two sites, three reports, a workspace and a linked project

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Aumtool/app/api/sites/evidence/route.ts | 19+++++++++++++++++++
Aumtool/app/api/sites/media/route.ts | 94+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/app/api/sites/workspace/route.ts | 23+++++++++++++++++++++++
Mumtool/app/browse/decisions/page.tsx | 4+++-
Aumtool/app/sites/[site]/page.tsx | 123+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/app/sites/page.tsx | 87+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/components/AppNav.tsx | 6+++++-
Aumtool/components/articles/ArticleTable.tsx | 103+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/components/articles/SiteChips.tsx | 14++++++++++++++
Aumtool/components/articles/WorkspacePanel.tsx | 110+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/e2e/fixtures/sites-fixture.mjs | 159+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Aumtool/lib/articles/article.ts | 178+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/articles/evidence.ts | 182+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/articles/files.ts | 77+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/articles/links.mjs | 99+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/articles/sites.ts | 177+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/articles/workspace.mjs | 65+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/lib/decisions.ts | 77+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/lib/projects.ts | 8+++++---
19 files changed, 1597 insertions(+), 8 deletions(-)

diff --git a/umtool/app/api/sites/evidence/route.ts b/umtool/app/api/sites/evidence/route.ts @@ -0,0 +1,19 @@ +import { citationEvidence } from "@/lib/articles/evidence"; +import { readReportFile, siteById } from "@/lib/articles/sites"; + +export const dynamic = "force-dynamic"; + +// GET ?site&report&cite -- one citation's evidence (lib/articles/evidence.ts): +// its quote and record, the transcript around it, and what can play it. +export async function GET(request: Request) { + const url = new URL(request.url); + const site = url.searchParams.get("site") ?? ""; + const reportId = url.searchParams.get("report") ?? ""; + const cite = url.searchParams.get("cite") ?? ""; + if (!siteById(site)) return Response.json({ error: `no site ${site}` }, { status: 404 }); + const read = await readReportFile(site, reportId); + if (!read.report) return Response.json({ error: read.problems[0]?.message ?? "no report" }, { status: 404 }); + const ev = await citationEvidence(site, read.report, cite); + if (!ev) return Response.json({ error: `no citation ${cite}` }, { status: 404 }); + return Response.json(ev, { headers: { "cache-control": "no-store" } }); +} diff --git a/umtool/app/api/sites/media/route.ts b/umtool/app/api/sites/media/route.ts @@ -0,0 +1,94 @@ +import { readFile, realpath, stat } from "node:fs/promises"; +import path from "node:path"; +import { reportMediaDir, reportMediaIndexFile, siteReportDir } from "yt-dlp-transcript-common/publish/reportMedia"; +import { rangeResponse } from "@/lib/report/serve.mjs"; +import { CHANNELS_DIR, REPORTS_ROOT, SITES_DIR, inside } from "@/lib/paths"; +import { sitesPaths } from "@/lib/articles/sites"; + +export const dynamic = "force-dynamic"; + +// Article media, READ-ONLY, with byte ranges (lib/report/serve.mjs +// rangeResponse, the one the cut player uses -- a <video> will not seek a +// stream it was not given a 206 for). +// +// ?site&report&file a file in the report's directory: video.mp4, +// poster.jpg, stills/… -- relative, no `..`, a media type, +// and its real path under SITES_DIR (or REPORTS_ROOT: a +// generator may link a take's preview in) +// ?site&moment the site's PREPARED evidence clip for a moment, found +// through report-media/index.json -- never a client path +// ?corpus=<abs> a clip window, saved video, audio or post capture in +// the corpus: lexically under CHANNELS_DIR (its real +// path is on whatever drive the media tier links to) +const TYPES: Record<string, string> = { + ".mp4": "video/mp4", + ".m4v": "video/mp4", + ".webm": "video/webm", + ".mkv": "video/x-matroska", + ".mov": "video/quicktime", + ".m4a": "audio/mp4", + ".mp3": "audio/mpeg", + ".opus": "audio/ogg", + ".ogg": "audio/ogg", + ".wav": "audio/wav", + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".png": "image/png", + ".webp": "image/webp", + ".gif": "image/gif", +}; +const SEGMENT = /^[a-z0-9][a-z0-9-]{0,63}$/; + +async function serve(request: Request, abs: string): Promise<Response> { + const type = TYPES[path.extname(abs).toLowerCase()]; + if (!type) return new Response("not a media file", { status: 400 }); + const st = await stat(/* turbopackIgnore: true */ abs).catch(() => null); + if (!st?.isFile()) return new Response("not found", { status: 404 }); + return rangeResponse(request, { + abs, + size: st.size, + headers: { "content-type": type, "accept-ranges": "bytes", "cache-control": "private, no-store" }, + }); +} + +async function reportFile(site: string, report: string, rel: string): Promise<string | null> { + if (!SEGMENT.test(site) || !SEGMENT.test(report)) return null; + if (!rel || rel.includes("\0") || rel.startsWith("/") || rel.split("/").some((s) => s === ".." || s === "" || s === ".")) return null; + const dir = siteReportDir(sitesPaths(), site, report); + const abs = path.join(/* turbopackIgnore: true */ dir, rel); + if (!inside(dir, abs)) return null; + const real = await realpath(/* turbopackIgnore: true */ abs).catch(() => null); + if (!real) return null; + const roots = await Promise.all([SITES_DIR, REPORTS_ROOT].map((r) => realpath(/* turbopackIgnore: true */ r).catch(() => r))); + return roots.some((r) => inside(r, real)) ? real : null; +} + +async function preparedFile(site: string, moment: string): Promise<string | null> { + if (!SEGMENT.test(site)) return null; + try { + const index = JSON.parse(await readFile(/* turbopackIgnore: true */ reportMediaIndexFile(sitesPaths(), site), "utf8")); + const entry = index?.moments?.[moment]; + if (!entry || typeof entry.file !== "string") return null; + const dir = reportMediaDir(sitesPaths(), site); + const abs = path.resolve(/* turbopackIgnore: true */ dir, entry.file); + return inside(dir, abs) ? abs : null; + } catch { + return null; + } +} + +export async function GET(request: Request) { + const url = new URL(request.url); + const q = (k: string) => url.searchParams.get(k) ?? ""; + if (url.searchParams.has("corpus")) { + const abs = path.resolve(/* turbopackIgnore: true */ q("corpus")); + if (abs !== q("corpus") || !inside(CHANNELS_DIR, abs)) return new Response("outside the corpus", { status: 400 }); + return serve(request, abs); + } + if (url.searchParams.has("moment")) { + const abs = await preparedFile(q("site"), q("moment")); + return abs ? serve(request, abs) : new Response("no prepared clip for that moment", { status: 404 }); + } + const abs = await reportFile(q("site"), q("report"), q("file")); + return abs ? serve(request, abs) : new Response("no such report file", { status: 404 }); +} diff --git a/umtool/app/api/sites/workspace/route.ts b/umtool/app/api/sites/workspace/route.ts @@ -0,0 +1,23 @@ +import { readFile } from "node:fs/promises"; +import { workspaceFile } from "@/lib/articles/workspace.mjs"; + +export const dynamic = "force-dynamic"; + +// GET ?ws=<name under REPORTS_ROOT>&rel=<listed file> -- one workspace file, +// READ-ONLY, only one that lib/articles/workspace.mjs lists. HTML is served +// under a CSP sandbox (no scripts, no same-origin) for the page's sandboxed +// iframe; markdown and JSON as plain text. +export async function GET(request: Request) { + const url = new URL(request.url); + const f = await workspaceFile(url.searchParams.get("ws") ?? "", url.searchParams.get("rel") ?? ""); + if (!f) return new Response("not a workspace file", { status: 404 }); + const body = await readFile(/* turbopackIgnore: true */ f.real); + const html = f.rel.endsWith(".html"); + return new Response(body, { + headers: { + "content-type": html ? "text/html; charset=utf-8" : f.rel.endsWith(".json") ? "application/json; charset=utf-8" : "text/plain; charset=utf-8", + "cache-control": "no-store", + ...(html ? { "content-security-policy": "sandbox; default-src 'none'; img-src data:; style-src 'unsafe-inline'" } : {}), + }, + }); +} diff --git a/umtool/app/browse/decisions/page.tsx b/umtool/app/browse/decisions/page.tsx @@ -122,7 +122,9 @@ export default async function DecisionsPage({ <section key={id} data-project={id}> <h2 className="mb-1.5 flex items-baseline gap-2"> <Link - href={`/browse/${id}`} + // An article's notes are listed under its page's path + // (lib/decisions.ts noteDecisions), not a project's. + href={id.startsWith("sites/") ? `/${id}` : `/browse/${id}`} className="font-mono text-[13px] text-[var(--color-text)] hover:text-[var(--color-sel)]" > {id} diff --git a/umtool/app/sites/[site]/page.tsx b/umtool/app/sites/[site]/page.tsx @@ -0,0 +1,123 @@ +import Link from "next/link"; +import { notFound } from "next/navigation"; +import BrowseHeader from "@/components/BrowseHeader"; +import ArticleTable, { articleHref } from "@/components/articles/ArticleTable"; +import SiteChips from "@/components/articles/SiteChips"; +import WorkspacePanel from "@/components/articles/WorkspacePanel"; +import { badgeVariants } from "@/components/ui/badge"; +import { readSiteRow, siteById } from "@/lib/articles/sites"; +import { openWorkspaceFile, siteWorkspaceListings, takeTally } from "@/lib/articles/files"; +import { videoProjects } from "@/lib/articles/links.mjs"; + +export const dynamic = "force-dynamic"; + +// One site: its articles, the videos they play, the umtool projects those +// videos were cut in (with their takes), and the workspace files the articles +// were written from. `?ws=&rel=` opens a workspace file below. + +export default async function SitePage({ + params, + searchParams, +}: { + params: Promise<{ site: string }>; + searchParams: Promise<{ ws?: string; rel?: string }>; +}) { + const { site: siteId } = await params; + const sp = await searchParams; + const site = siteById(siteId); + if (!site) notFound(); + const row = await readSiteRow(site); + + const projects = await videoProjects(); + const linkedIds = [...new Set(row.articles.flatMap((a) => a.projects.linked.map((p) => p.id)))]; + const linked = await Promise.all( + linkedIds.map(async (id) => { + const p = projects.find((x: { id: string }) => x.id === id) as { id: string; dir: string }; + return { id: p.id, tally: await takeTally(p.dir), articles: row.articles.filter((a) => a.projects.linked.some((l) => l.id === id)) }; + }), + ); + const videos = row.articles.filter((a) => a.hasVideo); + const workspaces = await siteWorkspaceListings(site.siteId, row.articles.map((a) => a.id)); + const opened = sp.ws && sp.rel ? await openWorkspaceFile(sp.ws, sp.rel) : null; + const hrefFor = (ws: string, rel: string) => + `/sites/${site.siteId}?${new URLSearchParams({ ws, rel }).toString()}#files`; + const media = (id: string, file: string) => + `/api/sites/media?site=${encodeURIComponent(site.siteId)}&report=${encodeURIComponent(id)}&file=${file}`; + + return ( + <div className="flex h-full flex-col"> + <BrowseHeader + active="sites" + crumbs={[{ href: "/sites", label: "sites" }, { label: row.title }]} + note={`${row.published} published · ${row.drafts} drafts · ${row.openNotes} open notes`} + /> + <main className="deck-main flex-1 space-y-6 p-4"> + <div className="flex flex-wrap items-baseline gap-2"> + <h1 className="text-[16px] font-semibold text-[var(--color-text)]">{row.title}</h1> + <span className="font-mono text-[11px] text-[var(--color-dim)]">{row.siteId}</span> + <SiteChips site={row} /> + </div> + + <section data-section="articles"> + <h2 className="micro mb-1.5">articles</h2> + <ArticleTable articles={row.articles} /> + </section> + + <section data-section="videos"> + <h2 className="micro mb-1.5">report videos {videos.length}</h2> + {videos.length === 0 ? ( + <p className="text-[12px] text-[var(--color-dim)]">none</p> + ) : ( + <div className="grid grid-cols-[repeat(auto-fill,minmax(280px,1fr))] gap-3"> + {videos.map((a) => ( + <figure key={a.id} data-report-video={a.id} className="rounded border border-[var(--color-line)] bg-[var(--color-panel)] p-2"> + <video + controls + preload="none" + src={media(a.id, "video.mp4")} + poster={a.hasPoster ? media(a.id, "poster.jpg") : undefined} + className="aspect-video w-full rounded bg-black" + /> + <figcaption className="mt-1 text-[12px]"> + <Link href={articleHref(a)} className="text-[var(--color-text)] hover:text-[var(--color-sel)]"> + {a.title} + </Link> + </figcaption> + </figure> + ))} + </div> + )} + </section> + + <section data-section="projects"> + <h2 className="micro mb-1.5">video projects {linked.length}</h2> + {linked.length === 0 ? ( + <p className="text-[12px] text-[var(--color-dim)]">none linked</p> + ) : ( + <ul className="space-y-1"> + {linked.map((p) => ( + <li key={p.id} data-video-project={p.id} className="flex flex-wrap items-baseline gap-2 rounded border border-[var(--color-line)] bg-[var(--color-panel)] px-3 py-1.5 text-[12px]"> + <Link href={`/browse/${p.id}`} className="font-mono text-[var(--color-sel)] hover:underline"> + {p.id} + </Link> + <span className="text-[var(--color-dim)]">{p.articles.map((a) => a.title).join(", ")}</span> + <Link href={`/browse/${p.id}/takes`} className="ml-auto text-[var(--color-dim)] hover:text-[var(--color-text)]" data-takes={p.tally.takes}> + {p.tally.takes} takes + </Link> + <span className={badgeVariants({ variant: "neutral", size: "sm" })}>like {p.tally.like}</span> + <span className={badgeVariants({ variant: "neutral", size: "sm" })}>maybe {p.tally.maybe}</span> + <span className={badgeVariants({ variant: "neutral", size: "sm" })}>no {p.tally.no}</span> + </li> + ))} + </ul> + )} + </section> + + <section data-section="files" id="files"> + <h2 className="micro mb-1.5">workspace files</h2> + <WorkspacePanel workspaces={workspaces} opened={opened} hrefFor={hrefFor} /> + </section> + </main> + </div> + ); +} diff --git a/umtool/app/sites/page.tsx b/umtool/app/sites/page.tsx @@ -0,0 +1,87 @@ +import Link from "next/link"; +import BrowseHeader from "@/components/BrowseHeader"; +import ArticleTable from "@/components/articles/ArticleTable"; +import SiteChips from "@/components/articles/SiteChips"; +import { badgeVariants } from "@/components/ui/badge"; +import { listSiteRows, type ArticleRow } from "@/lib/articles/sites"; + +export const dynamic = "force-dynamic"; + +// Every site's articles, private sites first. Zero client JS: the filters are +// links that change searchParams (/browse/decisions' idiom), so a filtered +// view is one pasteable URL. + +type Search = { site?: string; status?: string; notes?: string }; + +export default async function SitesPage({ searchParams }: { searchParams: Promise<Search> }) { + const sp = await searchParams; + const all = await listSiteRows(); + + const keep = (a: ArticleRow) => + (!sp.status || a.status === sp.status) && (sp.notes !== "open" || a.openNotes > 0); + const sites = all + .filter((s) => !sp.site || s.siteId === sp.site) + .map((s) => ({ ...s, shown: s.articles.filter(keep) })) + .filter((s) => s.shown.length > 0 || (!sp.status && !sp.notes)); + + const articles = all.flatMap((s) => s.articles); + const open = articles.reduce((n, a) => n + a.openNotes, 0); + const qs = (next: Partial<Search>) => { + const p = new URLSearchParams(); + for (const [k, v] of Object.entries({ ...sp, ...next })) if (v) p.set(k, String(v)); + const s = p.toString(); + return `/sites${s ? `?${s}` : ""}`; + }; + + return ( + <div className="flex h-full flex-col"> + <BrowseHeader active="sites" crumbs={[{ label: "sites" }]} note={`${all.length} sites · ${articles.length} articles · ${open} open notes`} /> + <main className="deck-main flex-1 p-4"> + <div className="mb-3 flex flex-wrap items-center gap-1.5"> + <span className="micro">site</span> + <Chip href={qs({ site: "" })} on={!sp.site} label={`all ${all.length}`} /> + {all.map((s) => ( + <Chip key={s.siteId} href={qs({ site: s.siteId })} on={sp.site === s.siteId} label={s.siteId} /> + ))} + <span className="micro ml-3">status</span> + <Chip href={qs({ status: "" })} on={!sp.status} label="all" /> + <Chip href={qs({ status: "published" })} on={sp.status === "published"} label={`published ${articles.filter((a) => a.status === "published").length}`} /> + <Chip href={qs({ status: "draft" })} on={sp.status === "draft"} label={`draft ${articles.filter((a) => a.status === "draft").length}`} /> + <span className="micro ml-3">notes</span> + <Chip href={qs({ notes: "" })} on={sp.notes !== "open"} label="all" /> + <Chip href={qs({ notes: "open" })} on={sp.notes === "open"} label={`open ${articles.filter((a) => a.openNotes > 0).length}`} /> + </div> + + {sites.length === 0 ? ( + <p className="text-[12px] text-[var(--color-dim)]">{all.length === 0 ? "no sites" : "nothing matches that filter"}</p> + ) : ( + <div className="space-y-6"> + {sites.map((s) => ( + <section key={s.siteId} data-site={s.siteId}> + <h2 className="mb-1.5 flex flex-wrap items-baseline gap-2"> + <Link href={`/sites/${s.siteId}`} className="text-[14px] font-semibold text-[var(--color-text)] hover:text-[var(--color-sel)]"> + {s.title} + </Link> + <span className="font-mono text-[11px] text-[var(--color-dim)]">{s.siteId}</span> + <SiteChips site={s} /> + <span className="micro" data-counts={`${s.published}/${s.drafts}`}> + {s.published} published · {s.drafts} draft{s.drafts === 1 ? "" : "s"} + </span> + </h2> + <ArticleTable articles={s.shown} /> + </section> + ))} + </div> + )} + </main> + </div> + ); +} + +function Chip({ href, on, label }: { href: string; on: boolean; label: string }) { + return ( + <Link href={href} aria-current={on ? "true" : undefined} className={badgeVariants({ variant: on ? "on" : "neutral" })}> + {label} + </Link> + ); +} diff --git a/umtool/components/AppNav.tsx b/umtool/components/AppNav.tsx @@ -6,7 +6,8 @@ import NavGroup from "./NavGroup"; // is waiting, the two benches that are not a project (mix, find), and the song // piles folded under one entry. // -// SEVEN visible entries, and the cap is still NINE. A tenth wraps the header on +// SEVEN visible entries (home, browse, decisions, sites, mix, find, song ▸), +// and the cap is still NINE. A tenth wraps the header on // a laptop, and a nav that wraps stops reading as one row of places and starts // reading as a list. The next tool goes UNDER one of these, not beside them -- // which is exactly what happened to the four judging piles and the sources @@ -30,6 +31,9 @@ export default function AppNav({ active }: { active: string }) { // The worklist across every project, not a sixth pile. It sits beside // browse because that is where every decision it names gets settled. { href: "/browse/decisions", label: "decisions" }, + // Every site's articles -- published and drafts -- with their notes, their + // evidence and the workspace they were written in. + { href: "/sites", label: "sites" }, { href: "/mix", label: "mix" }, // Every occurrence of a word across the corpus. It sits with browse because // what it retrieves is raw material for a build, not a pile to judge. diff --git a/umtool/components/articles/ArticleTable.tsx b/umtool/components/articles/ArticleTable.tsx @@ -0,0 +1,103 @@ +import Link from "next/link"; +import { badgeVariants } from "@/components/ui/badge"; +import { fmtAgo } from "@/lib/format"; +import type { ArticleRow } from "@/lib/articles/sites"; + +// One row per article: what it is, where it stands, what is waiting on it, +// and where it came from. Server-rendered, zero JS -- the /browse idiom. + +const posterUrl = (a: ArticleRow) => + `/api/sites/media?site=${encodeURIComponent(a.site)}&report=${encodeURIComponent(a.id)}&file=poster.jpg`; + +const baseName = (p: string) => p.split("/").pop() ?? p; + +export function articleHref(a: { site: string; id: string }) { + return `/sites/${a.site}/${a.id}`; +} + +export default function ArticleTable({ articles }: { articles: ArticleRow[] }) { + if (articles.length === 0) return <p className="text-[12px] text-[var(--color-dim)]">no articles</p>; + return ( + <table className="w-full border-collapse text-[12px]" data-testid="article-table"> + <thead> + <tr className="text-left text-[11px] text-[var(--color-dim)]"> + <th className="w-[72px] py-1 font-normal" /> + <th className="py-1 font-normal">article</th> + <th className="py-1 font-normal">status</th> + <th className="py-1 font-normal">updated</th> + <th className="py-1 text-right font-normal">cites</th> + <th className="py-1 text-right font-normal">notes</th> + <th className="py-1 pl-3 font-normal">video project</th> + <th className="py-1 font-normal">source</th> + </tr> + </thead> + <tbody> + {articles.map((a) => ( + <tr + key={`${a.site}/${a.id}`} + data-article={`${a.site}/${a.id}`} + data-status={a.status} + className="border-t border-[var(--color-line)] align-top" + > + <td className="py-1.5 pr-2"> + {a.hasPoster ? ( + // eslint-disable-next-line @next/next/no-img-element + <img src={posterUrl(a)} alt="" loading="lazy" className="h-9 w-16 rounded object-cover" /> + ) : ( + <div className="h-9 w-16 rounded bg-[var(--color-panel-2)]" /> + )} + </td> + <td className="py-1.5 pr-3"> + <Link href={articleHref(a)} className="text-[var(--color-text)] hover:text-[var(--color-sel)]"> + {a.title} + </Link> + <div className="font-mono text-[11px] text-[var(--color-dim)]"> + {a.id} + {a.series ? ` · ${a.series}` : ""} + {a.problem ? <span className="text-[var(--color-bad)]"> · {a.problems} problem{a.problems === 1 ? "" : "s"}</span> : null} + </div> + </td> + <td className="py-1.5 pr-3"> + <span className={badgeVariants({ variant: a.status === "published" ? "on" : "neutral", size: "sm" })}> + {a.status} + </span> + </td> + <td className="num py-1.5 pr-3 text-[var(--color-dim)]"> + {a.updated ?? (a.mtimeMs ? fmtAgo(a.mtimeMs) : "—")} + </td> + <td className="num py-1.5 pr-3 text-right">{a.citations}</td> + <td className="num py-1.5 text-right"> + {a.openNotes > 0 ? ( + <Link + href={`${articleHref(a)}?status=open`} + data-open-notes={a.openNotes} + className={badgeVariants({ variant: "open", size: "sm" })} + > + {a.openNotes} open + </Link> + ) : ( + <span className="text-[var(--color-dim)]">{a.notes || "—"}</span> + )} + </td> + <td className="py-1.5 pl-3 pr-3"> + {a.projects.linked.map((p) => ( + <Link key={p.id} href={`/browse/${p.id}`} data-project-link={p.id} className="block font-mono text-[11px] text-[var(--color-sel)] hover:underline"> + {p.id} + </Link> + ))} + {a.projects.possible.map((p) => ( + <Link key={p.id} href={`/browse/${p.id}`} title="shares the slug; not linked" className="block font-mono text-[11px] text-[var(--color-dim)] hover:underline"> + {p.id} (possible) + </Link> + ))} + {a.projects.linked.length + a.projects.possible.length === 0 && <span className="text-[var(--color-dim)]">—</span>} + </td> + <td className="py-1.5 font-mono text-[11px] text-[var(--color-dim)]" title={a.source?.how ?? ""}> + {a.source?.draft ? baseName(a.source.draft) : a.source?.generator ? baseName(a.source.generator) : "—"} + </td> + </tr> + ))} + </tbody> + </table> + ); +} diff --git a/umtool/components/articles/SiteChips.tsx b/umtool/components/articles/SiteChips.tsx @@ -0,0 +1,14 @@ +import { badgeVariants } from "@/components/ui/badge"; + +// A site's audience, listing and search, as three short chips. +export default function SiteChips({ site }: { site: { private: boolean; listed: boolean; search: boolean } }) { + return ( + <span className="inline-flex gap-1"> + <span data-audience={site.private ? "private" : "public"} className={badgeVariants({ variant: site.private ? "meter" : "neutral", size: "sm" })}> + {site.private ? "private" : "public"} + </span> + <span className={badgeVariants({ variant: "info", size: "sm" })}>{site.listed ? "listed" : "unlisted"}</span> + <span className={badgeVariants({ variant: "info", size: "sm" })}>{site.search ? "search" : "cited only"}</span> + </span> + ); +} diff --git a/umtool/components/articles/WorkspacePanel.tsx b/umtool/components/articles/WorkspacePanel.tsx @@ -0,0 +1,110 @@ +import Link from "next/link"; +import { Markdown } from "@/lib/markdown"; +import { fmtAgo, fmtBytes } from "@/lib/format"; +import type { OpenedFile, WorkspaceListing } from "@/lib/articles/files"; + +// The files an article was written from, and the one that is open. Zero JS: +// each file is a link that sets `?ws=&rel=` on the page it sits on; markdown +// renders, a draft's JSON is pretty-printed with each top-level key folded, +// and HTML opens in a sandboxed iframe (no scripts, no same origin). + +export default function WorkspacePanel({ + workspaces, + opened, + hrefFor, + highlight, +}: { + workspaces: WorkspaceListing[]; + opened: OpenedFile | null; + hrefFor: (ws: string, rel: string) => string; + /** A file to mark (the article's own draft), as `<ws>/<rel>`. */ + highlight?: string | null; +}) { + if (workspaces.length === 0) return <p className="text-[12px] text-[var(--color-dim)]">no workspace found</p>; + return ( + <div className="grid gap-4 min-[1100px]:grid-cols-[minmax(240px,320px)_1fr]" data-testid="workspace-panel"> + <div className="space-y-3"> + {workspaces.map((w) => ( + <div key={w.name} data-workspace={w.name}> + <div className="mb-1 font-mono text-[11px] text-[var(--color-dim)]">{w.dir}</div> + <ul className="space-y-0.5"> + {w.files.map((f) => { + const on = opened?.ws === w.name && opened.rel === f.rel; + const mine = highlight === `${w.name}/${f.rel}`; + return ( + <li key={f.rel} className="flex items-baseline gap-2 text-[12px]"> + <Link + href={hrefFor(w.name, f.rel)} + aria-current={on ? "true" : undefined} + data-file={f.rel} + className={`truncate font-mono ${on ? "text-[var(--color-sel)]" : mine ? "text-[var(--color-text)]" : "text-[var(--color-dim)] hover:text-[var(--color-text)]"}`} + > + {f.rel} + </Link> + {mine && <span className="micro">this article</span>} + <span className="num ml-auto shrink-0 text-[10px] text-[var(--color-dim)]"> + {fmtBytes(f.bytes)} · {fmtAgo(f.mtimeMs)} + </span> + </li> + ); + })} + </ul> + </div> + ))} + </div> + <div className="min-w-0">{opened ? <Opened file={opened} /> : <p className="text-[12px] text-[var(--color-dim)]">pick a file</p>}</div> + </div> + ); +} + +function Opened({ file }: { file: OpenedFile }) { + const head = <div className="mb-2 font-mono text-[11px] text-[var(--color-dim)]">{file.rel}</div>; + if (file.kind === "error") { + return ( + <div> + {head} + <p className="text-[12px] text-[var(--color-bad)]">{file.message}</p> + </div> + ); + } + if (file.kind === "html") { + return ( + <div> + {head} + <iframe title={file.rel} src={file.url} sandbox="" className="h-[70vh] w-full rounded border border-[var(--color-line)] bg-white" /> + </div> + ); + } + if (file.kind === "json") { + const v = file.value; + const entries = v && typeof v === "object" && !Array.isArray(v) ? Object.entries(v as Record<string, unknown>) : null; + return ( + <div data-opened="json"> + {head} + {entries ? ( + <div className="space-y-1"> + {entries.map(([k, val]) => ( + <details key={k} open={typeof val !== "object" || val === null} className="rounded border border-[var(--color-line)] bg-[var(--color-panel)]"> + <summary className="cursor-pointer px-2 py-1 font-mono text-[12px] text-[var(--color-text)]"> + {k} + {Array.isArray(val) ? <span className="micro ml-2">{val.length} items</span> : null} + </summary> + <pre className="max-h-[50vh] overflow-auto whitespace-pre-wrap px-2 pb-2 font-mono text-[11px] text-[var(--color-dim)]"> + {JSON.stringify(val, null, 2)} + </pre> + </details> + ))} + </div> + ) : ( + <pre className="overflow-auto whitespace-pre-wrap font-mono text-[11px]">{JSON.stringify(v, null, 2)}</pre> + )} + </div> + ); + } + return ( + <div data-opened="md"> + {head} + <Markdown text={file.text} /> + </div> + ); +} diff --git a/umtool/e2e/fixtures/sites-fixture.mjs b/umtool/e2e/fixtures/sites-fixture.mjs @@ -5,14 +5,167 @@ // // Called by make-fixture.mjs after the projects and the channels exist, so a // report here can cite the fixture's channels and link its projects. -import { mkdirSync } from "node:fs"; +// +// sites/priv private, cited-only. Reports: +// polemic-alpha PUBLISHED: summary, two sections, cites sitechan/sv1 +// (a clip window on disk: plays) and sitechan/sv2 (cues, +// no media: the fetch_clip line); video.mp4 + poster.jpg +// polemic-beta a DRAFT (not in site.json's reports) +// sites/pub public. Reports: +// gamma PUBLISHED, one section, no citations +// <dest>/sitews/ the WORKSPACE (a direct child of REPORTS_ROOT, which the +// e2e server takes as <dest>): polemics/drafts/{alpha,beta}.json, +// polemics/make-site.py naming `priv`, polemics/out/alpha.{md,html}, +// NOTES.md +// <dest>/sitews/polemic-alpha/ +// the article's report-video project: a manifest whose slug +// is polemic-alpha, two takes, one verdict +// channels/sitechan/data/{sv1,sv2} the cited records +import { spawnSync } from "node:child_process"; +import { mkdirSync, writeFileSync } from "node:fs"; import path from "node:path"; +const put = (file, value) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, typeof value === "string" ? value : JSON.stringify(value, null, 2) + "\n"); +}; + +const ff = (args) => { + const r = spawnSync("ffmpeg", ["-nostdin", "-v", "error", "-y", ...args], { encoding: "utf8" }); + if (r.status !== 0) throw new Error(`ffmpeg failed: ${r.stderr || r.status}`); +}; + +const clip = (out, seconds, hz) => { + mkdirSync(path.dirname(out), { recursive: true }); + ff([ + "-f", "lavfi", "-i", `testsrc=size=320x180:rate=15:duration=${seconds}`, + "-f", "lavfi", "-i", `sine=frequency=${hz}:duration=${seconds}`, + "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", "-shortest", "-movflags", "+faststart", out, + ]); +}; + +export const SITE_CUES = { + sv1: [ + [0, 3, "Opening line of the first source."], + [3, 6, "The vote was rigged and everybody knew."], + [6, 9, "Nobody checked it at the time."], + [9, 12, "Later the story changed again."], + [12, 15, "A fifth line for context."], + [15, 18, "And a sixth to close."], + ], + sv2: [ + [0, 4, "The second source starts here."], + [4, 8, "She said it on a different show."], + [8, 12, "Then she said the opposite."], + ], +}; + +const ALPHA_BODY_1 = + "She said [the vote was rigged](cite:c1) in the first stream. Nobody checked the claim at the time, and the clip shows it.\n\nThe second paragraph repeats the claim for the record."; +const ALPHA_BODY_2 = "Two years later [she said the opposite](cite:c2), on a different show."; + +const siteJson = (siteId, title, extra) => ({ + siteId, + siteTitle: title, + groups: [{ id: "default", name: "All channels", selectedByDefault: true, order: 0, inline: true }], + defaultGroupId: "default", + channels: [{ slug: "sitechan", order: 0 }], + ...extra, +}); + /** * @param {{ dest: string, reports: string, channels: string }} at */ -export function makeSitesFixture({ dest }) { +export function makeSitesFixture({ dest, channels }) { const sites = path.join(dest, "sites"); mkdirSync(sites, { recursive: true }); - return { sites }; + + // ---- the cited records ---------------------------------------------------- + for (const [vid, rows] of Object.entries(SITE_CUES)) { + put(path.join(channels, "sitechan", "data", vid, "transcript.cues.json"), { + title: `Site source ${vid}`, + uploadDate: "20240315", + channel: "Site Channel", + webpageUrl: `https://www.youtube.com/watch?v=${vid}`, + duration: rows[rows.length - 1][1], + cues: rows.map(([start, end, text]) => ({ start, end, text })), + }); + } + // sv1 has a clip window the editor fetched: [0, 12] holds the cited [3, 6]. + clip(path.join(channels, "sitechan", "data", "sv1", "clips", "0.00-12.00.mp4"), 12, 523); + + // ---- the sites ------------------------------------------------------------ + put(path.join(sites, "priv", "site.json"), siteJson("priv", "Private Fixture", { audience: "private", search: false, reports: ["polemic-alpha"] })); + put(path.join(sites, "pub", "site.json"), siteJson("pub", "Public Fixture", { reports: ["gamma"] })); + + const alphaDir = path.join(sites, "priv", "reports", "polemic-alpha"); + put(path.join(alphaDir, "report.json"), { + format: "archilyzer-report", + version: 1, + id: "polemic-alpha", + kind: "sweep", + series: "Polemics", + title: "Alpha: the rigged vote", + subtitle: "What she said, and when", + summary: "She said one thing in 2020 and the opposite later.", + published: "2026-10-01", + updated: "2026-10-07", + video: { src: "video.mp4", poster: "poster.jpg" }, + citations: { + c1: { kind: "video", channel: "sitechan", id: "sv1", start: 3, end: 6, quote: "The vote was rigged and everybody knew.", speaker: "Site Channel", date: "2024-03-15" }, + c2: { kind: "video", channel: "sitechan", id: "sv2", start: 8, end: 12, quote: "Then she said the opposite.", speaker: "Site Channel", date: "2024-03-15" }, + }, + sections: [ + { id: "first", title: "The first claim", body: ALPHA_BODY_1 }, + { id: "later", title: "Later", body: ALPHA_BODY_2 }, + ], + }); + clip(path.join(alphaDir, "video.mp4"), 4, 440); + ff(["-f", "lavfi", "-i", "testsrc=size=320x180:rate=1:duration=1", "-frames:v", "1", path.join(alphaDir, "poster.jpg")]); + + put(path.join(sites, "priv", "reports", "polemic-beta", "report.json"), { + format: "archilyzer-report", + version: 1, + id: "polemic-beta", + kind: "sweep", + title: "Beta: a draft", + summary: "A draft nobody has published.", + sections: [{ id: "only", title: "Only section", body: "The draft's only paragraph." }], + }); + + put(path.join(sites, "pub", "reports", "gamma", "report.json"), { + format: "archilyzer-report", + version: 1, + id: "gamma", + kind: "sweep", + title: "Gamma on the public site", + published: "2026-09-01", + sections: [{ id: "g", title: "Gamma", body: "A public article with nothing cited." }], + }); + + // ---- the workspace they were written in ------------------------------------ + const ws = path.join(dest, "sitews"); + put(path.join(ws, "polemics", "drafts", "alpha.json"), { id: "polemic-alpha", title: "Alpha: the rigged vote", sections: [] }); + put(path.join(ws, "polemics", "drafts", "beta.json"), { id: "polemic-beta", title: "Beta: a draft", sections: [] }); + put(path.join(ws, "polemics", "make-site.py"), 'SITE = "priv"\nfor f in DRAFTS.glob("drafts/*.json"):\n rid = f"polemic-{f.stem}"\n'); + put(path.join(ws, "polemics", "out", "alpha.md"), "# Alpha\n\nThe rendered draft of **alpha**.\n"); + put(path.join(ws, "polemics", "out", "alpha.html"), "<!doctype html><h1>Alpha html</h1><script>document.title='ran'</script>\n"); + put(path.join(ws, "NOTES.md"), "# Workspace notes\n\n- one\n- two\n"); + + // ---- the article's video project ------------------------------------------- + const proj = path.join(ws, "polemic-alpha"); + put(path.join(proj, "video.manifest.json"), { + schemaVersion: 1, + slug: "polemic-alpha", + title: "Alpha: the rigged vote", + generatedBy: "polemics/video/make-videos.py", + provenance: { channelSlug: "sitechan", siteOrigin: "https://priv.example" }, + timeline: [{ type: "clip", id: "a1", video: "sv1", channel: "sitechan", start: 3, end: 6, quote: "The vote was rigged" }], + }); + for (const [id, order] of [["deck", 1], ["tight", 2]]) { + put(path.join(proj, "takes", id, "take.json"), { id, group: "cut", order, label: id, kind: order === 1 ? "reference" : "similar", preview: "preview.mp4" }); + } + put(path.join(proj, "takes", "verdicts.json"), { deck: { verdict: "like", note: "", at: "2026-10-07T00:00:00Z" } }); + + return { sites, workspace: ws, project: proj }; } diff --git a/umtool/lib/articles/article.ts b/umtool/lib/articles/article.ts @@ -0,0 +1,178 @@ +import { readFile, stat } from "node:fs/promises"; +import path from "node:path"; +import { + buildReportPageView, + REPORT_PAGE_FORMAT, + REPORT_VIEWS_VERSION, + type RecordView, + type ReportPageView, +} from "yt-dlp-transcript-common/lib/report/views"; +import type { Report } from "yt-dlp-transcript-common/lib/report/schema"; +import { platformMomentUrl } from "yt-dlp-transcript-common/lib/momentUrl"; +import { readAllPosts } from "yt-dlp-transcript-common/lib/posts-server"; +import type { Post } from "yt-dlp-transcript-common/lib/posts"; +import { readCues } from "@/lib/projects/report.mjs"; +import { CHANNELS_DIR } from "@/lib/paths"; + +// A report's PAGE VIEW, built the way the export site builds it +// (common/lib/report/views.ts buildReportPageView) but resolved against what +// umtool can read without the LMDB index or a compose run: each cited record's +// own files on disk. compose's resolveSiteReports is not reused: it verifies, +// prepares and THROWS on any problem, and half of what this page is for is +// reading drafts that have problems. +// +// A record resolves from its transcript.cues.json (title, date, the uploader's +// display name, webpageUrl -- through lib/projects/report.mjs readCues, which is +// memoised on the file's mtime), else its metadata.info.json, else the +// citation's own label, speaker and date. A post resolves from the channel's +// posts. Nothing here fails the page: a record that cannot be read is a card +// with less on it. + +const isoDay = (d: unknown): string | undefined => { + const s = typeof d === "string" ? d : ""; + if (/^\d{8}$/.test(s)) return `${s.slice(0, 4)}-${s.slice(4, 6)}-${s.slice(6, 8)}`; + if (/^\d{4}-\d{2}-\d{2}/.test(s)) return s.slice(0, 10); + return undefined; +}; + +export const recordDir = (channel: string, id: string) => + path.join(/* turbopackIgnore: true */ CHANNELS_DIR, channel, "data", id); + +type RecordMeta = { title?: string; date?: string; channelTitle?: string; webpageUrl?: string }; + +const metaMemo = new Map<string, { key: string; value: RecordMeta | null }>(); + +/** What a record says about itself; null when its directory holds neither file. */ +export async function recordMeta(channel: string, id: string): Promise<RecordMeta | null> { + if (!/^[A-Za-z0-9_.@-]+$/.test(channel) || !/^[A-Za-z0-9_.@-]+$/.test(id)) return null; + const dir = recordDir(channel, id); + const cues = (await readCues(path.join(/* turbopackIgnore: true */ dir, "transcript.cues.json"))) as + | { title?: string; uploadDate?: string; webpageUrl?: string; channel?: string } + | null; + if (cues && (cues.title || cues.webpageUrl)) { + return { title: cues.title, date: isoDay(cues.uploadDate), channelTitle: cues.channel, webpageUrl: cues.webpageUrl }; + } + const file = path.join(/* turbopackIgnore: true */ dir, "metadata.info.json"); + const st = await stat(/* turbopackIgnore: true */ file).catch(() => null); + if (!st) return null; + const key = `${Math.round(st.mtimeMs)}-${st.size}`; + const hit = metaMemo.get(file); + if (hit?.key === key) return hit.value; + let value: RecordMeta | null = null; + try { + const m = JSON.parse(await readFile(/* turbopackIgnore: true */ file, "utf8")); + value = { + title: typeof m.title === "string" ? m.title : undefined, + date: isoDay(m.upload_date), + channelTitle: typeof m.uploader === "string" ? m.uploader : typeof m.channel === "string" ? m.channel : undefined, + webpageUrl: typeof m.webpage_url === "string" ? m.webpage_url : undefined, + }; + } catch { + value = null; + } + metaMemo.set(file, { key, value }); + return value; +} + +// A channel's posts, by id. A big X archive is thousands of posts, so one read +// per channel per minute. +const postsMemo = new Map<string, { at: number; value: Promise<Map<string, Post>> }>(); +export function channelPosts(channel: string): Promise<Map<string, Post>> { + const hit = postsMemo.get(channel); + if (hit && Date.now() - hit.at < 60_000) return hit.value; + const value = readAllPosts(path.join(/* turbopackIgnore: true */ CHANNELS_DIR, channel)) + .then((list) => new Map(list.map((p) => [p.id, p]))) + .catch(() => new Map<string, Post>()); + postsMemo.set(channel, { at: Date.now(), value }); + return value; +} + +/** The capture screenshot of a post, if the editor took one. */ +export async function postShotFile(channel: string, id: string): Promise<string | null> { + if (!/^[A-Za-z0-9_.@-]+$/.test(channel) || !/^[A-Za-z0-9_.@-]+$/.test(id)) return null; + const file = path.join(/* turbopackIgnore: true */ CHANNELS_DIR, channel, "posts-media", id, "shot.png"); + return (await stat(/* turbopackIgnore: true */ file).catch(() => null))?.isFile() ? file : null; +} + +export const corpusMediaUrl = (abs: string) => `/api/sites/media?corpus=${encodeURIComponent(abs)}`; + +type Cite = NonNullable<Report["citations"]>[string]; + +async function recordViewOf(c: Cite): Promise<RecordView | undefined> { + if (c.kind === "video" || c.kind === "audio") { + const m = await recordMeta(c.channel, c.id); + return { + channel: c.channel, + id: c.id, + ...(m?.channelTitle || c.speaker ? { channelTitle: m?.channelTitle ?? c.speaker } : {}), + ...(m?.title || c.label ? { title: m?.title ?? c.label } : {}), + ...(m?.date || c.date ? { date: m?.date ?? c.date } : {}), + ...(m?.webpageUrl ? { originalUrl: platformMomentUrl(m.webpageUrl, null, c.start) ?? m.webpageUrl } : {}), + }; + } + if (c.kind === "post") { + const post = (await channelPosts(c.channel)).get(c.id); + return { + channel: c.channel, + id: c.id, + ...(post?.authorName || post?.author ? { channelTitle: post.authorName ?? post.author } : {}), + ...(post?.createdAt ? { date: post.createdAt.slice(0, 10) } : c.date ? { date: c.date } : {}), + ...(post?.platform ? { platform: post.platform } : {}), + ...(post?.url ? { originalUrl: post.url } : {}), + }; + } + return undefined; +} + +/** + * The page view of a report, or -- when the report's citations cannot be built + * into one (a draft naming a source it does not define) -- the same view with + * no citations, and the reason. + */ +export async function articleView(report: Report): Promise<{ view: ReportPageView; error: string | null }> { + const records = new Map<Cite, RecordView>(); + const posts = new Map<Cite, { author?: string; text?: string; shot?: string }>(); + for (const c of Object.values(report.citations ?? {})) { + const r = await recordViewOf(c); + if (r) records.set(c, r); + if (c.kind === "post") { + const post = (await channelPosts(c.channel)).get(c.id); + const shot = await postShotFile(c.channel, c.id); + posts.set(c, { + ...(post ? { author: post.authorName ?? post.author, text: post.text } : {}), + ...(shot ? { shot: corpusMediaUrl(shot) } : {}), + }); + } + } + try { + const view = buildReportPageView(report, { + record: (c) => records.get(c as Cite) ?? { channel: c.channel, id: c.id }, + post: (c) => posts.get(c as Cite), + }); + return { view, error: null }; + } catch (err) { + const view = { + format: REPORT_PAGE_FORMAT, + version: REPORT_VIEWS_VERSION, + id: report.id, + kind: report.kind, + ...(report.series ? { series: report.series } : {}), + title: report.title, + ...(report.subtitle ? { subtitle: report.subtitle } : {}), + ...(report.summary ? { summary: report.summary } : {}), + ...(report.method ? { method: report.method } : {}), + ...(report.published ? { published: report.published } : {}), + ...(report.updated ? { updated: report.updated } : {}), + sources: {}, + verdicts: {}, + citations: {}, + sections: report.sections.map((s) => ({ + id: s.id, + title: s.title, + ...(s.body ? { body: s.body } : {}), + claims: (s.claims ?? []).map((cl) => ({ ...cl, citations: cl.citations ?? [] })), + })), + } as unknown as ReportPageView; + return { view, error: err instanceof Error ? err.message : String(err) }; + } +} diff --git a/umtool/lib/articles/evidence.ts b/umtool/lib/articles/evidence.ts @@ -0,0 +1,182 @@ +import { readFile } from "node:fs/promises"; +import path from "node:path"; +import type { Report } from "yt-dlp-transcript-common/lib/report/schema"; +import { momentKeyOf } from "yt-dlp-transcript-common/lib/citations/moments"; +import { evidenceSpan, resolveEvidenceSource } from "yt-dlp-transcript-common/lib/evidenceClip-server"; +import { reportMediaDir, reportMediaIndexFile } from "yt-dlp-transcript-common/publish/reportMedia"; +import { readCues } from "@/lib/projects/report.mjs"; +import { CHANNELS_DIR } from "@/lib/paths"; +import { channelPosts, corpusMediaUrl, postShotFile, recordDir, recordMeta } from "./article"; +import { sitesPaths } from "./sites"; + +// What a citation's EVIDENCE panel shows: the cited seconds with the transcript +// around them, and something to play -- found, never fetched. +// +// Playback, best first: +// 1. prepared the site's own evidence clip (.export-index/sites/<site>/ +// report-media/, what the build publishes), when prepare has run; +// 2. window a clip window the editor fetched into data/<id>/clips/; +// 3. saved a saved video or audio file in data/<id>/ (through its media +// tier link -- read, never written); +// 4. otherwise nothing to play, and the line that fetches it through the +// editor: the MCP `fetch_clip` tool. umtool's own fetch client +// (api/report/fetch) is a manifest's, so it cannot ask for a +// window no project names. Never yt-dlp. + +export const CONTEXT_CUES = 6; + +export type EvidenceCue = { start: number; end: number; text: string; cited: boolean }; + +export type EvidencePlay = { + kind: "prepared" | "window" | "saved" | "audio"; + url: string; + /** Seconds into the file where the cited span starts. */ + offset: number; + audio: boolean; + label: string; +}; + +export type Evidence = { + cite: string; + kind: string; + quote: string; + speaker?: string; + date?: string; + label?: string; + originalUrl?: string; + record?: { channel: string; id: string; title?: string; channelTitle?: string }; + start?: number; + end?: number; + cues: EvidenceCue[]; + cuesNote?: string; + play: EvidencePlay | null; + fetchLine?: string; + post?: { author?: string; text?: string; url?: string; shot?: string }; +}; + +type Cite = NonNullable<Report["citations"]>[string]; + +/** The ±CONTEXT_CUES cues around [start, end], the overlapping ones marked. */ +export async function cueContext(channel: string, id: string, start: number, end: number) { + const file = path.join(/* turbopackIgnore: true */ recordDir(channel, id), "transcript.cues.json"); + const doc = (await readCues(file)) as { cues?: { start: number; end: number; text?: string }[] } | null; + const cues = doc?.cues ?? []; + if (!cues.length) return { cues: [] as EvidenceCue[], note: doc ? "no cues" : "no transcript.cues.json" }; + const EPS = 0.05; + let first = cues.findIndex((c) => c.end > start + EPS); + if (first < 0) first = cues.length - 1; + let last = first; + while (last + 1 < cues.length && cues[last + 1].start < end - EPS) last += 1; + const from = Math.max(0, first - CONTEXT_CUES); + const to = Math.min(cues.length - 1, last + CONTEXT_CUES); + return { + cues: cues.slice(from, to + 1).map((c, i) => ({ + start: c.start, + end: c.end, + text: String(c.text ?? "").replace(/\s+/g, " ").trim(), + cited: from + i >= first && from + i <= last, + })), + }; +} + +async function preparedClip(siteId: string, c: Cite): Promise<EvidencePlay | null> { + if (c.kind !== "video" && c.kind !== "audio") return null; + const key = momentKeyOf(c); + if (!key) return null; + try { + const index = JSON.parse(await readFile(/* turbopackIgnore: true */ reportMediaIndexFile(sitesPaths(), siteId), "utf8")); + const entry = index?.moments?.[key]; + if (!entry || (entry.kind !== "video" && entry.kind !== "audio") || typeof entry.file !== "string") return null; + const abs = path.join(/* turbopackIgnore: true */ reportMediaDir(sitesPaths(), siteId), entry.file); + let from = evidenceSpan(c).from; + try { + const side = JSON.parse(await readFile(/* turbopackIgnore: true */ abs.replace(/\.(mp4|m4a)$/, ".json"), "utf8")); + if (typeof side?.span?.from === "number") from = side.span.from; + } catch { + // no sidecar: the citation's own pad is the best guess + } + return { + kind: "prepared", + url: `/api/sites/media?site=${encodeURIComponent(siteId)}&moment=${encodeURIComponent(key)}`, + offset: Math.max(0, c.start - from), + audio: entry.kind === "audio", + label: "prepared evidence clip", + }; + } catch { + return null; + } +} + +async function corpusClip(c: Cite): Promise<EvidencePlay | null> { + if (c.kind !== "video" && c.kind !== "audio") return null; + const span = { from: c.start, to: c.end }; + for (const audio of c.kind === "audio" ? [true] : [false, true]) { + const hit = await resolveEvidenceSource({ channelsDir: CHANNELS_DIR, slug: c.channel, id: c.id, span, audio }).catch(() => null); + if (!hit) continue; + const kind = hit.kind === "corpus-window" ? "window" : hit.kind === "saved-video" ? "saved" : "audio"; + return { + kind, + url: corpusMediaUrl(hit.path), + offset: Math.max(0, c.start - hit.windowStart), + audio: hit.kind === "audio", + label: kind === "window" ? `clip window ${hit.name}` : kind === "saved" ? `saved ${hit.name}` : `audio ${hit.name}`, + }; + } + return null; +} + +/** The MCP line that fetches this span through the editor. */ +export function fetchClipLine(c: { channel: string; id: string; start: number; end: number }, reason: string): string { + const s = (n: number) => Number(n.toFixed(2)); + return `fetch_clip ${JSON.stringify({ channel: c.channel, video: c.id, start: s(c.start), end: s(c.end), reason })}`; +} + +export async function citationEvidence(siteId: string, report: Report, citeId: string): Promise<Evidence | null> { + const c = report.citations?.[citeId]; + if (!c) return null; + const base: Evidence = { + cite: citeId, + kind: c.kind, + quote: c.quote, + ...(c.speaker ? { speaker: c.speaker } : {}), + ...(c.date ? { date: c.date } : {}), + ...(c.label ? { label: c.label } : {}), + cues: [], + play: null, + }; + if (c.kind === "video" || c.kind === "audio") { + const meta = await recordMeta(c.channel, c.id); + const ctx = await cueContext(c.channel, c.id, c.start, c.end); + const play = (await preparedClip(siteId, c)) ?? (await corpusClip(c)); + return { + ...base, + start: c.start, + end: c.end, + record: { channel: c.channel, id: c.id, ...(meta?.title ? { title: meta.title } : {}), ...(meta?.channelTitle ? { channelTitle: meta.channelTitle } : {}) }, + ...(meta?.webpageUrl ? { originalUrl: meta.webpageUrl } : {}), + cues: ctx.cues, + ...(ctx.note ? { cuesNote: ctx.note } : {}), + play, + ...(play ? {} : { fetchLine: fetchClipLine(c, `${siteId}/${report.id} ${citeId}`) }), + }; + } + if (c.kind === "post") { + const post = (await channelPosts(c.channel)).get(c.id); + const shot = await postShotFile(c.channel, c.id); + return { + ...base, + record: { channel: c.channel, id: c.id }, + ...(post?.url ? { originalUrl: post.url } : {}), + post: { + ...(post ? { author: post.authorName ?? post.author, text: post.text, url: post.url } : {}), + ...(shot ? { shot: corpusMediaUrl(shot) } : {}), + }, + }; + } + if (c.kind === "page") return { ...base, originalUrl: c.url }; + if (c.kind === "source") { + const s = report.sources?.[c.source]; + return { ...base, ...(s?.url ? { originalUrl: s.url } : {}), ...(s?.title ? { label: c.label ?? s.title } : {}) }; + } + return base; +} diff --git a/umtool/lib/articles/files.ts b/umtool/lib/articles/files.ts @@ -0,0 +1,77 @@ +import { readFile } from "node:fs/promises"; +import path from "node:path"; +import { REPORTS_ROOT } from "@/lib/paths"; +import { listTakes, readVerdicts } from "@/lib/report/takes.mjs"; +import { siteWorkspaces, tildify } from "./sources.mjs"; +import { untildify } from "./links.mjs"; +import { workspaceFile, workspaceFiles } from "./workspace.mjs"; + +// The page side of lib/articles/workspace.mjs and the video projects' takes. + +export type WorkspaceListing = { + /** The workspace's name under REPORTS_ROOT (what /api/sites/workspace takes). */ + name: string; + dir: string; + files: { rel: string; kind: "draft" | "out" | "doc"; bytes: number; mtimeMs: number }[]; +}; + +export async function listingOf(dir: string): Promise<WorkspaceListing> { + const files = (await workspaceFiles(dir)).map(({ rel, kind, bytes, mtimeMs }: { rel: string; kind: WorkspaceListing["files"][number]["kind"]; bytes: number; mtimeMs: number }) => ({ rel, kind, bytes, mtimeMs })); + return { name: path.basename(dir), dir: tildify(dir), files }; +} + +/** Every workspace a site's articles were written in. */ +export async function siteWorkspaceListings(siteId: string, reportIds: string[]): Promise<WorkspaceListing[]> { + const dirs: string[] = await siteWorkspaces(siteId, reportIds, { reportsRoot: REPORTS_ROOT }); + return Promise.all(dirs.map(listingOf)); +} + +/** One article's workspace (its source's), or null. */ +export async function articleWorkspaceListing(workspace: string | undefined | null): Promise<WorkspaceListing | null> { + if (!workspace) return null; + const abs = untildify(workspace); + if (path.dirname(abs) !== path.resolve(REPORTS_ROOT)) return null; + return listingOf(abs); +} + +export type OpenedFile = + | { ws: string; rel: string; kind: "md"; text: string } + | { ws: string; rel: string; kind: "json"; value: unknown; text: string } + | { ws: string; rel: string; kind: "html"; url: string } + | { ws: string; rel: string; kind: "error"; message: string }; + +const MAX = 2 * 1024 * 1024; + +/** A workspace file opened for the page, or an error saying why not. */ +export async function openWorkspaceFile(ws: string, rel: string): Promise<OpenedFile> { + const f = await workspaceFile(ws, rel); + if (!f) return { ws, rel, kind: "error", message: "not a listed workspace file" }; + if (rel.endsWith(".html")) { + return { ws, rel, kind: "html", url: `/api/sites/workspace?ws=${encodeURIComponent(ws)}&rel=${encodeURIComponent(rel)}` }; + } + if (f.bytes > MAX) return { ws, rel, kind: "error", message: `${f.bytes} bytes; too big to show` }; + const text = await readFile(/* turbopackIgnore: true */ f.real, "utf8"); + if (rel.endsWith(".json")) { + try { + return { ws, rel, kind: "json", value: JSON.parse(text), text }; + } catch (err) { + return { ws, rel, kind: "error", message: `not JSON: ${(err as Error).message}` }; + } + } + return { ws, rel, kind: "md", text }; +} + +export type TakeTally = { takes: number; like: number; maybe: number; no: number; skipped: number }; + +/** How many takes a video project has, and how they were judged. */ +export async function takeTally(dir: string): Promise<TakeTally> { + const [t, v] = await Promise.all([listTakes(dir), readVerdicts(dir)]); + const rows = Object.values(v) as { verdict: string | null }[]; + return { + takes: t.takes.length, + like: rows.filter((r) => r.verdict === "like").length, + maybe: rows.filter((r) => r.verdict === "maybe").length, + no: rows.filter((r) => r.verdict === "no").length, + skipped: t.skipped.length, + }; +} diff --git a/umtool/lib/articles/links.mjs b/umtool/lib/articles/links.mjs @@ -0,0 +1,99 @@ +// Which umtool report-video project is an article's video. +// +// Two ways, in order: +// +// 1. The manifest says so: a top-level `"article": "<site>/<report>"` in +// video.manifest.json. build-video.mjs never reads top-level keys it does +// not know (it reads `generatedBy` no more than this), so the key costs +// the render nothing. A manifest that names an article is linked to it +// and to nothing else. +// 2. The slug matches: the manifest's `slug` (else the project directory's +// name) is the report id, `polemic-<id>`, or the id without `polemic-` -- +// and the project lives in the same workspace as the article's draft +// (lib/articles/sources.mjs). A UNIQUE match is linked; two or more are +// only "possible", and the page says so rather than picking one. +// +// Plain ESM, so `umtool notes` can name an article's video too. +import { readFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { REPORTS_ROOT } from "../paths.mjs"; +import { projectRefs } from "../projects/core.mjs"; +import { idKeys } from "./sources.mjs"; + +const CACHE_MS = 30_000; +/** @type {Map<string, { at: number, value: Promise<any[]> }>} */ +const cache = new Map(); + +/** `~/x` back to an absolute path. */ +export function untildify(p) { + if (typeof p !== "string") return p; + return p === "~" || p.startsWith("~/") ? path.join(/* turbopackIgnore: true */ os.homedir(), p.slice(2)) : p; +} + +/** + * Every report-video project with what linking needs from its manifest: + * `{ id, dir, name, slug, article, generatedBy, title }`. + * + * @param {string} [reportsRoot] + */ +export function videoProjects(reportsRoot = REPORTS_ROOT) { + const hit = cache.get(reportsRoot); + if (hit && Date.now() - hit.at < CACHE_MS) return hit.value; + const value = (async () => { + const out = []; + for (const p of await projectRefs(reportsRoot)) { + if (p.kind !== "report-video") continue; + let m = {}; + try { + m = JSON.parse(await readFile(/* turbopackIgnore: true */ path.join(/* turbopackIgnore: true */ p.dir, "video.manifest.json"), "utf8")); + } catch { + // a project with no readable manifest still links by its directory name + } + out.push({ + id: p.id, + dir: p.dir, + name: p.name, + slug: typeof m.slug === "string" && m.slug ? m.slug : p.name, + article: typeof m.article === "string" ? m.article : null, + generatedBy: typeof m.generatedBy === "string" ? m.generatedBy : null, + title: typeof m.title === "string" ? m.title : p.name, + }); + } + return out.sort((a, b) => a.id.localeCompare(b.id)); + })(); + cache.set(reportsRoot, { at: Date.now(), value }); + value.catch(() => cache.delete(reportsRoot)); + return value; +} + +export function clearLinksCache() { + cache.clear(); +} + +const inside = (root, p) => p === root || p.startsWith(root + path.sep); + +/** + * The projects linked to one article: `{ linked, possible, how }`. + * + * @param {string} siteId + * @param {string} reportId + * @param {{ reportsRoot?: string, workspace?: string | null, projects?: any[] }} [opts] + * `workspace` is the article's workspace dir (sourceFor's, `~` allowed); without + * one, slug matches are only ever "possible". + */ +export async function linkedProjects(siteId, reportId, { reportsRoot = REPORTS_ROOT, workspace = null, projects } = {}) { + const all = projects ?? (await videoProjects(reportsRoot)); + const key = `${siteId}/${reportId}`; + const declared = all.filter((p) => p.article === key); + if (declared.length) return { linked: declared, possible: [], how: "manifest names the article" }; + + const keys = idKeys(reportId); + // A manifest that names a DIFFERENT article is never a slug match. + const bySlug = all.filter((p) => !p.article && keys.has(p.slug)); + const ws = workspace ? untildify(workspace) : null; + const inWs = ws ? bySlug.filter((p) => inside(ws, p.dir)) : []; + if (inWs.length === 1) return { linked: inWs, possible: [], how: `slug ${inWs[0].slug} in the article's workspace` }; + const possible = inWs.length > 1 ? inWs : bySlug; + return { linked: [], possible, how: possible.length ? `${possible.length} project(s) share the slug` : "no project" }; +} diff --git a/umtool/lib/articles/sites.ts b/umtool/lib/articles/sites.ts @@ -0,0 +1,177 @@ +import { readFile, stat } from "node:fs/promises"; +import path from "node:path"; +import { getPaths, type Paths } from "yt-dlp-transcript-common/lib/paths"; +import { getSite, isListedSite, isPrivateSite, listSites, type Site } from "yt-dlp-transcript-common/lib/site"; +import { listReportDirs, siteReportDir } from "yt-dlp-transcript-common/publish/reportMedia"; +import { parseReport } from "yt-dlp-transcript-common/lib/report/validate"; +import type { Report } from "yt-dlp-transcript-common/lib/report/schema"; +import { CHANNELS_DIR, REPORTS_ROOT, SITES_DIR } from "@/lib/paths"; +import { readNotes } from "@/lib/annotations/store.mjs"; +import { corpusNotesFile } from "@/lib/paths"; +import type { NotesDoc } from "@/lib/annotations/types"; +import { sourceFor } from "./sources.mjs"; +import { linkedProjects, videoProjects } from "./links.mjs"; + +// Every site's articles, as umtool reads them: the site list from common +// (site.json through getSite, so a default is the editor's default), each +// report directory through common's listReportDirs (the editor's report tab +// uses the same enumerator), each report.json through the report document's +// own validator, and published-or-draft from the site's `reports` order. Plus +// what only umtool knows: its notes, its source draft, its video project. +// +// READ-ONLY. The one thing umtool writes under SITES_DIR is a notes.json, and +// that goes through lib/annotations, never here. + +/** common's Paths, with the two roots umtool resolves itself (and e2e confines). */ +export function sitesPaths(): Paths { + return { ...getPaths(), sitesDir: SITES_DIR, channelsDir: CHANNELS_DIR }; +} + +export type ArticleStatus = "published" | "draft"; + +export type ProjectLinkRow = { id: string; title: string; slug: string }; + +export type ArticleRow = { + site: string; + id: string; + title: string; + series: string | null; + kind: string | null; + status: ArticleStatus; + published: string | null; + updated: string | null; + /** report.json's mtime, for "updated" when the report names no date. */ + mtimeMs: number | null; + citations: number; + notes: number; + openNotes: number; + hasVideo: boolean; + hasPoster: boolean; + /** report.json missing, unparseable, or invalid: the first problem, else null. */ + problem: string | null; + problems: number; + source: { draft?: string; generator?: string; how?: string; workspace?: string } | null; + projects: { linked: ProjectLinkRow[]; possible: ProjectLinkRow[] }; +}; + +export type SiteRow = { + siteId: string; + title: string; + private: boolean; + listed: boolean; + search: boolean; + published: number; + drafts: number; + openNotes: number; + articles: ArticleRow[]; +}; + +const exists = (p: string) => stat(/* turbopackIgnore: true */ p).then((s) => s.isFile(), () => false); + +export type ArticleRead = { + report: Report | null; + problems: { path?: string; message: string }[]; + mtimeMs: number | null; +}; + +/** One report.json, read and validated; never throws. */ +export async function readReportFile(siteId: string, reportId: string): Promise<ArticleRead> { + const file = path.join(/* turbopackIgnore: true */ siteReportDir(sitesPaths(), siteId, reportId), "report.json"); + let text: string; + let mtimeMs: number | null = null; + try { + const [t, st] = await Promise.all([readFile(/* turbopackIgnore: true */ file, "utf8"), stat(/* turbopackIgnore: true */ file)]); + text = t; + mtimeMs = Math.round(st.mtimeMs); + } catch { + return { report: null, problems: [{ message: "no report.json" }], mtimeMs: null }; + } + let raw: unknown; + try { + raw = JSON.parse(text); + } catch (err) { + return { report: null, problems: [{ message: `report.json is not JSON: ${(err as Error).message}` }], mtimeMs }; + } + const parsed = parseReport(raw, { id: reportId }); + if (!parsed.ok) return { report: null, problems: parsed.problems, mtimeMs }; + return { report: parsed.value, problems: parsed.problems, mtimeMs }; +} + +export async function readArticleNotes(siteId: string, reportId: string): Promise<{ doc: NotesDoc | null; token: string; error?: string }> { + const file = corpusNotesFile(siteId, reportId); + if (!file) return { doc: null, token: "absent" }; + return readNotes(file); +} + +async function articleRow(site: Site, id: string, published: Set<string>, projects: Awaited<ReturnType<typeof videoProjects>>): Promise<ArticleRow> { + const [read, notes, source] = await Promise.all([ + readReportFile(site.siteId, id), + readArticleNotes(site.siteId, id), + sourceFor(site.siteId, id, { reportsRoot: REPORTS_ROOT }), + ]); + const dir = siteReportDir(sitesPaths(), site.siteId, id); + const r = read.report; + const links = await linkedProjects(site.siteId, id, { workspace: source?.workspace ?? null, projects }); + const row = (p: { id: string; title: string; slug: string }) => ({ id: p.id, title: p.title, slug: p.slug }); + return { + site: site.siteId, + id, + title: r?.title ?? id, + series: r?.series ?? null, + kind: r?.kind ?? null, + status: published.has(id) ? "published" : "draft", + published: r?.published ?? null, + updated: r?.updated ?? r?.published ?? null, + mtimeMs: read.mtimeMs, + citations: Object.keys(r?.citations ?? {}).length, + notes: notes.doc?.notes.length ?? 0, + openNotes: notes.doc?.notes.filter((n) => n.status === "open").length ?? 0, + hasVideo: !!r?.video?.src && (await exists(path.join(/* turbopackIgnore: true */ dir, r.video.src))), + hasPoster: !!r?.video?.poster && (await exists(path.join(/* turbopackIgnore: true */ dir, r.video.poster))), + problem: read.problems[0]?.message ?? null, + problems: read.problems.length, + source, + projects: { linked: links.linked.map(row), possible: links.possible.map(row) }, + }; +} + +function siteFlags(site: Site) { + return { private: isPrivateSite(site), listed: isListedSite(site), search: site.search !== false }; +} + +/** One site with every article, published (in the site's order) then drafts (by id). */ +export async function readSiteRow(site: Site, projects?: Awaited<ReturnType<typeof videoProjects>>): Promise<SiteRow> { + const all = projects ?? (await videoProjects(REPORTS_ROOT)); + const published = new Set(site.reports ?? []); + const dirs = await listReportDirs(sitesPaths(), site.siteId); + const ids = [...(site.reports ?? []), ...dirs.filter((d) => !published.has(d))]; + const articles = await Promise.all(ids.map((id) => articleRow(site, id, published, all))); + return { + siteId: site.siteId, + title: site.siteTitle || site.siteId, + ...siteFlags(site), + published: articles.filter((a) => a.status === "published").length, + drafts: articles.filter((a) => a.status === "draft").length, + openNotes: articles.reduce((n, a) => n + a.openNotes, 0), + articles, + }; +} + +/** Every site, private first, then by id. */ +export async function listSiteRows(): Promise<SiteRow[]> { + const projects = await videoProjects(REPORTS_ROOT); + const sites = listSites(sitesPaths()); + const rows = await Promise.all(sites.map((s) => readSiteRow(s, projects))); + return rows.sort((a, b) => Number(b.private) - Number(a.private) || a.siteId.localeCompare(b.siteId)); +} + +/** A site by id, or null (a bad id or no site.json). */ +export function siteById(siteId: string): Site | null { + try { + const paths = sitesPaths(); + if (!listSites(paths).some((s) => s.siteId === siteId)) return null; + return getSite(siteId, paths); + } catch { + return null; + } +} diff --git a/umtool/lib/articles/workspace.mjs b/umtool/lib/articles/workspace.mjs @@ -0,0 +1,65 @@ +// The files an article was WRITTEN from, for reading beside it: a workspace's +// drafts, the rendered drafts and briefs under polemics/, and the notes a +// workspace keeps at its top level. Read-only -- this lists and serves, it +// never writes a workspace. +// +// polemics/drafts/*.json the drafts (source of truth) +// polemics/out/*.{md,html} what the drafts render to +// polemics/*.md BRIEF.md, NOTES.md, PRIVACY-SWEEP.md, … +// <ws>/{NOTES,BRIEF,LEADS,PLAN,PRIVACY-SWEEP}.md +// <ws>/site/PLAN.md, <ws>/site/MERGE-PLAN.md +import { readdir, realpath, stat } from "node:fs/promises"; +import path from "node:path"; +import { REPORTS_ROOT } from "../paths.mjs"; + +export const TOP_LEVEL = ["NOTES.md", "BRIEF.md", "LEADS.md", "PLAN.md", "PRIVACY-SWEEP.md"]; +const SITE_LEVEL = ["PLAN.md", "MERGE-PLAN.md"]; +export const WORKSPACE_EXT = /\.(md|html|json)$/; + +const statFile = (p) => stat(/* turbopackIgnore: true */ p).then((s) => (s.isFile() ? s : null), () => null); + +/** + * Every listed file of one workspace: `{ rel, abs, kind, bytes, mtimeMs }`, + * `kind` one of "draft" | "out" | "doc". Sorted: drafts, docs, outputs; by name. + * + * @param {string} wsDir + */ +export async function workspaceFiles(wsDir) { + const out = []; + const add = async (rel, kind) => { + const abs = path.join(/* turbopackIgnore: true */ wsDir, rel); + const st = await statFile(abs); + if (st) out.push({ rel, abs, kind, bytes: st.size, mtimeMs: Math.round(st.mtimeMs) }); + }; + const ls = (rel) => readdir(/* turbopackIgnore: true */ path.join(/* turbopackIgnore: true */ wsDir, rel)).catch(() => []); + for (const f of await ls("polemics/drafts")) if (f.endsWith(".json")) await add(`polemics/drafts/${f}`, "draft"); + for (const f of await ls("polemics/out")) if (/\.(md|html)$/.test(f)) await add(`polemics/out/${f}`, "out"); + for (const f of await ls("polemics")) if (f.endsWith(".md")) await add(`polemics/${f}`, "doc"); + for (const f of TOP_LEVEL) await add(f, "doc"); + for (const f of SITE_LEVEL) await add(`site/${f}`, "doc"); + const rank = { draft: 0, doc: 1, out: 2 }; + return out.sort((a, b) => rank[a.kind] - rank[b.kind] || a.rel.localeCompare(b.rel)); +} + +/** + * A workspace file a client named, or null: `ws` must be a directory directly + * under REPORTS_ROOT and `rel` one of the files workspaceFiles lists for it, + * and its REAL path must stay inside the workspace. + * + * @param {string} ws the workspace's name under REPORTS_ROOT + * @param {string} rel + * @param {{ reportsRoot?: string }} [opts] + */ +export async function workspaceFile(ws, rel, { reportsRoot = REPORTS_ROOT } = {}) { + if (typeof ws !== "string" || !/^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(ws) || ws === "..") return null; + if (typeof rel !== "string" || rel.includes("\0") || rel.split("/").some((s) => s === ".." || s === "" || s === ".")) return null; + const dir = path.join(/* turbopackIgnore: true */ reportsRoot, ws); + const listed = (await workspaceFiles(dir)).find((f) => f.rel === rel); + if (!listed) return null; + const [realDir, realAbs] = await Promise.all([ + realpath(/* turbopackIgnore: true */ dir).catch(() => null), + realpath(/* turbopackIgnore: true */ listed.abs).catch(() => null), + ]); + if (!realDir || !realAbs || !realAbs.startsWith(realDir + path.sep)) return null; + return { ...listed, real: realAbs }; +} diff --git a/umtool/lib/decisions.ts b/umtool/lib/decisions.ts @@ -4,6 +4,7 @@ import { DEFAULT_TARGET, loudnessVerdict } from "./loudness-types"; import { buildStatus, readManifest } from "./manifest"; import { readSpec, validateSpec } from "./spec"; import { readNotes, type NoteMap } from "./notes"; +import { listNotesFiles } from "./annotations/targets.mjs"; import { acceptedFor, readThumbAccepted, readThumbManifest, thumbNamesFor } from "./thumbs"; // --------------------------------------------------------------------------- @@ -298,3 +299,79 @@ export function decisionsMarkdown(items: Decision[]): string[] { } return out; } + +// --------------------------------------------------------------------------- +// Open NOTES -- on an article (sites/<site>/reports/<id>/notes.json) or on a +// report-video project (<project>/notes.json) -- are open decisions: somebody +// asked for a change and nobody has answered it. One row per open note, kind +// `open-note`, linked to the note on its page. A notes.json that does not +// parse is blocking: nothing can write to it until somebody fixes it by hand. +// +// An ARTICLE is not a project, so its row's `project` is its page's path +// (`sites/<site>/<report>`), which is also where its href points. +// --------------------------------------------------------------------------- + +const firstLine = (s: string, max = 140) => { + const line = s.split("\n").find((l) => l.trim()) ?? ""; + return line.length > max ? `${line.slice(0, max - 1)}…` : line; +}; + +function anchorLabel(a: { kind: string; [k: string]: unknown }): string { + switch (a.kind) { + case "text": + return `“${firstLine(String(a.quote ?? ""), 48)}”`; + case "cite": + return `cite ${a.cite}`; + case "section": + return `section ${a.section}`; + case "moment": + return `${a.file} @ ${Number(a.t).toFixed(1)}s`; + case "entry": + return `entry ${a.entry}`; + case "take": + return `take ${a.take}`; + case "edit": + return `edit ${a.field}${a.entry ? ` on ${a.entry}` : ""}`; + default: + return "whole"; + } +} + +export async function noteDecisions(): Promise<Decision[]> { + const files = await listNotesFiles().catch(() => []); + const out: Decision[] = []; + for (const f of files) { + const article = f.kind === "article"; + const project = article ? `sites/${f.id}` : f.id; + const page = article ? `/sites/${f.id}` : `/browse/${f.id}`; + if (f.error || !f.doc) { + out.push({ + kind: "unreadable-notes", + project, + projectKind: article ? "article" : "report-video", + target: "notes.json", + why: `notes.json does not parse (${f.error ?? "unknown"}); nothing will write to it until it is fixed`, + href: page, + severity: "blocking", + at: Date.now(), + }); + continue; + } + for (const n of f.doc.notes) { + if (n.status !== "open") continue; + const replies = n.replies.length ? ` · ${n.replies.length} repl${n.replies.length === 1 ? "y" : "ies"}` : ""; + const at = Date.parse(n.updatedAt); + out.push({ + kind: "open-note", + project, + projectKind: article ? "article" : "report-video", + target: anchorLabel(n.anchor as { kind: string }), + why: `${firstLine(n.text)}${replies}`, + href: `${page}?note=${encodeURIComponent(n.id)}`, + severity: "open", + at: Number.isFinite(at) ? at : 0, + }); + } + } + return out; +} diff --git a/umtool/lib/projects.ts b/umtool/lib/projects.ts @@ -5,7 +5,7 @@ import { collapseFolders, foldersFor, walkProjects } from "./projects/walk.mjs"; import { SONG_KIND } from "./projects/song.mjs"; import { openIndex, signRecord } from "./projects/index-db.mjs"; import { BROWSE_ROOT } from "./browse"; -import { decisionsForSong } from "./decisions"; +import { decisionsForSong, noteDecisions } from "./decisions"; import { listMedia, listMediaUnder, type MediaRow } from "./media"; import type { Decision } from "./decisions"; import type { @@ -302,8 +302,10 @@ const RANK: Record<string, number> = { blocking: 0, open: 1, info: 2 }; export async function openDecisions(): Promise<Decision[]> { const refs = await projectRefs(); const per = await Promise.all(refs.map(decisionsForProject)); - return per - .flat() + // Open notes on articles and report videos (lib/decisions.ts noteDecisions): + // an article is not a project, so they are added here, not per project. + const notes = await noteDecisions(); + return [...per.flat(), ...notes] .sort((a, b) => RANK[a.severity] - RANK[b.severity] || b.at - a.at); }