import { readFile, stat } from "node:fs/promises"; import path from "node:path"; import { getPaths, type Paths } from "yt-dlp-transcript-common/lib/paths"; import { getSite, isListedSite, isPrivateSite, listSites, type Site } from "yt-dlp-transcript-common/lib/site"; import { listReportDirs, siteReportDir } from "yt-dlp-transcript-common/publish/reportMedia"; import { parseReport } from "yt-dlp-transcript-common/lib/report/validate"; import type { Report } from "yt-dlp-transcript-common/lib/report/schema"; import { CHANNELS_DIR, REPORTS_ROOT, SITES_DIR } from "@/lib/paths"; import { readNotes } from "@/lib/annotations/store.mjs"; import { corpusNotesFile } from "@/lib/paths"; import type { NotesDoc } from "@/lib/annotations/types"; import { sourceFor } from "./sources.mjs"; import { linkedProjects, videoProjects } from "./links.mjs"; // Every site's articles, as umtool reads them: the site list from common // (site.json through getSite, so a default is the editor's default), each // report directory through common's listReportDirs (the editor's report tab // uses the same enumerator), each report.json through the report document's // own validator, and published-or-draft from the site's `reports` order. Plus // what only umtool knows: its notes, its source draft, its video project. // // READ-ONLY. The one thing umtool writes under SITES_DIR is a notes.json, and // that goes through lib/annotations, never here. /** common's Paths, with the two roots umtool resolves itself (and e2e confines). */ export function sitesPaths(): Paths { return { ...getPaths(), sitesDir: SITES_DIR, channelsDir: CHANNELS_DIR }; } export type ArticleStatus = "published" | "draft"; export type ProjectLinkRow = { id: string; title: string; slug: string }; export type ArticleRow = { site: string; id: string; title: string; series: string | null; kind: string | null; status: ArticleStatus; published: string | null; updated: string | null; /** report.json's mtime, for "updated" when the report names no date. */ mtimeMs: number | null; citations: number; notes: number; openNotes: number; hasVideo: boolean; hasPoster: boolean; /** report.json missing, unparseable, or invalid: the first problem, else null. */ problem: string | null; problems: number; source: { draft?: string; generator?: string; how?: string; workspace?: string } | null; projects: { linked: ProjectLinkRow[]; possible: ProjectLinkRow[] }; }; export type SiteRow = { siteId: string; title: string; private: boolean; listed: boolean; search: boolean; published: number; drafts: number; openNotes: number; articles: ArticleRow[]; }; const exists = (p: string) => stat(/* turbopackIgnore: true */ p).then((s) => s.isFile(), () => false); export type ArticleRead = { report: Report | null; problems: { path?: string; message: string }[]; mtimeMs: number | null; }; /** One report.json, read and validated; never throws. */ export async function readReportFile(siteId: string, reportId: string): Promise { const file = path.join(/* turbopackIgnore: true */ siteReportDir(sitesPaths(), siteId, reportId), "report.json"); let text: string; let mtimeMs: number | null = null; try { const [t, st] = await Promise.all([readFile(/* turbopackIgnore: true */ file, "utf8"), stat(/* turbopackIgnore: true */ file)]); text = t; mtimeMs = Math.round(st.mtimeMs); } catch { return { report: null, problems: [{ message: "no report.json" }], mtimeMs: null }; } let raw: unknown; try { raw = JSON.parse(text); } catch (err) { return { report: null, problems: [{ message: `report.json is not JSON: ${(err as Error).message}` }], mtimeMs }; } const parsed = parseReport(raw, { id: reportId }); if (!parsed.ok) return { report: null, problems: parsed.problems, mtimeMs }; return { report: parsed.value, problems: parsed.problems, mtimeMs }; } export async function readArticleNotes(siteId: string, reportId: string): Promise<{ doc: NotesDoc | null; token: string; error?: string }> { const file = corpusNotesFile(siteId, reportId); if (!file) return { doc: null, token: "absent" }; return readNotes(file); } async function articleRow(site: Site, id: string, published: Set, projects: Awaited>): Promise { const [read, notes, source] = await Promise.all([ readReportFile(site.siteId, id), readArticleNotes(site.siteId, id), sourceFor(site.siteId, id, { reportsRoot: REPORTS_ROOT }), ]); const dir = siteReportDir(sitesPaths(), site.siteId, id); const r = read.report; const links = await linkedProjects(site.siteId, id, { workspace: source?.workspace ?? null, projects }); const row = (p: { id: string; title: string; slug: string }) => ({ id: p.id, title: p.title, slug: p.slug }); return { site: site.siteId, id, title: r?.title ?? id, series: r?.series ?? null, kind: r?.kind ?? null, status: published.has(id) ? "published" : "draft", published: r?.published ?? null, updated: r?.updated ?? r?.published ?? null, mtimeMs: read.mtimeMs, citations: Object.keys(r?.citations ?? {}).length, notes: notes.doc?.notes.length ?? 0, openNotes: notes.doc?.notes.filter((n) => n.status === "open").length ?? 0, hasVideo: !!r?.video?.src && (await exists(path.join(/* turbopackIgnore: true */ dir, r.video.src))), hasPoster: !!r?.video?.poster && (await exists(path.join(/* turbopackIgnore: true */ dir, r.video.poster))), problem: read.problems[0]?.message ?? null, problems: read.problems.length, source, projects: { linked: links.linked.map(row), possible: links.possible.map(row) }, }; } function siteFlags(site: Site) { return { private: isPrivateSite(site), listed: isListedSite(site), search: site.search !== false }; } /** One site with every article, published (in the site's order) then drafts (by id). */ export async function readSiteRow(site: Site, projects?: Awaited>): Promise { const all = projects ?? (await videoProjects(REPORTS_ROOT)); const published = new Set(site.reports ?? []); const dirs = await listReportDirs(sitesPaths(), site.siteId); const ids = [...(site.reports ?? []), ...dirs.filter((d) => !published.has(d))]; const articles = await Promise.all(ids.map((id) => articleRow(site, id, published, all))); return { siteId: site.siteId, title: site.siteTitle || site.siteId, ...siteFlags(site), published: articles.filter((a) => a.status === "published").length, drafts: articles.filter((a) => a.status === "draft").length, openNotes: articles.reduce((n, a) => n + a.openNotes, 0), articles, }; } /** Every site, private first, then by id. */ export async function listSiteRows(): Promise { const projects = await videoProjects(REPORTS_ROOT); const sites = listSites(sitesPaths()); const rows = await Promise.all(sites.map((s) => readSiteRow(s, projects))); return rows.sort((a, b) => Number(b.private) - Number(a.private) || a.siteId.localeCompare(b.siteId)); } /** A site by id, or null (a bad id or no site.json). */ export function siteById(siteId: string): Site | null { try { const paths = sitesPaths(); if (!listSites(paths).some((s) => s.siteId === siteId)) return null; return getSite(siteId, paths); } catch { return null; } }