import { readFile, readdir, stat } from "node:fs/promises"; import path from "node:path"; import { songDir } from "./browse"; import { readState, readJson, type SongState } from "./state"; import { stateFile } from "./paths"; import { archiveMomentUrl } from "./archive"; // --------------------------------------------------------------------------- // Where every note in a finished cut came from. // // Two sources for the same rows: // // clips.csv written by render-poly.mjs beside a render, already sorted by // songTime. Cheap, exact, and present for exactly one song in // this tree (pokemon), so it is the fast path, not the plan. // plan/*.json voices[].plan[] -- the same information, plus the pitch and // the tuning error. Always there. // // The join key is the one the whole pipeline agrees on: // // clipId = `${video}@${srcStart.toFixed(2)}` // // It is what accepted.json, corepitch.json, keepside.json and reasons.json are // keyed by, and it is what /sort?only= takes -- so a row here can re-open the // exact clip in the judging deck. // --------------------------------------------------------------------------- export type ProvRow = { songTime: number; voice: string; clipId: string; video: string; srcStart: number; srcEnd: number; noteDur: number; /** Tuning error in semitones. The READMEs quote its p90. */ shift: number; keepSide: number; }; export type JoinedRow = ProvRow & { title: string; date: string; word: string; /** Which pile the clip is in now, or "" when the state files have never seen it. */ pile: "accepted" | "rejected" | "unclean" | "impure" | "provisional" | ""; /** The archive deep link, with the 3s lead-in. */ link: string | null; }; export type SourceVideo = { video: string; title: string; date: string; clips: number; /** The earliest moment used, so the group has something to link to. */ firstSrcStart: number; link: string | null; }; export type Provenance = { /** Which file the rows came from, for the UI to say so. */ source: { kind: "clips.csv" | "plan"; name: string } | null; /** Every plan in plan/, so the UI can offer the choice it must not make. */ available: string[]; rows: JoinedRow[]; videos: SourceVideo[]; /** p90 of |shift|, the figure the READMEs quote. Null with no rows. */ shiftP90: number | null; voices: string[]; }; const clipIdOf = (video: string, srcStart: number) => `${video}@${srcStart.toFixed(2)}`; // --------------------------------------------------------------------------- // Parsing, memoised by mtime. // // Plans are up to ~422KB / ~880 notes and NEVER change after a render, so this // is safe to hold. The JOIN side is deliberately not cached: readState() is a // few MB read per request by design, because the CLI scripts are the other // writer and a cached pile would show judgements that have since moved. // --------------------------------------------------------------------------- const parsed = new Map(); async function memo(abs: string, parse: (text: string) => ProvRow[]): Promise { let st; try { st = await stat(abs); } catch { return []; } const key = `${abs}|${Math.round(st.mtimeMs)}|${st.size}`; const hit = parsed.get(key); if (hit) return hit; let rows: ProvRow[] = []; try { rows = parse(await readFile(abs, "utf8")); } catch { rows = []; } rows.sort((a, b) => a.songTime - b.songTime); parsed.set(key, rows); return rows; } const num = (v: string | undefined) => { const n = Number(v); return Number.isFinite(n) ? n : 0; }; function parseClipsCsv(text: string): ProvRow[] { const lines = text.trim().split("\n"); const header = lines.shift()?.split(",") ?? []; const at = (name: string) => header.indexOf(name); const i = { songTime: at("songTime"), voice: at("voice"), clipId: at("clipId"), video: at("video"), srcStart: at("srcStart"), srcEnd: at("srcEnd"), noteDur: at("noteDur"), shift: at("shift"), keepSide: at("keepSide"), }; if (i.video < 0 || i.srcStart < 0) return []; const out: ProvRow[] = []; for (const line of lines) { if (!line.trim()) continue; const c = line.split(","); const video = c[i.video]; const srcStart = num(c[i.srcStart]); out.push({ songTime: num(c[i.songTime]), voice: c[i.voice] ?? "", // Recompute rather than trust the column: the CSV's own clipId is built // the same way, and deriving it here means one definition of the key. clipId: clipIdOf(video, srcStart), video, srcStart, srcEnd: num(c[i.srcEnd]), noteDur: num(c[i.noteDur]), shift: num(c[i.shift]), keepSide: num(c[i.keepSide]), }); } return out; } type PlanNote = { slotStart?: number; noteDur?: number; video?: string; srcStart?: number; srcEnd?: number; shift?: number; keepSide?: number; }; function parsePlan(text: string): ProvRow[] { const j = JSON.parse(text) as { voices?: { name?: string; plan?: PlanNote[] }[] }; const out: ProvRow[] = []; for (const v of j.voices ?? []) { for (const n of v.plan ?? []) { if (!n.video || typeof n.srcStart !== "number") continue; out.push({ songTime: n.slotStart ?? 0, voice: v.name ?? "", clipId: clipIdOf(n.video, n.srcStart), video: n.video, srcStart: n.srcStart, srcEnd: n.srcEnd ?? n.srcStart, noteDur: n.noteDur ?? 0, shift: n.shift ?? 0, keepSide: n.keepSide ?? 0, }); } } return out; } function pileOf(state: SongState, clipId: string): JoinedRow["pile"] { if (state.accepted.includes(clipId)) return "accepted"; if (state.rejected.includes(clipId)) return "rejected"; if (state.unclean.includes(clipId)) return "unclean"; if (state.impure.includes(clipId)) return "impure"; if (state.provisional.includes(clipId)) return "provisional"; return ""; } function p90(values: number[]): number | null { if (!values.length) return null; const s = [...values].sort((a, b) => a - b); return +s[Math.min(s.length - 1, Math.floor(0.9 * s.length))].toFixed(3); } /** * Rows for one cut. * * `planName` picks which of plan/'s files to read. Which plan produced which cut * is genuinely ambiguous for three of the five songs here -- mario-rpg's plan/ * holds a 2-voice and a 3-voice arrangement of the same tune, and nothing on * disk says which one shipped -- so this NEVER guesses. build.json names one * when a builder recorded it; otherwise the caller passes the user's choice, and * `available` is what the picker is built from. */ export async function readProvenance(id: string, planName?: string | null): Promise { const dir = songDir(id); const available = await readdir(path.join(dir, "plan")).then( (n) => n.filter((f) => f.endsWith(".json")).sort(), () => [] as string[], ); let rows: ProvRow[] = []; let source: Provenance["source"] = null; if (planName && available.includes(planName)) { rows = await memo(path.join(dir, "plan", planName), parsePlan); source = { kind: "plan", name: planName }; } else { const csv = path.join(dir, "clips.csv"); const fromCsv = await memo(csv, parseClipsCsv); if (fromCsv.length) { rows = fromCsv; source = { kind: "clips.csv", name: "clips.csv" }; } } if (!rows.length) return { source, available, rows: [], videos: [], shiftP90: null, voices: [] }; const [state, titles, dates] = await Promise.all([ readState(), readJson>(stateFile("titles.json"), {}), readJson>(stateFile("dates.json"), {}), ]); const joined: JoinedRow[] = rows.map((r) => ({ ...r, title: titles[r.video] ?? "", date: dates[r.video] ?? "", word: state.words[r.clipId]?.word ?? "", pile: pileOf(state, r.clipId), link: archiveMomentUrl(r.video, r.srcStart), })); const byVideo = new Map(); for (const r of joined) { const hit = byVideo.get(r.video); if (hit) { hit.clips += 1; hit.firstSrcStart = Math.min(hit.firstSrcStart, r.srcStart); } else { byVideo.set(r.video, { video: r.video, title: r.title, date: r.date, clips: 1, firstSrcStart: r.srcStart, link: r.link, }); } } const videos = [...byVideo.values()].sort((a, b) => b.clips - a.clips); // The group link should land on the FIRST moment used, not on whichever row // happened to be seen first. for (const v of videos) v.link = archiveMomentUrl(v.video, v.firstSrcStart); return { source, available, rows: joined, videos, shiftP90: p90(joined.map((r) => Math.abs(r.shift))), voices: [...new Set(joined.map((r) => r.voice))].filter(Boolean), }; }