import { readdir } from "node:fs/promises"; import path from "node:path"; import { songDir } from "./browse"; import { readProvenance, type JoinedRow } from "./provenance"; // --------------------------------------------------------------------------- // What changed between two arrangements. // // Three of the five songs here keep more than one plan -- mario-rpg's plan/ // holds a 2-voice and a 3-voice arrangement of the same tune, pokemon has a v11 // and a v12 -- and nothing compares them. The question that goes unanswered is // the only one worth asking of a re-arrange: which notes are actually different, // and are they the ones I re-sang? // // THE ALIGNMENT IS THE WHOLE DESIGN. You cannot pair by index (the plans have // different note counts) and you cannot pair on song time alone across voices // (a note in `melody` is never the same note as one in `bass`). So: // // 1. partition by VOICE. Voice names come from the arranger and are stable // across re-arranges; a voice present in only one plan is a whole-voice // add or drop, which is the headline rather than a diff detail. // 2. inside a voice, pair by SLOT within a tolerance. The arranger places // notes on a grid taken from the MIDI, so the same slot in two plans is // identical to the millisecond; 30 ms is slack, not a search radius. // 3. what is left over is paired by CLIP. A note that kept its clip and moved // is a `moved`, not a removal and an unrelated addition -- and this is // also what stops a whole plan shifted by a second from reading as 880 // deletions. // // The diff is on the CLIP, not on the slot: same slot, different clip is the // thing you want to see, because that is the note that was re-sung. // --------------------------------------------------------------------------- /** 30 ms. Beyond this it is a different slot, not the same note nudged. */ export const DEFAULT_TOLERANCE = 0.03; export type DeltaKind = "same" | "swapped" | "moved" | "added" | "removed"; export type PlanDelta = { voice: string; kind: DeltaKind; /** a's song time where there is one, else b's. */ songTime: number; a: JoinedRow | null; b: JoinedRow | null; /** Tuning difference, when both sides exist. */ dShift?: number; /** How far the slot moved, for a `moved`. */ dSlot?: number; }; export type VoiceDiff = { voice: string; /** Set when the voice exists in only one plan. */ only: "a" | "b" | null; same: number; swapped: number; moved: number; added: number; removed: number; }; export type PlanSide = { name: string; rows: number; videos: number; voices: string[]; shiftP90: number | null; }; export type PlanDiff = { song: string; tolerance: number; a: PlanSide; b: PlanSide; voices: VoiceDiff[]; totals: { same: number; swapped: number; moved: number; added: number; removed: number }; deltas: PlanDelta[]; /** Source episodes that left the palette, and ones that entered it. */ videosOnlyA: string[]; videosOnlyB: string[]; }; const byTime = (rows: JoinedRow[]) => [...rows].sort((a, b) => a.songTime - b.songTime); function diffVoice(voice: string, ar: JoinedRow[], br: JoinedRow[], tol: number): PlanDelta[] { const a = byTime(ar); const b = byTime(br); const out: PlanDelta[] = []; const leftA: JoinedRow[] = []; const leftB: JoinedRow[] = []; // Both lists are sorted, so one sweep pairs every slot that exists in both. let i = 0; let j = 0; while (i < a.length && j < b.length) { const d = a[i].songTime - b[j].songTime; if (Math.abs(d) <= tol) { const x = a[i]; const y = b[j]; out.push({ voice, kind: x.clipId === y.clipId ? "same" : "swapped", songTime: x.songTime, a: x, b: y, dShift: +(y.shift - x.shift).toFixed(3), }); i += 1; j += 1; } else if (d < 0) { leftA.push(a[i]); i += 1; } else { leftB.push(b[j]); j += 1; } } while (i < a.length) leftA.push(a[i++]); while (j < b.length) leftB.push(b[j++]); // Leftovers that share a clip are one note that moved, not two unrelated // events. First-come pairing is enough: within one voice a clip repeats // rarely, and when it does the nearest slot is the sorted-order one. const pool = new Map(); for (const y of leftB) { const list = pool.get(y.clipId); if (list) list.push(y); else pool.set(y.clipId, [y]); } const usedB = new Set(); for (const x of leftA) { const list = pool.get(x.clipId); const y = list?.shift(); if (y) { usedB.add(y); out.push({ voice, kind: "moved", songTime: x.songTime, a: x, b: y, dSlot: +(y.songTime - x.songTime).toFixed(3), dShift: +(y.shift - x.shift).toFixed(3), }); } else { out.push({ voice, kind: "removed", songTime: x.songTime, a: x, b: null }); } } for (const y of leftB) { if (usedB.has(y)) continue; out.push({ voice, kind: "added", songTime: y.songTime, a: null, b: y }); } return out; } /** The plans in plan/, which is what both names must be members of. */ async function planNames(id: string): Promise { return readdir(path.join(songDir(id), "plan")).then( (n) => n.filter((f) => f.endsWith(".json")).sort(), () => [] as string[], ); } /** * Diff two plans of one song, or null if either name is not in plan/. * * The null matters: readProvenance FALLS BACK TO clips.csv when it is handed a * name it does not recognise, so a typo would otherwise diff a plan against a * CSV and label the result with the name that was typed. */ export async function diffPlans( id: string, aName: string, bName: string, tolerance = DEFAULT_TOLERANCE, ): Promise { const available = await planNames(id); if (!available.includes(aName) || !available.includes(bName)) return null; // Memoised by mtime in lib/provenance.ts, so re-diffing the same pair is free. const [pa, pb] = await Promise.all([readProvenance(id, aName), readProvenance(id, bName)]); const voices = [...new Set([...pa.voices, ...pb.voices])].sort(); const deltas: PlanDelta[] = []; const perVoice: VoiceDiff[] = []; for (const voice of voices) { const ar = pa.rows.filter((r) => r.voice === voice); const br = pb.rows.filter((r) => r.voice === voice); const d = diffVoice(voice, ar, br, tolerance); deltas.push(...d); perVoice.push({ voice, only: ar.length && br.length ? null : ar.length ? "a" : "b", same: d.filter((x) => x.kind === "same").length, swapped: d.filter((x) => x.kind === "swapped").length, moved: d.filter((x) => x.kind === "moved").length, added: d.filter((x) => x.kind === "added").length, removed: d.filter((x) => x.kind === "removed").length, }); } const va = new Set(pa.videos.map((v) => v.video)); const vb = new Set(pb.videos.map((v) => v.video)); const side = (name: string, p: typeof pa): PlanSide => ({ name, rows: p.rows.length, videos: p.videos.length, voices: p.voices, shiftP90: p.shiftP90, }); // Swapped first: it is the answer to the question the tool was opened with. const rank: Record = { swapped: 0, moved: 1, added: 2, removed: 3, same: 4 }; return { song: id, tolerance, a: side(aName, pa), b: side(bName, pb), voices: perVoice, totals: { same: deltas.filter((d) => d.kind === "same").length, swapped: deltas.filter((d) => d.kind === "swapped").length, moved: deltas.filter((d) => d.kind === "moved").length, added: deltas.filter((d) => d.kind === "added").length, removed: deltas.filter((d) => d.kind === "removed").length, }, deltas: deltas.sort((x, y) => rank[x.kind] - rank[y.kind] || x.songTime - y.songTime), videosOnlyA: [...va].filter((v) => !vb.has(v)).sort(), videosOnlyB: [...vb].filter((v) => !va.has(v)).sort(), }; } /** The summary line, which is the product. Everything else is the evidence. */ export function summariseDiff(d: PlanDiff): string { const t = d.totals; return ( `${d.a.name} vs ${d.b.name}: ${d.voices.length} voice${d.voices.length === 1 ? "" : "s"}, ` + `${t.same + t.swapped + t.moved + t.removed} notes in ${d.a.name}. ` + `${t.same} same, ${t.swapped} re-sung, ${t.moved} moved, ${t.added} added, ${t.removed} dropped. ` + `|shift| p90 ${d.a.shiftP90?.toFixed(2) ?? "—"} → ${d.b.shiftP90?.toFixed(2) ?? "—"} st. ` + `${d.videosOnlyA.length} source episode${d.videosOnlyA.length === 1 ? "" : "s"} left, ` + `${d.videosOnlyB.length} entered.` ); }