import { stateFile } from "./paths"; import { readJson } from "./state"; import { archiveMomentUrl } from "./archive"; import { usageIndex, type VideoUse } from "./usage"; // --------------------------------------------------------------------------- // Whose voice is this, really. // // Every README here says the palette is drawn from a corpus of 300 episodes. // Nothing checks the shape of that draw. If 40% of the notes in a song come // from three episodes then the song is a portrait of three videos, not of a // channel -- which is a true and interesting thing about the work, and it is // also the difference between a claim that holds and one that does not. // // PLACEMENTS OR CLIPS -- both, always labelled. A placement is a note in a // finished render, so placements are how much of the RUNTIME comes from an // episode. Clips are how much of the PALETTE does. The same episode can be one // beloved clip used ninety times or ninety clips used once, and those are // different facts about the corpus. // --------------------------------------------------------------------------- export type Shares = { /** Units counted (placements or clips), and how many sources they came from. */ total: number; distinct: number; /** Share of the total held by the largest source, and by the largest five. */ top1: number; top5: number; /** How many sources it takes to cover half the total. */ half: number; /** Herfindahl: 1 is one source, 1/n is perfectly even. */ hhi: number; }; /** The arithmetic, over counts in any order. */ export function shares(counts: number[]): Shares { const sorted = [...counts].filter((n) => n > 0).sort((a, b) => b - a); const total = sorted.reduce((a, b) => a + b, 0); if (!total) return { total: 0, distinct: 0, top1: 0, top5: 0, half: 0, hhi: 0 }; let acc = 0; let half = 0; for (const n of sorted) { acc += n; half += 1; if (acc * 2 >= total) break; } return { total, distinct: sorted.length, top1: +(sorted[0] / total).toFixed(4), top5: +(sorted.slice(0, 5).reduce((a, b) => a + b, 0) / total).toFixed(4), half, hhi: +sorted.reduce((a, n) => a + (n / total) ** 2, 0).toFixed(4), }; } /** One line for a panel that already has the per-source counts in hand. */ export function concentrationLine(s: Shares, unit = "notes"): string { if (!s.total) return ""; const pc = (v: number) => `${Math.round(v * 100)}%`; return ( `${s.distinct} source episodes · the top one supplies ${pc(s.top1)} of the ${unit}, ` + `the top five ${pc(s.top5)} · ${s.half} cover half` ); } export type SourceRow = VideoUse & { title: string; date: string; link: string | null; }; export type Concentration = { videos: SourceRow[]; placements: Shares; clips: Shares; builds: number; }; /** * Every source episode across every plan in the tree. * * No new scan: this re-keys the usage index, which is already built from the * plan files' mtimes and is already in memory whenever a clip has been looked * at. Titles and dates join from the same two state files readProvenance uses. */ export async function corpusConcentration(): Promise { const idx = await usageIndex(); const [titles, dates] = await Promise.all([ readJson>(stateFile("titles.json"), {}), readJson>(stateFile("dates.json"), {}), ]); const videos: SourceRow[] = [...idx.byVideo.values()] .map((v) => ({ ...v, title: titles[v.video] ?? "", date: dates[v.video] ?? "", // No srcStart is known at this level -- the roll-up is over placements, // not moments -- so this lands at the top of the episode rather than // inventing a moment it cannot support. link: archiveMomentUrl(v.video, 0), })) .sort((a, b) => b.placements - a.placements || a.video.localeCompare(b.video)); return { videos, placements: shares(videos.map((v) => v.placements)), clips: shares(videos.map((v) => v.clips)), builds: idx.builds, }; }