diff --git a/README.md b/README.md index bd12f8c..83202f9 100644 --- a/README.md +++ b/README.md @@ -6,12 +6,13 @@ **[blog.psheon.me](https://blog.psheon.me)** · [English](https://blog.psheon.me/en) · [中文](https://blog.psheon.me/zh) · [RSS](https://blog.psheon.me/en/feed.xml) An interactive blog. Every article builds one thing from scratch — a CNN, a Transformer, a diffusion model, -neuroevolution, a quadruped's walking policy, a SLAM system, a city of autonomous agents, a task scheduler, a path tracer — and +neuroevolution, a quadruped's walking policy, a SLAM system, a city of autonomous agents, a task scheduler, a path tracer, +an AI composer — and ships with instruments you can train, open up and break right in the page. Apart from MuJoCo's physics in article 006, everything is written in TypeScript and runs in the reader's tab: no TensorFlow.js, no ONNX Runtime, no server. Bilingual (繁體中文 / English). 從零實作的互動部落格:每一篇都能在瀏覽器裡訓練、拆開、弄壞。CNN、Transformer、擴散模型、神經演化、 -機器狗走路策略、SLAM、自己過日子的小鎮居民、任務排程器、路徑追蹤器,全部用 TypeScript 從零寫起,不靠任何機器學習函式庫, +機器狗走路策略、SLAM、自己過日子的小鎮居民、任務排程器、路徑追蹤器、AI 作曲家,全部用 TypeScript 從零寫起,不靠任何機器學習函式庫, 也不靠伺服器。 ![paul.notebook](https://blog.psheon.me/en/opengraph-image) @@ -20,6 +21,7 @@ everything is written in TypeScript and runs in the reader's tab: no TensorFlow. | № | English | 中文 | | --- | --- | --- | +| 014 | [Train your own AI composer in three minutes: from noodling to calm](https://blog.psheon.me/en/posts/music-ai) | [三分鐘訓練你的 AI 作曲家:聽它從亂彈到療癒](https://blog.psheon.me/zh/posts/music-ai) | | 013 | [A mini Figure AI in three minutes: one camera, one arm](https://blog.psheon.me/en/posts/head-camera) | [三分鐘打造迷你 Figure AI:只給它一顆相機、一隻手臂](https://blog.psheon.me/zh/posts/head-camera) | | 012 | [Putting a path tracer in a 3D playground: crash a car, gun a plane, fly a helicopter](https://blog.psheon.me/en/posts/light-playground) | [把光追放進 3D 遊樂園:體驗開車碰撞、飛機加速、直升機飛行](https://blog.psheon.me/zh/posts/light-playground) | | 011 | [How a picture clears from snow: a WebGPU path tracer from scratch](https://blog.psheon.me/en/posts/light-from-noise) | [一張圖怎麼從雪花變清晰:從零寫一個 WebGPU 路徑追蹤器](https://blog.psheon.me/zh/posts/light-from-noise) | @@ -118,6 +120,7 @@ Conventions that keep pages fast and honest: | `styles/launch-ui.css` | Glass, fade and hairline utilities from Launch UI | | `public/sw.js` | The service worker (network first for pages, cache first for immutable assets) | | `scripts/train-mnist` | One-off PyTorch training for article 001; exports weights and golden values for the tests | +| `scripts/music` | Packs article 014's chorales and its trained composer into `public/posts/music-ai/*.bin` | | `scripts/light` | Packs the light series' assets (the playground, the vehicles, the character and its clips) into `public/posts/light-playground/*.bin`: geometry, skeleton and names only, never a texture | | `tests` | Vitest suites, grouped by library and by article | | `e2e` | Playwright smoke tests and axe accessibility checks | @@ -142,4 +145,7 @@ Layout primitives and CSS utilities adapted from [Launch UI](https://www.launchu AAPL price data in the trading article is daily closing prices for 2024. The Lite3 robot model and walking policy in article 006 come from DEEP Robotics; their licences are in `public/lite3/`. The light series' playground is the geometry of [Sketchbook](https://github.com/swift502/Sketchbook)'s world by -Jan Blaha (MIT); its textures are not used and `world.glb` is never committed. +Jan Blaha (MIT); its textures are not used and `world.glb` is never committed. Article 014 trains on +[Craig Sapp's digital edition of Bach's 370 four-part chorales](https://github.com/craigsapp/bach-370-chorales) +(CC BY-NC-SA 4.0); the tokens and the trained model in `public/posts/music-ai/` are derived from it and carry the +same licence. diff --git a/components/site/post-cover.tsx b/components/site/post-cover.tsx index efa3c5b..2bb91fe 100644 --- a/components/site/post-cover.tsx +++ b/components/site/post-cover.tsx @@ -318,6 +318,30 @@ function HeadCamera() { ); } +/** № 014: a piano roll that starts as scattered notes and settles into four voices, the way the composer learns. */ +function MusicAi() { + const rand = rng(23); + // Four voices on their own rows. On the left each note is thrown anywhere; by the right they have found their rows. + const notes: { x: number; y: number; w: number; v: number }[] = []; + const rows = [26, 42, 58, 74]; + for (let v = 0; v < 4; v++) { + for (let x = 12; x < 150; ) { + const settled = Math.min(1, Math.max(0, (x - 30) / 70)); + const stray = (rand() - 0.5) * 46 * (1 - settled); + const w = r1(6 + rand() * (4 + 12 * settled)); + notes.push({ x: r1(x), y: r1(rows[v] + stray + (settled > 0.6 ? (rand() - 0.5) * 7 : 0)), w, v }); + x += w + 3; + } + } + const ink = [S, S3, S2, "var(--chart-5)"]; + return ( + <> + {[12, 47, 82, 117, 150].map((x) => )} + {notes.map((n, i) => )} + + ); +} + const covers: Record ReactNode> = { "cnn-from-scratch": Cnn, "ai-flappy-bird": Flappy, @@ -332,6 +356,7 @@ const covers: Record ReactNode> = { "light-from-noise": Light, "light-playground": Playground, "head-camera": HeadCamera, + "music-ai": MusicAi, }; export function PostCover({ slug, no, className }: { slug: string; no: number; className?: string }) { diff --git a/content/posts/music-ai/components/blind-lab.tsx b/content/posts/music-ai/components/blind-lab.tsx new file mode 100644 index 0000000..2f50bb8 --- /dev/null +++ b/content/posts/music-ai/components/blind-lab.tsx @@ -0,0 +1,129 @@ +"use client"; + +import { Check, Heart, Pause, Play, Shuffle } from "lucide-react"; +import { useState } from "react"; +import { Button } from "@/components/ui/button"; +import { cn } from "@/lib/utils"; +import { TimbrePicker } from "./composer-lab"; +import { useLabels } from "./labels"; +import { PianoRoll, VoiceLegend } from "./piano-roll"; +import { usePlayer, useStopOnLeave } from "./player"; +import type { BlindItem } from "./protocol"; +import { useSharedWorker } from "./shared-worker"; +import type { Timbre } from "./synth"; + +type Who = BlindItem["who"]; +const WHO: Who[] = ["bach", "ai", "rules"]; +const LETTERS = ["A", "B", "C"]; + +/** Fig. 03: Bach, the AI and five lines of maths, unlabelled. Guess who wrote what, and say which one you liked. */ +export function BlindLab() { + const t = useLabels(); + const [state, setState] = useState<"idle" | "loading" | "listening" | "revealed" | "failed">("idle"); + const [items, setItems] = useState([]); + const [guesses, setGuesses] = useState<(Who | null)[]>([null, null, null]); + const [favourite, setFavourite] = useState(null); + const [timbre, setTimbre] = useState("pad"); + const player = usePlayer(); + useStopOnLeave(); + const send = useSharedWorker((data) => { + if (data.type === "loading") setState("loading"); + else if (data.type === "failed") setState("failed"); + else if (data.type === "blind") { setItems(data.items); setState("listening"); } + }); + + const deal = () => { + player.stop(); + setGuesses([null, null, null]); + setFavourite(null); + setItems([]); + send({ type: "blind" }); + }; + + if (state === "idle" || state === "loading" || state === "failed") + return ( +
+ +

{state === "failed" ? t.failed : state === "loading" ? t.judgeLoading : t.blindHow}

+
+ ); + + const revealed = state === "revealed"; + const right = items.filter((item, i) => guesses[i] === item.who).length; + + return ( +
+
+

{t.blindHow}

+ { setTimbre(v); player.restyle(v); }} /> +
+ +
+ {items.map((item, i) => { + const key = `blind-${i}-${item.who}`, playing = player.playing === key; + const correct = guesses[i] === item.who; + return ( +
+
+

{LETTERS[i]}

+ +
+ + {/* No roll before the answers: a chorale and a block-chord machine look nothing alike, which would give it away. */} + {revealed && } + +
+ {t.blindGuess} +
+ {WHO.map((who) => ( + + ))} +
+
+ + + + {revealed && ( +
+

+ {t.blindAnswer}: {t.blindWho[item.who]} · {correct ? t.blindCorrect : t.blindWrong} +

+

{t.parallels} {item.report.parallels.toFixed(1)} · {t.blindChords} {item.report.chords.toFixed(0)}

+
+ )} +
+ ); + })} +
+ + {revealed && } + +
+ {revealed ? ( + <> + {t.blindScore(right)} + {favourite !== null && items[favourite] && {t.blindYours} {t.blindWho[items[favourite].who]}} + + + ) : ( + <> + + {guesses.some((g) => g === null) ? t.blindNeedAll : t.blindWaiting} + + )} +
+
+ ); +} diff --git a/content/posts/music-ai/components/composer-lab.tsx b/content/posts/music-ai/components/composer-lab.tsx new file mode 100644 index 0000000..9ff36d9 --- /dev/null +++ b/content/posts/music-ai/components/composer-lab.tsx @@ -0,0 +1,152 @@ +"use client"; + +import { Pause, Play, Square } from "lucide-react"; +import { useEffect, useRef, useState } from "react"; +import { Readout } from "@/components/lab/readout"; +import { Sparkline } from "@/components/lab/sparkline"; +import { Button } from "@/components/ui/button"; +import { cn } from "@/lib/utils"; +import { useLabels } from "./labels"; +import { NGRAM_LOSS } from "./music"; +import { PianoRoll, VoiceLegend } from "./piano-roll"; +import { usePlayer, useStopOnLeave } from "./player"; +import type { Reply, Request, Snapshot } from "./protocol"; +import { TIMBRES, type Timbre } from "./synth"; +import { createMusicWorker } from "./worker-factory"; + +/** 2,400 steps of 8 windows: about three minutes (69 ms a step in Chromium on an M4 Pro; docs/research/music-ai). */ +const STEPS = 2400; + +export function TimbrePicker({ value, onChange }: { value: Timbre; onChange(t: Timbre): void }) { + const t = useLabels(); + return ( +
+ {t.timbre} + {TIMBRES.map((name) => ( + + ))} +
+ ); +} + +/** Fig. 01: the reader trains a composer in a worker and hears what it writes at six moments on the way. */ +export function ComposerLab() { + const t = useLabels(); + const worker = useRef(null); + const [state, setState] = useState<"idle" | "loading" | "training" | "done" | "failed">("idle"); + const [losses, setLosses] = useState([]), [progress, setProgress] = useState({ step: 0, seconds: 0 }); + const [snapshots, setSnapshots] = useState([]), [chosen, setChosen] = useState(null); + const [timbre, setTimbre] = useState("pad"), [autoplay, setAutoplay] = useState(true); + // Once the reader picks a snapshot themselves, a new one no longer takes the stage from under them. + const [pinned, setPinned] = useState(false), [unheard, setUnheard] = useState([]); + const player = usePlayer(); + useStopOnLeave(); + const auto = useRef({ autoplay, timbre, pinned }); + useEffect(() => { auto.current = { autoplay, timbre, pinned }; }, [autoplay, timbre, pinned]); + useEffect(() => () => worker.current?.terminate(), []); + + const start = () => { + worker.current?.terminate(); + const w = createMusicWorker(); + worker.current = w; + w.onmessage = ({ data }: MessageEvent) => { + if (data.type === "loading") setState("loading"); + else if (data.type === "failed") setState("failed"); + else if (data.type === "progress") { setState("training"); setProgress({ step: data.step, seconds: data.seconds }); setLosses((l) => [...l, data.loss]); } + else if (data.type === "snapshot") { + const s = data.snapshot; + setSnapshots((list) => [...list, s]); + if (auto.current.pinned) { setUnheard((u) => [...u, s.step]); return; } + setChosen(s.step); + if (auto.current.autoplay) void player.play(`snap-${s.step}`, s.piece, auto.current.timbre); + } else if (data.type === "done") { setState("done"); w.terminate(); worker.current = null; } + }; + setLosses([]); setSnapshots([]); setChosen(null); setProgress({ step: 0, seconds: 0 }); setPinned(false); setUnheard([]); + w.postMessage({ type: "train", seed: Math.floor(Math.random() * 2 ** 31), steps: STEPS } satisfies Request); + }; + const stop = () => { worker.current?.terminate(); worker.current = null; setState("idle"); }; + + const shown = snapshots.find((s) => s.step === chosen) ?? snapshots.at(-1) ?? null; + const key = shown ? `snap-${shown.step}` : "snap-none"; + const playing = player.playing === key; + const busy = state === "loading" || state === "training"; + const eta = state === "training" && progress.step > 0 ? (progress.seconds / progress.step) * (STEPS - progress.step) : null; + + return ( +
+
+ {busy ? ( + + ) : ( + + )} + {state === "loading" ? t.loading : state === "failed" ? t.failed : t.trainNote} +
+ +
+
+
+ +
+
+
+ + + + +
+
+

{t.loss}

+ +

{t.lossNote}

+
+
+ +
+
+

{t.snapshots}

+ +
+
+ {snapshots.map((s) => ( + + ))} +
+
+ {shown ? :

{t.waiting}

} +
+
+ + +
+ { setTimbre(v); player.restyle(v); }} /> + {shown && ( +
+ + + + + +
+ )} + {shown && ( +

{t.metricsNote}

+ )} +
+
+
+ ); +} diff --git a/content/posts/music-ai/components/index.ts b/content/posts/music-ai/components/index.ts new file mode 100644 index 0000000..e9d1e3c --- /dev/null +++ b/content/posts/music-ai/components/index.ts @@ -0,0 +1,11 @@ +"use client"; + +import dynamic from "next/dynamic"; + +/* The article's instruments as one lazy chunk: see content/posts/slam-2d/components/index.ts for why it is done this way. */ +const loading = typeof window === "undefined" ? null : import("./labs"); +const labs = () => loading ?? import("./labs"); + +export const ComposerLab = dynamic(() => labs().then((m) => m.ComposerLab)); +export const JudgeLab = dynamic(() => labs().then((m) => m.JudgeLab)); +export const BlindLab = dynamic(() => labs().then((m) => m.BlindLab)); diff --git a/content/posts/music-ai/components/judge-lab.tsx b/content/posts/music-ai/components/judge-lab.tsx new file mode 100644 index 0000000..f9529e1 --- /dev/null +++ b/content/posts/music-ai/components/judge-lab.tsx @@ -0,0 +1,96 @@ +"use client"; + +import { Check, Pause, Play, RotateCcw, Volume2 } from "lucide-react"; +import { useEffect, useRef, useState } from "react"; +import { Readout } from "@/components/lab/readout"; +import { Button } from "@/components/ui/button"; +import { useLabels } from "./labels"; +import type { Piece, Report } from "./music"; +import { PianoRoll, VoiceLegend } from "./piano-roll"; +import { usePlayer, useStopOnLeave } from "./player"; +import { useSharedWorker } from "./shared-worker"; + +import { TimbrePicker } from "./composer-lab"; +import type { Timbre } from "./synth"; + +/** Fig. 02: the reader is the judge. Two candidates, one pick, one DPO step; the numbers show what the picks did. */ +export function JudgeLab() { + const t = useLabels(); + const [state, setState] = useState<"idle" | "loading" | "ready" | "thinking" | "failed">("idle"); + const [pair, setPair] = useState<[Piece, Piece] | null>(null), [round, setRound] = useState(0); + const [stats, setStats] = useState<{ choices: number; now: Report; before: Report } | null>(null); + const [timbre, setTimbre] = useState("pad"); + const player = usePlayer(); + useStopOnLeave(); + // The reply arrives long after the click, so the worker callback reads the timbre through a ref. + const timbreRef = useRef(timbre); + useEffect(() => { timbreRef.current = timbre; }, [timbre]); + const send = useSharedWorker((data) => { + if (data.type === "loading") setState("loading"); + else if (data.type === "failed") setState("failed"); + else if (data.type === "pair") { setPair(data.pieces); setRound((r) => r + 1); setState("ready"); } + else if (data.type === "judged") setStats({ choices: data.choices, now: data.now, before: data.before }); + else if (data.type === "sample") void player.play("judge-fresh", data.piece, timbreRef.current); + }); + const load = () => { setStats(null); send({ type: "judge" }); }; + const pick = (winner: 0 | 1) => { + player.stop(); + setState("thinking"); + send({ type: "choose", winner }); + }; + + if (state === "idle" || state === "loading" || state === "failed") + return ( +
+ +

{state === "loading" ? t.judgeLoading : state === "failed" ? t.failed : `${t.judgeNote} ${t.judgeHint}`}

+
+ ); + + return ( +
+
+

{state === "thinking" ? t.thinking : t.judgeReady}

+ { setTimbre(v); player.restyle(v); }} /> +
+
+ {[0, 1].map((i) => { + const key = `judge-${round}-${i}`, piece = pair?.[i] ?? null, playing = player.playing === key; + return ( +
+
+

{t.candidate(i)}

+ +
+ {piece && } + +
+ ); + })} +
+
+ +
+ + +
+
+ +
+

{stats ? `${t.choices(stats.choices)} · ${t.since}` : t.judgeHint}

+ {stats && ( +
+ + + + +
+ )} +
+
+ ); +} diff --git a/content/posts/music-ai/components/labels.ts b/content/posts/music-ai/components/labels.ts new file mode 100644 index 0000000..7d1430a --- /dev/null +++ b/content/posts/music-ai/components/labels.ts @@ -0,0 +1,101 @@ +"use client"; + +import { useLocaleLabels } from "@/components/lab/use-locale-labels"; + +const zh = { + train: "開始訓練", again: "重新訓練", stop: "停止", + trainNote: "在你的瀏覽器裡從零訓練,大約三分鐘。可以邊練邊聽,也可以繼續往下讀。", + loading: "下載 333 首聖詠中…", failed: "資料載入失敗,重新整理再試一次。", + step: "步數", seconds: "已經過", secondsUnit: "秒", loss: "猜下一個音的誤差", + lossNote: "曲線是訓練時的誤差;虛線是 5-gram 在驗證資料上的成績(1.46),拿來和每個片段旁邊的驗證誤差比。", + autoplay: "有新片段就自動播放", + snapshots: "它在這些時刻即興了一段", + snapshotAt: (step: number) => `第 ${step} 步`, + play: "播放", playing: "播放中", stopPlay: "停止播放", + timbre: "音色", timbres: { pad: "溫暖墊音", epiano: "柔和電鋼琴", flute: "長笛" }, + roll: (step: number) => `第 ${step} 步寫的六小節:一條線是一個聲部,由上到下是女高音、女低音、男高音、男低音`, + voices: ["女高音", "女低音", "男高音", "男低音"], + inKey: "在調上", crossings: "聲部交錯", held: "拖長的音", parallels: "平行五度", copied: "最長抄襲", + per100: "次/百格", eighths: "格", + val: "驗證誤差", + waiting: "按「開始訓練」,第 0 步的亂彈會出現在這裡。", + + stopTrain: "停止訓練", + eta: "大約還要", + metricsNote: "一格=一個八分音符,八格是一小節。驗證誤差是它猜下一個音的平均誤差,越低越好,5-gram 的成績是 1.46。", + rollPlain: "四個聲部的樂譜,由上到下是女高音、女低音、男高音、男低音", + isNew: "新", + judgeHint: "大約選 20 次就聽得出差別;選太多次它會只剩持續音。", + judgeListen: "聽它現在寫的", + judgeReset: "從頭開始", + blindNeedAll: "三段都猜完才能對答案", + blindWaiting: "聽完三段,再猜誰是誰", + blindStart: "出三段給我聽", + blindHow: "三段各四小節,一段是巴赫本人、一段是練好 1000 步的 AI、一段是五行數學規則,順序是隨機的。聽完再猜。", + blindGuess: "你覺得這段是誰寫的", + blindWho: { bach: "巴赫", ai: "AI", rules: "數學規則" }, + blindFavourite: "最喜歡這段", + blindReveal: "對答案", + blindAgain: "換三段", + blindScore: (right: number) => `猜對 ${right} / 3`, + blindYours: "你最喜歡的是", + blindCorrect: "猜對", blindWrong: "猜錯", blindAnswer: "答案", + blindChords: "和弦種類", + judgeStart: "載入一個練好的作曲家", judgeLoading: "下載模型中…", judgeReady: "它寫了兩段,聽完選你比較喜歡的", + pick: "選這段", candidate: (i: number): string => (i === 0 ? "A 段" : "B 段"), + choices: (n: number) => `你已經選了 ${n} 次`, + since: "從你開始選到現在", + thinking: "它正在照你的選擇改自己…", + judgeNote: "每選一次,模型往你選的那段靠近一點、離另一段遠一點(DPO)。它同時被拉在原本的樣子附近,不會一下子跑太遠。", + before: "調教前", now: "現在", +}; + +const en: typeof zh = { + train: "Start training", again: "Train again", stop: "Stop", + trainNote: "Trained from scratch in your browser, in about three minutes. Listen as it learns, or keep reading.", + loading: "Downloading 333 chorales…", failed: "The data failed to load. Reload the page to try again.", + step: "Step", seconds: "Elapsed", secondsUnit: "s", loss: "Error guessing the next note", + lossNote: "The curve is the training error; the dashed line is a 5-gram's validation score (1.46), to compare with each snapshot's validation error.", + autoplay: "Play each new snapshot", + snapshots: "It improvised at these moments", + snapshotAt: (step: number) => `Step ${step}`, + play: "Play", playing: "Playing", stopPlay: "Stop", + timbre: "Sound", timbres: { pad: "Warm pad", epiano: "Soft e-piano", flute: "Flute" }, + roll: (step: number) => `Six bars written at step ${step}: one line per voice, soprano, alto, tenor and bass from the top`, + voices: ["Soprano", "Alto", "Tenor", "Bass"], + inKey: "In key", crossings: "Voices crossing", held: "Held notes", parallels: "Parallel fifths", copied: "Longest copy", + per100: "/100 steps", eighths: "steps", + val: "Validation error", + waiting: "Press Start training: step 0's noodling will appear here.", + + stopTrain: "Stop training", + eta: "About", + metricsNote: "One step is an eighth note; eight of them make a bar. The validation error is how well it guesses the next note, lower is better; a 5-gram gets 1.46.", + rollPlain: "A score for four voices: soprano, alto, tenor and bass from the top", + isNew: "new", + judgeHint: "About 20 picks is enough to hear the difference; many more and it holds every note.", + judgeListen: "Hear what it writes now", + judgeReset: "Start over", + blindNeedAll: "Guess all three to see the answers", + blindWaiting: "Listen to all three, then guess", + blindStart: "Give me three pieces", + blindHow: "Four bars each: one by Bach, one by the AI trained for 1,000 steps, one by the five lines of maths, in a random order. Listen first, then guess.", + blindGuess: "Who wrote this one?", + blindWho: { bach: "Bach", ai: "The AI", rules: "The maths" }, + blindFavourite: "My favourite", + blindReveal: "Show the answers", + blindAgain: "Three more", + blindScore: (right: number) => `${right} of 3 right`, + blindYours: "You liked", + blindCorrect: "right", blindWrong: "wrong", blindAnswer: "Answer", + blindChords: "Distinct chords", + judgeStart: "Load a trained composer", judgeLoading: "Downloading the model…", judgeReady: "It wrote two pieces. Listen, then pick the one you like", + pick: "Pick this one", candidate: (i: number) => (i === 0 ? "A" : "B"), + choices: (n: number) => `${n} picks so far`, + since: "Since your first pick", + thinking: "Changing itself to match your pick…", + judgeNote: "Each pick moves the model a little toward the piece you chose and away from the other (DPO), while a leash keeps it near where it started.", + before: "Before", now: "Now", +}; + +export const useLabels = () => useLocaleLabels(zh, en); diff --git a/content/posts/music-ai/components/labs.ts b/content/posts/music-ai/components/labs.ts new file mode 100644 index 0000000..c340ac5 --- /dev/null +++ b/content/posts/music-ai/components/labs.ts @@ -0,0 +1,3 @@ +export { BlindLab } from "./blind-lab"; +export { ComposerLab } from "./composer-lab"; +export { JudgeLab } from "./judge-lab"; diff --git a/content/posts/music-ai/components/music.ts b/content/posts/music-ai/components/music.ts new file mode 100644 index 0000000..d825a9c --- /dev/null +++ b/content/posts/music-ai/components/music.ts @@ -0,0 +1,278 @@ +import { Adam, Mat, Tape, Transformer, type Rng, type TransformerConfig } from "@/lib/ml"; + +/* + * Music as tokens, a Transformer that composes, and what we measure about what it writes. The recipe and every number + * the article quotes come from docs/research/music-ai (run B: d 48, 2 layers, a 16-step window). + * + * A chorale is a grid of eighth notes. Each step has four voices, soprano to bass, and each voice-step is one token: + * a new note is its pitch, a held one HOLD, silence REST. So the model writes S A T B, S A T B, … and knows which voice + * is next only from its position: windows must always start on a soprano. + */ +export const LO = 31, HI = 84; +export const HOLD = HI - LO + 1, REST = HOLD + 1, VOCAB = REST + 1; +export const VOICES = 4; +export const STEPS_IN_WINDOW = 16; +export const CONFIG: TransformerConfig = { vocab: VOCAB, ctx: STEPS_IN_WINDOW * VOICES, d: 48, heads: 4, layers: 2 }; +/** Every composition starts from the same C major chord: C5 G4 E4 C3. */ +export const PROMPT = [72 - LO, 67 - LO, 64 - LO, 48 - LO]; +/** The 5-gram's validation loss on the same split (nats per token): the line the Transformer has to cross. */ +export const NGRAM_LOSS = 1.462; + +export interface Piece { pitches: number[][]; onset: number[][] } + +/** chorales.bin: u16 train count, u16 validation count, u16 length per chorale, then one byte per token. */ +export function parseChorales(buffer: ArrayBuffer): { train: number[][]; val: number[][] } { + const view = new DataView(buffer), bytes = new Uint8Array(buffer); + const nTrain = view.getUint16(0, true), nVal = view.getUint16(2, true), all: number[][] = []; + let at = 4 + 2 * (nTrain + nVal); + for (let i = 0; i < nTrain + nVal; i++) { + const len = view.getUint16(4 + 2 * i, true); + all.push(Array.from(bytes.subarray(at, at + len))); + at += len; + } + return { train: all.slice(0, nTrain), val: all.slice(nTrain) }; +} + +/** Tokens back to what sounds: the pitch of each voice at each step (−1 silent), and whether it starts there. */ +export function decode(ids: ArrayLike): Piece { + const pitches: number[][] = [], onset: number[][] = [], last = [-1, -1, -1, -1]; + for (let t = 0; t * VOICES < ids.length; t++) { + const p: number[] = [], o: number[] = []; + for (let v = 0; v < VOICES; v++) { + const id = ids[t * VOICES + v]; + if (id === undefined || id === REST) { if (id === REST) last[v] = -1; p.push(-1); o.push(0); } + else if (id === HOLD) { p.push(last[v]); o.push(0); } + else { last[v] = id + LO; p.push(last[v]); o.push(1); } + } + pitches.push(p); onset.push(o); + } + return { pitches, onset }; +} + +/** The context for the next token: the latest ≤ ctx tokens, cut so that it still starts on a soprano. */ +export function context(ids: number[], ctx = CONFIG.ctx): number[] { + return ids.slice(Math.max(0, Math.ceil((ids.length - ctx) / VOICES) * VOICES)); +} + +export function compose(model: Transformer, steps: number, temperature: number, rng: Rng, prompt = PROMPT): number[] { + const ids = [...prompt]; + while (ids.length < steps * VOICES) { + const ctx = context(ids, model.config.ctx); + const { logits } = model.forward(new Tape(), ctx); + const row = logits.data.subarray((ctx.length - 1) * VOCAB, ctx.length * VOCAB); + let max = -Infinity; + for (const x of row) max = Math.max(max, x); + const p = Array.from(row, (x) => Math.exp((x - max) / temperature)); + let u = rng() * p.reduce((a, b) => a + b, 0), k = 0; + while (u > p[k] && k < p.length - 1) u -= p[k++]; + ids.push(k); + } + return ids; +} + +/** A random training window of one context plus one token, starting on a step boundary. */ +export function window(pool: number[][], rng: Rng, ctx = CONFIG.ctx): number[] { + const seq = pool[Math.floor(rng() * pool.length)]; + const steps = seq.length / VOICES, len = Math.min(ctx + VOICES, seq.length); + const start = Math.floor(rng() * Math.max(1, steps - len / VOICES + 1)) * VOICES; + return seq.slice(start, start + len); +} + +/** Next-token training: Adam, lr 3e-3 with a 100-step warm-up and a cosine decay to a tenth, batches of 8. */ +export class Composer { + readonly model: Transformer; + private readonly adam: Adam; + step = 0; + + constructor(private readonly train: number[][], private readonly rng: Rng, readonly steps: number, weights?: Float32Array) { + this.model = new Transformer(CONFIG, rng); + if (weights) load(this.model, weights); + this.adam = new Adam(this.model.params); + } + + /** One optimiser step. Returns the batch's mean loss in nats per token. */ + learn(batch = 8, lr = 3e-3): number { + this.model.zeroGrad(); + const tape = new Tape(); + let loss = 0; + for (let b = 0; b < batch; b++) { + const w = window(this.train, this.rng); + loss += tape.crossEntropy(this.model.forward(tape, w.slice(0, -1).slice(0, CONFIG.ctx)).logits, w.slice(1, CONFIG.ctx + 1), 1 / batch) / batch; + } + tape.backward(); + const s = this.step++; + this.adam.step(lr * Math.min(1, (s + 1) / 100) * (0.1 + 0.9 * 0.5 * (1 + Math.cos((Math.PI * s) / this.steps)))); + return loss; + } +} + +/** Mean loss on fixed validation windows (the same ones every time, so the curve is comparable with itself). */ +export function validate(model: Transformer, windows: number[][]): number { + let s = 0; + for (const w of windows) s += new Tape().crossEntropy(model.forward(new Tape(), w.slice(0, -1).slice(0, CONFIG.ctx)).logits, w.slice(1, CONFIG.ctx + 1)); + return s / windows.length; +} + +export function load(model: Transformer, weights: Float32Array) { + let k = 0; + for (const p of Object.values(model.params)) for (let i = 0; i < p.data.length; i++) p.data[i] = weights[k++]; +} + +/** + * The opponent with no AI in it: a soprano that walks a C pentatonic scale by 1/f noise (Voss's dice trick — four + * dice rerolled at halving rates, which is how the pitch of real music wanders), over I–vi–IV–V, a bar each. + * It never plays a wrong note and never crosses a voice; the article's numbers say what it pays for that. + */ +export function rulesPiece(steps: number, rng: Rng): Piece { + const scale = [60, 62, 64, 67, 69, 72, 74, 76, 79, 81]; + const chords = [[48, 55, 64], [45, 57, 64], [41, 53, 60], [43, 55, 62]]; // C, Am, F, G: bass, tenor, alto + const dice = [rng(), rng(), rng(), rng()]; + const pitches: number[][] = [], onset: number[][] = []; + let soprano = -1; + for (let t = 0; t < steps; t++) { + if (t % 2 === 0) for (let d = 0; d < dice.length; d++) if (((t / 2) & ((1 << d) - 1)) === 0) dice[d] = rng(); + const chord = chords[Math.floor(t / 8) % chords.length]; + const next = scale[Math.min(scale.length - 1, Math.floor((dice.reduce((a, b) => a + b, 0) / dice.length) * scale.length))]; + const sings = t % 2 === 0 && next !== soprano; + if (sings) soprano = next; + pitches.push([soprano, chord[2], chord[1], chord[0]]); + onset.push([sings ? 1 : 0, ...(t % 8 === 0 ? [1, 1, 1] : [0, 0, 0])]); + } + return { pitches, onset }; +} + +// --------------------------------------------------------------------------------------------------------------- +// What we measure. None of it says "beautiful": it says whether the texture does what four-part writing does. + +/** C major's pitch classes, plus F# and G# (A minor's raised sixth and seventh). Every chorale was moved to C or A minor. */ +const KEY = new Set([0, 2, 4, 5, 7, 9, 11, 6, 8]); + +export interface Report { + /** share of new notes in the key */ + inKey: number; + /** parallel fifths or octaves per 100 steps */ + parallels: number; + /** share of steps where a lower voice is above a higher one */ + crossings: number; + /** share of sounding voice-steps that are held */ + held: number; + /** distinct chords (pitch-class sets) per 100 steps */ + chords: number; +} + +export function report({ pitches, onset }: Piece): Report { + let notes = 0, inKey = 0, parallels = 0, crossings = 0, held = 0, cells = 0; + const chords = new Set(); + for (let t = 0; t < pitches.length; t++) { + const p = pitches[t]; + for (let v = 0; v < VOICES; v++) { + if (p[v] < 0) continue; + cells++; + if (onset[t][v]) { notes++; if (KEY.has(p[v] % 12)) inKey++; } else held++; + } + for (let v = 0; v < VOICES - 1; v++) if (p[v] >= 0 && p[v + 1] >= 0 && p[v + 1] > p[v]) { crossings++; break; } + chords.add(p.filter((x) => x >= 0).map((x) => x % 12).sort((a, b) => a - b).join(",")); + if (t === 0) continue; + const q = pitches[t - 1]; + for (let a = 0; a < VOICES; a++) + for (let b = a + 1; b < VOICES; b++) { + if (p[a] < 0 || p[b] < 0 || q[a] < 0 || q[b] < 0) continue; + const now = (p[a] - p[b]) % 12, before = (q[a] - q[b]) % 12; + const together = p[a] !== q[a] && p[b] !== q[b] && Math.sign(p[a] - q[a]) === Math.sign(p[b] - q[b]); + if (together && now === before && (now === 7 || now === 0)) parallels++; + } + } + const n = pitches.length; + return { inKey: notes ? inKey / notes : 0, parallels: (100 * parallels) / n, crossings: crossings / n, held: cells ? held / cells : 0, chords: (100 * chords.size) / n }; +} + +/** Which training steps (all four voices) sound the same, for finding the longest stretch copied from the chorales. */ +export class CopyFinder { + private readonly index = new Map(); + private readonly pieces: number[][][]; + + constructor(train: number[][]) { + this.pieces = train.map((t) => decode(t).pitches); + this.pieces.forEach((c, ci) => c.forEach((p, t) => { + const k = p.join(","); + const list = this.index.get(k); + if (list) list.push([ci, t]); else this.index.set(k, [[ci, t]]); + })); + } + + /** The longest run of steps in `pitches` that also occurs, in the same order, in one training chorale. */ + longest(pitches: number[][]): { steps: number; chorale: number; at: number; from: number } { + let best = { steps: 0, chorale: -1, at: 0, from: 0 }; + for (let t = 0; t < pitches.length; t++) + for (const [ci, s] of this.index.get(pitches[t].join(",")) ?? []) { + const c = this.pieces[ci]; + let n = 0; + while (t + n < pitches.length && s + n < c.length && c[s + n].join(",") === pitches[t + n].join(",")) n++; + if (n > best.steps) best = { steps: n, chorale: ci, at: s, from: t }; + } + return best; + } +} + +// --------------------------------------------------------------------------------------------------------------- +// Preference tuning (DPO). The reader hears two short pieces and picks one; the model moves toward the pick and away +// from the other, measured against a frozen copy of itself so that it cannot drift far from Bach (β is the leash). + +/** Sum of log-probabilities of the tokens after the prompt, and what crossEntropy needs to push on them. */ +function logProb(model: Transformer, ids: number[], tape: Tape): { value: number; logits: Mat; targets: number[]; count: number } { + const { logits } = model.forward(tape, ids.slice(0, -1)); + const targets = ids.slice(1).map((t, i) => (i + 1 < PROMPT.length ? -1 : t)); + let value = 0, count = 0; + for (let r = 0; r < targets.length; r++) { + if (targets[r] < 0) continue; + const row = logits.data.subarray(r * VOCAB, (r + 1) * VOCAB); + let max = -Infinity; + for (const x of row) max = Math.max(max, x); + let s = 0; + for (const x of row) s += Math.exp(x - max); + value += row[targets[r]] - max - Math.log(s); + count++; + } + return { value, logits, targets, count }; +} + +export class Judge { + readonly policy: Transformer; + private readonly reference: Transformer; + private readonly adam: Adam; + choices = 0; + + /** + * β 0.5 and lr 1e-4 per pick: 20 picks of "more held notes" took held notes from 48 % to 64 % with the validation loss + * unchanged; at 3e-4 the same 20 went to 88 % and the loss rose from 1.195 to 1.257 (docs/research/music-ai). + */ + constructor(weights: Float32Array, private readonly beta = 0.5, private readonly lr = 1e-4) { + this.policy = new Transformer(CONFIG); + this.reference = new Transformer(CONFIG); + load(this.policy, weights); + load(this.reference, weights); + this.adam = new Adam(this.policy.params); + } + + /** Two candidates of one window each (16 steps, 8 seconds at 60 bpm), sampled at temperature 1. */ + pair(rng: Rng): [number[], number[]] { + const n = CONFIG.ctx + 1; + return [compose(this.policy, Math.ceil(n / VOICES), 1, rng).slice(0, n), compose(this.policy, Math.ceil(n / VOICES), 1, rng).slice(0, n)]; + } + + /** One DPO step on one choice: L = −log σ(β[(log π(win) − log π₀(win)) − (log π(lose) − log π₀(lose))]). */ + choose(win: number[], lose: number[]) { + this.policy.zeroGrad(); + const tape = new Tape(); + const pw = logProb(this.policy, win, tape), pl = logProb(this.policy, lose, tape); + const rw = logProb(this.reference, win, new Tape()).value, rl = logProb(this.reference, lose, new Tape()).value; + const z = this.beta * (pw.value - rw - (pl.value - rl)); + const g = this.beta * (1 - 1 / (1 + Math.exp(-z))); + // crossEntropy's gradient is weight × d(mean NLL), and log p = −count × mean NLL. + tape.crossEntropy(pw.logits, pw.targets, g * pw.count); + tape.crossEntropy(pl.logits, pl.targets, -g * pl.count); + tape.backward(); + this.adam.step(this.lr); + this.choices++; + } +} diff --git a/content/posts/music-ai/components/music.worker.ts b/content/posts/music-ai/components/music.worker.ts new file mode 100644 index 0000000..864b567 --- /dev/null +++ b/content/posts/music-ai/components/music.worker.ts @@ -0,0 +1,145 @@ +/// +import { mulberry32 } from "@/lib/ml"; +import { Transformer } from "@/lib/ml"; +import { CONFIG, Composer, CopyFinder, Judge, VOICES, compose, decode, load, parseChorales, report, rulesPiece, validate, window, type Report } from "./music"; +import type { BlindItem, Reply, Request } from "./protocol"; +import { render } from "./synth"; + +/* + * Training, composing, preference tuning and the synthesiser all run here, so the page keeps scrolling. Snapshots are + * taken where docs/research/music-ai saw the model change its mind: it drones first, then the voices stop crossing, + * then the key and the rhythm settle. + */ +const SNAPSHOTS = [0, 100, 250, 500, 1000, 2400]; +/** A composition is six bars: 48 eighths, 24 seconds. The blind test uses four, so three in a row stay listenable. */ +const PIECE_STEPS = 48; +const BLIND_STEPS = 32; +const EVERY = 10; + +const send = (reply: Reply, transfer: Transferable[] = []) => (self as unknown as Worker).postMessage(reply, transfer); +let run = 0; +let data: ReturnType | null = null; +let finder: CopyFinder | null = null; +let judge: Judge | null = null; +let candidates: [number[], number[]] | null = null; +let before: Report | null = null; +let trained: Transformer | null = null; +let weights: Float32Array | null = null; + +/** Run B at step 1,000, as shipped in judge.bin: the blind test composes with it and the judge figure tunes a copy. */ +async function judgeWeights(): Promise { + weights ??= new Float32Array(await (await fetch("/posts/music-ai/judge.bin")).arrayBuffer()); + return weights; +} + +async function trainedModel(): Promise { + if (!trained) { + trained = new Transformer(CONFIG); + load(trained, await judgeWeights()); + } + return trained; +} + +async function chorales() { + if (!data) data = parseChorales(await (await fetch("/posts/music-ai/chorales.bin")).arrayBuffer()); + finder ??= new CopyFinder(data.train); + return data; +} + +/** How the model writes right now, from eight seeds (the same eight every time, so a change is the model's): what the judge figure shows. */ +function measure(model: Composer["model"]): Report { + const reps = [1, 2, 3, 4, 5, 6, 7, 8].map((seed) => report(decode(compose(model, 32, 0.7, mulberry32(seed))))); + const mean = (k: keyof Report) => reps.reduce((s, r) => s + r[k], 0) / reps.length; + return { inKey: mean("inKey"), parallels: mean("parallels"), crossings: mean("crossings"), held: mean("held"), chords: mean("chords") }; +} + +self.onmessage = async ({ data: request }: MessageEvent) => { + // Every reply goes back tagged with whoever asked, because two figures share this worker. + const post = (reply: Reply, transfer: Transferable[] = []) => send({ ...reply, from: request.from }, transfer); + if (request.type === "stop") { run++; return; } + + if (request.type === "render") { + const audio = render(request.piece, request.timbre); + post({ type: "audio", id: request.id, ...audio }, [audio.left.buffer, audio.right.buffer]); + return; + } + + if (request.type === "train") { + const id = ++run; + post({ type: "loading" }); + let d: Awaited>; + try { d = await chorales(); } catch { post({ type: "failed" }); return; } + const rng = mulberry32(request.seed), composer = new Composer(d.train, rng, request.steps); + const valRng = mulberry32(99), valWindows = Array.from({ length: 24 }, () => window(d.val, valRng)); + const t0 = performance.now(); + let losses = 0, count = 0; + const snapshot = () => { + const piece = decode(compose(composer.model, PIECE_STEPS, 0.7, mulberry32(request.seed + composer.step))); + post({ type: "snapshot", snapshot: { step: composer.step, seconds: (performance.now() - t0) / 1000, val: validate(composer.model, valWindows), piece, report: report(piece), copied: finder!.longest(piece.pitches).steps } }); + }; + snapshot(); + const chunk = () => { + if (id !== run) return; + for (let i = 0; i < EVERY && composer.step < request.steps; i++) { losses += composer.learn(); count++; } + post({ type: "progress", step: composer.step, steps: request.steps, loss: losses / count, seconds: (performance.now() - t0) / 1000 }); + losses = 0; count = 0; + if (SNAPSHOTS.includes(composer.step) || composer.step === request.steps) snapshot(); + if (composer.step < request.steps) setTimeout(chunk, 0); + else post({ type: "done" }); + }; + setTimeout(chunk, 0); + return; + } + + if (request.type === "blind") { + post({ type: "loading" }); + try { + const d = await chorales(), model = await trainedModel(); + const rng = mulberry32(Math.floor(Math.random() * 2 ** 31)); + // Bach is an excerpt of a validation chorale — one the model never saw — starting on a bar line. + const chorale = d.val[Math.floor(rng() * d.val.length)]; + const bars = Math.max(1, Math.floor(chorale.length / VOICES / 8) - BLIND_STEPS / 8); + const from = Math.floor(rng() * bars) * 8 * VOICES; + const items: BlindItem[] = [ + { who: "bach" as const, piece: decode(chorale.slice(from, from + BLIND_STEPS * VOICES)) }, + { who: "ai" as const, piece: decode(compose(model, BLIND_STEPS, 0.7, rng)) }, + { who: "rules" as const, piece: rulesPiece(BLIND_STEPS, rng) }, + ].map((x) => ({ ...x, report: report(x.piece) })); + // Shuffled, so that A, B and C say nothing. + for (let i = items.length - 1; i > 0; i--) { const j = Math.floor(rng() * (i + 1)); [items[i], items[j]] = [items[j], items[i]]; } + post({ type: "blind", items }); + } catch { post({ type: "failed" }); } + return; + } + + if (request.type === "judge") { + post({ type: "loading" }); + try { + judge = new Judge(await judgeWeights()); + before = measure(judge.policy); + // The first pair comes with the model, so the figure has something to play the moment it is ready. + candidates = judge.pair(mulberry32(Math.floor(Math.random() * 2 ** 31))); + post({ type: "pair", pieces: [decode(candidates[0].slice(0, -1)), decode(candidates[1].slice(0, -1))] }); + } catch { post({ type: "failed" }); } + return; + } + + if (!judge) return; + if (request.type === "sample") { + post({ type: "sample", piece: decode(compose(judge.policy, PIECE_STEPS, 0.7, mulberry32(Math.floor(Math.random() * 2 ** 31)))) }); + return; + } + if (request.type === "pair") { + candidates = judge.pair(mulberry32(request.seed)); + post({ type: "pair", pieces: [decode(candidates[0].slice(0, -1)), decode(candidates[1].slice(0, -1))] }); + return; + } + if (request.type === "choose" && candidates) { + const [a, b] = candidates; + judge.choose(request.winner === 0 ? a : b, request.winner === 0 ? b : a); + // The next pair first, so the reader can listen while the numbers are worked out. + candidates = judge.pair(mulberry32(Math.floor(Math.random() * 2 ** 31))); + post({ type: "pair", pieces: [decode(candidates[0].slice(0, -1)), decode(candidates[1].slice(0, -1))] }); + post({ type: "judged", choices: judge.choices, now: measure(judge.policy), before: before! }); + } +}; diff --git a/content/posts/music-ai/components/piano-roll.tsx b/content/posts/music-ai/components/piano-roll.tsx new file mode 100644 index 0000000..16eca82 --- /dev/null +++ b/content/posts/music-ai/components/piano-roll.tsx @@ -0,0 +1,57 @@ +"use client"; + +import { useRef } from "react"; +import type { Piece } from "./music"; +import { usePlayhead } from "./player"; + +/** Soprano to bass. Each voice wears one series colour (DESIGN §2's order), and the legend names them. */ +export const VOICE_COLOURS = ["var(--signal)", "var(--signal-3)", "var(--signal-2)", "var(--chart-5)"]; +const LOW = 36, HIGH = 84, STEP_W = 8, ROW_H = 2.4; + +/** A piano roll: time to the right, pitch upward, one bar per note. A playhead crosses it while `playKey` plays. */ +export function PianoRoll({ piece, label, playKey, className }: { piece: Piece | null; label: string; playKey: string; className?: string }) { + const head = useRef(null); + const steps = Math.max(piece?.pitches.length ?? 48, 1); + const W = steps * STEP_W, H = (HIGH - LOW) * ROW_H; + usePlayhead(playKey, (f) => { + const line = head.current; + if (!line) return; + line.style.opacity = f === null ? "0" : "1"; + if (f !== null) line.setAttribute("transform", `translate(${(f * W).toFixed(1)} 0)`); + }); + + const notes: { x: number; y: number; w: number; v: number }[] = []; + if (piece) + for (let v = 0; v < 4; v++) + for (let t = 0; t < piece.pitches.length; ) { + const p = piece.pitches[t][v]; + if (p < 0) { t++; continue; } + let len = 1; + while (t + len < piece.pitches.length && piece.pitches[t + len][v] === p && !piece.onset[t + len][v]) len++; + notes.push({ x: t * STEP_W, y: (HIGH - Math.min(HIGH, Math.max(LOW, p))) * ROW_H, w: len * STEP_W, v }); + t += len; + } + + return ( + + {/* A faint line at every bar (eight eighths), and C3 / C4 / C5 as the only pitch rules. */} + {Array.from({ length: Math.floor(steps / 8) + 1 }, (_, i) => )} + {[48, 60, 72].map((c) => )} + {notes.map((n, i) => )} + + + ); +} + +export function VoiceLegend({ names }: { names: string[] }) { + return ( +

+ {names.map((n, i) => ( + + + {n} + + ))} +

+ ); +} diff --git a/content/posts/music-ai/components/player.ts b/content/posts/music-ai/components/player.ts new file mode 100644 index 0000000..1157ff0 --- /dev/null +++ b/content/posts/music-ai/components/player.ts @@ -0,0 +1,119 @@ +"use client"; + +import { useCallback, useEffect, useRef, useState } from "react"; +import type { Piece } from "./music"; +import type { Reply, Request } from "./protocol"; +import type { Timbre } from "./synth"; +import { createMusicWorker } from "./worker-factory"; + +/* + * One synthesiser worker and one AudioContext for the page, shared by every figure: playing something stops whatever + * was playing. The worker is separate from the training one, so a note never waits behind a training step. The + * AudioContext is made on the first press of Play (a browser only lets a page make sound after a gesture). + */ +let synth: Worker | null = null, audio: AudioContext | null = null, source: AudioBufferSourceNode | null = null; +/** Fires when the music itself ends, a little before the buffer does: the reverb tail rings on, the controls do not wait for it. */ +let endTimer: ReturnType | null = null; +let nextId = 0, current: { key: string; started: number; seconds: number } | null = null; +/** What is playing, kept so that changing the timbre can re-render the same piece at once. */ +let last: { key: string; piece: Piece } | null = null; +const listeners = new Set<() => void>(); +const changed = () => listeners.forEach((f) => f()); + +function worker(): Worker { + synth ??= createMusicWorker(); + return synth; +} + +export function stopAll() { + if (endTimer) { clearTimeout(endTimer); endTimer = null; } + source?.stop(); + source = null; + current = null; + changed(); +} + +/** Counts the figures using the player: when the last one goes (the reader navigated away), the sound stops with it. */ +let users = 0; + +export function useStopOnLeave() { + useEffect(() => { + users++; + return () => { + if (--users > 0) return; + stopAll(); + synth?.terminate(); + synth = null; + void audio?.close(); + audio = null; + }; + }, []); +} + +/** Swap the timbre of what is playing without going back to the beginning. Silence stays silent. */ +export function restyle(timbre: Timbre) { + const at = current && current.started >= 0 && audio ? audio.currentTime - current.started : 0; + if (last && current) void play(last.key, last.piece, timbre, at); +} + +async function play(key: string, piece: Piece, timbre: Timbre, offset = 0) { + // Changing the timbre of what is already playing keeps the old sound going until the new one is ready, so the + // music does not stop for the second or so the synthesiser takes. Anything else stops at once. + const swap = current?.key === key && !!source; + const previous = source; + if (!swap) stopAll(); + last = { key, piece }; + audio ??= new AudioContext(); + if (audio.state === "suspended") await audio.resume(); + const id = ++nextId, w = worker(); + current = { key, started: -1, seconds: 0 }; + changed(); + const reply = await new Promise>((resolve) => { + const on = ({ data }: MessageEvent) => { if (data.type === "audio" && data.id === id) { w.removeEventListener("message", on); resolve(data); } }; + w.addEventListener("message", on); + w.postMessage({ type: "render", id, piece, timbre } satisfies Request); + }); + if (current?.key !== key || id !== nextId || !audio) return; // something else was asked for meanwhile + const buffer = audio.createBuffer(2, reply.left.length, 44100); + buffer.copyToChannel(reply.left as Float32Array, 0); + buffer.copyToChannel(reply.right as Float32Array, 1); + if (swap && previous) { previous.onended = null; previous.stop(); if (source === previous) source = null; } + const node = audio.createBufferSource(); + node.buffer = buffer; + node.connect(audio.destination); + node.onended = () => { if (source === node) { source = null; current = null; changed(); } }; + node.start(0, Math.max(0, Math.min(offset, buffer.duration - 0.05))); + source = node; + current = { key, started: audio.currentTime - offset, seconds: reply.seconds }; + // The buffer carries three seconds of reverb after the last note. Count the piece as played when the music ends + // (plus a moment), or Play stays "Stop" for several silent seconds. + if (endTimer) clearTimeout(endTimer); + endTimer = setTimeout(() => { endTimer = null; current = null; changed(); }, Math.max(0, reply.seconds + 0.8 - offset) * 1000); + changed(); +} + +/** Which key is playing (or being prepared), and a function for how far through it is (0–1, or null before it starts). */ +export function usePlayer() { + const [, force] = useState(0); + useEffect(() => { + const f = () => force((n) => n + 1); + listeners.add(f); + return () => { listeners.delete(f); }; + }, []); + const progress = useCallback(() => (current && current.started >= 0 && audio ? Math.min(1, (audio.currentTime - current.started) / current.seconds) : null), []); + return { playing: current?.key ?? null, play, stop: stopAll, restyle, progress }; +} + +/** Moves a playhead while `key` plays: calls `draw` every frame with the fraction played, and with null when it stops. */ +export function usePlayhead(key: string, draw: (fraction: number | null) => void) { + const { playing, progress } = usePlayer(); + const drawRef = useRef(draw); + useEffect(() => { drawRef.current = draw; }); + useEffect(() => { + if (playing !== key) { drawRef.current(null); return; } + let frame = 0; + const tick = () => { drawRef.current(progress()); frame = requestAnimationFrame(tick); }; + frame = requestAnimationFrame(tick); + return () => cancelAnimationFrame(frame); + }, [playing, key, progress]); +} diff --git a/content/posts/music-ai/components/protocol.ts b/content/posts/music-ai/components/protocol.ts new file mode 100644 index 0000000..bb24c74 --- /dev/null +++ b/content/posts/music-ai/components/protocol.ts @@ -0,0 +1,43 @@ +import type { Piece, Report } from "./music"; +import type { Timbre } from "./synth"; + +/** One candidate in the blind test: who wrote it is sent with it, and the page keeps it hidden until the reader answers. */ +export interface BlindItem { who: "bach" | "ai" | "rules"; piece: Piece; report: Report } + +export interface Snapshot { + step: number; + seconds: number; + val: number; + piece: Piece; + report: Report; + /** the longest stretch, in eighth notes, found note for note in one training chorale */ + copied: number; +} + +/** Both the judge and the blind test talk to one worker, so every message carries who asked: a reply meant for one figure must not reset the other. */ +export type Request = { from?: number } & RequestBody; + +type RequestBody = + | { type: "train"; seed: number; steps: number } + | { type: "stop" } + | { type: "render"; id: number; piece: Piece; timbre: Timbre } + | { type: "judge" } + | { type: "pair"; seed: number } + | { type: "choose"; winner: 0 | 1 } + | { type: "sample" } + | { type: "blind" }; + +export type Reply = { from?: number } & ReplyBody; + +type ReplyBody = + | { type: "loading" } + | { type: "failed" } + | { type: "progress"; step: number; steps: number; loss: number; seconds: number } + | { type: "snapshot"; snapshot: Snapshot } + | { type: "done" } + | { type: "audio"; id: number; left: Float32Array; right: Float32Array; seconds: number } + | { type: "ready" } + | { type: "pair"; pieces: [Piece, Piece] } + | { type: "judged"; choices: number; now: Report; before: Report } + | { type: "blind"; items: BlindItem[] } + | { type: "sample"; piece: Piece }; diff --git a/content/posts/music-ai/components/shared-worker.ts b/content/posts/music-ai/components/shared-worker.ts new file mode 100644 index 0000000..9bf9d7e --- /dev/null +++ b/content/posts/music-ai/components/shared-worker.ts @@ -0,0 +1,32 @@ +"use client"; + +import { useEffect, useRef, useState } from "react"; +import type { Reply, Request } from "./protocol"; +import { createMusicWorker } from "./worker-factory"; + +/* + * Figures 02 and 03 use the same trained composer (judge.bin, 223 KB) and the same chorales, so they share one worker + * and one copy of both instead of fetching and parsing them twice. Training keeps a worker of its own: it holds the + * thread for minutes at a time. + */ +let shared: Worker | null = null, users = 0, ids = 0; + +/** The shared worker, with `onReply` attached for as long as the component lives. */ +export function useSharedWorker(onReply: (reply: Reply) => void): (request: Request) => void { + // One identity per figure, fixed for its lifetime. + const [id] = useState(() => ++ids); + const handler = useRef(onReply); + useEffect(() => { handler.current = onReply; }); + useEffect(() => { + users++; + shared ??= createMusicWorker(); + const listener = ({ data }: MessageEvent) => { if (data.from === id) handler.current(data); }; + shared.addEventListener("message", listener); + const worker = shared; + return () => { + worker.removeEventListener("message", listener); + if (--users === 0) { worker.terminate(); shared = null; } + }; + }, [id]); + return (request: Request) => shared?.postMessage({ ...request, from: id }); +} diff --git a/content/posts/music-ai/components/synth.ts b/content/posts/music-ai/components/synth.ts new file mode 100644 index 0000000..8fc4265 --- /dev/null +++ b/content/posts/music-ai/components/synth.ts @@ -0,0 +1,125 @@ +import type { Piece } from "./music"; + +/* + * The instrument the model's scores are played on, computed sample by sample (no Web Audio oscillators or effects): + * three soft voices, a damped stereo reverb laid out like Freeverb, and a gentle low-pass on the mix. The first + * version was an organ of five harmonics with a fast attack and an undamped echo; Paul found it sharp, and it had + * 14 dB more energy in the 1–4 kHz band the ear is most sensitive to (docs/research/music-ai, "Timbre"). + */ +export type Timbre = "pad" | "epiano" | "flute"; +export const TIMBRES: Timbre[] = ["pad", "epiano", "flute"]; +export const SAMPLE_RATE = 44100; +export const BPM = 60; +/** Seconds per step: an eighth note at 60 beats a minute. */ +export const STEP_SECONDS = 60 / BPM / 2; + +export interface Audio { left: Float32Array; right: Float32Array; seconds: number } + +function noise(seed: number) { + let a = seed | 0; + return () => { a = (Math.imul(a, 1664525) + 1013904223) | 0; return a / 2147483648; }; +} + +/** One note into the two channels. Lower voices sit a little louder; the four are spread across the stereo field. */ +function note(L: Float32Array, R: Float32Array, timbre: Timbre, midi: number, start: number, dur: number, voice: number) { + const f = 440 * 2 ** ((midi - 69) / 12); + const level = [0.5, 0.55, 0.6, 0.75][voice], pan = [0.35, 0.6, 0.4, 0.5][voice]; + const gl = Math.cos((pan * Math.PI) / 2), gr = Math.sin((pan * Math.PI) / 2); + const rnd = noise(midi * 131 + Math.round(start * 1000)); + const attack = timbre === "pad" ? 0.35 : timbre === "flute" ? 0.12 : 0.008; + const release = timbre === "pad" ? 1.4 : timbre === "flute" ? 0.35 : 1.2; + const total = Math.round((dur + release) * SAMPLE_RATE), a0 = Math.round(start * SAMPLE_RATE); + const w = 2 * Math.PI * f, detune = 2 ** (4 / 1200); + let breath = 0; + for (let i = 0; i < total && a0 + i < L.length; i++) { + const t = i / SAMPLE_RATE; + const sustain = t < dur ? 1 : Math.exp(-(t - dur) / (release / 4)); + let env: number, x: number; + if (timbre === "pad") { + // The fundamental and a soft octave, each doubled 4 cents apart: the slow beating between them is the warmth. + env = Math.min(1, t / attack) ** 2 * sustain; + x = 0.5 * (Math.sin(w * t) + Math.sin(w * detune * t)) + 0.12 * (Math.sin(2 * w * t) + Math.sin((2 * w * t) / detune)) + 0.03 * Math.sin(3 * w * t); + } else if (timbre === "flute") { + // A sine with a little second harmonic, a breath of noise on the attack, and a vibrato that fades in. + env = Math.min(1, t / attack) * sustain; + const vib = 1 + 0.003 * Math.sin(2 * Math.PI * 4.8 * t) * Math.min(1, t / 0.6); + breath = 0.97 * breath + 0.03 * rnd(); + x = Math.sin(w * vib * t) + 0.08 * Math.sin(2 * w * vib * t) + 3 * breath * Math.exp(-t / 0.08); + } else { + // An electric piano by two-operator FM: the modulation index decays, so the tone rounds off as it sounds. + env = Math.min(1, t / attack) * Math.exp(-t / 2.2) * sustain; + x = Math.sin(w * t + 1.1 * Math.exp(-t / 0.35) * Math.sin(w * t)); + } + const s = 0.16 * level * env * x; + L[a0 + i] += s * gl; + R[a0 + i] += s * gr; + } +} + +/** Freeverb's layout: eight feedback combs with a low-pass inside the loop (the tail darkens as it decays), four all-passes. */ +function reverb(L: Float32Array, R: Float32Array, room = 0.84, damp = 0.45, wet = 0.32) { + const combs = [1116, 1188, 1277, 1356, 1422, 1491, 1557, 1617], passes = [556, 441, 341, 225]; + const side = (input: Float32Array, spread: number) => { + const out = new Float32Array(input.length); + for (const d0 of combs) { + const d = d0 + spread, buf = new Float32Array(d); + let idx = 0, store = 0; + for (let i = 0; i < input.length; i++) { + const y = buf[idx]; + store = y * (1 - damp) + store * damp; + buf[idx] = input[i] * 0.015 + store * room; + out[i] += y; + if (++idx === d) idx = 0; + } + } + for (const d0 of passes) { + const d = d0 + spread, buf = new Float32Array(d); + let idx = 0; + for (let i = 0; i < out.length; i++) { + const b = buf[idx], x = out[i]; + out[i] = b - x; + buf[idx] = x + b * 0.5; + if (++idx === d) idx = 0; + } + } + return out; + }; + const wl = side(L, 0), wr = side(R, 23); + for (let i = 0; i < L.length; i++) { + L[i] = L[i] * (1 - wet) + wl[i] * wet * 3; + R[i] = R[i] * (1 - wet) + wr[i] * wet * 3; + } +} + +/** Two one-pole low-passes: 12 dB per octave above `hz`. */ +function lowpass(x: Float32Array, hz: number) { + const a = Math.exp((-2 * Math.PI * hz) / SAMPLE_RATE); + for (let pass = 0; pass < 2; pass++) { + let y = 0; + for (let i = 0; i < x.length; i++) x[i] = y = (1 - a) * x[i] + a * y; + } +} + +export function render({ pitches, onset }: Piece, timbre: Timbre): Audio { + const seconds = pitches.length * STEP_SECONDS; + const n = Math.ceil((seconds + 3) * SAMPLE_RATE); + const L = new Float32Array(n), R = new Float32Array(n); + for (let v = 0; v < 4; v++) { + for (let t = 0; t < pitches.length; ) { + const p = pitches[t][v]; + if (p < 0) { t++; continue; } + let len = 1; + while (t + len < pitches.length && pitches[t + len][v] === p && !onset[t + len][v]) len++; + note(L, R, timbre, p, t * STEP_SECONDS, len * STEP_SECONDS, v); + t += len; + } + } + reverb(L, R); + lowpass(L, 3200); + lowpass(R, 3200); + let peak = 0; + for (let i = 0; i < n; i++) peak = Math.max(peak, Math.abs(L[i]), Math.abs(R[i])); + const g = 0.7 / (peak || 1); + for (let i = 0; i < n; i++) { L[i] *= g; R[i] *= g; } + return { left: L, right: R, seconds }; +} diff --git a/content/posts/music-ai/components/worker-factory.ts b/content/posts/music-ai/components/worker-factory.ts new file mode 100644 index 0000000..2abc237 --- /dev/null +++ b/content/posts/music-ai/components/worker-factory.ts @@ -0,0 +1,6 @@ +/** + * The one place a music worker is made. Turbopack wires `new Worker(new URL(…))` at build time, and three separate + * call sites for the same worker file left the production bundle with "Missing worker bootstrap config": in + * development every figure worked, in a production build none of them did. + */ +export const createMusicWorker = () => new Worker(new URL("./music.worker.ts", import.meta.url), { type: "module" }); diff --git a/content/posts/music-ai/en.mdx b/content/posts/music-ai/en.mdx new file mode 100644 index 0000000..86b533d --- /dev/null +++ b/content/posts/music-ai/en.mdx @@ -0,0 +1,165 @@ +--- +title: "Train your own AI composer in three minutes: from noodling to calm" +description: "Train a small Transformer to write four-part chorales from scratch, in your browser, and listen to it turn noodling into music. Then judge it: twenty clicks make it calmer — and show you how it games the judge." +seoTitle: "Train an AI composer in three minutes" +seoDescription: "Train a Transformer from scratch in your browser until it writes four-part chorales, hear it learn, tune it to your taste, and watch it game the judge." +date: 2026-09-23 +tags: [generative, llm, from-scratch] +no: 14 +interactive: true +--- + +import { BlindLab, ComposerLab, JudgeLab } from "./components"; + +The composer below knows nothing at all. Press train and it reads 333 of Bach's four-part chorales in your browser, +improvising a piece for you at six moments as it learns, starting with the noodling of step 0. + + + + + +First, what this is not. It is not Suno: it does not sing and it does not produce audio. It writes a score — which +voice, which note, how long — and a synthesiser written from scratch plays it. The model has sixty-five thousand +parameters and its weights are 260 KB, about a tenth of a three-minute MP3. + +## Turning music into words + +A chorale is cut into a grid of eighth notes. Each step has four voices, soprano down to bass, and each voice-step is +one token: a new note is its pitch, a held note is `HOLD`, silence is `REST`. So what the model writes looks like +this: soprano, alto, tenor, bass, soprano, alto, tenor, bass, and so on. + +```ts title="content/posts/music-ai/components/music.ts" +export const HOLD = HI - LO + 1, REST = HOLD + 1, VOCAB = REST + 1; // 54 pitches + 2 = 56 words +``` + +From here it is exactly [№ 004's Transformer](/en/posts/transformer-from-scratch), the same code: look at the words +so far, guess the next one. There it learned to write numbers backwards; here it learns Bach. + +There is a trap in this: the only way the model knows whose turn it is, is *which position* a word is in. So every +window of the last 64 words has to start on the soprano, or the model reads every voice as its +neighbour.I walked into it the first time I let it compose: the bass climbed above the tenor and the voices +crossed six steps out of ten. Training was never affected, only composing. + +## What it learns first + +Train it a few times in figure 1 and the order barely changes. Here are two runs of figure 1's own recipe, five pieces measured at each +moment, averaged. The validation error is how well it guesses the next note, lower being better; parallel fifths are +four-part writing's cardinal sin, two voices a fifth apart moving the same way:The table was measured in +node, running the same code and the same 2,400 steps as the browser. On my M4 Pro a step takes about 69 ms in Chrome, +so the whole run is under three minutes. Scripts and output are in `docs/research/music-ai/`. + +| step | val. error | in key | voices crossing | held notes | parallel fifths (per 100 steps) | +| --- | --- | --- | --- | --- | --- | +| 0 | 4.51 | 82% | 87% | 0.5% | 3.9 | +| 100 | 2.01 | 99% | 36% | 78% | 1.1 | +| 250 | 1.60 | 99.7% | 1.9% | 50% | 9.8 | +| 500 | 1.39 | 99.6% | 1.1% | 55% | 10.0 | +| 1000 | 1.21 | 98.7% | 1.8% | 54% | 12.0 | +| 2000 | 1.07 | 99.2% | 1.3% | 53% | 8.9 | +| 2400 | 1.05 | 99.5% | 0% | 65% | 7.2 | +| Bach himself | | 97.7% | 2.6% | 45% | 0.16 | + +1. **Holding notes comes first.** At step 100, 78% of the notes are held: it has found that "carry on with the + previous note" is the best guess going, so it drones. +2. **Then the voices learn their ranges.** Crossings fall from 87% to 2% by step 250, and stay cleaner than Bach's. +3. **The key settles; the rhythm does not quite.** In-key notes hold at 99%, and held notes come back to about 50%, a little above Bach's 45% — then rise to 65% by the end of the run, which is slower than Bach. +4. **Parallel fifths never arrive.** Bach breaks the rule 0.16 times per hundred steps; the model spends the whole + run between 7 and 12. + +In my browser the validation error passes 1.46 inside a minute, which is the best a 5-gram — look at the four words +before, count how often each word follows — manages on the same data. The 5-gram composes far worse: it cannot see +past four words, so a voice that has held a note for one step has forgotten which note it is on, and the voices cross +72% of the time. + +## Is it composing, or remembering? + +The "longest copy" in figure 1 takes every step it wrote and looks for the longest run that appears, note for note, in +one of the 333 training chorales. At 1,000 steps it is 3 to 6 steps (a step is an eighth note, eight of them a bar) — +less than a bar. A chorale runs about 107 steps. + +So it is mostly composing. But the longer it trains, the longer the pieces it remembers: at 2,400 steps the longest +copy is 5 to 10 steps — and one piece in ten copied 32 steps, four whole bars, so it can hold a passage verbatim. + +## You be the judge + +The "calm" in the title is not something it knows. Bach's chorales are neither slow nor quiet; slowing down and +quietening are what you are about to tune into it. + +The trained model writes like Bach, which is not necessarily what you want to hear. Your turn: it writes two pieces, +you pick one, and it moves a little towards the one you picked and away from the other. Models like ChatGPT are tuned +on human preferences (RLHF); this is DPO, a simpler version that came later, with you as the judge. + + + + + +Keep picking the slower, quieter one and it slows down. I tried it with an automatic judge, set up exactly like this +figure: always pick the piece with more held notes, and after 20 picks held notes go from 48% to 64%; after 40 they +reach 78%, the chords lose half their variety, and it is on its way to a drone. + +### It finds the hole in the judge + +Then I had the automatic judge mark parallel fifths instead: always pick the piece that breaks the rule less. After 40 +picks the rate barely moves, 12.3 per hundred steps down to 10.9 — but held notes are up from 48% to 58% and the +chords have lost a quarter of their variety. By 80 picks the rate is 8.0, held notes 63%, and distinct chords are down +from 45 to 29. + +It really has brought the mistakes down, and the way it did it is to **move less**: a voice that stays put cannot walk +into a parallel fifth. + +This is reward hacking, one of the hardest problems in training large language models: the judge measures one thing, +the model optimises "keep the judge happy", and any gap between the two is a gap it will find. The textbook has two +answers: + +- **A score that cannot be gamed that way.** Count the mistakes *per step on which voices move*, so standing still + buys nothing. +- **A shorter leash.** DPO has a β that decides how far the model may drift from the model it started as. + +At the strength this figure uses, the two answers buy very little: over the same 80 picks the hack-proof judge gets +parallel fifths to 7.0 against the gameable judge's 8.0, and both end up with 63% held notes. Both still learn to move +less first. + +The difference shows when you push harder. Updating once per four picks, at three times the learning rate: the +gameable judge gets to 5.9, at the price of 73% held notes and half the chords; the hack-proof judge gets to 7.0 with +52% held notes and 41 chords, near where it started. **The harder you optimise, the more gaming pays** — which is +exactly the position anyone tuning a large model is in. The figure above is deliberately gentle, so that twenty clicks +do not wreck it. + +## Can it beat five lines of maths? + +Its last opponent uses no AI at all and fits in five lines: a 1/f random walk over a C pentatonic scale, above a fixed +C–Am–F–G, a bar each. + +```ts title="simplified from content/posts/music-ai/components/music.ts" +const scale = [60, 62, 64, 67, 69, 72, 74, 76, 79, 81]; // C pentatonic +const chords = [[48, 55, 64], [45, 57, 64], [41, 53, 60], [43, 55, 62]]; +for (let d = 0; d < 4; d++) if (t % (2 << d) === 0) dice[d] = rng(); // four dice +const soprano = scale[Math.floor(mean(dice) * scale.length)]; // a 1/f melody +voices.push([soprano, ...chords[Math.floor(t / 8) % 4]]); // a chord a bar +``` + +It never plays a wrong note and its voices never cross. + +But it breaks the parallel-fifths rule 19.8 times per hundred steps, over a hundred times Bach's rate, and it has half +his harmonic variety. Four voices shifting together *are* a run of parallel fifths. + +So: can you hear it? Three pieces below, four bars each — one by Bach (from the 37 chorales the model never saw), one +by the model after 1,000 steps, one by the five lines of maths, shuffled every time. + + + + + +Once you have the answers, the two numbers under each card tell you that the three differ by more than how they sound. + +## The data and the sound + +- The data is Craig Sapp's edition of [Bach's 370 four-part chorales](https://github.com/craigsapp/bach-370-chorales), + licensed CC BY-NC-SA 4.0; the tokens and the trained model this page uses carry the same licence. Every chorale is + moved to C major or A minor and sampled onto a grid of eighth notes: 333 to train on, 37 kept back to check against. +- The synthesiser is written from scratch, with none of the browser's own instruments or effects: the warm pad is a + fundamental and a soft octave, each doubled 4 cents apart, with a slow attack; the reverb follows Freeverb's layout, + so the tail darkens as it fades. The first version was an organ of five harmonics, and the first thing I heard was + how sharp it was: 6 to 16 dB more in the 1–4 kHz band than this one. +- Everything here that measures "like Bach" is a proxy: in key, voices crossing, parallel fifths, held notes, chord + variety. They can say whether a piece follows the rules. They cannot say whether it is any good. diff --git a/content/posts/music-ai/zh.mdx b/content/posts/music-ai/zh.mdx new file mode 100644 index 0000000..473d97d --- /dev/null +++ b/content/posts/music-ai/zh.mdx @@ -0,0 +1,117 @@ +--- +title: 三分鐘訓練你的 AI 作曲家:聽它從亂彈到療癒 +description: 在瀏覽器裡從零訓練一個會寫四部合唱的小 Transformer,邊練邊聽它從亂彈變成音樂。然後你當評審,點二十下把它調得更放鬆;也看它怎麼找到評審的漏洞。 +seoTitle: 三分鐘訓練你的 AI 作曲家 +seoDescription: 在瀏覽器裡從零訓練一個會寫四部合唱的小 Transformer,邊練邊聽它進步,再當評審把它調得更放鬆,看它怎麼鑽評審的漏洞。 +date: 2026-09-23 +tags: [generative, llm, from-scratch] +no: 14 +interactive: true +--- + +import { BlindLab, ComposerLab, JudgeLab } from "./components"; + +下面這個作曲家現在什麼都不會。按下訓練,它會在你的瀏覽器裡讀巴赫的 333 首四部合唱,一邊學,一邊在六個時刻即興一段給你聽,從第 0 步的亂彈開始。 + + + + + +先說這不是什麼。這不是 Suno:它不會唱歌,也不產生聲音檔。它寫的是樂譜,「哪個聲部、什麼音、拖多長」,再由一個從零寫的合成器演奏出來。模型只有六萬五千個參數,權重檔 260 KB,大約是一首三分鐘 MP3 的十分之一。 + +## 音樂怎麼變成字 + +一首合唱被切成一格一格的八分音符。每一格有四個聲部,從女高音到男低音,每個聲部是一個 token:新的音就是它的音高,延續上一個音是 `HOLD`,休止是 `REST`。所以模型寫的東西長這樣:女高、女低、男高、男低,女高、女低、男高、男低…… + +```ts title="content/posts/music-ai/components/music.ts" +export const HOLD = HI - LO + 1, REST = HOLD + 1, VOCAB = REST + 1; // 54 個音高 + 2 = 56 個字 +``` + +接下來就和 [№ 004 的 Transformer](/zh/posts/transformer-from-scratch) 一模一樣,連程式都是同一份:看前面的字,猜下一個字。那篇它學的是把數字倒過來寫,這篇學的是巴赫。 + +這裡有個陷阱:模型只靠「這是第幾個字」分辨聲部,所以每次取最近的 64 個字,開頭都必須剛好是女高音,否則它會把每個聲部都當成隔壁那一個。我第一次讓它作曲時就踩到了:男低音一路爬到男高音上面,六成的時間聲部都交錯。訓練本身沒受影響,只有作曲時會錯。 + +## 它先學會什麼 + +在圖 1 多練幾次,會發現它每次學會的順序都差不多。我用圖 1 的配方練了兩次,每個時刻寫五段、量同樣的東西,下面是兩次的平均。驗證誤差是它猜下一個音的平均誤差,越低越好;平行五度是四部寫作的頭號禁忌,指兩個聲部隔著純五度一起往同一個方向走:表是在 node 裡跑的,和瀏覽器裡是同一份程式、同樣 2400 步。在我的 M4 Pro 上,Chrome 裡一步約 69 毫秒,整輪不到三分鐘。量法和輸出在 `docs/research/music-ai/`。 + +| 步數 | 驗證誤差 | 在調上 | 聲部交錯 | 拖長的音 | 平行五度(每百格) | +| --- | --- | --- | --- | --- | --- | +| 0 | 4.51 | 82% | 87% | 0.5% | 3.9 | +| 100 | 2.01 | 99% | 36% | 78% | 1.1 | +| 250 | 1.60 | 99.7% | 1.9% | 50% | 9.8 | +| 500 | 1.39 | 99.6% | 1.1% | 55% | 10.0 | +| 1000 | 1.21 | 98.7% | 1.8% | 54% | 12.0 | +| 2000 | 1.07 | 99.2% | 1.3% | 53% | 8.9 | +| 2400 | 1.05 | 99.5% | 0% | 65% | 7.2 | +| 巴赫本人 | | 97.7% | 2.6% | 45% | 0.16 | + +1. **先學會拖長音。** 第 100 步有 78% 的音都是拖著的:它發現「繼續上一個音」最常猜對,於是幾乎只會持續音。 +2. **再學會各守各的音域。** 聲部交錯從 87% 掉到 250 步的 2%,之後就一直比巴赫還乾淨。 +3. **調性穩下來,節奏沒有完全穩。** 調內音一路維持在 99%;拖長的音回到 50% 上下,比巴赫的 45% 多一點,到訓練尾聲又升到 65%,聽起來會比巴赫慢。 +4. **平行五度學不會。** 巴赫一百格才犯 0.16 次,它整輪都在 7 到 12 次之間晃。 + +在我的瀏覽器裡,驗證誤差不到一分鐘就低於 1.46,那是一個 5-gram(只看前面四個字、把次數數起來)在同一份資料上的最好成績。5-gram 的作曲更慘:它看不到四個字以前,一個聲部拖了一格,它就忘了自己在哪個音上,72% 的時間聲部交錯。 + +## 它在作曲,還是在背譜? + +圖 1 裡的「最長抄襲」,是拿它寫的每一格去比對 333 首訓練曲目,找出和其中一首一模一樣、連續最長的一段。練到 1000 步時是 3 到 6 格(一格是一個八分音符,八格一小節),不到一小節;一首聖詠平均有 107 格。 + +所以它多半是在作曲。但練得越久,記住的片段越長:2400 步時變成 5 到 10 格。而且十段裡出現過一段整整抄了 32 格、也就是四小節的例外——它確實有能力把整段譜背下來。 + +## 你當評審 + +標題說的「療癒」不是它本來就會的。巴赫的合唱不慢也不安靜;慢下來、安靜下來,是接下來你自己調出來的。 + +練好的模型寫得像巴赫,但不一定是你想聽的。現在換你:它一次寫兩段,你選一段,它就往你選的那段靠近一點,離另一段遠一點。ChatGPT 這類模型就是用人的偏好調教出來的(RLHF);這裡用的是它後來一個比較簡單的版本 DPO,只是評審換成你。 + + + + + +你一直選比較慢、比較安靜的那段,它就會慢下來。我用一個自動評審試過,設定和這張圖一模一樣:每次選「拖長的音比較多」的那段,選 20 次,拖長音從 48% 升到 64%;選 40 次變成 78%,和弦的種類掉了一半,開始只剩持續音。 + +### 它會找評審的漏洞 + +然後我讓自動評審改考平行五度:每次選犯得比較少的那段。選 40 次,平行五度從每百格 12.3 次只降到 10.9 次,幾乎沒動;拖長音倒是從 48% 升到 58%,和弦的種類少了四分之一。選到 80 次,平行五度降到 8.0,拖長音 63%、和弦種類從 45 掉到 29。 + +它確實把犯規壓下來了,但用的辦法是**少動**:不動的聲部不可能走出平行五度。 + +這叫 reward hacking,是訓練大型語言模型時最頭痛的問題之一:評審量的是一件事,模型最佳化的是「讓評審滿意」,兩者只要有一點縫,它就會找到。書上的修法有兩招: + +- **換一個鑽不了漏洞的評分。** 改算「每次有聲部移動時,犯了幾次」,不動就拿不到好處。 +- **把繩子拉緊。** DPO 裡有一個 β,決定模型可以離原本的自己多遠。 + +在這張圖的力道下,這兩招只買到一點點:同樣選 80 次,防鑽的評審把平行五度壓到 7.0,可鑽的是 8.0,而兩者的拖長音都是 63%。它們都還是先學會少動。 + +差別要推得更用力才看得出來。我把更新改成累積四次一起做、學習率調成三倍:可鑽的評審把平行五度壓到 5.9,代價是拖長音衝到 73%、和弦種類只剩一半;防鑽的評審壓到 7.0,拖長音 52%、和弦種類 41,和原本差不多。**越用力優化,鑽漏洞越划算**,這正是大模型調教時的處境。上面那張圖故意留在溫和的力道,免得你點個二十下它就跑偏。 + +## 它贏得過五行數學嗎? + +最後是一個不用任何 AI 的對手,五行就寫得完:五聲音階(C D E G A)上的一條 1/f 隨機旋律,底下墊著固定的 C–Am–F–G 四個和弦。 + +```ts title="簡化自 content/posts/music-ai/components/music.ts" +const scale = [60, 62, 64, 67, 69, 72, 74, 76, 79, 81]; // C 五聲音階 +const chords = [[48, 55, 64], [45, 57, 64], [41, 53, 60], [43, 55, 62]]; +for (let d = 0; d < 4; d++) if (t % (2 << d) === 0) dice[d] = rng(); // 四顆骰子 +const soprano = scale[Math.floor(mean(dice) * scale.length)]; // 1/f 旋律 +voices.push([soprano, ...chords[Math.floor(t / 8) % 4]]); // 每小節換和弦 +``` + +它永遠不會彈錯音,聲部也永遠不會交錯。 + +但它的平行五度是每百格 19.8 次,巴赫的一百多倍,和弦種類只有巴赫的一半。四個和弦一起平移,就是一連串的平行五度。 + +所以,聽得出來嗎?下面三段各四小節,一段是巴赫本人(模型沒看過的那 37 首裡的一段)、一段是練好 1000 步的 AI、一段是五行數學,順序每次都重新洗過。 + + + + + +對完答案,卡片下面那兩個數字會告訴你,三段的差別不只在聽感上。 + +## 資料和聲音 + +- 資料是 Craig Sapp 整理的[巴赫 370 首四部聖詠](https://github.com/craigsapp/bach-370-chorales),授權是 CC BY-NC-SA 4.0;這一頁用到的樂譜資料和練好的模型,也以同樣的授權提供。每一首都移到 C 大調或 a 小調,切成八分音符的格子,333 首拿來訓練,37 首留著驗證。 +- 合成器是自己寫的,沒有用瀏覽器內建的樂器或效果:溫暖墊音是基音加一個輕輕的八度,各疊兩個相差 4 音分的音,慢慢起音;殘響照 Freeverb 的結構,尾音會越來越暗。第一版是五個泛音的風琴,我第一次聽就覺得太尖:1–4 kHz 比現在多了 6 到 16 dB。 +- 所有的「像不像巴赫」都是代理指標:在不在調上、聲部有沒有交錯、平行五度、拖長音的比例、和弦的種類。它們說得出一段音樂「守不守規矩」,說不出好不好聽。 diff --git a/docs/COMMITS.md b/docs/COMMITS.md index 70a0753..b8f4d7b 100644 --- a/docs/COMMITS.md +++ b/docs/COMMITS.md @@ -40,7 +40,8 @@ The area, lowercase, `[a-z0-9._/-]`. Optional only when the change really belong (`docs: production is blog.psheon.me`). - **An article**: its slug or the short name already in the log: `cnn`, `transformer`, `diffusion`, `flappy`, `lite3`, - `slam`, `city`, `scheduler`, `light`, `playground` (the figure), `light-playground` (the article's text). + `slam`, `city`, `scheduler`, `light`, `playground` (the figure), `light-playground` (the article's text), + `head-camera`, `music-ai`. - **Shared code**: `rt` (`lib/rt`), `ml`, `controls` (`components/lab`), `ui`, `site`, `home`, `header`, `search`, `post`. - **Everything else**: `e2e`, `deps`, `research` (`docs/research`), `repo`, `release` (release PR titles only). diff --git a/docs/DESIGN.md b/docs/DESIGN.md index bd4eeb7..27ea24e 100644 --- a/docs/DESIGN.md +++ b/docs/DESIGN.md @@ -130,6 +130,23 @@ Rules inside an instrument: - A stage whose content depends on colour (coloured point clouds) stays dark in both themes: wrap it in `className="dark bg-[#070918]"`. +### Instruments that make sound + +Article 014 is the first one that plays music, and its rules hold for the next one: + +- **One player for the page.** `content/posts/music-ai/components/player.ts` owns the single `AudioContext` and the + synthesiser worker; a figure asks it to play and whatever was playing stops. The context is created on the reader's + first press of Play, because a browser will not make sound before a gesture. +- **Sound stops when the figures go.** The last figure to unmount stops playback, terminates the worker and closes + the context, or the music plays on over the next article. +- **A piece ends when the music does.** The rendered buffer carries a reverb tail; treating the end of the buffer as + the end of playback left Play saying "Stop" over several seconds of silence. +- **Changing the instrument keeps the position** and the old sound plays until the new one is ready. +- **Nothing is heard by CI**, so the sound is made by pure functions with tests, and the E2E tests check everything + around it. A figure that only makes sound is not accessible: the piano roll, the numbers and the captions carry the + same information. +- Audio and models are fetched when the reader asks for them, never with the article (`e2e/music-ai.spec.ts` checks). + ### Controls for steering something by hand One thumb stick for the whole site: `components/lab/stick.tsx` (drag it, or focus it and use the arrows / W A S D; it diff --git a/docs/HANDOFF.md b/docs/HANDOFF.md index dd0ae2f..0e68450 100644 --- a/docs/HANDOFF.md +++ b/docs/HANDOFF.md @@ -1,4 +1,4 @@ -# Handoff — main session, last revised 2026-09-22 +# Handoff — main session, last revised 2026-09-23 For whoever picks this up next. Read this, then the memory files under `~/.claude/projects/-Users-paul-jiang-Desktop-Paul/memory/` (they are loaded automatically, this file is not), then @@ -14,16 +14,34 @@ For whoever picks this up next. Read this, then the memory files under | Production | (since 2026-09-21; DNS on Cloudflare, CNAME to Vercel, DNS only). Vercel project `paul-notebook`, deploys `main`. `paul-notebook.vercel.app` redirects 308 to it, path kept. Production env: `NEXT_PUBLIC_SITE_URL=https://blog.psheon.me` | | Dev server | `pnpm dev` on :3000 | | Other worktrees | None since 2026-09-22: `feat/city-of-agents`, `feat/sche` and `feat/vla` are merged and deleted. One writer per checkout | -| Tests | 323 unit tests (2 skipped) and 206 E2E runs (two projects: desktop, mobile), plus axe on every article. CI runs all of it on every push | +| Tests | 323 unit tests (2 skipped) and 218 E2E runs (two projects: desktop, mobile), plus axe on every article. CI runs all of it on every push | Published, in both languages: 001 CNN, 002 Flappy Bird, 003 trading agent, 004 Transformer, 005 HydraNet, 006 Lite3, 007 point-cloud diffusion, 008 2D SLAM, 009 city of agents, 010 a task scheduler from scratch (published 2026-09-21; built on `feat/sche` by another session), 011 and 012 the light series (PR #14), 013 `head-camera` (PR #15, merged from -`feat/vla`). All thirteen had their copy polished with Paul on 2026-09-22, item by item. The earlier PCB-flip +`feat/vla`), 014 `music-ai` (2026-09-23). Articles 001–013 had their copy polished with Paul on 2026-09-22, item by +item, and 014 the same way as it was written. The earlier PCB-flip VLA draft (№ 014) was dropped the same day, Paul found it dull; its simulation (arm, rasteriser, world) lives on as `content/posts/head-camera/components/sim`, which 013 imports, with its tests in `tests/head-camera/`; its notes stay in `docs/research/pcb-flip-vla/`. A draft shows only in `next dev`, with a mark in the page's language ("草稿" / "DRAFT"). +### № 014, the AI composer (2026-09-23) + +A small Transformer (65k parameters, `lib/ml` again) learns Bach's four-part chorales in the reader's browser in about +three minutes, and a synthesiser written sample by sample plays what it writes. Three figures: train and listen, +tune it with your own picks (DPO, and it games the judge), and a blind test against Bach and five lines of maths. + +- `docs/research/music-ai/RESULTS.md` holds every number, both recipes (the research schedule and the page's), the + timbre work and the measurements that were thrown away. Copies of the research scripts sit beside it. +- The data is **CC BY-NC-SA 4.0** (Craig Sapp's edition of the chorales), and so are `public/posts/music-ai/*.bin` + and anything else derived from them: attribution is in the README, the article and the licence note. +- `scripts/music/convert.py` and `scripts/music/pack.ts` rebuild those two files byte for byte from the corpus clone + and a checkpoint, both of which live outside git in `Desktop/Paul/music-work/` (with every run and the MP3 packs). +- Sound has its own rules now: DESIGN.md §4 "Instruments that make sound". The player is one `AudioContext` and one + worker for the page; figures 02 and 03 share a second worker, and every message carries which figure asked. +- CI cannot hear: `e2e/music-ai.spec.ts` checks training, tuning, the blind test's secrecy and that nothing heavy is + fetched before the reader asks. + ### The light series (two articles, published 2026-09-22) `light-from-noise` (№ 011, part one: what path tracing is) and `light-playground` (№ 012, part two: a playground you diff --git a/docs/research/music-ai/RESULTS.md b/docs/research/music-ai/RESULTS.md new file mode 100644 index 0000000..4bbe70d --- /dev/null +++ b/docs/research/music-ai/RESULTS.md @@ -0,0 +1,238 @@ +# Music AI spike — lab notebook + +Question: can a reader train a small music model in the browser, in minutes, that makes calm music worth listening +to? And the honest follow-ups: does it compose or copy, and does it beat a few lines of rules? + +Everything here ran in node with `lib/ml` (the code the browser runs), on an M4 Pro. Scripts, data and runs are in +`Desktop/Paul/music-work/` (outside git); copies of the scripts are kept beside this file. + +## Data (2026-09-22) + +- **Source: Craig Sapp's digital edition of 370 Bach chorales**, , + **CC BY-NC-SA 4.0**. Non-commercial use with attribution is fine for this site; anything derived from it that we + publish (the token file, trained weights) must carry the same licence. +- Rejected: `czhuang/JSB-Chorales-dataset` (the one papers use) declares no licence; music21's chorales are + distributed with permission for music21 only; the Nottingham folk set is GPL-3.0 and mostly jigs and reels, not calm. +- `convert.py` (music21) parses the kern files, transposes each chorale to C major or A minor (the smaller shift, at + most a tritone), and samples an eighth-note grid: per step, the MIDI pitch of S A T B, whether it starts there, and + fermatas. All 370 convert. 39,602 steps (107 per chorale on average), 194 major, 176 minor; 331 in 4/4, 38 in 3/4. + Rests are 1.0 % of voice-steps. Spot check: chor001 (G major, shifted +5) opens C3 E4 G4 C5, which is its GG B d g. +- Tokens: one per voice per step, S A T B: a new note is its pitch (MIDI 31–84, 54 ids), a held one HOLD, silence + REST. Vocabulary 56. Split by chorale with a fixed seed: 333 train (143,232 tokens), 37 validation (15,176). + +## Yardsticks (`baseline.ts`) + +None of these says "beautiful". They say whether the texture does what four-part writing does. Reference values are +Bach's own validation chorales. + +| | in key | parallel 5ths/8ves per 100 steps | voice crossings | roughness | distinct chords per 100 steps | +| --- | --- | --- | --- | --- | --- | +| Bach (37 validation chorales) | 97.7 % | 0.16 | 2.6 % | 0.67 | 40 | +| Rules: pentatonic 1/f melody over I–vi–IV–V (5 seeds) | 100 % | 19.8 | 0 % | 0.50 | 19 | +| 5-gram, T 0.7 (3 seeds) | 79.5 % | 2.1 | 72 % | 0.90 | 83 | + +- In key: C major's pitch classes plus F# and G# (A minor's raised 6th and 7th). Roughness: Plomp–Levelt after + Sethares, 6 harmonics at 1/k per note, summed over all partial pairs, averaged per step. +- The rule composer never plays a wrong note and is the smoothest, but moves its block chords in parallel + constantly: 120 times Bach's rate. It also has half the harmonic variety. +- n-gram validation loss (nats per token, add-0.05 smoothing, the context includes which voice is next): + 1-gram 2.133, 2-gram 1.695, 3-gram 1.510, 4-gram 1.472, **5-gram 1.462**, 6-gram 1.595, 8-gram 1.927. + **1.46 is the line the Transformer has to cross.** +- The 5-gram composes badly despite that loss: voices cross on 72 % of steps. Its window is the previous step's four + tokens, and a held note is the token HOLD, so a voice that has held for one step has forgotten its own pitch. + +## Transformer runs (`train.ts`) + +Adam, lr 3e-3 with 100 warm-up steps and cosine decay to a tenth, batch 8 windows, 4 heads, windows start on a step +boundary. Samples start from a C major chord (C5 G4 E4 C3) and run 64 steps (8 bars); three seeds per temperature. + +Speed (node, one thread): d 64, 2 layers, 32-step window (128 tokens), 114,944 parameters: about 290 ms per step. +d 48, 2 layers, 16-step window: about 145 ms per step. So three minutes in the page is roughly 600 or 1,250 steps. + +**A bug to remember:** the first samples crossed voices on 60 % of steps. Sampling slid the context window one token +at a time, so position 0 stopped being the soprano and every voice read as its neighbour. The window must slide by +whole steps (`ceil((len − ctx) / 4) × 4`). Training was never affected; `eval.ts` re-scores saved checkpoints with the +fix. Any number logged by `train.ts` before the fix (its `log.txt` checkpoint lines) is void. + +### Run B: d 48, 2 layers, 16-step window (64 tokens), 4,000 steps, 300 s in all + +`eval.ts`, 5 seeds × 64 steps from the C major chord, temperature 0.7: + +| step | ≈ time | val loss | in key | parallels /100 | crossings | held | chords /100 | longest copy (steps) | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | +| 0 | 0 | 4.11 | 88 % | 5.3 | 84 % | 1 % | 96 | 1 | +| 100 | 15 s | 1.98 | 99 % | 1.9 | 35 % | 77 % | 31 | 2–8 | +| 250 | 35 s | 1.58 | 99.6 % | 10.3 | 4.4 % | 45 % | 52 | 2–4 | +| 500 | 70 s | 1.40 | 99 % | 11.9 | 0.3 % | 49 % | 35 | 4–8 | +| 1000 | 145 s | 1.20 | 99.4 % | 7.5 | 1.9 % | 47 % | 41 | 4–6 | +| 2000 | | 1.08 | 99.7 % | 6.6 | 1.9 % | 48 % | 42 | 4–6 | +| 4000 | | 0.98 | 98.9 % | 6.6 | 0.3 % | 55 % | 36 | 6–9 | + +(The times assume ~145 ms per step; this run shared the machine with run A, so its own wall clock was not clean.) + +- It passes the 5-gram's 1.46 before step 500, a little over a minute. +- Order of learning, as the table shows it: first "hold notes" (step 100 holds 77 %: it drones), then voices stop + crossing (by 250–500), key and rhythm settle; parallel fifths are last and never get near Bach (6.6 vs 0.16). +- Copying grows with training: at 1,000 steps the longest stretch found verbatim in the training set is 4–6 eighths + (under a bar); at 4,000, 6–9. For comparison, a chorale is about 107 steps long. +- At temperature 1.0 everything is looser (crossings 4–15 %, in key 97 %). + +### Preference tuning (`dpo.ts`): the reader's clicks, simulated + +DPO on the step-1000 checkpoint of run B. Each "click" is a pair of samples (16 steps, temperature 1.0) and a judge +picks one; ties are skipped. 4 pairs per Adam update, lr 3e-4. Evaluated on 8 fresh 64-step samples at 0.7. +Before: parallels 12.3 per 100 steps, held 48 %, chords 45 (these 8 seeds; the 5-seed table above says 7.5). + +| judge | β | pairs | parallels | held | chords | val loss | reading | +| --- | --- | --- | --- | --- | --- | --- | --- | +| fewest parallels | 0.2 | 40 | 5.9 | **73 %** | **25** | 1.199 | reward hacking: it learned to move less | +| fewest parallels | 1.0 | 40 | 10.5 | 55 % | 36 | 1.199 | a tighter leash: hacks less, learns less | +| fewest parallels **per step on which voices move** | 0.2 | 40 | 12.7 | 47 % | 44 | 1.205 | no effect at 40 | +| same | 0.5 | 80 | **7.0** | 52 % | 41 | 1.209 | learned the rule without gaming it | +| more held notes ("calm") | 0.5 | 20 | 6.1 | 68 % | 27 | 1.204 | 20 clicks audibly slower | +| more held notes | 0.5 | 40 | 1.6 | 89 % | 11 | 1.225 | too far: a drone | + +- 40 pairs take about 12 s in node; each pair draws two samples, so a human-paced version is bound by listening, + not computing. **Twenty choices is enough to move the music audibly**, which is a count a reader will make. +- The first judge is the article's best moment: asked for fewer mistakes, the model found that the way to make none + is to play nothing. The fix is the one real RLHF uses: a judge that cannot be gamed that way, and a leash (β). +- Single runs, 8 samples each: the direction of every row is clear, the second digit is not. Seeds before prose. + +### Listening pack + +`music-work/listen/` (MP3, 35–40 s each, 60 bpm eighths = 0.5 s, soft additive organ + feedback-delay reverb): +Bach's chor127 (validation), the rules, the 5-gram, run B at 0 / 100 / 250 / 500 / 1,000 / 4,000 steps, and the four +DPO results above, and run A at 4,000 steps. **Nobody has listened yet; the metrics are proxies.** Paul's ears decide. + +### Run A: d 64, 2 layers, 32-step window (128 tokens), 114,944 parameters, 4,000 steps, 1,185 s + +Same evaluation, temperature 0.7: step 1,000 val 1.09, parallels 7.5, crossings 1.3 %, chords 29; step 4,000 val +**0.93**, in key 98.6 %, parallels **5.0**, crossings 0.3 %, held 48 %, chords 45, longest copy 6–10. +Twice the context and 1.4× the width buys 0.05 nats and a little on parallels; it does not close the gap to Bach's +0.16. At about 300 ms per step (sharing the machine) it gets 600 steps in three minutes against B's 1,250, and B at +1,250 is ahead of A at 600. **For the page, B's size is the right one.** + +### Open + +- Timing in a real browser (a worker): node's V8 should be close; measure it. +- Seeds for the DPO table; the judge a reader would really be (a human) is slower and noisier than these. +- Whether the samples are pleasant at all. + +## Timbre (2026-09-22, after Paul's first listen: "the key is fine but it is too sharp") + +The first organ was 5 harmonics at 1/k², a 60 ms attack and four undamped feedback delays (a metallic ring). +`synth.ts` has three softer voices, all through a Freeverb-style reverb (8 damped combs + 4 all-passes, stereo) and a +12 dB/octave low-pass at 3.2 kHz on the mix: + +- **pad**: fundamental + soft octave, each doubled 4 cents apart (slow beating), 350 ms attack, 1.4 s release. +- **epiano**: two-operator FM whose index decays (bright for a moment, then round), 2.2 s decay. +- **flute**: sine + a little 2nd harmonic, a breath of noise on the attack, 4.8 Hz vibrato fading in. + +Band levels relative to the whole spectrum (`hiband.ts`, 4096-point Hann FFT frames; Bach's chor127 / the AI sample): + +| voice | 1–4 kHz | above 2 kHz | +| --- | --- | --- | +| old organ | −19.0 / −19.7 dB | −33.4 / −32.5 dB | +| pad | −25.1 / −25.1 dB | −82.9 / −54.7 dB | +| epiano | −25.9 / −26.8 dB | −59.2 / −51.8 dB | +| flute | −34.7 / −34.9 dB | −49.2 / −50.1 dB | + +Two measures I tried first and threw away: the spectral centroid (475 → ~340 Hz) is dominated by the fundamentals +and says little about harshness; "signal minus a low-pass" is not a high-pass (the low-pass's phase lag leaves the +low end in), and reported 20–30 % above 2 kHz for every voice. Listening files: `music-work/listen/timbres/`. + +## In the page (draft № 014, 2026-09-22) + +- Chromium (Playwright, M4 Pro), the worker in `content/posts/music-ai/components/music.worker.ts`: 1,200 steps in + 83 s, **69 ms per step**, validation loss 1.17 at the end (24 fixed windows). Half node's 145 ms, which was measured + while run A shared the machine. The figure now trains 2,400 steps (about 2.8 minutes). +- The article's learning-order table is run B's (cosine over 4,000 steps); the page's schedule decays over 2,400. + Rerun the table with the page's exact recipe before publishing. + +## Fig 03 and the player (2026-09-23) + +- The blind test ships as fig 03: four bars each of Bach (an excerpt of a validation chorale, never trained on), the + step-1,000 model at temperature 0.7, and the rules composer, shuffled. The piano rolls stay hidden until the reader + answers, because a chorale and a block-chord machine are obvious on sight. One live round measured, for a sanity + check: Bach parallels 0.0 / 72 chords per 100 steps, rules 18.8 / 31, the AI 9.4 / 66 — the same ordering as the + offline table. +- Three bugs Paul found while playing with the draft, all in the shared player: + 1. Changing the timbre only restarted the sound in fig 01, because each figure did it for itself. The player now + owns "what is playing" and every figure asks it to restyle, keeping the position and with no gap in the sound + (the old buffer plays until the new one is ready). + 2. The music kept playing after navigating away: nothing stopped the page-level player when the figures unmounted. + The last figure to leave now stops it, terminates the synth worker and closes the AudioContext (verified: the + context goes from `running` to `closed`). + 3. Play stayed "Stop" for three or four silent seconds at the end, because the buffer carries a three-second + reverb tail. Playback now counts as finished when the music ends plus 0.8 s: a 24-second piece flips back at 25 s. + +## UI review of the draft (2026-09-23) + +Checked at 1440 and 390, both themes, with axe (no violations before or after) and the network log. Thirteen things +changed; the ones worth remembering: + +- **A shared worker needs to say who asked.** Figures 02 and 03 now use one worker (judge.bin 223 KB and the + chorales are fetched and parsed once, not twice), and the first version broadcast every reply to both: loading the + judge reset the blind test to its loading screen. Every message carries the id of the figure that asked. +- Auto-play used to take the stage: pick an earlier snapshot to listen to and the next one stole it. Choosing a + snapshot by hand now pins it, and later ones are marked "new" instead. +- Figure 02 could only be read, not heard — the whole article is about listening. It now has "hear what it writes + now" and "start over", and says that about twenty picks is where the difference becomes audible. +- The numbers had no explanation anywhere ("6.3 per 100 steps" of what?). One line under them now says a step is an + eighth note, eight make a bar, and what the validation error means. +- The piano roll's alternative text in figures 02 and 03 claimed "six bars written at step 0", copied from figure 01. +- Training's "Stop" sat beside playback's "Stop": it is "Stop training" now. The progress bar has an estimate again. +- The auto-play checkbox was 13 px with a 16 px label; the label is now 24 px tall with the site's `tap` hit area. + +## DPO at the page's own recipe (2026-09-23) + +The table above updates once per four choices at lr 3e-4. The page updates on every choice at lr 1e-4, so the +article's numbers were re-measured with the page's recipe (8 samples of 64 steps at 0.7; before: parallels 12.3, +held 48 %, chords 45, val 1.195): + +| judge | picks | parallels | held | chords | val | +| --- | --- | --- | --- | --- | --- | +| more held notes ("calm") | 20 | 6.1 | 64 % | 28 | 1.194 | +| more held notes | 40 | 3.9 | 78 % | 21 | 1.196 | +| fewest parallels (gameable) | 40 | 10.9 | 58 % | 35 | 1.200 | +| fewest parallels (gameable) | 80 | 8.0 | 63 % | 29 | 1.199 | +| parallels per moving step | 80 | 7.0 | 63 % | 29 | 1.199 | + +**The clean reward-hacking contrast does not reproduce at this gentler push.** Both judges drift toward holding +(63 % either way) and the hack-proof one only buys 8.0 → 7.0. The sharp version (gameable 5.9 with 73 % held and 25 +chords, versus hack-proof 7.0 with 52 % held and 41 chords) needs the harder push. The article now says exactly +that: the harder you optimise, the more gaming pays, which is the real lesson; the figure stays gentle so that +twenty clicks do not wreck the model. Paul chose to keep the page's learning rate as it is. + +## The page's own recipe, measured twice (2026-09-23) — the table the article prints + +`train.ts page-2400 2400 48 2 16 8 3e-3`, twice (seeds 7 and 11), 195 s each in node; `eval.ts`, five pieces of 64 +steps at temperature 0.7 per checkpoint. The article's table is the mean of the two runs. + +| step | val | in key | crossings | held | parallels | chords | longest copy | +| --- | --- | --- | --- | --- | --- | --- | --- | +| 0 | 4.51 | 82 % | 87 % | 0.5 % | 3.9 | 97 | 1 | +| 100 | 2.01 | 99 % | 36 % | 78 % | 1.1 | 30 | 2–5 | +| 250 | 1.60 | 99.7 % | 1.9 % | 50 % | 9.8 | 45 | 2–4 | +| 500 | 1.39 | 99.6 % | 1.1 % | 55 % | 10.0 | 34 | 2–6 | +| 1000 | 1.21 | 98.7 % | 1.8 % | 54 % | 12.0 | 38 | 3–6 | +| 2000 | 1.07 | 99.2 % | 1.3 % | 53 % | 8.9 | 38 | 6–10 | +| 2400 | 1.05 | 99.5 % | 0 % | 65 % | 7.2 | 28 | 5–10, once 32 | + +- Held notes rise again at the end of the cosine decay (70 % and 59 % in the two runs), so the model the reader ends + up with is slower than Bach's 45 %. The article says so rather than quoting the mid-training number. +- **One sample in ten copied 32 steps — four whole bars — from a training chorale** (run 7, step 2,400). The other + nine were 5 to 10. It is in the article: the model can hold a passage verbatim, and the copy detector is the way to + catch it. +- Parallel fifths never settle: 7 to 12 per hundred steps the whole way, against Bach's 0.16. +- Run B (the 4,000-step schedule) is still the reference for the DPO work, because judge.bin is its step-1,000 + checkpoint. + +## Where everything lives + +- In the repo: the article (`content/posts/music-ai/`), the data files (`public/posts/music-ai/*.bin`, CC BY-NC-SA + 4.0), the packing scripts (`scripts/music/convert.py`, `scripts/music/pack.ts`), these notes and copies of the + research scripts beside them. +- Outside the repo, in `Desktop/Paul/music-work/`: the corpus clone (`bach-370-chorales/`), `chorales.json`, every + run under `runs/`, the rendered WAV and MP3 packs (`listen/`, `timbres/`). `pack.ts` in the repo reproduces both + `.bin` files from `chorales.json` and a checkpoint, byte for byte. diff --git a/docs/research/music-ai/baseline.ts.txt b/docs/research/music-ai/baseline.ts.txt new file mode 100644 index 0000000..e0061e2 --- /dev/null +++ b/docs/research/music-ai/baseline.ts.txt @@ -0,0 +1,74 @@ +// Reference numbers: Bach's own validation chorales, n-gram models, and a rule-based "five lines of maths" composer. +import { HOLD, REST, VOCAB, VOICES, decode, longestCopy, renderWav, report, trainTokens, valSet, valTokens } from "./music"; +import { mulberry32 } from "../Blog/lib/ml/neuroevolution"; + +const mean = (rs: ReturnType[]) => Object.fromEntries(Object.keys(rs[0]).map((k) => [k, +(rs.reduce((s, r) => s + (r as any)[k], 0) / rs.length).toFixed(3)])); +console.log("train tokens", trainTokens.reduce((s, t) => s + t.length, 0), "val tokens", valTokens.reduce((s, t) => s + t.length, 0), "vocab", VOCAB); +console.log("bach (validation)", mean(valSet.map((c) => report(c.pitches, c.onset)))); + +// n-gram with add-k smoothing, context = previous n-1 tokens. Validation cross-entropy in nats per token. +function ngram(n: number, k = 0.05) { + const counts = new Map(); + for (const seq of trainTokens) + for (let t = 0; t < seq.length; t++) { + const ctx = seq.slice(Math.max(0, t - n + 1), t).join(",") + "|" + (t % VOICES); + if (!counts.has(ctx)) counts.set(ctx, new Float64Array(VOCAB)); + counts.get(ctx)![seq[t]]++; + } + const prob = (ctx: string, id: number) => { + const c = counts.get(ctx); + if (!c) return 1 / VOCAB; + const total = c.reduce((a, b) => a + b, 0); + return (c[id] + k) / (total + k * VOCAB); + }; + let nll = 0, count = 0; + for (const seq of valTokens) + for (let t = 0; t < seq.length; t++) { nll -= Math.log(prob(seq.slice(Math.max(0, t - n + 1), t).join(",") + "|" + (t % VOICES), seq[t])); count++; } + return { nll: nll / count, counts, prob }; +} +for (const n of [1, 2, 3, 4, 5, 6, 8]) console.log(`${n}-gram val nats/token`, ngram(n).nll.toFixed(3)); + +// Rule composer: pentatonic C (C D E G A) soprano as a 1/f (Voss) walk, a fixed I–vi–IV–V progression under it. +function rules(steps: number, seed: number) { + const rng = mulberry32(seed); + const penta = [60, 62, 64, 67, 69, 72, 74, 76, 79, 81]; + const dice = [0, 0, 0, 0].map(() => rng()); + const chords = [[48, 55, 64], [45, 57, 64], [41, 53, 60], [43, 55, 62]]; // C, Am, F, G (bass, tenor, alto) + const pitches: number[][] = [], onset: number[][] = []; + let last = -1; + for (let t = 0; t < steps; t++) { + if (t % 2 === 0) for (let b = 0; b < 4; b++) if (((t / 2) & ((1 << b) - 1)) === 0) dice[b] = rng(); + const idx = Math.floor((dice.reduce((a, b) => a + b, 0) / 4) * penta.length); + const s = penta[Math.min(penta.length - 1, idx)], ch = chords[Math.floor(t / 8) % 4]; + const newS = t % 2 === 0 && s !== last; + pitches.push([newS || t % 2 === 1 ? (t % 2 === 0 ? s : last) : last, ch[2], ch[1], ch[0]]); + onset.push([newS ? 1 : 0, t % 8 === 0 ? 1 : 0, t % 8 === 0 ? 1 : 0, t % 8 === 0 ? 1 : 0]); + if (newS) last = s; + } + return { pitches, onset }; +} +const rs = [1, 2, 3, 4, 5].map((s) => rules(96, s)); +console.log("rules", mean(rs.map((r) => report(r.pitches, r.onset))), "copy", Math.max(...rs.map((r) => longestCopy(r.pitches)))); +renderWav("out/rules.wav", rs[0].pitches, rs[0].onset); +renderWav("out/bach-val.wav", valSet[0].pitches, valSet[0].onset); +console.log("bach val chorale for listening:", valSet[0].id); + +// Let the 5-gram compose, from the same C major chord the Transformer starts from. +{ + const { prob } = ngram(5); + for (const temp of [0.7, 1.0]) { + const out = [1, 2, 3].map((seed) => { + const r = mulberry32(seed), ids = [72 - 31, 67 - 31, 64 - 31, 48 - 31]; + while (ids.length < 64 * VOICES) { + const ctx = ids.slice(-4).join(",") + "|" + (ids.length % VOICES); + const p = Array.from({ length: VOCAB }, (_, k) => prob(ctx, k) ** (1 / temp)); + let u = r() * p.reduce((a, b) => a + b, 0), k = 0; + while (u > p[k] && k < VOCAB - 1) u -= p[k++]; + ids.push(k); + } + return decode(ids); + }); + console.log(`5-gram T${temp}`, mean(out.map((d) => report(d.pitches, d.onset))), "copy", out.map((d) => longestCopy(d.pitches))); + renderWav(`out/5gram-T${temp}.wav`, out[0].pitches, out[0].onset); + } +} diff --git a/docs/research/music-ai/convert.py.txt b/docs/research/music-ai/convert.py.txt new file mode 100644 index 0000000..b9d6ac6 --- /dev/null +++ b/docs/research/music-ai/convert.py.txt @@ -0,0 +1,52 @@ +# Bach chorales (Craig Sapp's edition, CC BY-NC-SA 4.0) -> an eighth-note grid, one JSON file. +# uv run --with music21 python convert.py +# Each chorale is transposed to C major or A minor. A step holds, for S A T B, the MIDI pitch sounding (or -1), +# whether that note starts on this step, and whether a fermata (phrase end) is on it. +import glob, json, math, os, sys +from music21 import converter, interval, pitch, note, expressions + +STEP = 0.5 # quarter lengths: an eighth note +out, skipped = [], [] +for path in sorted(glob.glob("bach-370-chorales/kern/*.krn")): + name = os.path.basename(path)[:-4] + try: + score = converter.parse(path) + parts = list(score.parts) + if len(parts) != 4: + skipped.append((name, f"{len(parts)} parts")); continue + key = score.analyze("key") + tonic = "C" if key.mode == "major" else "A" + shift = interval.Interval(key.tonic, pitch.Pitch(tonic)) + # Keep transpositions within a tritone so voices stay in their ranges. + if shift.semitones > 6: shift = interval.Interval(shift.semitones - 12) + if shift.semitones < -6: shift = interval.Interval(shift.semitones + 12) + score = score.transpose(shift) + parts = list(score.parts) + # kern lists bass first; order voices top-down by mean pitch to be sure. + def mean(p): + ps = [n.pitch.midi for n in p.recurse().notes if n.isNote] + return sum(ps) / max(1, len(ps)) + parts.sort(key=mean, reverse=True) + length = max(p.highestTime for p in parts) + n = int(math.ceil(length / STEP)) + pitches = [[-1] * 4 for _ in range(n)] + onset = [[0] * 4 for _ in range(n)] + fermata = [0] * n + for v, p in enumerate(parts): + for el in p.flatten().notesAndRests: + if not el.isNote: continue + a = int(round(el.offset / STEP)); b = int(round((el.offset + el.quarterLength) / STEP)) + tied_in = el.tie is not None and el.tie.type in ("continue", "stop") + for t in range(a, min(b, n)): + pitches[t][v] = el.pitch.midi + onset[t][v] = 1 if (t == a and not tied_in) else 0 + if any(isinstance(e, expressions.Fermata) for e in el.expressions) and b - 1 < n: + fermata[max(a, b - 1)] = 1 + ts = score.recurse().getElementsByClass("TimeSignature") + out.append({"id": name, "mode": key.mode, "shift": shift.semitones, "meter": ts[0].ratioString if ts else "4/4", + "pitches": pitches, "onset": onset, "fermata": fermata}) + except Exception as e: # report and go on + skipped.append((name, repr(e)[:80])) +json.dump(out, open("chorales.json", "w")) +print(len(out), "chorales,", len(skipped), "skipped") +for s in skipped[:20]: print(" skipped", *s) diff --git a/docs/research/music-ai/dpo.ts.txt b/docs/research/music-ai/dpo.ts.txt new file mode 100644 index 0000000..8bb667b --- /dev/null +++ b/docs/research/music-ai/dpo.ts.txt @@ -0,0 +1,121 @@ +// Preference tuning (DPO) on a trained checkpoint, with an automatic judge standing in for the reader. +// npx tsx dpo.ts [pairs] [pairsPerUpdate] [beta] [lr] +// The reader's click is simulated: of two samples, the judge prefers the one with fewer parallel fifths/octaves +// ("parallels") or with more held notes ("calm"). Ties are skipped (a reader would skip them too). +import fs from "node:fs"; +import { Mat, Tape } from "../Blog/lib/ml/autograd"; +import { Adam, Transformer } from "../Blog/lib/ml/transformer"; +import { mulberry32 } from "../Blog/lib/ml/neuroevolution"; +import { LO, VOCAB, VOICES, decode, longestCopy, renderWav, report, valTokens } from "./music"; + +const [file, d, ctxSteps, judge = "parallels", pairsArg = "40", perArg = "4", betaArg = "0.2", lrArg = "3e-4"] = process.argv.slice(2); +const CTX = +ctxSteps * VOICES, PAIRS = +pairsArg, PER = +perArg, BETA = +betaArg, LR = +lrArg; +const config = { vocab: VOCAB, ctx: CTX, d: +d, heads: 4, layers: 2 }; +const policy = new Transformer(config), ref = new Transformer(config); +const w = JSON.parse(fs.readFileSync(file, "utf8")); +for (const m of [policy, ref]) for (const [k, p] of Object.entries(m.params)) p.data.set(w[k]); +const adam = new Adam(policy.params); +const PROMPT = [72 - LO, 67 - LO, 64 - LO, 48 - LO]; +const LEN = CTX; // a whole sample fits in one window, so its probability is one forward pass + +function sample(model: Transformer, tokens: number, temperature: number, r: () => number): number[] { + const ids = [...PROMPT]; + while (ids.length < tokens) { + const ctx = ids.slice(Math.max(0, Math.ceil((ids.length - CTX) / VOICES) * VOICES)); + const { logits } = model.forward(new Tape(), ctx); + const row = logits.data.subarray((ctx.length - 1) * VOCAB, ctx.length * VOCAB); + let max = -Infinity; for (const x of row) max = Math.max(max, x); + const p = Array.from(row, (x) => Math.exp((x - max) / temperature)); + let u = r() * p.reduce((a, b) => a + b, 0), k = 0; + while (u > p[k] && k < p.length - 1) u -= p[k++]; + ids.push(k); + } + return ids; +} + +/** Sum of log-probabilities of the tokens after the prompt. */ +function logProb(model: Transformer, ids: number[], tape: Tape): { value: number; logits: Mat; targets: number[] } { + const { logits } = model.forward(tape, ids.slice(0, -1)); + const targets = ids.slice(1).map((t, i) => (i + 1 < PROMPT.length ? -1 : t)); + let lp = 0; + for (let r = 0; r < targets.length; r++) { + if (targets[r] < 0) continue; + const row = logits.data.subarray(r * VOCAB, (r + 1) * VOCAB); + let max = -Infinity; for (const x of row) max = Math.max(max, x); + let s = 0; for (const x of row) s += Math.exp(x - max); + lp += row[targets[r]] - max - Math.log(s); + } + return { value: lp, logits, targets }; +} + +/** How often two voices move in parallel fifths/octaves, per step on which at least two voices move: holding still cannot game it. */ +function perMove(pitches: number[][]): number { + let moves = 0, par = 0; + for (let t = 1; t < pitches.length; t++) { + const p = pitches[t], q = pitches[t - 1]; + if (p.filter((x, v) => x !== q[v]).length >= 2) moves++; + for (let a = 0; a < 4; a++) for (let c = a + 1; c < 4; c++) { + if (p[a] < 0 || p[c] < 0 || q[a] < 0 || q[c] < 0) continue; + const now = (p[a] - p[c]) % 12, was = (q[a] - q[c]) % 12; + if (p[a] !== q[a] && p[c] !== q[c] && Math.sign(p[a] - q[a]) === Math.sign(p[c] - q[c]) && now === was && (now === 7 || now === 0)) par++; + } + } + return moves ? par / moves : 0; +} +const score = (ids: number[]) => { + const g = decode(ids), r = report(g.pitches, g.onset); + if (judge === "calm") return r.holdShare; + if (judge === "permove") return -perMove(g.pitches); + return -r.parallels; +}; + +function evaluate(label: string) { + const reps = [1, 2, 3, 4, 5, 6, 7, 8].map((seed) => { const g = decode(sample(policy, 64 * VOICES, 0.7, mulberry32(1000 + seed))); return { r: report(g.pitches, g.onset), copy: longestCopy(g.pitches), g }; }); + const mean = (k: keyof (typeof reps)[0]["r"]) => +(reps.reduce((s, x) => s + x.r[k], 0) / reps.length).toFixed(3); + const valRng = mulberry32(99); + let val = 0; + for (let i = 0; i < 64; i++) { + const seq = valTokens[Math.floor(valRng() * valTokens.length)]; + const steps = seq.length / VOICES, len = Math.min(CTX + VOICES, seq.length); + const start = Math.floor(valRng() * Math.max(1, steps - len / VOICES + 1)) * VOICES; + const win = seq.slice(start, start + len); + val += new Tape().crossEntropy(policy.forward(new Tape(), win.slice(0, -1).slice(0, CTX)).logits, win.slice(1, CTX + 1)); + } + console.log(label, JSON.stringify({ val: +(val / 64).toFixed(3), inKey: mean("inKey"), parallels: mean("parallels"), crossings: mean("crossings"), hold: mean("holdShare"), chords: mean("distinctChords"), copy: reps.map((x) => x.copy) })); + return reps[0].g; +} + +const tag = `${judge}-p${PAIRS}-b${BETA}-lr${LR}`; +const before = evaluate("before"); +renderWav(`out/dpo-${tag}-before.wav`, before.pitches, before.onset); +const r = mulberry32(5); +let used = 0, skipped = 0, sampled = 0; +const start = Date.now(); +while (used < PAIRS) { + policy.zeroGrad(); + const tape = new Tape(); + let n = 0; + while (n < PER && used < PAIRS) { + const a = sample(policy, LEN + 1, 1.0, r), b = sample(policy, LEN + 1, 1.0, r); + sampled += 2; + const sa = score(a), sb = score(b); + if (sa === sb) { skipped++; continue; } + const [win, lose] = sa > sb ? [a, b] : [b, a]; + const pw = logProb(policy, win, tape), pl = logProb(policy, lose, tape); + const rw = logProb(ref, win, new Tape()).value, rl = logProb(ref, lose, new Tape()).value; + const z = BETA * (pw.value - rw - (pl.value - rl)); + const g = BETA * (1 - 1 / (1 + Math.exp(-z))) / PER; // −dL/dz · β, averaged over the update's pairs + const count = (t: number[]) => t.filter((x) => x >= 0).length; + // crossEntropy's gradient is weight × d(mean NLL); log p = −count × mean NLL. + tape.crossEntropy(pw.logits, pw.targets, g * count(pw.targets)); + tape.crossEntropy(pl.logits, pl.targets, -g * count(pl.targets)); + n++; used++; + } + tape.backward(); + adam.step(LR); + if (used % 20 === 0 || used === PAIRS) console.log(`pairs ${used} (skipped ties ${skipped}) ${((Date.now() - start) / 1000).toFixed(0)} s`); +} +const after = evaluate("after"); +renderWav(`out/dpo-${tag}-after.wav`, after.pitches, after.onset); +fs.writeFileSync(`out/dpo-${tag}.json`, JSON.stringify(Object.fromEntries(Object.entries(policy.params).map(([k, m]) => [k, Array.from(m.data, (x) => +x.toFixed(5))])))); +console.log("samples drawn", sampled); diff --git a/docs/research/music-ai/eval.ts.txt b/docs/research/music-ai/eval.ts.txt new file mode 100644 index 0000000..25d7ac5 --- /dev/null +++ b/docs/research/music-ai/eval.ts.txt @@ -0,0 +1,57 @@ +// Re-evaluate saved checkpoints: validation loss, samples at two temperatures (5 seeds, 64 steps), metrics, WAVs. +// npx tsx eval.ts [layers] +import fs from "node:fs"; +import { Tape } from "../Blog/lib/ml/autograd"; +import { Transformer } from "../Blog/lib/ml/transformer"; +import { mulberry32 } from "../Blog/lib/ml/neuroevolution"; +import { LO, VOCAB, VOICES, decode, longestCopy, renderWav, report, valTokens } from "./music"; + +const [dir, d, ctxSteps, layers = "2"] = process.argv.slice(2); +const CTX = +ctxSteps * VOICES; +const model = new Transformer({ vocab: VOCAB, ctx: CTX, d: +d, heads: 4, layers: +layers }); + +export function sample(steps: number, temperature: number, seed: number, prompt = [72 - LO, 67 - LO, 64 - LO, 48 - LO]): number[] { + const r = mulberry32(seed), ids = [...prompt]; + while (ids.length < steps * VOICES) { + const ctx = ids.slice(Math.max(0, Math.ceil((ids.length - CTX) / VOICES) * VOICES)); + const { logits } = model.forward(new Tape(), ctx); + const row = logits.data.subarray((ctx.length - 1) * VOCAB, ctx.length * VOCAB); + let max = -Infinity; for (const x of row) max = Math.max(max, x); + const p = Array.from(row, (x) => Math.exp((x - max) / temperature)); + let u = r() * p.reduce((a, b) => a + b, 0), k = 0; + while (u > p[k] && k < p.length - 1) u -= p[k++]; + ids.push(k); + } + return ids; +} + +// The same 64 validation windows as train.ts. +function window(pool: number[][], r: () => number): number[] { + const seq = pool[Math.floor(r() * pool.length)]; + const steps = seq.length / VOICES, len = Math.min(CTX + VOICES, seq.length); + const start = Math.floor(r() * Math.max(1, steps - len / VOICES + 1)) * VOICES; + return seq.slice(start, start + len); +} +const valRng = mulberry32(99), valWindows = Array.from({ length: 64 }, () => window(valTokens, valRng)); + +const files = fs.readdirSync(dir).filter((f) => /^model-\d+\.json$/.test(f)).sort((a, b) => parseInt(a.slice(6)) - parseInt(b.slice(6))); +const rows: string[] = []; +for (const f of files) { + const step = parseInt(f.slice(6)); + const w = JSON.parse(fs.readFileSync(`${dir}/${f}`, "utf8")); + for (const [k, m] of Object.entries(model.params)) m.data.set(w[k]); + let val = 0; + for (const win of valWindows) val += new Tape().crossEntropy(model.forward(new Tape(), win.slice(0, -1).slice(0, CTX)).logits, win.slice(1, CTX + 1)); + val /= valWindows.length; + const line: Record = { step, val: +val.toFixed(3) }; + for (const temp of [0.7, 1.0]) { + const rs = [1, 2, 3, 4, 5].map((seed) => decode(sample(64, temp, seed))); + const reps = rs.map((r) => report(r.pitches, r.onset)); + const mean = (k: keyof (typeof reps)[0]) => +(reps.reduce((s, r) => s + r[k], 0) / reps.length).toFixed(3); + line[`T${temp}`] = { inKey: mean("inKey"), parallels: mean("parallels"), crossings: mean("crossings"), rough: mean("rough"), hold: mean("holdShare"), chords: mean("distinctChords"), copy: rs.map((r) => longestCopy(r.pitches)) }; + renderWav(`${dir}/eval-step${step}-T${temp}.wav`, rs[0].pitches, rs[0].onset); + } + rows.push(JSON.stringify(line)); + console.log(rows.at(-1)); +} +fs.writeFileSync(`${dir}/eval.txt`, rows.join("\n") + "\n"); diff --git a/docs/research/music-ai/hiband.ts.txt b/docs/research/music-ai/hiband.ts.txt new file mode 100644 index 0000000..ded6c74 --- /dev/null +++ b/docs/research/music-ai/hiband.ts.txt @@ -0,0 +1,27 @@ +// Share of spectral energy above 2 kHz (and above 4 kHz), from 4096-point Hann-windowed FFT frames. WAV 16-bit in. +import fs from "node:fs"; +function fft(re: Float64Array, im: Float64Array) { + const n = re.length; + for (let i = 1, j = 0; i < n; i++) { let bit = n >> 1; for (; j & bit; bit >>= 1) j ^= bit; j ^= bit; if (i < j) { [re[i], re[j]] = [re[j], re[i]]; [im[i], im[j]] = [im[j], im[i]]; } } + for (let len = 2; len <= n; len <<= 1) { + const ang = (-2 * Math.PI) / len; + for (let i = 0; i < n; i += len) for (let k = 0; k < len / 2; k++) { + const wr = Math.cos(ang * k), wi = Math.sin(ang * k), a = i + k, b = a + len / 2; + const xr = re[b] * wr - im[b] * wi, xi = re[b] * wi + im[b] * wr; + re[b] = re[a] - xr; im[b] = im[a] - xi; re[a] += xr; im[a] += xi; + } + } +} +const bands = (f: string) => { + const b = fs.readFileSync(f), ch = b.readUInt16LE(22), n = (b.length - 44) / 2 / ch, N = 4096; + let e2 = 0, e4 = 0, all = 0, mid = 0; + for (let s = 0; s + N <= n; s += N) { + const re = new Float64Array(N), im = new Float64Array(N); + for (let i = 0; i < N; i++) { let v = 0; for (let c = 0; c < ch; c++) v += b.readInt16LE(44 + ((s + i) * ch + c) * 2); re[i] = (v / ch) * (0.5 - 0.5 * Math.cos((2 * Math.PI * i) / N)); } + fft(re, im); + for (let k = 1; k < N / 2; k++) { const e = re[k] ** 2 + im[k] ** 2, hz = (k * 44100) / N; all += e; if (hz > 2000) e2 += e; if (hz > 4000) e4 += e; if (hz > 1000 && hz < 4000) mid += e; } + } + const db = (x: number) => (10 * Math.log10(x / all)).toFixed(1).padStart(6); + return `1–4 kHz ${db(mid)} dB >2 kHz ${db(e2)} dB >4 kHz ${db(e4)} dB`; +}; +for (const f of process.argv.slice(2)) console.log(f.padEnd(44), bands(f)); diff --git a/docs/research/music-ai/music.ts.txt b/docs/research/music-ai/music.ts.txt new file mode 100644 index 0000000..e209529 --- /dev/null +++ b/docs/research/music-ai/music.ts.txt @@ -0,0 +1,162 @@ +// Shared pieces of the music-AI spike: tokens, the split, sampling, the metrics, a WAV writer. +import fs from "node:fs"; +import { mulberry32 } from "../Blog/lib/ml/neuroevolution"; + +export interface Chorale { id: string; mode: string; meter: string; pitches: number[][]; onset: number[][]; fermata: number[] } +export const LO = 31, HI = 84; +export const HOLD = HI - LO + 1, REST = HOLD + 1, VOCAB = REST + 1; +export const VOICES = 4; + +export const chorales: Chorale[] = JSON.parse(fs.readFileSync(new URL("./chorales.json", import.meta.url), "utf8")); + +/** One token per voice per eighth: S A T B. A new note is its pitch, a held one HOLD, silence REST. */ +export function tokens(c: Chorale): number[] { + const out: number[] = []; + c.pitches.forEach((step, t) => step.forEach((p, v) => out.push(p < 0 ? REST : c.onset[t][v] || t === 0 ? p - LO : HOLD))); + return out; +} + +/** Split by chorale, seeded: 90 % train, 10 % validation. */ +const rng = mulberry32(2026); +const order = chorales.map((_, i) => i).sort(() => rng() - 0.5); +const nVal = Math.round(chorales.length * 0.1); +export const valSet = order.slice(0, nVal).map((i) => chorales[i]); +export const trainSet = order.slice(nVal).map((i) => chorales[i]); +export const trainTokens = trainSet.map(tokens); +export const valTokens = valSet.map(tokens); + +/** Tokens back to sounding pitches per step and voice (-1 silent), and onsets. */ +export function decode(ids: number[]): { pitches: number[][]; onset: number[][] } { + const pitches: number[][] = [], onset: number[][] = []; + const last = [-1, -1, -1, -1]; + for (let t = 0; t * VOICES < ids.length; t++) { + const p: number[] = [], o: number[] = []; + for (let v = 0; v < VOICES; v++) { + const id = ids[t * VOICES + v]; + if (id === undefined) { p.push(-1); o.push(0); continue; } + if (id === HOLD) { p.push(last[v]); o.push(0); } + else if (id === REST) { last[v] = -1; p.push(-1); o.push(0); } + else { last[v] = id + LO; p.push(id + LO); o.push(1); } + } + pitches.push(p); onset.push(o); + } + return { pitches, onset }; +} + +// --------------------------------------------------------------------------------------------------------------- +// Metrics. None of them says "beautiful"; they say whether the texture obeys what four-part writing obeys. + +/** C major's pitch classes, plus F# and G#: A minor's raised sixth and seventh. Every chorale is in C major or A minor. */ +const KEY = new Set([0, 2, 4, 5, 7, 9, 11, 6, 8]); + +/** Plomp–Levelt roughness of a chord of sine-rich tones (6 harmonics, 1/k amplitudes), after Sethares. */ +export function roughness(midi: number[]): number { + const partials: [number, number][] = []; + for (const m of midi) if (m >= 0) for (let k = 1; k <= 6; k++) partials.push([440 * 2 ** ((m - 69) / 12) * k, 1 / k]); + let r = 0; + for (let i = 0; i < partials.length; i++) + for (let j = i + 1; j < partials.length; j++) { + const [f1, a1] = partials[i], [f2, a2] = partials[j]; + const s = 0.24 / (0.021 * Math.min(f1, f2) + 19), d = Math.abs(f2 - f1); + r += Math.min(a1, a2) * (Math.exp(-3.5 * s * d) - Math.exp(-5.75 * s * d)); + } + return r; +} + +export interface Report { + inKey: number; // share of note onsets whose pitch class belongs to C major or A minor (with raised 6, 7) + parallels: number; // parallel fifths/octaves per 100 steps (between any two voices, on moving steps) + crossings: number; // share of steps where a lower voice sounds above a higher one + rough: number; // mean chord roughness + holdShare: number; // share of voice-steps that are held (rhythm) + distinctChords: number; // distinct pitch-class sets per 100 steps (variety) +} + +export function report(pitches: number[][], onset: number[][]): Report { + let notes = 0, inKey = 0, parallels = 0, crossings = 0, rough = 0, held = 0, cells = 0; + const chords = new Set(); + const key = KEY; + for (let t = 0; t < pitches.length; t++) { + const p = pitches[t]; + for (let v = 0; v < 4; v++) { + if (p[v] < 0) continue; + cells++; + if (onset[t][v]) { notes++; if (key.has(p[v] % 12)) inKey++; } else held++; + } + for (let v = 0; v < 3; v++) if (p[v] >= 0 && p[v + 1] >= 0 && p[v + 1] > p[v]) { crossings++; break; } + rough += roughness(p); + chords.add(p.filter((x) => x >= 0).map((x) => x % 12).sort((a, b) => a - b).join(",")); + if (t > 0) { + const q = pitches[t - 1]; + for (let a = 0; a < 4; a++) + for (let b = a + 1; b < 4; b++) { + if (p[a] < 0 || p[b] < 0 || q[a] < 0 || q[b] < 0) continue; + const now = (p[a] - p[b]) % 12, before = (q[a] - q[b]) % 12; + const moved = p[a] !== q[a] && p[b] !== q[b] && Math.sign(p[a] - q[a]) === Math.sign(p[b] - q[b]); + if (moved && now === before && (now === 7 || now === 0)) parallels++; + } + } + } + const n = pitches.length; + return { inKey: inKey / notes, parallels: (100 * parallels) / n, crossings: crossings / n, rough: rough / n, holdShare: held / cells, distinctChords: (100 * chords.size) / n }; +} + +/** Longest run of steps (all four voices equal) that also occurs somewhere in the training set. */ +export function longestCopy(pitches: number[][]): number { + const key = (p: number[]) => p.join(","); + const index = new Map(); + trainSet.forEach((c, ci) => c.pitches.forEach((p, t) => { const k = key(p); if (!index.has(k)) index.set(k, []); index.get(k)!.push([ci, t]); })); + let best = 0; + for (let t = 0; t < pitches.length; t++) { + for (const [ci, s] of index.get(key(pitches[t])) ?? []) { + const c = trainSet[ci].pitches; + let n = 0; + while (t + n < pitches.length && s + n < c.length && key(c[s + n]) === key(pitches[t + n])) n++; + if (n > best) best = n; + } + } + return best; +} + +// --------------------------------------------------------------------------------------------------------------- +// Sound: a soft additive organ with a slow attack, and a simple feedback-delay reverb. Mono, 44.1 kHz, 16-bit. + +export function renderWav(file: string, pitches: number[][], onset: number[][], { bpm = 60, sampleRate = 44100 } = {}) { + const stepSec = 60 / bpm / 2; // an eighth + const n = Math.ceil((pitches.length * stepSec + 3) * sampleRate); + const out = new Float32Array(n); + for (let v = 0; v < 4; v++) { + let t = 0; + while (t < pitches.length) { + const p = pitches[t][v]; + if (p < 0) { t++; continue; } + let len = 1; + while (t + len < pitches.length && pitches[t + len][v] === p && !onset[t + len][v]) len++; + const f = 440 * 2 ** ((p - 69) / 12), a0 = Math.round(t * stepSec * sampleRate), dur = len * stepSec; + const total = Math.round((dur + 0.6) * sampleRate); + for (let i = 0; i < total && a0 + i < n; i++) { + const s = i / sampleRate; + const env = Math.min(1, s / 0.06) * (s < dur ? 1 : Math.exp(-(s - dur) / 0.18)); + let x = 0; + for (let k = 1; k <= 5; k++) x += Math.sin(2 * Math.PI * f * k * s) / (k * k); + out[a0 + i] += 0.12 * env * x; + } + t += len; + } + } + // Reverb: four feedback delays, a little of the dry signal fed in. + const wet = new Float32Array(n); + for (const [ms, g] of [[29.7, 0.72], [37.1, 0.7], [41.1, 0.68], [43.7, 0.66]] as const) { + const d = Math.round((ms / 1000) * sampleRate), buf = new Float32Array(n); + for (let i = 0; i < n; i++) buf[i] = out[i] + (i >= d ? g * buf[i - d] : 0); + for (let i = 0; i < n; i++) wet[i] += buf[i] * 0.18; + } + let peak = 0; + for (let i = 0; i < n; i++) { out[i] = out[i] * 0.75 + wet[i] * 0.35; peak = Math.max(peak, Math.abs(out[i])); } + const buf = Buffer.alloc(44 + n * 2); + buf.write("RIFF", 0); buf.writeUInt32LE(36 + n * 2, 4); buf.write("WAVE", 8); buf.write("fmt ", 12); + buf.writeUInt32LE(16, 16); buf.writeUInt16LE(1, 20); buf.writeUInt16LE(1, 22); buf.writeUInt32LE(sampleRate, 24); + buf.writeUInt32LE(sampleRate * 2, 28); buf.writeUInt16LE(2, 32); buf.writeUInt16LE(16, 34); buf.write("data", 36); buf.writeUInt32LE(n * 2, 40); + for (let i = 0; i < n; i++) buf.writeInt16LE(Math.round((out[i] / (peak || 1)) * 0.85 * 32767), 44 + i * 2); + fs.writeFileSync(file, buf); +} diff --git a/docs/research/music-ai/synth.ts.txt b/docs/research/music-ai/synth.ts.txt new file mode 100644 index 0000000..ae8c6b1 --- /dev/null +++ b/docs/research/music-ai/synth.ts.txt @@ -0,0 +1,122 @@ +// Softer voices for the listening tests: three timbres, a damped stereo reverb (Freeverb's layout), a gentle +// low-pass on the mix, and the spectral centroid of the result as a number for "how bright". +import fs from "node:fs"; + +export type Timbre = "pad" | "epiano" | "flute"; +const SR = 44100; + +/** Deterministic noise for breath and detune. */ +function noise(seed: number) { let a = seed; return () => { a = (a * 1664525 + 1013904223) | 0; return a / 2147483648; }; } + +/** One note into L/R. Voice 0 is the soprano; lower voices sit a little louder and wider apart in the stereo field. */ +function note(L: Float32Array, R: Float32Array, timbre: Timbre, midi: number, start: number, dur: number, voice: number) { + const f = 440 * 2 ** ((midi - 69) / 12); + const level = [0.5, 0.55, 0.6, 0.75][voice]; + const pan = [0.35, 0.6, 0.4, 0.5][voice]; // 0 left … 1 right + const gl = Math.cos((pan * Math.PI) / 2), gr = Math.sin((pan * Math.PI) / 2); + const rnd = noise(midi * 131 + Math.round(start * 1000)); + const attack = timbre === "pad" ? 0.35 : timbre === "flute" ? 0.12 : 0.008; + const release = timbre === "pad" ? 1.4 : timbre === "flute" ? 0.35 : 1.2; + const total = Math.round((dur + release) * SR), a0 = Math.round(start * SR); + let breath = 0; + for (let i = 0; i < total && a0 + i < L.length; i++) { + const t = i / SR; + let env: number, x: number; + const on = t < dur ? 1 : Math.exp(-(t - dur) / (release / 4)); + if (timbre === "pad") { + // Fundamental and a soft octave, each doubled 4 cents apart: the slow beating is the warmth. + env = Math.min(1, t / attack) ** 2 * on; + const d = 2 ** (4 / 1200); + x = 0.5 * (Math.sin(2 * Math.PI * f * t) + Math.sin(2 * Math.PI * f * d * t)) + 0.12 * (Math.sin(4 * Math.PI * f * t) + Math.sin(4 * Math.PI * f / d * t)) + 0.03 * Math.sin(6 * Math.PI * f * t); + } else if (timbre === "flute") { + // A sine, a little second harmonic, a breath of filtered noise at the start, and a slow vibrato. + env = Math.min(1, t / attack) * on; + const vib = 1 + 0.003 * Math.sin(2 * Math.PI * 4.8 * t) * Math.min(1, t / 0.6); + breath = 0.97 * breath + 0.03 * rnd(); + x = Math.sin(2 * Math.PI * f * vib * t) + 0.08 * Math.sin(4 * Math.PI * f * vib * t) + 3 * breath * Math.exp(-t / 0.08); + } else { + // Electric piano by two-operator FM: the modulation index decays, so the tone softens as it sounds. + env = Math.min(1, t / attack) * Math.exp(-t / 2.2) * on; + const index = 1.1 * Math.exp(-t / 0.35); + x = Math.sin(2 * Math.PI * f * t + index * Math.sin(2 * Math.PI * f * t)); + } + const s = 0.16 * level * env * x; + L[a0 + i] += s * gl; R[a0 + i] += s * gr; + } +} + +/** Freeverb: eight damped feedback combs and four all-passes per side, the right side's delays a little longer. */ +function reverb(inL: Float32Array, inR: Float32Array, room = 0.84, damp = 0.45, wet = 0.32) { + const combs = [1116, 1188, 1277, 1356, 1422, 1491, 1557, 1617], alls = [556, 441, 341, 225]; + const side = (input: Float32Array, spread: number) => { + const out = new Float32Array(input.length); + for (const d0 of combs) { + const d = d0 + spread, buf = new Float32Array(d); let idx = 0, store = 0; + for (let i = 0; i < input.length; i++) { + const y = buf[idx]; + store = y * (1 - damp) + store * damp; // the damping is a low-pass inside the loop: the tail darkens as it decays + buf[idx] = input[i] * 0.015 + store * room; + out[i] += y; + idx = (idx + 1) % d; + } + } + for (const d0 of alls) { + const d = d0 + spread, buf = new Float32Array(d); let idx = 0; + for (let i = 0; i < out.length; i++) { + const b = buf[idx], x = out[i]; + out[i] = b - x; buf[idx] = x + b * 0.5; + idx = (idx + 1) % d; + } + } + return out; + }; + const wl = side(inL, 0), wr = side(inR, 23); + for (let i = 0; i < inL.length; i++) { inL[i] = inL[i] * (1 - wet) + wl[i] * wet * 3; inR[i] = inR[i] * (1 - wet) + wr[i] * wet * 3; } +} + +/** Two one-pole low-passes in a row at `hz`: a gentle 12 dB/octave roll-off over the whole mix. */ +function lowpass(x: Float32Array, hz: number) { + const a = Math.exp((-2 * Math.PI * hz) / SR); + for (let pass = 0; pass < 2; pass++) { let y = 0; for (let i = 0; i < x.length; i++) x[i] = y = (1 - a) * x[i] + a * y; } +} + +/** Amplitude-weighted mean frequency of the whole signal (Hz), from 2048-point frames: higher sounds brighter. */ +export function centroid(x: Float32Array): number { + const N = 2048; let num = 0, den = 0; + for (let s = 0; s + N <= x.length; s += N * 8) { + for (let k = 1; k < N / 2; k += 2) { + let re = 0, im = 0; + for (let n = 0; n < N; n++) { const w = 0.5 - 0.5 * Math.cos((2 * Math.PI * n) / N), v = x[s + n] * w; re += v * Math.cos((2 * Math.PI * k * n) / N); im -= v * Math.sin((2 * Math.PI * k * n) / N); } + const mag = Math.hypot(re, im); num += mag * ((k * SR) / N); den += mag; + } + } + return den ? num / den : 0; +} + +export function renderStereo(file: string, pitches: number[][], onset: number[][], timbre: Timbre, { bpm = 60, octave = 0 } = {}) { + const stepSec = 60 / bpm / 2, n = Math.ceil((pitches.length * stepSec + 4) * SR); + const L = new Float32Array(n), R = new Float32Array(n); + for (let v = 0; v < 4; v++) { + let t = 0; + while (t < pitches.length) { + const p = pitches[t][v]; + if (p < 0) { t++; continue; } + let len = 1; + while (t + len < pitches.length && pitches[t + len][v] === p && !onset[t + len][v]) len++; + note(L, R, timbre, p + 12 * octave, t * stepSec, len * stepSec, v); + t += len; + } + } + reverb(L, R); + lowpass(L, 3200); lowpass(R, 3200); + let peak = 0; for (let i = 0; i < n; i++) peak = Math.max(peak, Math.abs(L[i]), Math.abs(R[i])); + const g = (0.8 / (peak || 1)); + const buf = Buffer.alloc(44 + n * 4); + buf.write("RIFF", 0); buf.writeUInt32LE(36 + n * 4, 4); buf.write("WAVE", 8); buf.write("fmt ", 12); + buf.writeUInt32LE(16, 16); buf.writeUInt16LE(1, 20); buf.writeUInt16LE(2, 22); buf.writeUInt32LE(SR, 24); + buf.writeUInt32LE(SR * 4, 28); buf.writeUInt16LE(4, 32); buf.writeUInt16LE(16, 34); buf.write("data", 36); buf.writeUInt32LE(n * 4, 40); + for (let i = 0; i < n; i++) { buf.writeInt16LE(Math.round(L[i] * g * 32767), 44 + i * 4); buf.writeInt16LE(Math.round(R[i] * g * 32767), 46 + i * 4); } + fs.writeFileSync(file, buf); + const mono = new Float32Array(n); for (let i = 0; i < n; i++) mono[i] = (L[i] + R[i]) / 2; + return centroid(mono); +} diff --git a/docs/research/music-ai/train.ts.txt b/docs/research/music-ai/train.ts.txt new file mode 100644 index 0000000..e91604c --- /dev/null +++ b/docs/research/music-ai/train.ts.txt @@ -0,0 +1,88 @@ +// Train a small Transformer on the chorales with lib/ml (the same code the browser runs), sample at checkpoints. +// npx tsx train.ts [steps] [d] [layers] [ctxSteps] [batch] [lr] +import fs from "node:fs"; +import { Tape } from "../Blog/lib/ml/autograd"; +import { Adam, Transformer } from "../Blog/lib/ml/transformer"; +import { mulberry32 } from "../Blog/lib/ml/neuroevolution"; +import { LO, VOCAB, VOICES, decode, longestCopy, renderWav, report, trainTokens, valTokens } from "./music"; + +const [name = "run", stepsArg = "2000", dArg = "64", layersArg = "2", ctxStepsArg = "32", batchArg = "8", lrArg = "3e-3"] = process.argv.slice(2); +const STEPS = +stepsArg, D = +dArg, LAYERS = +layersArg, CTX = +ctxStepsArg * VOICES, BATCH = +batchArg, LR = +lrArg; +const dir = `runs/${name}`; +fs.mkdirSync(dir, { recursive: true }); +const log = (s: string) => { console.log(s); fs.appendFileSync(`${dir}/log.txt`, s + "\n"); }; + +const rng = mulberry32(7); +const model = new Transformer({ vocab: VOCAB, ctx: CTX, d: D, heads: 4, layers: LAYERS }, rng); +const adam = new Adam(model.params); +log(`params ${model.parameterCount()} ctx ${CTX} batch ${BATCH} lr ${LR}`); + +/** A random window that starts on a step boundary, so the voice order S A T B is always in phase. */ +function window(pool: number[][], r: () => number): number[] { + const seq = pool[Math.floor(r() * pool.length)]; + const steps = seq.length / VOICES, len = Math.min(CTX + VOICES, seq.length); + const start = Math.floor(r() * Math.max(1, steps - len / VOICES + 1)) * VOICES; + return seq.slice(start, start + len); +} + +function lossOn(ids: number[], tape: Tape, weight: number) { + const input = ids.slice(0, -1).slice(0, CTX), targets = ids.slice(1, CTX + 1); + return tape.crossEntropy(model.forward(tape, input).logits, targets, weight); +} + +/** Validation loss on fixed windows: 64 of them, the same every time. */ +const valRng = mulberry32(99), valWindows = Array.from({ length: 64 }, () => window(valTokens, valRng)); +function validate(): number { + let s = 0; + for (const w of valWindows) s += lossOn(w, new Tape(), 1); + return s / valWindows.length; +} + +function sample(steps: number, temperature: number, seed: number): number[] { + const r = mulberry32(seed); + const ids = [72 - LO, 67 - LO, 64 - LO, 48 - LO]; // a C major chord to start from + while (ids.length < steps * VOICES) { + // Slide by whole steps: position 0 must stay the soprano, or every voice reads as its neighbour. + const ctx = ids.slice(Math.max(0, Math.ceil((ids.length - CTX) / VOICES) * VOICES)); + const { logits } = model.forward(new Tape(), ctx); + const row = logits.data.subarray((ctx.length - 1) * VOCAB, ctx.length * VOCAB); + let max = -Infinity; for (const x of row) max = Math.max(max, x); + const p = Array.from(row, (x) => Math.exp((x - max) / temperature)); + let u = r() * p.reduce((a, b) => a + b, 0), k = 0; + while (u > p[k] && k < p.length - 1) u -= p[k++]; + ids.push(k); + } + return ids; +} + +function checkpoint(step: number) { + const val = validate(); + const out: Record = { step, val: +val.toFixed(4) }; + for (const temp of [0.7, 1.0]) { + const rs = [1, 2, 3].map((seed) => decode(sample(64, temp, seed))); + const mean = (k: string) => +(rs.reduce((s, r) => s + (report(r.pitches, r.onset) as any)[k], 0) / rs.length).toFixed(3); + out[`T${temp}`] = { inKey: mean("inKey"), parallels: mean("parallels"), crossings: mean("crossings"), rough: mean("rough"), holdShare: mean("holdShare"), chords: mean("distinctChords"), copy: Math.max(...rs.map((r) => longestCopy(r.pitches))) }; + renderWav(`${dir}/step${step}-T${temp}.wav`, rs[0].pitches, rs[0].onset); + } + log(JSON.stringify(out)); + fs.writeFileSync(`${dir}/model-${step}.json`, JSON.stringify(Object.fromEntries(Object.entries(model.params).map(([k, m]) => [k, Array.from(m.data, (x) => +x.toFixed(5))])))); +} + +const checkpoints = new Set([0, 100, 250, 500, 1000, 2000, 4000, 8000, 16000].filter((s) => s <= STEPS).concat([STEPS])); +const start = Date.now(); +let running = 0; +for (let step = 0; step <= STEPS; step++) { + if (checkpoints.has(step)) { log(`t=${((Date.now() - start) / 1000).toFixed(0)}s`); checkpoint(step); } + if (step === STEPS) break; + model.zeroGrad(); + const tape = new Tape(); + let loss = 0; + for (let b = 0; b < BATCH; b++) loss += lossOn(window(trainTokens, rng), tape, 1 / BATCH) / BATCH; + tape.backward(); + // Short warm-up, then a cosine decay to a tenth. + const lr = LR * Math.min(1, (step + 1) / 100) * (0.1 + 0.9 * 0.5 * (1 + Math.cos((Math.PI * step) / STEPS))); + adam.step(lr); + running = step === 0 ? loss : 0.98 * running + 0.02 * loss; + if (step % 50 === 0) log(`step ${step} loss ${running.toFixed(3)} ${((Date.now() - start) / (step + 1)).toFixed(0)} ms/step`); +} +log(`done in ${((Date.now() - start) / 1000).toFixed(0)} s`); diff --git a/e2e/a11y.spec.ts b/e2e/a11y.spec.ts index 8594d4c..f65b32d 100644 --- a/e2e/a11y.spec.ts +++ b/e2e/a11y.spec.ts @@ -19,6 +19,8 @@ export const pages = [ "/en/posts/city-of-agents", "/zh/posts/task-scheduler", "/en/posts/task-scheduler", + "/zh/posts/music-ai", + "/en/posts/music-ai", ]; for (const theme of ["dark", "light"] as const) { diff --git a/e2e/music-ai.spec.ts b/e2e/music-ai.spec.ts new file mode 100644 index 0000000..e682333 --- /dev/null +++ b/e2e/music-ai.spec.ts @@ -0,0 +1,82 @@ +import AxeBuilder from "@axe-core/playwright"; +import { expect, test } from "@playwright/test"; + +/* + * № 014, music-ai. The figures make sound, which CI cannot hear; what is checked here is everything around it — + * that training really runs and produces a score, that a pick changes the model, and that the blind test keeps its + * answers to itself until the reader has guessed. + */ +const ZH = "/zh/posts/music-ai", EN = "/en/posts/music-ai"; + +test.beforeEach(async ({ request }) => { + test.skip((await request.get(ZH)).status() === 404, "music-ai is still a draft"); +}); + +test("fig. 01: training writes a score and measures it", async ({ page }) => { + // A hundred steps of a 65k-parameter Transformer in a worker: about 7 s alone, much more with the suite running. + test.setTimeout(180_000); + await page.goto(ZH); + const figure = page.locator('figure[data-instrument="composer / train"]'); + await figure.scrollIntoViewIfNeeded(); + await expect(figure.getByRole("alert")).toHaveCount(0); + await figure.getByTestId("music-train").click(); + // Step 0's noodling arrives before any training, then the first real snapshot. + await expect(figure.getByRole("button", { name: /第 0 步/ })).toBeVisible({ timeout: 60_000 }); + await expect(figure.getByRole("button", { name: /第 100 步/ })).toBeVisible({ timeout: 120_000 }); + // The roll is drawn from what it wrote: notes, not an empty box. + await expect(figure.locator("svg[role=img] rect").first()).toBeVisible(); + await expect(figure.getByTestId("music-report")).toContainText("%"); + await figure.getByRole("button", { name: "停止訓練" }).click(); + await expect(figure.getByTestId("music-train")).toBeVisible(); +}); + +test("fig. 02: a pick tunes the model and the numbers say what changed", async ({ page }) => { + test.setTimeout(180_000); + await page.goto(ZH); + const figure = page.locator('figure[data-instrument="composer / judge"]'); + await figure.scrollIntoViewIfNeeded(); + await figure.getByTestId("music-judge-load").click(); + await expect(figure.getByTestId("music-pick-0")).toBeEnabled({ timeout: 60_000 }); + await figure.getByTestId("music-pick-0").click(); + await expect(figure.getByTestId("music-judge-stats")).toContainText("→", { timeout: 60_000 }); + await expect(figure.getByText(/你已經選了 1 次/)).toBeVisible(); +}); + +test("fig. 03: the blind test hides the answers until all three are guessed", async ({ page }) => { + test.setTimeout(180_000); + await page.goto(ZH); + const figure = page.locator('figure[data-instrument="composer / blind"]'); + await figure.scrollIntoViewIfNeeded(); + await figure.getByTestId("music-blind-deal").click(); + await expect(figure.getByTestId("music-blind-play-2")).toBeVisible({ timeout: 60_000 }); + // Nothing on screen says who wrote what, and the rolls are not drawn yet ("對答案" is the button, "答案:" is the reveal). + await expect(figure.getByText(/答案:/)).toHaveCount(0); + await expect(figure.locator("svg[role=img]")).toHaveCount(0); + await expect(figure.getByTestId("music-blind-reveal")).toBeDisabled(); + // The guesses live in a fieldset each; the timbre picker is a group too, so ask for the fieldsets. + for (let i = 0; i < 3; i++) await figure.locator("fieldset").nth(i).getByRole("button", { name: "巴赫" }).click(); + await figure.getByTestId("music-blind-reveal").click(); + await expect(figure.getByTestId("music-blind-score")).toContainText("/ 3"); + await expect(figure.locator("svg[role=img]")).toHaveCount(3); +}); + +test("nothing heavy is fetched before the reader asks for it", async ({ page }) => { + const requests: string[] = []; + page.on("request", (r) => requests.push(r.url())); + await page.goto(ZH, { waitUntil: "networkidle" }); + // The chorales and the trained model arrive on Start training / Load, not with the article. + expect(requests.filter((url) => /music-ai\/\w+\.bin/.test(url))).toEqual([]); +}); + +for (const theme of ["dark", "light"] as const) { + for (const path of [ZH, EN]) { + test(`axe: ${path} (${theme})`, async ({ page }) => { + await page.addInitScript((value) => localStorage.setItem("theme", value), theme); + await page.goto(path); + await expect(page.locator("html")).toHaveClass(new RegExp(theme)); + await page.waitForTimeout(1500); + const { violations } = await new AxeBuilder({ page }).withTags(["wcag2a", "wcag2aa", "wcag21a", "wcag21aa"]).analyze(); + expect(violations.map((v) => `${v.id}: ${v.nodes.map((n) => n.target.join(" ")).join(" | ")}`)).toEqual([]); + }); + } +} diff --git a/e2e/smoke.spec.ts b/e2e/smoke.spec.ts index 6f216d2..57259cb 100644 --- a/e2e/smoke.spec.ts +++ b/e2e/smoke.spec.ts @@ -133,8 +133,8 @@ test("feeds and sitemap are served", async ({ request }) => { const sitemap = await (await request.get("/sitemap.xml")).text(); expect(sitemap).toContain("/zh/tags/from-scratch"); // A tag with one article is left out, and its page asks not to be indexed. - expect(sitemap).not.toContain("/zh/tags/llm"); - expect(await (await request.get("/zh/tags/llm")).text()).toContain('content="noindex, follow"'); + expect(sitemap).not.toContain("/zh/tags/multi-task"); + expect(await (await request.get("/zh/tags/multi-task")).text()).toContain('content="noindex, follow"'); }); // Next replaces `alternates` and `openGraph` wholesale when a page sets them, which once cost article diff --git a/public/posts/music-ai/chorales.bin b/public/posts/music-ai/chorales.bin new file mode 100644 index 0000000..00f0365 Binary files /dev/null and b/public/posts/music-ai/chorales.bin differ diff --git a/public/posts/music-ai/judge.bin b/public/posts/music-ai/judge.bin new file mode 100644 index 0000000..33e57cd Binary files /dev/null and b/public/posts/music-ai/judge.bin differ diff --git a/public/sw.js b/public/sw.js index 5936854..1c5bac9 100644 --- a/public/sw.js +++ b/public/sw.js @@ -97,5 +97,9 @@ self.addEventListener("fetch", (event) => { const { request } = event, url = new URL(request.url); // Only our own GETs. Vercel's analytics endpoints and anything cross-origin go straight to the network. if (request.method !== "GET" || url.origin !== self.location.origin || url.pathname.startsWith("/_vercel/")) return; + // A worker's own script goes straight to the network, never through a cache. The bundler hands a worker its chunk + // list in the URL's fragment (`…/turbopack-worker-…js#params=…`), and a fragment is not part of a Request: answering + // one from here loses it, and every worker on the site dies with "Missing worker bootstrap config". + if (request.destination === "worker" || request.destination === "sharedworker") return; event.respondWith(isAsset(url) ? cacheFirst(request) : networkFirst(request)); }); diff --git a/scripts/music/convert.py b/scripts/music/convert.py new file mode 100644 index 0000000..dc353ee --- /dev/null +++ b/scripts/music/convert.py @@ -0,0 +1,56 @@ +# Bach chorales (Craig Sapp's edition, CC BY-NC-SA 4.0) -> an eighth-note grid, one JSON file. +# Clone the corpus first; it is not part of this repo: +# git clone --depth 1 https://github.com/craigsapp/bach-370-chorales +# uv run --with music21 python scripts/music/convert.py [kern dir] [out.json] +# Each chorale is transposed to C major or A minor. A step holds, for S A T B, the MIDI pitch sounding (or -1), +# whether that note starts on this step, and whether a fermata (phrase end) is on it. +import glob, json, math, os, sys +from music21 import converter, interval, pitch, note, expressions + +STEP = 0.5 # quarter lengths: an eighth note +out, skipped = [], [] +KERN = sys.argv[1] if len(sys.argv) > 1 else "bach-370-chorales/kern" +OUT = sys.argv[2] if len(sys.argv) > 2 else "chorales.json" +for path in sorted(glob.glob(f"{KERN}/*.krn")): + name = os.path.basename(path)[:-4] + try: + score = converter.parse(path) + parts = list(score.parts) + if len(parts) != 4: + skipped.append((name, f"{len(parts)} parts")); continue + key = score.analyze("key") + tonic = "C" if key.mode == "major" else "A" + shift = interval.Interval(key.tonic, pitch.Pitch(tonic)) + # Keep transpositions within a tritone so voices stay in their ranges. + if shift.semitones > 6: shift = interval.Interval(shift.semitones - 12) + if shift.semitones < -6: shift = interval.Interval(shift.semitones + 12) + score = score.transpose(shift) + parts = list(score.parts) + # kern lists bass first; order voices top-down by mean pitch to be sure. + def mean(p): + ps = [n.pitch.midi for n in p.recurse().notes if n.isNote] + return sum(ps) / max(1, len(ps)) + parts.sort(key=mean, reverse=True) + length = max(p.highestTime for p in parts) + n = int(math.ceil(length / STEP)) + pitches = [[-1] * 4 for _ in range(n)] + onset = [[0] * 4 for _ in range(n)] + fermata = [0] * n + for v, p in enumerate(parts): + for el in p.flatten().notesAndRests: + if not el.isNote: continue + a = int(round(el.offset / STEP)); b = int(round((el.offset + el.quarterLength) / STEP)) + tied_in = el.tie is not None and el.tie.type in ("continue", "stop") + for t in range(a, min(b, n)): + pitches[t][v] = el.pitch.midi + onset[t][v] = 1 if (t == a and not tied_in) else 0 + if any(isinstance(e, expressions.Fermata) for e in el.expressions) and b - 1 < n: + fermata[max(a, b - 1)] = 1 + ts = score.recurse().getElementsByClass("TimeSignature") + out.append({"id": name, "mode": key.mode, "shift": shift.semitones, "meter": ts[0].ratioString if ts else "4/4", + "pitches": pitches, "onset": onset, "fermata": fermata}) + except Exception as e: # report and go on + skipped.append((name, repr(e)[:80])) +json.dump(out, open(OUT, "w")) +print(len(out), "chorales,", len(skipped), "skipped") +for s in skipped[:20]: print(" skipped", *s) diff --git a/scripts/music/pack.ts b/scripts/music/pack.ts new file mode 100644 index 0000000..cb127ef --- /dev/null +++ b/scripts/music/pack.ts @@ -0,0 +1,56 @@ +// Article 014's data files, from the chorales convert.py writes and a checkpoint train.ts saves: +// +// npx tsx scripts/music/pack.ts +// +// public/posts/music-ai/chorales.bin u16 train count, u16 validation count, u16 length per chorale, then one +// byte per token (training chorales first). Bach, Craig Sapp's edition, +// CC BY-NC-SA 4.0 — so is this file. +// public/posts/music-ai/judge.bin float32 parameters in the Transformer's own order, for figures 02 and 03. +// +// The corpus, the research scripts and the runs live outside this repo (docs/research/music-ai explains where). +import fs from "node:fs"; +import path from "node:path"; +import { Transformer, mulberry32 } from "@/lib/ml"; +import { CONFIG, HOLD, LO, REST } from "@/content/posts/music-ai/components/music"; + +interface Chorale { pitches: number[][]; onset: number[][] } + +const [source, checkpoint] = process.argv.slice(2); +if (!source || !checkpoint) throw new Error("usage: pack.ts "); +const out = path.join(process.cwd(), "public/posts/music-ai"); +fs.mkdirSync(out, { recursive: true }); + +/** One token per voice per eighth: a new note is its pitch, a held one HOLD, silence REST. */ +function tokens(c: Chorale): number[] { + const ids: number[] = []; + c.pitches.forEach((step, t) => step.forEach((p, v) => ids.push(p < 0 ? REST : c.onset[t][v] || t === 0 ? p - LO : HOLD))); + return ids; +} + +// The same seeded split the research used: 90 % to train on, 10 % kept back. +const chorales: Chorale[] = JSON.parse(fs.readFileSync(source, "utf8")); +const rng = mulberry32(2026); +const order = chorales.map((_, i) => i).sort(() => rng() - 0.5); +const nVal = Math.round(chorales.length * 0.1); +const val = order.slice(0, nVal).map((i) => tokens(chorales[i])); +const train = order.slice(nVal).map((i) => tokens(chorales[i])); + +const all = [...train, ...val]; +const bytes = all.reduce((n, t) => n + t.length, 0); +const buffer = Buffer.alloc(4 + 2 * all.length + bytes); +buffer.writeUInt16LE(train.length, 0); +buffer.writeUInt16LE(val.length, 2); +all.forEach((t, i) => buffer.writeUInt16LE(t.length, 4 + 2 * i)); +let at = 4 + 2 * all.length; +for (const t of all) for (const id of t) buffer[at++] = id; +fs.writeFileSync(`${out}/chorales.bin`, buffer); + +// The checkpoint is a map of parameter name to numbers; the page reads them in the Transformer's own order. +const saved: Record = JSON.parse(fs.readFileSync(checkpoint, "utf8")); +const model = new Transformer(CONFIG); +const weights = new Float32Array(model.parameterCount()); +let k = 0; +for (const name of Object.keys(model.params)) for (const x of saved[name]) weights[k++] = x; +fs.writeFileSync(`${out}/judge.bin`, Buffer.from(weights.buffer)); + +console.log(`${train.length} + ${val.length} chorales, ${bytes} tokens, ${buffer.length} bytes; ${weights.length} parameters, ${weights.byteLength} bytes`);