jevstrudel.git / .claude / workflows / song-review.js
1export const meta = {
2  name: 'song-review',
3  description: 'Adversarially review one jevstrudel song from four angles, verify every finding, and optionally apply the fixes',
4  whenToUse:
5    'Before a new or reworked song reaches the user or a deploy. args: { song: "jev/<song>", apply: false, pitch: "the user\'s intent, if the spec may have drifted from it" }',
6  phases: [
7    { title: 'Scout', detail: 'measure the song: map, clip lengths, masks, a headless play' },
8    { title: 'Critique', detail: 'four adversarial lenses in parallel' },
9    { title: 'Verify', detail: 'a skeptic per lens tries to refute each finding' },
10    { title: 'Plan', detail: 'dedupe and rank what survived' },
11    { title: 'Apply', detail: 'spec first, one worktree, then re-check (apply: true only)' },
12  ],
13}
14
15// Usage: Workflow({ name: 'song-review', args: { song: 'jev/eleven-thousand-volumes' } })
16// Returns { facts, findings, plan, applied? }. With apply: true, the fixes land
17// on a worktree branch that `applied` names; the caller merges it, plays
18// the song in a silent tab, and lets the user listen before any deploy.
19//
20// Every lesson below was paid for once (2026-09-24/25, songs/CLAUDE.md holds
21// the durable ones): keep this list in step with it.
22
23const song = args?.song
24if (!song || !/^[a-z0-9-]+\/[a-z0-9-]+$/.test(song)) throw new Error('args.song must be "<theme>/<song>", e.g. "jev/lightning-in-a-bottle"')
25const dir = `/home/nixos/strudel/songs/${song}`
26// shared: false (several reviews at once): the apply step edits only this
27// song and its samples, and returns the shared changes it wanted instead.
28const shared = args?.shared !== false
29const pitch = args?.pitch ? `\nThe user's intent, in their words: "${args.pitch}". Where the spec disagrees with it, the intent wins.` : ''
30
31const BASE = `You are reviewing the jevstrudel song in ${dir} (SPEC.md, song.js, README.md, CLAUDE.md), its samples (samples/<name>/ and their rows in /home/nixos/strudel/samples/README.md), and the site's Jev API (/home/nixos/strudel/website/src/jev/jevCore.mjs and its README). Read /home/nixos/strudel/songs/CLAUDE.md first: its rules are the house rules.${pitch}
32The song is only what the user asked for: never propose characters, films or franchises the pitch does not name, and never reintroduce anything the spec says the user withdrew.
33Measure rather than recall: run the code, transcribe the audio, fetch the source. Transcribe only with 'nix run .#transcribe -- FILE...', which prints each file as small.en and medium.en hear it and runs one transcription at a time on this machine; never load Whisper yourself (openai-whisper, a whisper-cpp model of your own, python): several reviews each loading one ran the host out of memory. A claim with no evidence is not a finding.\nSeveral reviews may be playing songs at once on the same dev server: a rate limit or backoff during a headless play is test noise, not a finding about this song.\nTo wait for a long command (measure, check-songs, whisper), run it in the background and wait for its completion notice, or run it in the foreground with a timeout. Never poll with 'pgrep -f <pattern>': the polling shell's own command line contains the pattern, so the loop matches itself and never ends (two review agents hung this way, 2026-09-25).`
34
35const FINDINGS = {
36  type: 'object',
37  properties: {
38    findings: {
39      type: 'array',
40      items: {
41        type: 'object',
42        properties: {
43          title: { type: 'string' },
44          severity: { type: 'string', enum: ['high', 'medium', 'low'] },
45          file: { type: 'string' },
46          line: { type: 'number' },
47          evidence: { type: 'string', description: 'what you ran, read or heard that shows it, with sources graded authoritative/secondary/anecdotal' },
48          fix: { type: 'string', description: 'concrete: the spec change first, then the code change' },
49        },
50        required: ['title', 'severity', 'evidence', 'fix'],
51      },
52    },
53  },
54  required: ['findings'],
55}
56
57const VERDICTS = {
58  type: 'object',
59  properties: {
60    verdicts: {
61      type: 'array',
62      items: {
63        type: 'object',
64        properties: {
65          title: { type: 'string' },
66          real: { type: 'boolean' },
67          why: { type: 'string' },
68        },
69        required: ['title', 'real', 'why'],
70      },
71    },
72  },
73  required: ['verdicts'],
74}
75
76const LENSES = [
77  {
78    key: 'concept',
79    prompt: `Critique it as CONCEPT and COMEDY, and fact-check it.
80- Does the joke land for a LISTENER? A joke that lives only in SPEC prose does not count: it must be audible.
81- Does every character have a distinct voice? One TTS voice playing two roles (Jev and the store, say) blurs the joke.
82- Does the song say something true about Jev? Jev is TypeSafe's System One: a fast classifier answering choice/score/yes-no questions, never an agent (docs: https://docs.typesafe.ai/llms-full.txt; a saved copy may be in the session scratchpad). A song that has Jev "do" things an agent does is wrong unless mocking that misuse.
83- Does the premise contradict itself (e.g. "too fast" while the song's joke is that everything is slow)?
84- Fact-check every quote, speaker, episode, film and technical claim against real sources. Transcribe the actual sample files with 'nix run .#transcribe -- FILE...' (from the repo root) and compare with what the spec says each one says and who says it.
85- A sung name must survive two whisper models (small.en and medium.en), per songs/CLAUDE.md: a line whose name comes back as Jeff or Jeb is a finding.
86- The brief is the user's words with typos fixed and nothing added; the spec carries no session history (earlier attempts, who got what wrong).`,
87  },
88  {
89    key: 'music',
90    prompt: `Critique it as MUSIC and MIX, as a producer would.
91- Is it the genre the spec claims (rhythm, harmony, instrumentation), and idiomatic? Is it a genre the set already has too much of (count the other songs' tempos and styles)?
92- Does the form build: do the drops lift, does anything repeat without development?
93- Vocal clips: 1 bar = 240/bpm seconds. Measure every clip (ffprobe) and check none overlaps another line, and none sits under a hit (crash, thunder, ending chord) that buries it.
94- Pitfalls paid for already: Dirt-Samples names lie (cr is a ride; hh27:1 is the crash, 15 dB quieter; list a folder with gh api before trusting it); sub-bass under ~45 Hz vanishes on laptops; .segment(n) retriggers oscillators (a chopped "pad"); a build's ramps need .slow(n) to rise across the build instead of restarting every bar; clicks on the downbeat are masked by the kick.
95- Loudness: samples near -16 LUFS for voices; anything clipping (true peak > 0 dBTP) or buried. The MIX must peak at or under -1 dBFS as measured (nix run .#measure): superdough's output goes straight to the speakers with no limiter, so a measured peak over 0 dBFS clips, and the art critic rightly scores craft near zero for it. Fix it with a master level on the song's final stack, .mul(postgain(x)), and re-measure. The measure tool (nix run .#measure -- --song ${song}) needs the dev server.
96- Speech must be understood: a stutter, click or arpeggio layer in the speech band ducks under every line (mask it out for the line's length); beeps and chirps land in the gaps, not on words.
97- Can Jev's per-part levels or transitions wreck it (drums out in a drop, a fill or riser over a long clip, a riser doubling a build)? The lines the story needs (the punchline, the turning point) belong outside part(), like the ending, so Jev cannot mute them; reactions and colour can stay under Jev.
98- A repeated gag stops being funny: count how often each joke sample plays (a stutter, a catchphrase) and whether it builds, rests and pays off, or just runs wall to wall. Ration it; save the densest use for the climax.
99- An acceleration or build must not fall back at the drop; a "door slam" or hit the spec promises must actually sound; no dead air before the ending that sounds like the song is over.
100- Harmony: voice leading that makes the progression audible (a sequence's top line should move), no doubled leading tones, a lead line that is more than doubled chord tones.`,
101  },
102  {
103    key: 'code',
104    prompt: `Review the CODE and the FORM, and measure: evaluate the song headlessly in Node the way website/src/jev/allSongs.test.mjs does (JEV_SONGS_DIR can point at a scratch copy under the session scratchpad), and drive several paths with a stand-in Jev (offline; the written order with every transition a dropout; a loop to the cap; late detours).
105- Every mask sums to the written length; every line starts in the section the spec says.
106- walk(): every section reachable, loops end, and the form never forces an ending over a running vocal.
107- Every loop within one kind of section (a next back inside a groove or a drop, or a section followed by itself) has a \`times\` on the section it returns to, per songs/CLAUDE.md. Drive a stand-in Jev that always takes the way back and report the longest run of one kind of section it gets (in bars): without \`times\`, borrow-checker-blues looped a groove for 44 bars in review and could reach 56, select-star 80 bars of keyUp.
108- An \`allow\` rule must handle the opening, where there is no section before (\`playingNow\` is undefined): borrow-checker-blues' phrase(playingNow) threw there, and the opening's whole arrangement fell back silently until a headless check showed it (allSongs.test.mjs now fails on it).
109- The dropout gap is audible (held sounds and voices ring through it: say so or stop them); the ending's last beat survives; nothing plays after the ending; .early(1) transitions after the last section yield nothing.
110- Levels and energy multiply velocity without overriding a part's own; each part has its own orbit (the meter reads parts by orbit); effects on one orbit don't conflict (one reverb size per orbit).
111- Every question text Jev reads describes what actually plays (a "closing filter" that opens, "rising" ramps that restart each bar are lies Jev decides on).
112- House rules: Jev text in single quotes; .add(note(n)), never a bare number; no "e-v-a-l" substring.`,
113  },
114  {
115    key: 'jev',
116    prompt: `Review how the song USES JEV, and what it costs.
117- Is each question one TypeSafe would endorse (independent questions in one request; dependent ones asked after; control flow in code)? Are fallbacks the written order?
118- Is sample: temperature ever combined with minConfidence (refused by .every(); TypeSafe: use probabilities for a statistical algorithm)?
119- Rate: compute calls per minute from setcps, the .every(n) cadence and the number of jev() requests, against the relay's per-visitor limit (worker/wrangler.json); the busiest song must stay under half of it.
120- State: does what Jev is told (song state, section descriptions, measuredSound) give it what it needs to decide well, and nothing false?
121- Run it for real if the dev server is up: nix develop -c buck2 run //:check-songs -- ${song.split('/')[1]} and read the path Jev took: late answers, fallbacks, sections Jev never picks, parts it always silences.`,
122  },
123]
124
125phase('Scout')
126const facts = await agent(
127  `${BASE}
128Produce the FACTS every critic will share, measured, not recalled:
1291. The section map as written (bars, 1-based) and the walk() rules.
1302. Every vocal/sample line: its sample, bar and beat, and the clip's measured length in seconds and bars (ffprobe on samples/<name>/<name>.mp3; 1 bar = 240/bpm s).
1313. Every mask's sum.
1324. The jev() questions as Jev reads them.
1335. Every sample's level: integrated LUFS and true peak (ffmpeg ebur128=peak=true), the gain the song applies to it, and the level as heard (LUFS + 20*log10(gain)). A clip under 0.4 s reads -70 LUFS: measure it looped ten times instead. Flag voices more than 2 dB from the song's other voices, anything over -1 dBTP, and film clips raised by more than 15 dB (their background comes up with them).
1346. If the dev server answers on http://localhost:4322, the mix as heard: nix run .#measure -- --song ${song} (per-part loudness; voices buried under the band show here), and a real headless play: nix develop -c buck2 run //:check-songs -- ${song.split('/')[1]} (report the path, late answers, fallbacks, warnings verbatim). If it is down, say so; do not start it.
135Return plain text, compact, tables where they fit.`,
136  { label: 'scout', phase: 'Scout', effort: 'medium' },
137)
138
139const verified = await pipeline(
140  LENSES,
141  (lens) =>
142    agent(`${BASE}\n\nShared facts, measured by a scout:\n${facts}\n\nYour lens:\n${lens.prompt}\n\nBe harsh and specific; rank findings; every one needs evidence and a concrete fix. Read-only: do not edit the repo.`, {
143      label: `critique:${lens.key}`,
144      phase: 'Critique',
145      schema: FINDINGS,
146    }),
147  (found, lens) => {
148    if (!found?.findings?.length) return { lens: lens.key, findings: [] }
149    return agent(
150      `${BASE}\n\nA critic (lens: ${lens.key}) claims these findings. Try to REFUTE each one: reproduce it yourself (run the code, measure the clip, transcribe the audio, open the source). A finding is real only if you can reproduce it; default to real=false when you cannot.\n\n${JSON.stringify(found.findings, null, 2)}`,
151      { label: `verify:${lens.key}`, phase: 'Verify', schema: VERDICTS },
152    ).then((v) => {
153      const real = new Set((v?.verdicts ?? []).filter((x) => x.real).map((x) => x.title))
154      const dropped = found.findings.filter((f) => !real.has(f.title))
155      if (dropped.length) log(`${lens.key}: ${dropped.length} finding(s) refuted: ${dropped.map((f) => f.title).join('; ')}`)
156      return { lens: lens.key, findings: found.findings.filter((f) => real.has(f.title)) }
157    })
158  },
159)
160
161const findings = verified.filter(Boolean).flatMap((r) => r.findings.map((f) => ({ ...f, lens: r.lens })))
162log(`${findings.length} verified finding(s)`)
163if (!findings.length) return { facts, findings, plan: 'nothing survived verification' }
164
165phase('Plan')
166const plan = await agent(
167  `${BASE}\n\nThese findings survived adversarial verification:\n${JSON.stringify(findings, null, 2)}\n\nMerge duplicates across lenses, resolve conflicts (say which fix wins and why), and rank by what a listener would notice first. Separate: (a) fixes to make now, spec first; (b) findings that need the user's decision (taste, scope, anything the pitch does not settle): state each as one sentence with a recommendation. Return plain text.`,
168  { label: 'plan', phase: 'Plan' },
169)
170
171if (!args?.apply) return { facts, findings, plan }
172
173phase('Apply')
174const applied = await agent(
175  `${BASE}\n\nApply part (a) of this plan, and nothing from part (b):\n${plan}\n\n${shared ? '' : `Other reviews run at the same time: edit ONLY ${dir}/ and this song's own sample folders (samples/<prefix>*/) and their rows in samples/README.md. Do not edit website/, worker/, tools/, songs/CLAUDE.md or any other song. If a fix needs a shared change, do not make it: list it under a heading SHARED CHANGES WANTED in your reply, with the file, the change and why.\n\n`}House rules for the change: spec first, in the same commit; a new revisions entry in SPEC.md (prompt: a clean brief of what changed about the music, no session talk); samples/README.md rows for any sample touched. Then prove it: node --check a .mjs copy; the frontmatter parses (nix shell --impure --expr '(import (builtins.getFlake "nixpkgs") {}).python3.withPackages (p: [ p.pyyaml ])' -c python3); run website/src/jev/allSongs.test.mjs from /home/nixos/strudel with JEV_SONGS_DIR pointed at your worktree's songs. Commit with a quoted-heredoc message. Do not deploy, do not play in the user's tab. Return the branch name, the commits, and what you could not verify.`,
176  { label: 'apply', phase: 'Apply', isolation: 'worktree' },
177)
178return { facts, findings, plan, applied }