1export const meta = { 2 name: 'song-review', 3 description: 'Adversarially review one jevstrudel song from four angles, verify every finding, and optionally apply the fixes', 4 whenToUse: 5 'Before a new or reworked song reaches the user or a deploy. args: { song: "jev/<song>", apply: false, pitch: "the user\'s intent, if the spec may have drifted from it" }', 6 phases: [ 7 { title: 'Scout', detail: 'measure the song: map, clip lengths, masks, a headless play' }, 8 { title: 'Critique', detail: 'four adversarial lenses in parallel' }, 9 { title: 'Verify', detail: 'a skeptic per lens tries to refute each finding' }, 10 { title: 'Plan', detail: 'dedupe and rank what survived' }, 11 { title: 'Apply', detail: 'spec first, one worktree, then re-check (apply: true only)' }, 12 ], 13} 14 15// Usage: Workflow({ name: 'song-review', args: { song: 'jev/eleven-thousand-volumes' } }) 16// Returns { facts, findings, plan, applied? }. With apply: true, the fixes land 17// on a worktree branch that `applied` names; the caller merges it, plays 18// the song in a silent tab, and lets the user listen before any deploy. 19// 20// Every lesson below was paid for once (2026-09-24/25, songs/CLAUDE.md holds 21// the durable ones): keep this list in step with it. 22 23const song = args?.song 24if (!song || !/^[a-z0-9-]+\/[a-z0-9-]+$/.test(song)) throw new Error('args.song must be "<theme>/<song>", e.g. "jev/lightning-in-a-bottle"') 25const dir = `/home/nixos/strudel/songs/${song}` 26// shared: false (several reviews at once): the apply step edits only this 27// song and its samples, and returns the shared changes it wanted instead. 28const shared = args?.shared !== false 29const pitch = args?.pitch ? `\nThe user's intent, in their words: "${args.pitch}". Where the spec disagrees with it, the intent wins.` : '' 30 31const BASE = `You are reviewing the jevstrudel song in ${dir} (SPEC.md, song.js, README.md, CLAUDE.md), its samples (samples/<name>/ and their rows in /home/nixos/strudel/samples/README.md), and the site's Jev API (/home/nixos/strudel/website/src/jev/jevCore.mjs and its README). Read /home/nixos/strudel/songs/CLAUDE.md first: its rules are the house rules.${pitch} 32The song is only what the user asked for: never propose characters, films or franchises the pitch does not name, and never reintroduce anything the spec says the user withdrew. 33Measure rather than recall: run the code, transcribe the audio, fetch the source. Transcribe only with 'nix run .#transcribe -- FILE...', which prints each file as small.en and medium.en hear it and runs one transcription at a time on this machine; never load Whisper yourself (openai-whisper, a whisper-cpp model of your own, python): several reviews each loading one ran the host out of memory. A claim with no evidence is not a finding.\nSeveral reviews may be playing songs at once on the same dev server: a rate limit or backoff during a headless play is test noise, not a finding about this song.\nTo wait for a long command (measure, check-songs, whisper), run it in the background and wait for its completion notice, or run it in the foreground with a timeout. Never poll with 'pgrep -f <pattern>': the polling shell's own command line contains the pattern, so the loop matches itself and never ends (two review agents hung this way, 2026-09-25).` 34 35const FINDINGS = { 36 type: 'object', 37 properties: { 38 findings: { 39 type: 'array', 40 items: { 41 type: 'object', 42 properties: { 43 title: { type: 'string' }, 44 severity: { type: 'string', enum: ['high', 'medium', 'low'] }, 45 file: { type: 'string' }, 46 line: { type: 'number' }, 47 evidence: { type: 'string', description: 'what you ran, read or heard that shows it, with sources graded authoritative/secondary/anecdotal' }, 48 fix: { type: 'string', description: 'concrete: the spec change first, then the code change' }, 49 }, 50 required: ['title', 'severity', 'evidence', 'fix'], 51 }, 52 }, 53 }, 54 required: ['findings'], 55} 56 57const VERDICTS = { 58 type: 'object', 59 properties: { 60 verdicts: { 61 type: 'array', 62 items: { 63 type: 'object', 64 properties: { 65 title: { type: 'string' }, 66 real: { type: 'boolean' }, 67 why: { type: 'string' }, 68 }, 69 required: ['title', 'real', 'why'], 70 }, 71 }, 72 }, 73 required: ['verdicts'], 74} 75 76const LENSES = [ 77 { 78 key: 'concept', 79 prompt: `Critique it as CONCEPT and COMEDY, and fact-check it. 80- Does the joke land for a LISTENER? A joke that lives only in SPEC prose does not count: it must be audible. 81- Does every character have a distinct voice? One TTS voice playing two roles (Jev and the store, say) blurs the joke. 82- Does the song say something true about Jev? Jev is TypeSafe's System One: a fast classifier answering choice/score/yes-no questions, never an agent (docs: https://docs.typesafe.ai/llms-full.txt; a saved copy may be in the session scratchpad). A song that has Jev "do" things an agent does is wrong unless mocking that misuse. 83- Does the premise contradict itself (e.g. "too fast" while the song's joke is that everything is slow)? 84- Fact-check every quote, speaker, episode, film and technical claim against real sources. Transcribe the actual sample files with 'nix run .#transcribe -- FILE...' (from the repo root) and compare with what the spec says each one says and who says it. 85- A sung name must survive two whisper models (small.en and medium.en), per songs/CLAUDE.md: a line whose name comes back as Jeff or Jeb is a finding. 86- The brief is the user's words with typos fixed and nothing added; the spec carries no session history (earlier attempts, who got what wrong).`, 87 }, 88 { 89 key: 'music', 90 prompt: `Critique it as MUSIC and MIX, as a producer would. 91- Is it the genre the spec claims (rhythm, harmony, instrumentation), and idiomatic? Is it a genre the set already has too much of (count the other songs' tempos and styles)? 92- Does the form build: do the drops lift, does anything repeat without development? 93- Vocal clips: 1 bar = 240/bpm seconds. Measure every clip (ffprobe) and check none overlaps another line, and none sits under a hit (crash, thunder, ending chord) that buries it. 94- Pitfalls paid for already: Dirt-Samples names lie (cr is a ride; hh27:1 is the crash, 15 dB quieter; list a folder with gh api before trusting it); sub-bass under ~45 Hz vanishes on laptops; .segment(n) retriggers oscillators (a chopped "pad"); a build's ramps need .slow(n) to rise across the build instead of restarting every bar; clicks on the downbeat are masked by the kick. 95- Loudness: samples near -16 LUFS for voices; anything clipping (true peak > 0 dBTP) or buried. The MIX must peak at or under -1 dBFS as measured (nix run .#measure): superdough's output goes straight to the speakers with no limiter, so a measured peak over 0 dBFS clips, and the art critic rightly scores craft near zero for it. Fix it with a master level on the song's final stack, .mul(postgain(x)), and re-measure. The measure tool (nix run .#measure -- --song ${song}) needs the dev server. 96- Speech must be understood: a stutter, click or arpeggio layer in the speech band ducks under every line (mask it out for the line's length); beeps and chirps land in the gaps, not on words. 97- Can Jev's per-part levels or transitions wreck it (drums out in a drop, a fill or riser over a long clip, a riser doubling a build)? The lines the story needs (the punchline, the turning point) belong outside part(), like the ending, so Jev cannot mute them; reactions and colour can stay under Jev. 98- A repeated gag stops being funny: count how often each joke sample plays (a stutter, a catchphrase) and whether it builds, rests and pays off, or just runs wall to wall. Ration it; save the densest use for the climax. 99- An acceleration or build must not fall back at the drop; a "door slam" or hit the spec promises must actually sound; no dead air before the ending that sounds like the song is over. 100- Harmony: voice leading that makes the progression audible (a sequence's top line should move), no doubled leading tones, a lead line that is more than doubled chord tones.`, 101 }, 102 { 103 key: 'code', 104 prompt: `Review the CODE and the FORM, and measure: evaluate the song headlessly in Node the way website/src/jev/allSongs.test.mjs does (JEV_SONGS_DIR can point at a scratch copy under the session scratchpad), and drive several paths with a stand-in Jev (offline; the written order with every transition a dropout; a loop to the cap; late detours). 105- Every mask sums to the written length; every line starts in the section the spec says. 106- walk(): every section reachable, loops end, and the form never forces an ending over a running vocal. 107- Every loop within one kind of section (a next back inside a groove or a drop, or a section followed by itself) has a \`times\` on the section it returns to, per songs/CLAUDE.md. Drive a stand-in Jev that always takes the way back and report the longest run of one kind of section it gets (in bars): without \`times\`, borrow-checker-blues looped a groove for 44 bars in review and could reach 56, select-star 80 bars of keyUp. 108- An \`allow\` rule must handle the opening, where there is no section before (\`playingNow\` is undefined): borrow-checker-blues' phrase(playingNow) threw there, and the opening's whole arrangement fell back silently until a headless check showed it (allSongs.test.mjs now fails on it). 109- The dropout gap is audible (held sounds and voices ring through it: say so or stop them); the ending's last beat survives; nothing plays after the ending; .early(1) transitions after the last section yield nothing. 110- Levels and energy multiply velocity without overriding a part's own; each part has its own orbit (the meter reads parts by orbit); effects on one orbit don't conflict (one reverb size per orbit). 111- Every question text Jev reads describes what actually plays (a "closing filter" that opens, "rising" ramps that restart each bar are lies Jev decides on). 112- House rules: Jev text in single quotes; .add(note(n)), never a bare number; no "e-v-a-l" substring.`, 113 }, 114 { 115 key: 'jev', 116 prompt: `Review how the song USES JEV, and what it costs. 117- Is each question one TypeSafe would endorse (independent questions in one request; dependent ones asked after; control flow in code)? Are fallbacks the written order? 118- Is sample: temperature ever combined with minConfidence (refused by .every(); TypeSafe: use probabilities for a statistical algorithm)? 119- Rate: compute calls per minute from setcps, the .every(n) cadence and the number of jev() requests, against the relay's per-visitor limit (worker/wrangler.json); the busiest song must stay under half of it. 120- State: does what Jev is told (song state, section descriptions, measuredSound) give it what it needs to decide well, and nothing false? 121- Run it for real if the dev server is up: nix develop -c buck2 run //:check-songs -- ${song.split('/')[1]} and read the path Jev took: late answers, fallbacks, sections Jev never picks, parts it always silences.`, 122 }, 123] 124 125phase('Scout') 126const facts = await agent( 127 `${BASE} 128Produce the FACTS every critic will share, measured, not recalled: 1291. The section map as written (bars, 1-based) and the walk() rules. 1302. Every vocal/sample line: its sample, bar and beat, and the clip's measured length in seconds and bars (ffprobe on samples/<name>/<name>.mp3; 1 bar = 240/bpm s). 1313. Every mask's sum. 1324. The jev() questions as Jev reads them. 1335. Every sample's level: integrated LUFS and true peak (ffmpeg ebur128=peak=true), the gain the song applies to it, and the level as heard (LUFS + 20*log10(gain)). A clip under 0.4 s reads -70 LUFS: measure it looped ten times instead. Flag voices more than 2 dB from the song's other voices, anything over -1 dBTP, and film clips raised by more than 15 dB (their background comes up with them). 1346. If the dev server answers on http://localhost:4322, the mix as heard: nix run .#measure -- --song ${song} (per-part loudness; voices buried under the band show here), and a real headless play: nix develop -c buck2 run //:check-songs -- ${song.split('/')[1]} (report the path, late answers, fallbacks, warnings verbatim). If it is down, say so; do not start it. 135Return plain text, compact, tables where they fit.`, 136 { label: 'scout', phase: 'Scout', effort: 'medium' }, 137) 138 139const verified = await pipeline( 140 LENSES, 141 (lens) => 142 agent(`${BASE}\n\nShared facts, measured by a scout:\n${facts}\n\nYour lens:\n${lens.prompt}\n\nBe harsh and specific; rank findings; every one needs evidence and a concrete fix. Read-only: do not edit the repo.`, { 143 label: `critique:${lens.key}`, 144 phase: 'Critique', 145 schema: FINDINGS, 146 }), 147 (found, lens) => { 148 if (!found?.findings?.length) return { lens: lens.key, findings: [] } 149 return agent( 150 `${BASE}\n\nA critic (lens: ${lens.key}) claims these findings. Try to REFUTE each one: reproduce it yourself (run the code, measure the clip, transcribe the audio, open the source). A finding is real only if you can reproduce it; default to real=false when you cannot.\n\n${JSON.stringify(found.findings, null, 2)}`, 151 { label: `verify:${lens.key}`, phase: 'Verify', schema: VERDICTS }, 152 ).then((v) => { 153 const real = new Set((v?.verdicts ?? []).filter((x) => x.real).map((x) => x.title)) 154 const dropped = found.findings.filter((f) => !real.has(f.title)) 155 if (dropped.length) log(`${lens.key}: ${dropped.length} finding(s) refuted: ${dropped.map((f) => f.title).join('; ')}`) 156 return { lens: lens.key, findings: found.findings.filter((f) => real.has(f.title)) } 157 }) 158 }, 159) 160 161const findings = verified.filter(Boolean).flatMap((r) => r.findings.map((f) => ({ ...f, lens: r.lens }))) 162log(`${findings.length} verified finding(s)`) 163if (!findings.length) return { facts, findings, plan: 'nothing survived verification' } 164 165phase('Plan') 166const plan = await agent( 167 `${BASE}\n\nThese findings survived adversarial verification:\n${JSON.stringify(findings, null, 2)}\n\nMerge duplicates across lenses, resolve conflicts (say which fix wins and why), and rank by what a listener would notice first. Separate: (a) fixes to make now, spec first; (b) findings that need the user's decision (taste, scope, anything the pitch does not settle): state each as one sentence with a recommendation. Return plain text.`, 168 { label: 'plan', phase: 'Plan' }, 169) 170 171if (!args?.apply) return { facts, findings, plan } 172 173phase('Apply') 174const applied = await agent( 175 `${BASE}\n\nApply part (a) of this plan, and nothing from part (b):\n${plan}\n\n${shared ? '' : `Other reviews run at the same time: edit ONLY ${dir}/ and this song's own sample folders (samples/<prefix>*/) and their rows in samples/README.md. Do not edit website/, worker/, tools/, songs/CLAUDE.md or any other song. If a fix needs a shared change, do not make it: list it under a heading SHARED CHANGES WANTED in your reply, with the file, the change and why.\n\n`}House rules for the change: spec first, in the same commit; a new revisions entry in SPEC.md (prompt: a clean brief of what changed about the music, no session talk); samples/README.md rows for any sample touched. Then prove it: node --check a .mjs copy; the frontmatter parses (nix shell --impure --expr '(import (builtins.getFlake "nixpkgs") {}).python3.withPackages (p: [ p.pyyaml ])' -c python3); run website/src/jev/allSongs.test.mjs from /home/nixos/strudel with JEV_SONGS_DIR pointed at your worktree's songs. Commit with a quoted-heredoc message. Do not deploy, do not play in the user's tab. Return the branch name, the commits, and what you could not verify.`, 176 { label: 'apply', phase: 'Apply', isolation: 'worktree' }, 177) 178return { facts, findings, plan, applied }