Which Workers AI model is the LLM: the cheapest one that writes a correct Jev tool call for every case.
Sends the Worker's own request (llm::request) to each candidate over
Cloudflare's REST API and reads the reply with the Worker's own parser
(llm::parse), so a pass here is a pass there. Candidates are tried
cheapest first and the run stops at the first that passes every case.
lmjtfy-eval # stop at the first model that passes
lmjtfy-eval --all # run every candidate
lmjtfy-eval <model> # run one model, verbosely
op-env-run -- lmjtfy-eval facts # the questions Jev is asked first (facts.rs)
lmjtfy-eval is the devshell's wrapper: cargo run -p eval with the
account and the token's 1Password reference in its environment.
It spends real neurons from the account's daily allocation: about 15 requests a model. The token comes from 1Password per run and is never printed.
An input, and the sets of tool calls that would answer it. Most inputs have one right shape; a few are fairly read two ways.
44const CASES: &[Case] = &[ 45 Case { input: "is water wet?", accept: &[&[Noul]] }, 46 Case { input: "should I rewrite it in rust", accept: &[&[Noul]] }, 47 Case { input: "can penguins fly", accept: &[&[Noul]] }, 48 Case { input: "is a hot dog a sandwich?", accept: &[&[Noul]] }, 49 Case { input: "what is the best text editor", accept: &[&[Choice]] }, 50 Case { input: "which programming language should I learn first?", accept: &[&[Choice]] }, 51 Case { input: "who would win in a fight, a bear or a shark", accept: &[&[Choice]] }, 52 Case { input: "best pizza topping", accept: &[&[Choice]] }, 53 Case { input: "how good is the movie The Matrix?", accept: &[&[Score]] }, 54 // A Noul's probability is itself "how likely". 55 Case { input: "how likely is it to rain in Seattle in November", accept: &[&[Noul]] }, 56 Case { input: "rate my startup idea: uber for dogs", accept: &[&[Score]] }, 57 Case { input: "how spicy is a jalapeño", accept: &[&[Score]] }, 58 // Quotes and symbols have to survive the model's JSON. 59 Case { input: "is \"C++\" better than C#?", accept: &[&[Noul], &[Choice]] }, 60 // Two separate questions in one input. 61 Case { 62 input: "is python slower than rust, and which of the two is better for a beginner?", 63 accept: &[&[Noul, Choice], &[Noul, Noul]], 64 }, 65]; 66 67fn kind(draft: &Draft) -> Kind { 68 match draft { 69 Draft::Noul { .. } => Noul, 70 Draft::Choice { .. } => Choice, 71 Draft::Score { .. } => Score, 72 } 73}
Why a reply does not answer its case, or the neurons it cost.
76fn judge(case: &Case, body: &str) -> (Result<(), String>, llm::Usage) { 77 let reply = match llm::parse(body) { 78 Ok(reply) => reply, 79 Err(error) => return (Err(error), llm::Usage::default()), 80 }; 81 let verdict = (|| { 82 if reply.calls.is_empty() { 83 return Err("made no tool call".to_owned()); 84 } 85 if reply.dropped > 0 { 86 return Err(format!("made {} calls, over the cap", reply.calls.len() + reply.dropped)); 87 } 88 let mut kinds = Vec::new(); 89 for call in &reply.calls { 90 let draft = call.draft.as_ref().map_err(|e| format!("{}: {e}", call.name))?; 91 // The last word is Jev's own protocol. 92 ask::check(draft).map_err(|e| format!("{}: {e}", call.name))?; 93 kinds.push(kind(draft)); 94 } 95 kinds.sort(); 96 let accepted = case.accept.iter().any(|want| { 97 let mut want = want.to_vec(); 98 want.sort(); 99 want == kinds 100 }); 101 if accepted { Ok(()) } else { Err(format!("called {kinds:?}, wanted one of {:?}", case.accept)) } 102 })(); 103 (verdict, reply.usage) 104}
106struct Cloudflare { 107 agent: ureq::Agent, 108 account: String, 109 token: String, 110} 111 112impl Cloudflare { 113 fn from_env() -> Result<Self, String> { 114 let var = |name: &str| std::env::var(name).map_err(|_| format!("{name} is not set; run `lmjtfy-eval` from the devshell")); 115 let account = var("CLOUDFLARE_ACCOUNT_ID")?; 116 let reference = format!("{}/Token", var("LMJTFY_OP_ITEM_REF")?); 117 // op.exe under WSL (it has the desktop app's session), op elsewhere. 118 let op = if Command::new("op.exe").arg("--version").output().is_ok() { "op.exe" } else { "op" }; 119 let read = Command::new(op).args(["read", &reference]).output().map_err(|e| format!("{op}: {e}"))?; 120 let token = String::from_utf8_lossy(&read.stdout).trim().to_owned(); 121 if !read.status.success() || token.is_empty() { 122 return Err(format!("{op} read gave no token")); 123 } 124 let agent = ureq::Agent::config_builder() 125 .http_status_as_error(false) 126 .timeout_global(Some(Duration::from_secs(120))) 127 .build() 128 .into(); 129 Ok(Cloudflare { agent, account, token }) 130 } 131 132 fn run(&self, model: &str, request: &str) -> Result<String, String> { 133 let url = format!("https://api.cloudflare.com/client/v4/accounts/{}/ai/run/{model}", self.account); 134 let mut response = self 135 .agent 136 .post(&url) 137 .header("authorization", &format!("Bearer {}", self.token)) 138 .header("content-type", "application/json") 139 .send(request) 140 .map_err(|e| e.to_string())?; 141 response.body_mut().read_to_string().map_err(|e| e.to_string()) 142 } 143} 144 145struct Scored { 146 passed: usize, 147 neurons: f64, 148 took: Duration, 149} 150 151fn score(cloudflare: &Cloudflare, model: &Model, verbose: bool) -> Scored { 152 let mut scored = Scored { passed: 0, neurons: 0.0, took: Duration::ZERO }; 153 for case in CASES { 154 let started = Instant::now(); 155 let (verdict, usage) = match cloudflare.run(model.id, &llm::request(case.input, &[])) { 156 Ok(body) => { 157 if verbose { 158 println!(" {body}"); 159 } 160 judge(case, &body) 161 } 162 Err(error) => (Err(error), llm::Usage::default()), 163 }; 164 scored.took += started.elapsed(); 165 scored.neurons += model.neurons(usage); 166 match verdict { 167 Ok(()) => { 168 scored.passed += 1; 169 println!(" pass {}", case.input); 170 } 171 Err(why) => println!(" FAIL {}\n {why}", case.input), 172 } 173 } 174 scored 175} 176 177fn main() -> ExitCode { 178 let args: Vec<String> = std::env::args().skip(1).collect(); 179 if args.first().is_some_and(|arg| arg == "facts") { 180 return facts::run(); 181 } 182 let all = args.iter().any(|arg| arg == "--all"); 183 let named: Vec<&Model> = args.iter().filter_map(|arg| Model::find(arg)).collect(); 184 if let Some(unknown) = args.iter().find(|arg| *arg != "--all" && Model::find(arg).is_none()) { 185 eprintln!("eval: {unknown} is not a candidate (see packages/llm/src/models.rs)"); 186 return ExitCode::FAILURE; 187 } 188 let cloudflare = match Cloudflare::from_env() { 189 Ok(cloudflare) => cloudflare, 190 Err(error) => { 191 eprintln!("eval: {error}"); 192 return ExitCode::FAILURE; 193 } 194 }; 195 let models: Vec<&Model> = if named.is_empty() { CANDIDATES.iter().collect() } else { named.clone() }; 196 197 let mut table = Vec::new(); 198 let mut winner = None; 199 for model in models { 200 println!("\n{}", model.id); 201 let scored = score(&cloudflare, model, !named.is_empty()); 202 let perfect = scored.passed == CASES.len(); 203 table.push(format!( 204 "{:>2}/{} {:>7.1} neurons/call {:>5} ms/call {}", 205 scored.passed, 206 CASES.len(), 207 scored.neurons / CASES.len() as f64, 208 scored.took.as_millis() / CASES.len() as u128, 209 model.id 210 )); 211 if perfect && winner.is_none() { 212 winner = Some(model.id); 213 if !all && named.is_empty() { 214 break; 215 } 216 } 217 } 218 println!("\n{}", table.join("\n")); 219 match winner { 220 Some(id) => println!("\ncheapest model that passed every case: {id}"), 221 None => println!("\nno model passed every case"), 222 } 223 ExitCode::SUCCESS 224}