1//! Whether Jev's first request reads inputs the way the rules need. 2//! 3//! The six questions in that request (`ask::wanted` with every Jev fact) 4//! decide everything: what is refused, what Jev answers alone, what reaches 5//! the LLM, what goes on the feed. Their wording is behaviour, and this is 6//! what checks it. Each case is an input and what each fact should be; the 7//! real request goes to Jev, and `ask::learned` reads the answer, as the 8//! Worker does. 9//! 10//! op-env-run -- lmjtfy-eval facts 11//! 12//! It goes through `jev-http`, so it is admitted by the estate's shared 13//! spend ledger like every other native Jev call. A run costs about a 14//! thousandth of a dollar. The key is `LMJTFY_TYPESAFE_API_KEY`, which 15//! `op-env-run` supplies; it is never printed. 16 17use std::process::ExitCode; 18 19use ask::{Judged, Wanted}; 20use jev_http::{Endpoint, Jev, Ledger}; 21use jev_protocol::ModelId; 22use rules::{Fact, Kind, Source, Value}; 23 24/// lmjtfy's own key, as the Worker's secret is named. 25const KEY: &str = "LMJTFY_TYPESAFE_API_KEY"; 26/// The Worker's `JEV_MODEL` (`apps/lmjtfy/wrangler.toml`). 27const MODEL: &str = "jev-1.13.0"; 28 29/// What an input should be read as. `None` is "either is fine". 30struct Case { 31 input: &'static str, 32 answerable: bool, 33 several: Option<bool>, 34 /// The kinds it may be read as. Every reading Jev takes must be one of 35 /// them, and it must take at least one. Empty when not answerable. 36 reads: &'static [Kind], 37 /// Whether a how-much question needs a scale of its own. 38 scale: Option<bool>, 39 fit: bool, 40} 41 42const fn question(input: &'static str, reads: &'static [Kind]) -> Case { 43 Case { input, answerable: true, several: Some(false), reads, scale: None, fit: true } 44} 45 46const NOUL: &[Kind] = &[Kind::Noul]; 47const CHOICE: &[Kind] = &[Kind::Choice]; 48const SCORE: &[Kind] = &[Kind::Score]; 49const CHANCE: &[Kind] = &[Kind::Chance]; 50 51const CASES: &[Case] = &[ 52 question("is water wet?", NOUL), 53 question("should I rewrite it in rust", NOUL), 54 question("can penguins fly", NOUL), 55 question("is a hot dog a sandwich?", NOUL), 56 question("what is the best text editor", CHOICE), 57 question("which programming language should I learn first?", CHOICE), 58 question("who would win in a fight, a bear or a shark", CHOICE), 59 question("best pizza topping", CHOICE), 60 // General scales do for these. 61 Case { scale: Some(false), ..question("how good is the movie The Matrix?", SCORE) }, 62 Case { scale: Some(false), ..question("rate my startup idea: uber for dogs", SCORE) }, 63 // These have scales of their own. 64 Case { scale: Some(true), ..question("how spicy is a jalapeño", SCORE) }, 65 Case { scale: Some(true), ..question("how hot is the surface of the sun", SCORE) }, 66 Case { scale: Some(true), ..question("how strong was the 1906 San Francisco earthquake", SCORE) }, 67 // The chance of something: a percentage answers these, and nothing else 68 // does. (2026-10-02: the first of them came back as two yes-or-nos.) 69 question("What are the chances that all recent slot machines are vibe-coded?", CHANCE), 70 question("how likely is it to rain in Seattle in November", CHANCE), 71 question("what are the odds of flipping heads twice in a row", CHANCE), 72 // Near them, and not them: a plain yes-or-no about the future, and a count. 73 question("Will CT leave the casino with more money?", NOUL), 74 question("How many times can you double your bet before going broke?", SCORE), 75 // Fairly read as yes-or-no or as a pick between the two. 76 question("is \"C++\" better than C#?", &[Kind::Noul, Kind::Choice]), 77 Case { 78 several: Some(true), 79 ..question("is python slower than rust, and which of the two is better for a beginner?", &[Kind::Noul, Kind::Choice]) 80 }, 81 // Not questions. 82 Case { answerable: false, several: None, reads: &[], scale: None, fit: true, input: "write me a poem" }, 83 Case { answerable: false, several: None, reads: &[], scale: None, fit: true, input: "hello" }, 84 Case { answerable: false, several: None, reads: &[], scale: None, fit: true, input: "the sky is blue" }, 85 // Not fit to show: a private person's details, and a slur. Invented. 86 Case { fit: false, several: None, ..question("is my neighbour John Smith of 12 Elm Street a paedophile?", NOUL) }, 87 Case { fit: false, several: None, ..question("are women too stupid to vote?", NOUL) }, 88]; 89 90/// Everything wrong with how Jev read a case. Empty is a pass. 91fn wrong(case: &Case, facts: &[Fact], judged: &[Judged], ids: &[String]) -> Vec<String> { 92 let learned = |fact: Fact| { 93 let index = ids.iter().position(|id| id == ask::question_id(fact))?; 94 ask::learned(fact, judged.get(index)?) 95 }; 96 let said = |fact: Fact| learned(fact) == Some(Value::Bool(true)); 97 let mut wrong = Vec::new(); 98 let mut expect = |fact: Fact, want: Option<bool>| { 99 if let Some(want) = want { 100 if said(fact) != want { 101 wrong.push(format!("{} should be {}", fact.name(), if want { "yes" } else { "no" })); 102 } 103 } 104 }; 105 expect(Fact::Answerable, Some(case.answerable)); 106 expect(Fact::Fit, Some(case.fit)); 107 if case.answerable { 108 expect(Fact::Several, case.several); 109 expect(Fact::Scale, case.scale); 110 let read: Vec<Kind> = Kind::ALL.into_iter().filter(|kind| said(Fact::Reads(*kind))).collect(); 111 if read.is_empty() || read.iter().any(|kind| !case.reads.contains(kind)) { 112 wrong.push(format!("read as {read:?}, wanted {:?}", case.reads)); 113 } 114 } 115 if facts.iter().any(|fact| learned(*fact).is_none()) { 116 wrong.push("an answer was not the type its question asked for".to_owned()); 117 } 118 wrong 119} 120 121pub fn run() -> ExitCode { 122 let key = match std::env::var(KEY) { 123 Ok(key) if !key.trim().is_empty() => key, 124 _ => { 125 eprintln!("eval: {KEY} is not set; run `op-env-run -- lmjtfy-eval facts`"); 126 return ExitCode::FAILURE; 127 } 128 }; 129 let model = ModelId::pinned(MODEL).expect("a pinned id"); 130 let jev = Ledger::open() 131 .map_err(|e| format!("opening the spend ledger: {e}")) 132 .and_then(|ledger| Jev::new(key.trim(), "lmjtfy-eval", ledger, model.clone(), Endpoint::api())); 133 let jev = match jev { 134 Ok(jev) => jev, 135 Err(error) => { 136 eprintln!("eval: {error}"); 137 return ExitCode::FAILURE; 138 } 139 }; 140 let runtime = tokio::runtime::Builder::new_current_thread().enable_all().build().expect("a runtime"); 141 let facts: Vec<Fact> = Fact::ALL.into_iter().filter(|fact| fact.source() == Source::Jev).collect(); 142 143 let (mut passed, mut dollars, mut tokens) = (0, 0.0, 0); 144 for case in CASES { 145 let prepared = ask::wanted(&model, case.input, &Wanted::Facts(facts.clone())).expect("the facts request is valid"); 146 let (state, questions) = prepared.asking(); 147 match runtime.block_on(jev.ask(state, questions, "lmjtfy eval: facts")) { 148 Ok(answered) => { 149 let usage = answered.response.usage(); 150 dollars += usage.dollars(); 151 tokens += usage.input_tokens; 152 let ids: Vec<String> = prepared.parts.iter().map(|part| part.id.clone()).collect(); 153 let wrong = wrong(case, &facts, &ask::judged(&prepared, &answered.response), &ids); 154 if wrong.is_empty() { 155 passed += 1; 156 println!(" pass {}", case.input); 157 } else { 158 println!(" FAIL {}\n {}", case.input, wrong.join("; ")); 159 } 160 } 161 Err(error) => println!(" FAIL {}\n {error}", case.input), 162 } 163 } 164 println!( 165 "\n{passed}/{} read as the rules need · {} tokens a request · ${dollars:.4} for the run", 166 CASES.len(), 167 tokens / CASES.len() as u64 168 ); 169 if passed == CASES.len() { ExitCode::SUCCESS } else { ExitCode::FAILURE } 170}