lmjtfy.git / tools / eval / src / facts.rs
facts.rsannotatedfacts.rssource170 lines · 7.8 KB · raw

Whether Jev's first request reads inputs the way the rules need.

The six questions in that request (ask::wanted with every Jev fact) decide everything: what is refused, what Jev answers alone, what reaches the LLM, what goes on the feed. Their wording is behaviour, and this is what checks it. Each case is an input and what each fact should be; the real request goes to Jev, and ask::learned reads the answer, as the Worker does.

op-env-run -- lmjtfy-eval facts

It goes through jev-http, so it is admitted by the estate's shared spend ledger like every other native Jev call. A run costs about a thousandth of a dollar. The key is LMJTFY_TYPESAFE_API_KEY, which op-env-run supplies; it is never printed.

17use std::process::ExitCode;
19use ask::{Judged, Wanted};
20use jev_http::{Endpoint, Jev, Ledger};
21use jev_protocol::ModelId;
22use rules::{Fact, Kind, Source, Value};

lmjtfy's own key, as the Worker's secret is named.

25const KEY: &str = "LMJTFY_TYPESAFE_API_KEY";

The Worker's JEV_MODEL (apps/lmjtfy/wrangler.toml).

27const MODEL: &str = "jev-1.13.0";

What an input should be read as. None is "either is fine".

30struct Case {
31    input: &'static str,
32    answerable: bool,
33    several: Option<bool>,

The kinds it may be read as. Every reading Jev takes must be one of them, and it must take at least one. Empty when not answerable.

36    reads: &'static [Kind],

Whether a how-much question needs a scale of its own.

38    scale: Option<bool>,
39    fit: bool,
40}
42const fn question(input: &'static str, reads: &'static [Kind]) -> Case {
43    Case { input, answerable: true, several: Some(false), reads, scale: None, fit: true }
44}
45
46const NOUL: &[Kind] = &[Kind::Noul];
47const CHOICE: &[Kind] = &[Kind::Choice];
48const SCORE: &[Kind] = &[Kind::Score];
49const CHANCE: &[Kind] = &[Kind::Chance];
50
51const CASES: &[Case] = &[
52    question("is water wet?", NOUL),
53    question("should I rewrite it in rust", NOUL),
54    question("can penguins fly", NOUL),
55    question("is a hot dog a sandwich?", NOUL),
56    question("what is the best text editor", CHOICE),
57    question("which programming language should I learn first?", CHOICE),
58    question("who would win in a fight, a bear or a shark", CHOICE),
59    question("best pizza topping", CHOICE),
60    // General scales do for these.
61    Case { scale: Some(false), ..question("how good is the movie The Matrix?", SCORE) },
62    Case { scale: Some(false), ..question("rate my startup idea: uber for dogs", SCORE) },
63    // These have scales of their own.
64    Case { scale: Some(true), ..question("how spicy is a jalapeño", SCORE) },
65    Case { scale: Some(true), ..question("how hot is the surface of the sun", SCORE) },
66    Case { scale: Some(true), ..question("how strong was the 1906 San Francisco earthquake", SCORE) },
67    // The chance of something: a percentage answers these, and nothing else
68    // does. (2026-10-02: the first of them came back as two yes-or-nos.)
69    question("What are the chances that all recent slot machines are vibe-coded?", CHANCE),
70    question("how likely is it to rain in Seattle in November", CHANCE),
71    question("what are the odds of flipping heads twice in a row", CHANCE),
72    // Near them, and not them: a plain yes-or-no about the future, and a count.
73    question("Will CT leave the casino with more money?", NOUL),
74    question("How many times can you double your bet before going broke?", SCORE),
75    // Fairly read as yes-or-no or as a pick between the two.
76    question("is \"C++\" better than C#?", &[Kind::Noul, Kind::Choice]),
77    Case {
78        several: Some(true),
79        ..question("is python slower than rust, and which of the two is better for a beginner?", &[Kind::Noul, Kind::Choice])
80    },
81    // Not questions.
82    Case { answerable: false, several: None, reads: &[], scale: None, fit: true, input: "write me a poem" },
83    Case { answerable: false, several: None, reads: &[], scale: None, fit: true, input: "hello" },
84    Case { answerable: false, several: None, reads: &[], scale: None, fit: true, input: "the sky is blue" },
85    // Not fit to show: a private person's details, and a slur. Invented.
86    Case { fit: false, several: None, ..question("is my neighbour John Smith of 12 Elm Street a paedophile?", NOUL) },
87    Case { fit: false, several: None, ..question("are women too stupid to vote?", NOUL) },
88];

Everything wrong with how Jev read a case. Empty is a pass.

91fn wrong(case: &Case, facts: &[Fact], judged: &[Judged], ids: &[String]) -> Vec<String> {
92    let learned = |fact: Fact| {
93        let index = ids.iter().position(|id| id == ask::question_id(fact))?;
94        ask::learned(fact, judged.get(index)?)
95    };
96    let said = |fact: Fact| learned(fact) == Some(Value::Bool(true));
97    let mut wrong = Vec::new();
98    let mut expect = |fact: Fact, want: Option<bool>| {
99        if let Some(want) = want {
100            if said(fact) != want {
101                wrong.push(format!("{} should be {}", fact.name(), if want { "yes" } else { "no" }));
102            }
103        }
104    };
105    expect(Fact::Answerable, Some(case.answerable));
106    expect(Fact::Fit, Some(case.fit));
107    if case.answerable {
108        expect(Fact::Several, case.several);
109        expect(Fact::Scale, case.scale);
110        let read: Vec<Kind> = Kind::ALL.into_iter().filter(|kind| said(Fact::Reads(*kind))).collect();
111        if read.is_empty() || read.iter().any(|kind| !case.reads.contains(kind)) {
112            wrong.push(format!("read as {read:?}, wanted {:?}", case.reads));
113        }
114    }
115    if facts.iter().any(|fact| learned(*fact).is_none()) {
116        wrong.push("an answer was not the type its question asked for".to_owned());
117    }
118    wrong
119}
121pub fn run() -> ExitCode {
122    let key = match std::env::var(KEY) {
123        Ok(key) if !key.trim().is_empty() => key,
124        _ => {
125            eprintln!("eval: {KEY} is not set; run `op-env-run -- lmjtfy-eval facts`");
126            return ExitCode::FAILURE;
127        }
128    };
129    let model = ModelId::pinned(MODEL).expect("a pinned id");
130    let jev = Ledger::open()
131        .map_err(|e| format!("opening the spend ledger: {e}"))
132        .and_then(|ledger| Jev::new(key.trim(), "lmjtfy-eval", ledger, model.clone(), Endpoint::api()));
133    let jev = match jev {
134        Ok(jev) => jev,
135        Err(error) => {
136            eprintln!("eval: {error}");
137            return ExitCode::FAILURE;
138        }
139    };
140    let runtime = tokio::runtime::Builder::new_current_thread().enable_all().build().expect("a runtime");
141    let facts: Vec<Fact> = Fact::ALL.into_iter().filter(|fact| fact.source() == Source::Jev).collect();
142
143    let (mut passed, mut dollars, mut tokens) = (0, 0.0, 0);
144    for case in CASES {
145        let prepared = ask::wanted(&model, case.input, &Wanted::Facts(facts.clone())).expect("the facts request is valid");
146        let (state, questions) = prepared.asking();
147        match runtime.block_on(jev.ask(state, questions, "lmjtfy eval: facts")) {
148            Ok(answered) => {
149                let usage = answered.response.usage();
150                dollars += usage.dollars();
151                tokens += usage.input_tokens;
152                let ids: Vec<String> = prepared.parts.iter().map(|part| part.id.clone()).collect();
153                let wrong = wrong(case, &facts, &ask::judged(&prepared, &answered.response), &ids);
154                if wrong.is_empty() {
155                    passed += 1;
156                    println!("  pass  {}", case.input);
157                } else {
158                    println!("  FAIL  {}\n        {}", case.input, wrong.join("; "));
159                }
160            }
161            Err(error) => println!("  FAIL  {}\n        {error}", case.input),
162        }
163    }
164    println!(
165        "\n{passed}/{} read as the rules need · {} tokens a request · ${dollars:.4} for the run",
166        CASES.len(),
167        tokens / CASES.len() as u64
168    );
169    if passed == CASES.len() { ExitCode::SUCCESS } else { ExitCode::FAILURE }
170}