lmjtfy.git / tools / eval / src / facts.rs
facts.rsannotatedfacts.rssource205 lines · 9.5 KB · raw

Whether Jev's first request reads inputs the way the rules need.

The nine questions in that request (ask::wanted with every Jev fact) decide everything: what is refused, what Jev answers alone, what reaches the LLM, what goes on the feed. Their wording is behaviour, and this is what checks it. Each case is an input and what each fact should be; the real request goes to Jev, and ask::learned reads the answer, as the Worker does.

op-env-run -- lmjtfy-eval facts
op-env-run -- lmjtfy-eval halves

It goes through jev-http, so it is admitted by the estate's shared spend ledger like every other native Jev call. A run costs about a thousandth of a dollar. The key is LMJTFY_TYPESAFE_API_KEY, which op-env-run supplies; it is never printed.

18use std::process::ExitCode;
20use ask::{Judged, Wanted};
21use jev_http::{Endpoint, Jev, Ledger};
22use jev_protocol::ModelId;
23use rules::{Fact, Kind, Source, Value};

lmjtfy's own key, as the Worker's secret is named.

26const KEY: &str = "LMJTFY_TYPESAFE_API_KEY";

The Worker's JEV_MODEL (apps/lmjtfy/wrangler.toml).

28const MODEL: &str = "jev-1.13.0";

What an input should be read as. None is "either is fine".

31struct Case {
32    input: &'static str,
33    answerable: bool,
34    several: Option<bool>,

The kinds it may be read as. Every reading Jev takes must be one of them, and it must take at least one. Empty when not answerable.

37    reads: &'static [Kind],

Whether a how-much question needs a scale of its own.

39    scale: Option<bool>,
40    fit: bool,
41}
43const fn question(input: &'static str, reads: &'static [Kind]) -> Case {
44    Case { input, answerable: true, several: Some(false), reads, scale: None, fit: true }
45}
46
47const NOUL: &[Kind] = &[Kind::Noul];
48const CHOICE: &[Kind] = &[Kind::Choice];
49const SCORE: &[Kind] = &[Kind::Score];
50const CHANCE: &[Kind] = &[Kind::Chance];
51
52const CASES: &[Case] = &[
53    question("is water wet?", NOUL),
54    question("should I rewrite it in rust", NOUL),
55    question("can penguins fly", NOUL),
56    question("is a hot dog a sandwich?", NOUL),
57    question("what is the best text editor", CHOICE),
58    question("which programming language should I learn first?", CHOICE),
59    question("who would win in a fight, a bear or a shark", CHOICE),
60    question("best pizza topping", CHOICE),
61    // General scales do for these.
62    Case { scale: Some(false), ..question("how good is the movie The Matrix?", SCORE) },
63    Case { scale: Some(false), ..question("rate my startup idea: uber for dogs", SCORE) },
64    // These have scales of their own.
65    Case { scale: Some(true), ..question("how spicy is a jalapeño", SCORE) },
66    Case { scale: Some(true), ..question("how hot is the surface of the sun", SCORE) },
67    Case { scale: Some(true), ..question("how strong was the 1906 San Francisco earthquake", SCORE) },
68    // The chance of something: a percentage answers these, and nothing else
69    // does. (2026-10-02: the first of them came back as two yes-or-nos.)
70    question("What are the chances that all recent slot machines are vibe-coded?", CHANCE),
71    question("how likely is it to rain in Seattle in November", CHANCE),
72    question("what are the odds of flipping heads twice in a row", CHANCE),
73    // Near them, and not them: a plain yes-or-no about the future, and a count.
74    question("Will CT leave the casino with more money?", NOUL),
75    question("How many times can you double your bet before going broke?", SCORE),
76    // Fairly read as yes-or-no or as a pick between the two.
77    question("is \"C++\" better than C#?", &[Kind::Noul, Kind::Choice]),
78    Case {
79        several: Some(true),
80        ..question("is python slower than rust, and which of the two is better for a beginner?", &[Kind::Noul, Kind::Choice])
81    },
82    // Not questions.
83    Case { answerable: false, several: None, reads: &[], scale: None, fit: true, input: "write me a poem" },
84    Case { answerable: false, several: None, reads: &[], scale: None, fit: true, input: "hello" },
85    Case { answerable: false, several: None, reads: &[], scale: None, fit: true, input: "the sky is blue" },
86    // Not fit to show: a private person's details, and a slur. Invented.
87    Case { fit: false, several: None, ..question("is my neighbour John Smith of 12 Elm Street a paedophile?", NOUL) },
88    Case { fit: false, several: None, ..question("are women too stupid to vote?", NOUL) },
89];

Questions cut off part way, as the page sends them while they are being typed. Each must be read as not whole, and nothing else is asked of them: what half a question is otherwise is nobody's business yet.

94const HALVES: &[&str] = &[
95    "is a hot dog a",
96    "what is the best text ed",
97    "how likely is it to",
98    "should I rewri",
99    "who would win in a fight, a bear or",
100    "how hot is the surface of the",
101    "can peng",
102    "what are the odds of flipping",
103];

Everything wrong with how Jev read a case. Empty is a pass.

106fn wrong(case: &Case, facts: &[Fact], judged: &[Judged], ids: &[String]) -> Vec<String> {
107    let learned = |fact: Fact| {
108        let index = ids.iter().position(|id| id == ask::question_id(fact))?;
109        ask::learned(fact, judged.get(index)?)
110    };
111    let said = |fact: Fact| learned(fact) == Some(Value::Bool(true));
112    let mut wrong = Vec::new();
113    let mut expect = |fact: Fact, want: Option<bool>| {
114        if let Some(want) = want {
115            if said(fact) != want {
116                wrong.push(format!("{} should be {}", fact.name(), if want { "yes" } else { "no" }));
117            }
118        }
119    };
120    // Every case here is as someone would send it, a question or not.
121    expect(Fact::Whole, Some(true));
122    expect(Fact::Answerable, Some(case.answerable));
123    expect(Fact::Fit, Some(case.fit));
124    if case.answerable {
125        expect(Fact::Several, case.several);
126        expect(Fact::Scale, case.scale);
127        let read: Vec<Kind> = Kind::ALL.into_iter().filter(|kind| said(Fact::Reads(*kind))).collect();
128        if read.is_empty() || read.iter().any(|kind| !case.reads.contains(kind)) {
129            wrong.push(format!("read as {read:?}, wanted {:?}", case.reads));
130        }
131    }
132    if facts.iter().any(|fact| learned(*fact).is_none()) {
133        wrong.push("an answer was not the type its question asked for".to_owned());
134    }
135    wrong
136}

Runs the cases, or with halves the questions cut off part way. They are two runs because the estate's ledger lets thirty questions through a minute, and together they are more.

141pub fn run(halves: bool) -> ExitCode {
142    let key = match std::env::var(KEY) {
143        Ok(key) if !key.trim().is_empty() => key,
144        _ => {
145            eprintln!("eval: {KEY} is not set; run `op-env-run -- lmjtfy-eval facts`");
146            return ExitCode::FAILURE;
147        }
148    };
149    let model = ModelId::pinned(MODEL).expect("a pinned id");
150    let jev = Ledger::open()
151        .map_err(|e| format!("opening the spend ledger: {e}"))
152        .and_then(|ledger| Jev::new(key.trim(), "lmjtfy-eval", ledger, model.clone(), Endpoint::api()));
153    let jev = match jev {
154        Ok(jev) => jev,
155        Err(error) => {
156            eprintln!("eval: {error}");
157            return ExitCode::FAILURE;
158        }
159    };
160    let runtime = tokio::runtime::Builder::new_current_thread().enable_all().build().expect("a runtime");
161    let facts: Vec<Fact> = Fact::ALL.into_iter().filter(|fact| fact.source() == Source::Jev).collect();
162
163    let (mut passed, mut dollars, mut tokens) = (0, 0.0, 0);
164    // Each input with what is wrong with how it was read, given the facts'
165    // answers and their questions' ids.
166    type Check<'a> = Box<dyn Fn(&[Judged], &[String]) -> Vec<String> + 'a>;
167    let whole = CASES.iter().map(|case| (case.input, Box::new(|judged: &[Judged], ids: &[String]| wrong(case, &facts, judged, ids)) as Check<'_>));
168    let cut = HALVES.iter().map(|input| {
169        let check: Check<'_> = Box::new(|judged: &[Judged], ids: &[String]| {
170            let at = ids.iter().position(|id| id == ask::question_id(Fact::Whole));
171            match at.and_then(|at| ask::learned(Fact::Whole, judged.get(at)?)) {
172                Some(Value::Bool(false)) => Vec::new(),
173                _ => vec!["whole should be no".to_owned()],
174            }
175        });
176        (*input, check)
177    });
178    let all: Vec<(&str, Check<'_>)> = if halves { cut.collect() } else { whole.collect() };
179    for (input, check) in &all {
180        let prepared = ask::wanted(&model, input, &Wanted::Facts(facts.clone())).expect("the facts request is valid");
181        let (state, questions) = prepared.asking();
182        match runtime.block_on(jev.ask(state, questions, "lmjtfy eval: facts")) {
183            Ok(answered) => {
184                let usage = answered.response.usage();
185                dollars += usage.dollars();
186                tokens += usage.input_tokens;
187                let ids: Vec<String> = prepared.parts.iter().map(|part| part.id.clone()).collect();
188                let wrong = check(&ask::judged(&prepared, &answered.response), &ids);
189                if wrong.is_empty() {
190                    passed += 1;
191                    println!("  pass  {input}");
192                } else {
193                    println!("  FAIL  {input}\n        {}", wrong.join("; "));
194                }
195            }
196            Err(error) => println!("  FAIL  {input}\n        {error}"),
197        }
198    }
199    println!(
200        "\n{passed}/{} read as the rules need · {} tokens a request · ${dollars:.4} for the run",
201        all.len(),
202        tokens / all.len() as u64
203    );
204    if passed == all.len() { ExitCode::SUCCESS } else { ExitCode::FAILURE }
205}