lmjtfy.git / packages / llm / src / lib.rs
lib.rsannotatedlib.rssource385 lines · 16.3 KB · raw
1//! The LLM that sits between a visitor and Jev: what it is told, the tools it
2//! is given, and what it wrote back.
3//!
4//! Jev cannot write text. It judges a question whose answer space somebody
5//! has already written down. The LLM's whole job is to write that question:
6//! one of three tool calls, one per Jev type. Nothing here does I/O; the
7//! Worker sends [`request`] through the Workers AI binding and the eval sends
8//! it over REST, and both hand the reply to [`parse`].
9#![forbid(unsafe_code)]
10
11use rules::Want;
12use serde::{Deserialize, Serialize};
13use serde_json::{Value, json};
14
15mod models;
16pub use models::{CANDIDATES, FREE_NEURONS_PER_DAY, Model};
17
18/// The most tool calls one input may become. More are dropped, not run.
19pub const MAX_CALLS: usize = 4;
20/// The most options a Choice may list: enough for a real field, few enough to
21/// read as bars on a phone. Jev itself takes up to 255.
22pub const MAX_OPTIONS: usize = 8;
23/// A Score's levels, as Jev takes them.
24pub const SCORE_LEVELS: std::ops::RangeInclusive<usize> = 2..=10;
25/// The reply's token ceiling, and so the worst case a call can cost. Qwen3
26/// reasons before it calls the tool: replies measured 2026-10-05 ran to 1320
27/// completion tokens, so the earlier 1200 cut about half of the how-spicy calls
28/// off with no tool call. Only the tokens used are billed.
29pub const MAX_TOKENS: u32 = 2400;
30
31const SYSTEM: &str = "\
32You sit between a person and Jev. Jev is a model that cannot write text. It only judges, in three ways:
33- jev_noul: a yes-or-no question, answered with the probability of yes.
34- jev_choice: one of several options that you list, answered with a probability for each.
35- jev_score: a position on a scale whose levels you write, lowest first.
36
37Turn the person's input into the tool call that answers it.
38- Always call a tool. Never answer the question yourself, and write no other text.
39- A yes-or-no question becomes jev_noul. So does \"how likely is X\": ask whether X, and the probability is the answer.
40- \"Which\", \"who\", \"what is the best\" and other questions with a best answer become jev_choice. \
41You choose 2 to 8 real, specific options and give each a one-line description.
42- \"How good\", \"how much\", \"how many\", \"how spicy\", \"rate this\" become jev_score with 3 to 7 levels, lowest first. Its levels cover every possible answer, lowest first: for a count or an amount the first is none or zero when that is possible and the last is open-ended (\"more than 5\"). Name each level by a range or a word, so that \"0\", \"1 to 2\", \"3 to 5\", \"more than 5\" is right and \"1\", \"2\", \"3\" is not.
43- Use one call. Use more only when the input plainly asks several separate things, and never more than 4.
44- Write `instructions` as one clear question. Jev sees the person's input beside it, and nothing else.";
45
46/// The three tools, in the OpenAI function format Workers AI takes.
47fn tools() -> Value {
48    let text = |description: &str| json!({ "type": "string", "description": description });
49    let tool = |name: &str, description: &str, properties: Value, required: &[&str]| {
50        json!({
51            "type": "function",
52            "function": {
53                "name": name,
54                "description": description,
55                "parameters": { "type": "object", "properties": properties, "required": required },
56            },
57        })
58    };
59    json!([
60        tool(
61            "jev_noul",
62            "Ask Jev a yes-or-no question. Jev answers with the probability that the answer is yes.",
63            json!({
64                "instructions": text("The yes-or-no question."),
65                "yes_means": text("What a yes means, in one sentence."),
66                "no_means": text("What a no means, in one sentence."),
67            }),
68            &["instructions", "yes_means", "no_means"],
69        ),
70        tool(
71            "jev_choice",
72            "Ask Jev to pick one of several options. Jev answers with a probability for every option.",
73            json!({
74                "instructions": text("The question the options answer."),
75                "options": {
76                    "type": "array",
77                    "description": "2 to 8 options. Labels are short and all different.",
78                    "items": {
79                        "type": "object",
80                        "properties": {
81                            "label": text("The option's short name."),
82                            "description": text("One line on what this option is."),
83                        },
84                        "required": ["label", "description"],
85                    },
86                },
87            }),
88            &["instructions", "options"],
89        ),
90        tool(
91            "jev_score",
92            "Ask Jev to place the input on a scale. Jev answers with a score and a probability for every level.",
93            json!({
94                "instructions": text("What is being scored."),
95                "levels": {
96                    "type": "array",
97                    "description": "2 to 10 levels, lowest first, covering every possible answer (a count starts at none or zero when that is possible). Each is one line naming a range or a word, not a bare number.",
98                    "items": { "type": "string" },
99                },
100            }),
101            &["instructions", "levels"],
102        ),
103    ])
104}
105
106/// The tools the LLM may call for `wants`: every one when the input is
107/// several questions (what each one is, is the LLM's to work out), and
108/// otherwise only the kinds the rules settled on. `None` is every tool.
109fn settled(wants: &[Want]) -> Option<Vec<&'static str>> {
110    if wants.is_empty() || wants.contains(&Want::Split) {
111        return None;
112    }
113    Some(
114        wants
115            .iter()
116            .map(|want| match want {
117                Want::Options => "jev_choice",
118                Want::Scale => "jev_score",
119                Want::Split => unreachable!("ruled out above"),
120            })
121            .collect(),
122    )
123}
124
125/// The LLM's instructions when the rules have settled what the input is.
126/// It is told about, and given, only the tools for that: Jev has already
127/// answered the other readings itself, and a second answer to one of them
128/// from here would only disagree with the first (2026-10-02: "what are the
129/// chances that…" got a yes-or-no from Jev and another, with a different
130/// probability, from a `jev_noul` the LLM wrote when a scale was wanted).
131fn settled_system(tools: &[&str]) -> String {
132    let has = |tool: &str| tools.contains(&tool);
133    let mut text = String::from(
134        "You sit between a person and Jev. Jev is a model that cannot write text. It only judges. \
135         For this input, like this:\n",
136    );
137    if has("jev_choice") {
138        text.push_str("- jev_choice: one of several options that you list, answered with a probability for each.\n");
139    }
140    if has("jev_score") {
141        text.push_str("- jev_score: a position on a scale whose levels you write, lowest first.\n");
142    }
143    text.push_str(match (has("jev_choice"), has("jev_score")) {
144        (true, true) => {
145            "\nThe person's input has already been read two ways: as a pick among possibilities, and as a \
146             how-much question. Write exactly two tool calls, one jev_choice and one jev_score.\n"
147        }
148        (true, false) => {
149            "\nThe person's input has already been read as a pick among possibilities. Write exactly one \
150             jev_choice call.\n"
151        }
152        _ => "\nThe person's input has already been read as a how-much question. Write exactly one jev_score call.\n",
153    });
154    text.push_str("- Always call a tool. Never answer the question yourself, and write no other text.\n");
155    if has("jev_choice") {
156        text.push_str("- For jev_choice, choose 2 to 8 real, specific options and give each a one-line description.\n");
157    }
158    if has("jev_score") {
159        text.push_str(
160            "- For jev_score, write 3 to 7 levels, lowest first, that fit what is asked: its own units, ranges \
161             or named grades where it has them, and levels of likelihood where it asks how likely. Its levels cover every possible answer, lowest first: for a count or an amount the first is none or zero when that is possible and the last is open-ended (\"more than 5\"). Name each level by a range or a word, so that \"0\", \"1 to 2\", \"3 to 5\", \"more than 5\" is right and \"1\", \"2\", \"3\" is not.\n",
162        );
163    }
164    text.push_str("- Write `instructions` as one clear question. Jev sees the person's input beside it, and nothing else.");
165    text
166}
167
168/// Whether a question the LLM wrote is of a kind it was asked for. One that
169/// is not, is not sent: Jev has answered that reading already, or the rules
170/// did not take it.
171pub fn takes(wants: &[Want], draft: &Draft) -> bool {
172    settled(wants).is_none_or(|tools| tools.contains(&draft.tool()))
173}
174
175/// The request body for `input`, as JSON text. The same bytes go to the
176/// binding and to the page's tool call panel. `wants` is what the rules say
177/// the LLM is to write ([`rules::Network::wants`]): sorted, no repeats.
178pub fn request(input: &str, wants: &[Want]) -> String {
179    let (system, tools) = match settled(wants) {
180        Some(names) => {
181            let given: Vec<Value> = tools()
182                .as_array()
183                .into_iter()
184                .flatten()
185                .filter(|tool| names.iter().any(|name| tool["function"]["name"] == *name))
186                .cloned()
187                .collect();
188            (settled_system(&names), Value::Array(given))
189        }
190        None => (SYSTEM.to_owned(), tools()),
191    };
192    json!({
193        "messages": [
194            { "role": "system", "content": system },
195            { "role": "user", "content": input },
196        ],
197        "tools": tools,
198        "max_tokens": MAX_TOKENS,
199        "temperature": 0,
200    })
201    .to_string()
202}
203
204/// A question for Jev as the LLM wrote it, checked for shape.
205#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
206pub enum Draft {
207    Noul { instructions: String, yes_means: String, no_means: String },
208    Choice { instructions: String, options: Vec<Opt> },
209    Score { instructions: String, levels: Vec<String> },
210}
211
212#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
213#[serde(deny_unknown_fields)]
214pub struct Opt {
215    pub label: String,
216    pub description: String,
217}
218
219impl Draft {
220    /// The tool that was called, which is also the Jev type.
221    pub fn tool(&self) -> &'static str {
222        match self {
223            Draft::Noul { .. } => "jev_noul",
224            Draft::Choice { .. } => "jev_choice",
225            Draft::Score { .. } => "jev_score",
226        }
227    }
228
229    pub fn instructions(&self) -> &str {
230        match self {
231            Draft::Noul { instructions, .. }
232            | Draft::Choice { instructions, .. }
233            | Draft::Score { instructions, .. } => instructions,
234        }
235    }
236}
237
238/// One tool call from the reply.
239#[derive(Clone, Debug, PartialEq)]
240pub struct ToolCall {
241    pub name: String,
242    /// The arguments exactly as the model wrote them.
243    pub arguments: String,
244    /// The question they describe, or why they do not describe one.
245    pub draft: Result<Draft, String>,
246}
247
248/// What a reply cost, as the API reported it.
249#[derive(Clone, Copy, Debug, Default, PartialEq)]
250pub struct Usage {
251    pub prompt_tokens: u64,
252    pub completion_tokens: u64,
253    /// Cloudflare's own count, when the reply carries one.
254    pub neurons: Option<f64>,
255}
256
257#[derive(Clone, Debug, PartialEq)]
258pub struct Reply {
259    /// At most [`MAX_CALLS`], in the order the model made them.
260    pub calls: Vec<ToolCall>,
261    /// How many calls the model made past the cap. They were dropped.
262    pub dropped: usize,
263    pub usage: Usage,
264}
265
266/// Reads a chat-completion reply: the binding's own, or REST's, which wraps
267/// it in `result`. An `Err` is a reply with no usable shape at all; a reply
268/// whose calls are malformed is `Ok`, with the reason on each call.
269pub fn parse(body: &str) -> Result<Reply, String> {
270    let root: Value = serde_json::from_str(body).map_err(|e| format!("the reply is not JSON: {e}"))?;
271    let root = root.get("result").unwrap_or(&root);
272    let message = root
273        .pointer("/choices/0/message")
274        .ok_or_else(|| "the reply has no choices[0].message".to_owned())?;
275    let made: &[Value] = message.get("tool_calls").and_then(Value::as_array).map_or(&[], Vec::as_slice);
276    let calls = made.iter().take(MAX_CALLS).map(tool_call).collect();
277    let usage = root.get("usage");
278    let count = |name: &str| usage.and_then(|u| u.get(name)).and_then(Value::as_u64).unwrap_or(0);
279    Ok(Reply {
280        calls,
281        dropped: made.len().saturating_sub(MAX_CALLS),
282        usage: Usage {
283            prompt_tokens: count("prompt_tokens"),
284            completion_tokens: count("completion_tokens"),
285            neurons: usage.and_then(|u| u.get("neurons")).and_then(Value::as_f64),
286        },
287    })
288}
289
290fn tool_call(call: &Value) -> ToolCall {
291    let name = call.pointer("/function/name").and_then(Value::as_str).unwrap_or_default().to_owned();
292    let arguments = match call.pointer("/function/arguments") {
293        Some(Value::String(text)) => text.clone(),
294        Some(other) => other.to_string(),
295        None => String::new(),
296    };
297    let draft = draft(&name, &arguments);
298    ToolCall { name, arguments, draft }
299}
300
301/// The arguments as an object. The format says `arguments` is JSON text of an
302/// object. One model (granite-4.0-h-micro, measured 2026-10-02) encodes that
303/// text a second time, so a string that decodes to a string is decoded once
304/// more. Nothing else is repaired.
305fn object(arguments: &str) -> Result<Value, String> {
306    let mut value: Value =
307        serde_json::from_str(arguments).map_err(|e| format!("the arguments are not JSON: {e}"))?;
308    if let Value::String(inner) = &value {
309        value = serde_json::from_str(inner).map_err(|e| format!("the arguments are not JSON: {e}"))?;
310    }
311    if value.is_object() { Ok(value) } else { Err("the arguments are not an object".to_owned()) }
312}
313
314fn draft(name: &str, arguments: &str) -> Result<Draft, String> {
315    #[derive(Deserialize)]
316    #[serde(deny_unknown_fields)]
317    struct NoulArgs {
318        instructions: String,
319        yes_means: String,
320        no_means: String,
321    }
322    #[derive(Deserialize)]
323    #[serde(deny_unknown_fields)]
324    struct ChoiceArgs {
325        instructions: String,
326        options: Vec<Opt>,
327    }
328    #[derive(Deserialize)]
329    #[serde(deny_unknown_fields)]
330    struct ScoreArgs {
331        instructions: String,
332        levels: Vec<String>,
333    }
334    fn read<T: for<'de> Deserialize<'de>>(value: Value) -> Result<T, String> {
335        serde_json::from_value(value).map_err(|e| e.to_string())
336    }
337    let filled = |what: &str, text: &str| {
338        if text.trim().is_empty() { Err(format!("{what} is empty")) } else { Ok(()) }
339    };
340
341    let value = object(arguments)?;
342    match name {
343        "jev_noul" => {
344            let NoulArgs { instructions, yes_means, no_means } = read(value)?;
345            filled("instructions", &instructions)?;
346            filled("yes_means", &yes_means)?;
347            filled("no_means", &no_means)?;
348            Ok(Draft::Noul { instructions, yes_means, no_means })
349        }
350        "jev_choice" => {
351            let ChoiceArgs { instructions, options } = read(value)?;
352            filled("instructions", &instructions)?;
353            if !(2..=MAX_OPTIONS).contains(&options.len()) {
354                return Err(format!("a choice takes 2 to {MAX_OPTIONS} options, not {}", options.len()));
355            }
356            for (i, option) in options.iter().enumerate() {
357                filled("an option's label", &option.label)?;
358                if options[..i].iter().any(|earlier| earlier.label == option.label) {
359                    return Err(format!("the option {:?} appears twice", option.label));
360                }
361            }
362            Ok(Draft::Choice { instructions, options })
363        }
364        "jev_score" => {
365            let ScoreArgs { instructions, levels } = read(value)?;
366            filled("instructions", &instructions)?;
367            if !SCORE_LEVELS.contains(&levels.len()) {
368                return Err(format!(
369                    "a score takes {} to {} levels, not {}",
370                    SCORE_LEVELS.start(),
371                    SCORE_LEVELS.end(),
372                    levels.len()
373                ));
374            }
375            for level in &levels {
376                filled("a level", level)?;
377            }
378            Ok(Draft::Score { instructions, levels })
379        }
380        other => Err(format!("there is no tool called {other:?}")),
381    }
382}
383
384#[cfg(test)]
385mod tests;