model.rsannotatedmodel.rssource322 lines · 12.6 KB · raw

A prompt, before its turn starts: which model the turn is given to.

One fact is the daemon's own: whether the user has switched model choice on. With it off, nothing is asked. With it on, Jev is asked one question, how much carrying the prompt out takes, and one fact is read from the answer: the cheapest model likely to be enough.

8use jev_facts::{Judged, Prepared, Source, Unlearned};
9use jev_protocol::{Json, ProtocolError, Question, Response, Score};
10use jevhooks_events::Tier;
11use rete::{Domain, Known, Network, Rule, Test, Then};
12use serde_json::json;
14use crate::{Never, SHOWN_CHARS, head, tail};

A prompt is given to the cheapest model that is at least this likely to be enough for it. Above a half on purpose: a model too small for the work costs a wasted turn, one too large costs a few cents.

19pub const ENOUGH_AT: f64 = 0.75;

What carrying out a prompt takes, lowest first: the Score's levels. Level N is the work Tier::ALL[N] is the cheapest sound choice for.

23const NEED_LEVELS: [&str; 4] = [
24    "A lookup or one mechanical step: answer a question about what is already in the conversation, run a \
25     command that is named, rename something, commit, reply yes or no, continue a list already agreed.",
26    "Routine work of a known shape: an edit in one place, a fix whose cause is stated, a test for code that \
27     exists, a summary of one file, following written steps.",
28    "Work that needs judgment: a change across several files, a bug whose cause is not known, a review, a \
29     refactor, a design inside a shape that already exists.",
30    "Open-ended or long work: designing something new, weighing trade-offs nobody has stated, research \
31     across many sources, or many steps to be carried through unattended.",
32];

The same levels as the user reads them.

35const NEED_PHRASES: [&str; 4] = ["a lookup", "routine work", "work that needs judgment", "open-ended work"];

The model for a prompt, from the probability of each need level: the cheapest one at least [ENOUGH_AT] likely to be enough.

39pub fn model_for(levels: &[f64]) -> Tier {
40    let mut enough = 0.0;
41    for (tier, probability) in Tier::ALL.into_iter().zip(levels) {
42        enough += probability;
43        if enough >= ENOUGH_AT {
44            return tier;
45        }
46    }
47    Tier::Fable
48}

The domain: a marker for the engine, and the source of the question.

51#[derive(Clone, Copy, Debug, PartialEq, Eq)]
52pub struct Model;

Something known about a prompt.

55#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)]
56pub enum Fact {

The user has switched model choice on. The daemon's.

58    Choosing,

The cheapest model at least [ENOUGH_AT] likely to be enough. Jev's.

60    Enough,
61}
63#[derive(Clone, Copy, Debug, PartialEq, Eq)]
64pub enum Value {
65    Yes,
66    No,
67    Cheapest(Tier),
68}
69
70const TIERS: [Value; 4] = [Value::Cheapest(Tier::Haiku), Value::Cheapest(Tier::Sonnet), Value::Cheapest(Tier::Opus), Value::Cheapest(Tier::Fable)];

How the matter ends.

73#[derive(Clone, Copy, Debug, PartialEq, Eq)]
74pub enum End {

The session's own model stands.

76    Stands,

The turn is given to this one.

78    Give(Tier),
79}
81impl Domain for Model {
82    type Fact = Fact;
83    type Value = Value;
84    type Effect = Never;
85    type End = End;
86    type Note = Never;
87
88    fn facts() -> &'static [Fact] {
89        &[Fact::Choosing, Fact::Enough]
90    }
91
92    fn fact_name(fact: Fact) -> &'static str {
93        match fact {
94            Fact::Choosing => "choosing is on",
95            Fact::Enough => "model",
96        }
97    }
98
99    fn values(fact: Fact) -> &'static [Value] {
100        match fact {
101            Fact::Choosing => &[Value::Yes, Value::No],
102            Fact::Enough => &TIERS,
103        }
104    }
105
106    fn value_name(value: Value) -> &'static str {
107        match value {
108            Value::Yes => "yes",
109            Value::No => "no",
110            Value::Cheapest(tier) => tier.name(),
111        }
112    }
113
114    fn asked_for(fact: Fact) -> bool {
115        fact == Fact::Enough
116    }
117
118    fn teaches(effect: Never) -> Fact {
119        match effect {}
120    }
121
122    fn effect_name(effect: Never) -> String {
123        match effect {}
124    }
125
126    fn end_name(end: End) -> String {
127        match end {
128            End::Stands => "no change".to_owned(),
129            End::Give(tier) => tier.name().to_owned(),
130        }
131    }
132
133    fn note_name(note: Never) -> String {
134        match note {}
135    }
136}
137
138const fn give(name: &'static str, tier: Tier) -> Rule<'static, Model> {
139    // Each begins with the same test, so the network shares it: with
140    // choosing off the four fail together and Jev is not asked.
141    Rule {
142        name,
143        when: match tier {
144            Tier::Haiku => &[Test::Is(Fact::Choosing, Value::Yes), Test::Is(Fact::Enough, Value::Cheapest(Tier::Haiku))],
145            Tier::Sonnet => &[Test::Is(Fact::Choosing, Value::Yes), Test::Is(Fact::Enough, Value::Cheapest(Tier::Sonnet))],
146            Tier::Opus => &[Test::Is(Fact::Choosing, Value::Yes), Test::Is(Fact::Enough, Value::Cheapest(Tier::Opus))],
147            Tier::Fable => &[Test::Is(Fact::Choosing, Value::Yes), Test::Is(Fact::Enough, Value::Cheapest(Tier::Fable))],
148        },
149        then: Then::End(End::Give(tier)),
150    }
151}

The rules. Which model you use is yours to hand over, so the first rule is that with choosing off nothing is chosen, and nothing is asked.

155pub const RULES: [Rule<'static, Model>; 5] = [
156    Rule { name: "choosing is off", when: &[Test::Is(Fact::Choosing, Value::No)], then: Then::End(End::Stands) },
157    give("a lookup", Tier::Haiku),
158    give("routine work", Tier::Sonnet),
159    give("work that needs judgment", Tier::Opus),
160    give("open-ended work", Tier::Fable),
161];

[RULES] as the network that runs.

164pub fn network() -> Network<Model> {
165    Network::compile(&RULES)
166}

What the daemon knows before anything is asked.

169pub fn before(choosing: bool) -> Known<Model> {
170    let mut known = Known::default();
171    known.learn(Fact::Choosing, if choosing { Value::Yes } else { Value::No });
172    known
173}

What Jev judges: the new prompt, the session's prompts before it, oldest first, and the end of the assistant's last message: "yes, do that" is only as hard as what it agrees to.

178pub fn state(prompt: &str, earlier: &[String], reply: Option<&str>) -> Result<Json, ProtocolError> {
179    let earlier: Vec<&str> = earlier.iter().map(|request| head(request, SHOWN_CHARS / 4)).collect();
180    let state = json!({
181        "developer_new_request": head(prompt, SHOWN_CHARS),
182        "developer_earlier_requests_oldest_first": earlier,
183        "assistant_last_message": reply.map(|reply| tail(reply, SHOWN_CHARS / 2)),
184    });
185    Json::verbatim(&state.to_string())
186}
188impl Source for Model {
189    type Fact = Fact;
190    type Value = Value;
191
192    fn question_id(&self, fact: Fact) -> String {
193        match fact {
194            Fact::Enough => "need",
195            Fact::Choosing => "choosing",
196        }
197        .to_owned()
198    }
199
200    fn question(&self, fact: Fact) -> Result<Question, ProtocolError> {
201        match fact {
202            Fact::Enough => Ok(Question::Score(Score::new(
203                Json::text(
204                    "A developer has just sent a coding assistant this new request. How much does carrying it out take? \
205                     Judge the work the request asks for, not how long or short its wording is: a short request that \
206                     agrees to a plan in the assistant's last message takes what that plan takes.",
207                ),
208                NEED_LEVELS.map(Json::text),
209            )?)),
210            Fact::Choosing => Err(ProtocolError::Invalid(format!("{} is not a fact Jev is asked for", Model::fact_name(fact)))),
211        }
212    }
213
214    fn learned(&self, fact: Fact, judged: &Judged) -> Option<Value> {
215        match (fact, judged) {
216            (Fact::Enough, Judged::Score(answer)) => Some(Value::Cheapest(model_for(&answer.probabilities))),
217            _ => None,
218        }
219    }
220}

What Jev said a prompt needs: the fact, and the numbers behind it.

223#[derive(Clone, Debug)]
224pub struct Asked {
225    pub known: Known<Model>,

The expected need level, 0 to 3.

227    pub need: f64,

The probability of each level, lowest first.

229    pub probabilities: Vec<f64>,
230}

What a response to prepared (a request for facts) adds to known.

233pub fn taught(mut known: Known<Model>, prepared: &Prepared, response: &Response, facts: &[Fact]) -> Result<Asked, Unlearned> {
234    let judged = prepared.judged(response);
235    for (fact, value) in jev_facts::learn(&Model, prepared, &judged, facts)? {
236        known.learn(fact, value);
237    }
238    let Some(Judged::Score(need)) = prepared.parts.iter().position(|part| part.id == "need").and_then(|index| judged.get(index)) else {
239        return Err(Unlearned::NotAsked { id: "need".to_owned() });
240    };
241    Ok(Asked { known, need: need.score, probabilities: need.probabilities.clone() })
242}
244impl Asked {

The line the user reads when the turn is given to tier.

246    pub fn line(&self, tier: Tier) -> String {
247        let level = (self.need.round() as usize).min(NEED_PHRASES.len() - 1);
248        format!("Jev: {} ({:.1} of 3); {}", NEED_PHRASES[level], self.need, tier.name())
249    }

The probability of each level, as the decision log keeps it.

252    pub fn json(&self) -> serde_json::Value {
253        json!({ "need": { "score": self.need, "probabilities": self.probabilities } })
254    }

An answer made up for a test, read the way a real response is.

257    #[cfg(feature = "made-up")]
258    pub fn made_up(probabilities: [f64; 4]) -> Self {
259        let facts = [Fact::Enough];
260        let model = crate::jev_model().expect("the pinned model");
261        let prepared = jev_facts::wanted(&Model, &model, state("a prompt", &[], None).expect("a state"), &facts).expect("a request");
262        let score: f64 = probabilities.iter().enumerate().map(|(level, p)| level as f64 * p).sum();
263        let each = |value: &dyn Fn(usize) -> String| (0..4).map(|level| format!(r#""{level}":{}"#, value(level))).collect::<Vec<_>>().join(",");
264        let answer = format!(
265            r#"{{"type":"score","score":{score},"confidence":1.0,"legend":{{{}}},"probabilities":{{{}}}}}"#,
266            each(&|level| format!(r#""level {level}""#)),
267            each(&|level| probabilities[level].to_string()),
268        );
269        let body = crate::made_up_body(&[("need", answer)]);
270        let response = Response::parse(&model, prepared.asking().1, body.as_bytes()).expect("a response Jev could send");
271        taught(before(true), &prepared, &response, &facts).expect("the fact asked for")
272    }
273}
275#[cfg(test)]
276mod tests {
277    use rete::Next;
278
279    use super::*;

When Jev is sure of a level, the prompt goes to that level's model.

282    #[test]
283    fn a_prompt_gets_the_cheapest_model_likely_to_be_enough() {
284        assert_eq!(model_for(&[0.9, 0.1, 0.0, 0.0]), Tier::Haiku);
285        assert_eq!(model_for(&[0.1, 0.8, 0.1, 0.0]), Tier::Sonnet);
286        assert_eq!(model_for(&[0.0, 0.1, 0.7, 0.2]), Tier::Opus);
287        assert_eq!(model_for(&[0.0, 0.0, 0.3, 0.7]), Tier::Fable);
288    }

The likeliest level is not enough to go on: a model too small costs the turn, so doubt goes to the next model up.

292    #[test]
293    fn doubt_about_a_prompt_rounds_up_not_down() {
294        // Likeliest a lookup, but only at 60%: a quarter of the time haiku
295        // would be too small, so the prompt goes to the next model up.
296        assert_eq!(model_for(&[0.6, 0.3, 0.1, 0.0]), Tier::Sonnet);
297        // An even spread is Jev not knowing; that is not a reason for haiku.
298        assert_eq!(model_for(&[0.25, 0.25, 0.25, 0.25]), Tier::Opus);
299        // Answers that do not add up still end somewhere.
300        assert_eq!(model_for(&[]), Tier::Fable);
301    }

The rules give the turn to the model the answer makes cheapest-enough, through the network that runs.

305    #[test]
306    fn the_network_gives_the_turn_to_that_model() {
307        let network = network();
308        assert_eq!(network.next(&Asked::made_up([0.9, 0.1, 0.0, 0.0]).known), Next::End(End::Give(Tier::Haiku)));
309        assert_eq!(network.next(&Asked::made_up([0.6, 0.3, 0.1, 0.0]).known), Next::End(End::Give(Tier::Sonnet)));
310        assert_eq!(network.next(&Asked::made_up([0.0, 0.0, 0.3, 0.7]).known), Next::End(End::Give(Tier::Fable)));
311        assert_eq!(Asked::made_up([0.9, 0.1, 0.0, 0.0]).line(Tier::Haiku), "Jev: a lookup (0.1 of 3); haiku");
312    }

Which model you use is yours to hand over: with choosing off the session's model stands and Jev is not asked; with it on, the one question is.

316    #[test]
317    fn with_choosing_off_nothing_is_asked() {
318        let network = network();
319        assert_eq!(network.next(&before(false)), Next::End(End::Stands));
320        assert_eq!(network.next(&before(true)), Next::Ask(vec![Fact::Enough]));
321    }
322}