jevcrates.git / jev-facts / src / tests.rs
tests.rsannotatedtests.rssource165 lines · 6.5 KB · raw

A small domain of this crate's own: a command, and four things to know about it. Two of the facts are readings of one Choice, which is the case the crate exists to get right.

5use jev_protocol::{Choice, Noul, Score};
7use super::*;
8
9#[derive(Clone, Copy, Debug, PartialEq)]
10enum Fact {

A Noul of its own.

12    Safe,

Two readings of the one Choice act.

14    Reads,
15    Deletes,

A Score of its own.

17    Load,
18}
20#[derive(Debug, PartialEq)]
21enum Value {
22    Bool(bool),
23    Level(f64),
24}
25
26struct Command;
27
28impl Source for Command {
29    type Fact = Fact;
30    type Value = Value;
31
32    fn question_id(&self, fact: Fact) -> String {
33        match fact {
34            Fact::Safe => "safe",
35            Fact::Reads | Fact::Deletes => "act",
36            Fact::Load => "load",
37        }
38        .to_owned()
39    }
40
41    fn question(&self, fact: Fact) -> Result<Question, ProtocolError> {
42        Ok(match fact {
43            Fact::Safe => Question::Noul(Noul::new(Json::text("Is the command safe to run?"))),
44            Fact::Reads | Fact::Deletes => Question::Choice(Choice::new(
45                Json::text("What does the command do?"),
46                [("read".to_owned(), None), ("delete".to_owned(), None)],
47            )?),
48            Fact::Load => Question::Score(Score::new(Json::text("How heavy is it?"), ["light", "heavy"].map(Json::text))?),
49        })
50    }
51
52    fn learned(&self, fact: Fact, judged: &Judged) -> Option<Value> {
53        match (fact, judged) {
54            (Fact::Safe, Judged::Noul(p)) => Some(Value::Bool(*p >= 0.5)),
55            (Fact::Reads, Judged::Choice(answer)) => Some(Value::Bool(answer.choice == "read")),
56            (Fact::Deletes, Judged::Choice(answer)) => Some(Value::Bool(answer.choice == "delete")),
57            (Fact::Load, Judged::Score(answer)) => Some(Value::Level(answer.score)),
58            _ => None,
59        }
60    }
61}
62
63fn model() -> ModelId {
64    ModelId::pinned("jev-1.13.0").unwrap()
65}
66
67fn state() -> Json {
68    Json::canonical(r#"{"command":"rm -rf target"}"#).unwrap()
69}

A response body with these answers, as Jev sends one.

72fn response(prepared: &Prepared, answers: &str) -> Response {
73    let body = format!(r#"{{"model":"jev-1.13.0","answers":{answers},"usage":{{"input_tokens":120,"output_tokens":1}}}}"#);
74    Response::parse(&model(), prepared.asking().1, body.as_bytes()).unwrap()
75}
77const ANSWERS: &str = r#"{
78    "safe":{"type":"noul","noul":0.2},
79    "act":{"type":"choice","choice":"delete","confidence":0.9,"probabilities":{"read":0.05,"delete":0.95}},
80    "load":{"type":"score","score":0.3,"confidence":0.7,"legend":{"0":"light","1":"heavy"},"probabilities":{"0":0.7,"1":0.3}}
81}"#;

The point of the crate: four facts wanted, one request, and the question two of them share asked once.

85#[test]
86fn every_wanted_fact_goes_in_one_request_and_a_shared_question_once() {
87    let prepared = wanted(&Command, &model(), state(), &[Fact::Safe, Fact::Reads, Fact::Deletes, Fact::Load]).unwrap();
88    let ids: Vec<&str> = prepared.parts.iter().map(|part| part.id.as_str()).collect();
89    assert_eq!(ids, ["safe", "act", "load"]);
90    let kinds: Vec<&str> = prepared.parts.iter().map(|part| part.kind).collect();
91    assert_eq!(kinds, ["noul", "choice", "score"]);
92    assert_eq!(prepared.request.matches("What does the command do?").count(), 1);
93    assert!(prepared.request.contains("rm -rf target"));
94    assert!(prepared.worst_case_dollars > 0.0);
95}

One answer teaches every fact that shares its question, each its own value.

98#[test]
99fn each_fact_reads_its_own_value_out_of_the_answers() {
100    let facts = [Fact::Deletes, Fact::Safe, Fact::Reads, Fact::Load];
101    let prepared = wanted(&Command, &model(), state(), &facts).unwrap();
102    let judged = prepared.judged(&response(&prepared, ANSWERS));
103    assert_eq!(judged.len(), 3);
104    let learned = learn(&Command, &prepared, &judged, &facts).unwrap();
105    assert_eq!(
106        learned,
107        [(Fact::Deletes, Value::Bool(true)), (Fact::Safe, Value::Bool(false)), (Fact::Reads, Value::Bool(false)), (Fact::Load, Value::Level(0.3))]
108    );
109}

Asking for fewer facts asks fewer questions: nothing a rule is not waiting on is paid for.

113#[test]
114fn only_what_is_wanted_is_asked() {
115    let prepared = wanted(&Command, &model(), state(), &[Fact::Load]).unwrap();
116    assert_eq!(prepared.parts.len(), 1);
117    assert!(!prepared.request.contains("safe to run"));
118}

A fact the request never asked about is an error that names it, never a default value: a host that learns facts it did not ask for has a bug.

122#[test]
123fn a_fact_that_was_not_asked_is_reported() {
124    let prepared = wanted(&Command, &model(), state(), &[Fact::Safe]).unwrap();
125    let judged = prepared.judged(&response(&prepared, r#"{"safe":{"type":"noul","noul":0.9}}"#));
126    assert_eq!(learn(&Command, &prepared, &judged, &[Fact::Safe, Fact::Load]), Err(Unlearned::NotAsked { id: "load".into() }));
127}

A source whose learned does not match its question is caught the first time it runs, with the question's id, and nothing is half-learned.

131#[test]
132fn an_answer_of_the_wrong_type_teaches_nothing() {
133    struct Confused;
134    impl Source for Confused {
135        type Fact = Fact;
136        type Value = Value;
137        fn question_id(&self, fact: Fact) -> String {
138            Command.question_id(fact)
139        }
140        fn question(&self, fact: Fact) -> Result<Question, ProtocolError> {
141            Command.question(fact)
142        }
143        // Reads a Score where a Noul was asked.
144        fn learned(&self, _: Fact, judged: &Judged) -> Option<Value> {
145            match judged {
146                Judged::Score(answer) => Some(Value::Level(answer.score)),
147                _ => None,
148            }
149        }
150    }
151    let facts = [Fact::Load, Fact::Safe];
152    let prepared = wanted(&Confused, &model(), state(), &facts).unwrap();
153    let judged = prepared.judged(&response(&prepared, r#"{"load":{"type":"score","score":1.0,"confidence":0.9,"legend":{"0":"light","1":"heavy"},"probabilities":{"0":0.0,"1":1.0}},"safe":{"type":"noul","noul":0.9}}"#));
154    assert_eq!(learn(&Confused, &prepared, &judged, &facts), Err(Unlearned::WrongType { id: "safe".into() }));
155}

prepare is also for questions that are not facts (ones a model drafted): given ids, in the order given.

159#[test]
160fn questions_that_are_not_facts_are_prepared_the_same_way() {
161    let asked = [("q2".to_owned(), Command.question(Fact::Load).unwrap()), ("q1".to_owned(), Command.question(Fact::Safe).unwrap())];
162    let prepared = prepare(&model(), state(), asked).unwrap();
163    let ids: Vec<&str> = prepared.parts.iter().map(|part| part.id.as_str()).collect();
164    assert_eq!(ids, ["q2", "q1"]);
165}