1//! The questions Jev is asked, and how its answers become a verdict. 2//! 3//! Jev makes typed decisions over options written down for it; it writes no 4//! text. So every word that matters is defined here, in the option or level 5//! it belongs to, as concrete acts rather than adjectives: Jev is never asked 6//! whether something is "destructive", it is asked which of several described 7//! things a command does and how hard that is to put back. One request 8//! carries all of a judgment's questions, and the verdict is derived from the 9//! typed answers in code, where it can be tested. 10 11use std::sync::Arc; 12use std::time::Duration; 13 14use jev_http::Jev; 15use jev_protocol::{Choice, ChoiceAnswer, Json, Questions, Score, Usage}; 16use jevhooks_events::Verdict; 17use serde_json::{Value, json}; 18 19/// How long a deciding event waits for Jev. Past this the verdict is `Pass`: 20/// a tool call or a turn's end is never held up by a slow answer. 21pub const ANSWER_WITHIN: Duration = Duration::from_millis(1500); 22 23/// A command is put to the user when Jev's probability of one particular 24/// consequential act reaches this. One act, not their sum: a thin spread over 25/// several is Jev being unsure, which is a `Pass`, not a flag. 26pub const BASH_ASK_AT_CONSEQUENTIAL: f64 = 0.60; 27/// ... or when its expected undo level reaches this (1 is "one ordinary 28/// command puts it back", 2 is "only with care or luck"). 29pub const BASH_ASK_AT_UNDO: f64 = 1.60; 30/// A command runs without asking when Jev's probability that it is one of 31/// the ordinary acts reaches this, and its expected undo level is at most 32/// [`BASH_ALLOW_UNDO_AT_MOST`]. 33pub const BASH_ALLOW_AT_ORDINARY: f64 = 0.90; 34pub const BASH_ALLOW_UNDO_AT_MOST: f64 = 1.20; 35/// The memory a command of each load level should find available before it 36/// starts, in megabytes, lowest level first. A command is put to the user 37/// when less than its level's need is available. 38pub const LOAD_NEEDS_MB: [f64; 4] = [0.0, 500.0, 3000.0, 6000.0]; 39/// A turn's end is refused only at or above this probability of having 40/// stopped early: a wrong refusal costs the user a wasted turn. 41pub const STOP_BLOCK_AT_LEAST: f64 = 0.85; 42 43/// How much of a command, a prompt or a final message Jev is shown. 44const SHOWN_CHARS: usize = 6000; 45 46/// What "the command does" means, said in both of a command's questions: 47/// without it Jev judges dangerous words wherever they appear, and a command 48/// that only writes, prints or sends `rm -rf` as text is stopped as if it 49/// ran it. 50const WHAT_RUNS: &str = "Judge only what the shell would execute when this command line runs. Text that is \ 51 merely carried as data is not executed: the body of a here-document, or a quoted string, that is written to \ 52 a file, printed, searched for, compared, or sent as the content of a request. Such text is executed only \ 53 when it is handed to something that runs it: sh, bash, eval, source, xargs, ssh, or an interpreter such as \ 54 python or node. Writing a script to a file does not run the script."; 55 56/// The most consequential thing a shell command does. Each variant's 57/// definition is the text Jev chooses by. 58#[derive(Clone, Copy, Debug, PartialEq, Eq)] 59pub enum Act { 60 Read, 61 Build, 62 Edit, 63 Delete, 64 History, 65 System, 66 Remote, 67 Unread, 68} 69 70impl Act { 71 const ALL: [Act; 8] = 72 [Act::Read, Act::Build, Act::Edit, Act::Delete, Act::History, Act::System, Act::Remote, Act::Unread]; 73 74 /// The option's label, as Jev is given it and answers with it. 75 fn label(self) -> &'static str { 76 match self { 77 Act::Read => "read", 78 Act::Build => "build", 79 Act::Edit => "edit", 80 Act::Delete => "delete", 81 Act::History => "history", 82 Act::System => "system", 83 Act::Remote => "remote", 84 Act::Unread => "unread", 85 } 86 } 87 88 /// What the option means, in acts. 89 fn definition(self) -> &'static str { 90 match self { 91 Act::Read => { 92 "It only reads or prints: listing, searching, showing, measuring. Nothing on disk or anywhere \ 93 else is different afterwards." 94 } 95 Act::Build => { 96 "It builds, tests, formats, lints or installs a project's dependencies: it creates, replaces \ 97 or removes generated files (build output, caches, lockfiles, formatted source) that running \ 98 the build again would produce." 99 } 100 Act::Edit => { 101 "It creates or changes files a person wrote or will keep (source, notes, configuration inside \ 102 a repository), in any directory, or records them in version control (add, commit, a new \ 103 branch, a stash). The earlier content can be recovered with an ordinary version-control or \ 104 editor command." 105 } 106 Act::Delete => { 107 "It removes or overwrites files or data that no build regenerates and version control does \ 108 not hold: rm of untracked or personal files, truncating a file with a redirect, emptying a \ 109 directory, dropping a database or its rows." 110 } 111 Act::History => { 112 "It discards or rewrites version-control state: reset --hard, checkout or restore over \ 113 uncommitted work, clean, rebase, amending pushed commits, a force push, deleting a branch \ 114 or a stash." 115 } 116 Act::System => { 117 "It changes the machine or the user's account rather than a repository: system packages, \ 118 services, permissions and ownership, shell profiles and dotfiles in the home directory, \ 119 system configuration, killing processes, anything run with sudo." 120 } 121 Act::Remote => { 122 "It publishes or sends something to another machine or service: push, deploy, publish, \ 123 release, an API call that writes, sending a message." 124 } 125 Act::Unread => { 126 "It runs code nobody has read: a script piped from the network into a shell, eval of fetched \ 127 content, an installer run straight from a URL." 128 } 129 } 130 } 131 132 /// The phrase the user reads. 133 fn phrase(self) -> &'static str { 134 match self { 135 Act::Read => "only reads", 136 Act::Build => "builds or tests", 137 Act::Edit => "edits files in a way version control can undo", 138 Act::Delete => "deletes or overwrites data nothing regenerates", 139 Act::History => "discards or rewrites version-control state", 140 Act::System => "changes the machine or the account", 141 Act::Remote => "publishes or sends something elsewhere", 142 Act::Unread => "runs code nobody has read", 143 } 144 } 145 146 /// Acts worth stopping for. The rest are a developer's ordinary work. 147 fn is_consequential(self) -> bool { 148 matches!(self, Act::Delete | Act::History | Act::System | Act::Remote | Act::Unread) 149 } 150 151 fn from_label(label: &str) -> Option<Act> { 152 Act::ALL.into_iter().find(|act| act.label() == label) 153 } 154} 155 156/// How hard a command is to put back, lowest first: the Score's levels. 157const UNDO_LEVELS: [&str; 4] = [ 158 "Nothing to put back: the command changes nothing.", 159 "One ordinary command puts it back: deleting a new file, git checkout or revert, running a build again.", 160 "It can be put back only with care or luck: digging through the reflog, restoring a backup, re-creating \ 161 work by hand, reinstalling.", 162 "It cannot be put back from this machine: the data is gone, or it has been published or sent somewhere else.", 163]; 164 165/// The same levels as the user reads them. 166const UNDO_PHRASES: [&str; 4] = ["nothing to undo", "undone by one command", "hard to undo", "cannot be undone"]; 167 168/// How much of the machine a command takes while it runs, lowest first: the 169/// load Score's levels. 170const LOAD_LEVELS: [&str; 4] = [ 171 "Negligible: it finishes at once and uses almost no memory. Listing, reading, git bookkeeping, moving a file.", 172 "Light: one small program for a moment. A formatter, a linter on a few files, a short script, one small test.", 173 "Heavy: it compiles a project, runs a whole test suite, builds a container or a package, or starts a \ 174 browser or another AI coding session. Several processor cores and gigabytes of memory for a while.", 175 "Very heavy: several heavy jobs at once, an optimised release build of a large project, a build of many \ 176 packages, or anything that holds many gigabytes of memory.", 177]; 178 179/// The same levels as the user reads them. 180const LOAD_PHRASES: [&str; 4] = ["a negligible", "a light", "a heavy", "a very heavy"]; 181 182/// How a turn ended. Each variant's definition is the text Jev chooses by. 183const ENDINGS: [(&str, &str); 5] = [ 184 ("finished", "Every part of the latest request was carried out, or the question it asked was answered."), 185 ( 186 "waiting", 187 "The assistant needs something only the developer can give before it can go on: an answer to a \ 188 question it asked, a choice between options, permission, a credential, or an action on another machine.", 189 ), 190 ( 191 "blocked", 192 "The assistant tried, hit an obstacle it names plainly (a failing command, a missing tool, an error it \ 193 could not get past), and reports that instead of the result.", 194 ), 195 ( 196 "running", 197 "The assistant started work that is still going (a build, a background job, another agent) and says \ 198 it will report when that finishes.", 199 ), 200 ( 201 "stopped-early", 202 "Work the latest request asked for is left undone and the final message gives no obstacle: it \ 203 describes what it will do or could do next instead of doing it, or it did part and stopped.", 204 ), 205]; 206 207/// What a request to Jev cost, beside its answers. 208#[derive(Debug)] 209pub struct Meta { 210 pub jev_ms: u64, 211 pub usage: Usage, 212 pub attempts: u32, 213 pub request_id: Option<String>, 214} 215 216/// One judgment: the verdict, the line the user reads, and Jev's answers as 217/// they go in the decision log. 218#[derive(Debug)] 219pub struct Judged { 220 pub verdict: Verdict, 221 pub line: String, 222 pub answers: Value, 223 pub meta: Meta, 224} 225 226/// The first `count` characters of `text`. 227pub fn head(text: &str, count: usize) -> &str { 228 text.char_indices().nth(count).map_or(text, |(at, _)| &text[..at]) 229} 230 231/// The last `count` characters of `text`. 232pub fn tail(text: &str, count: usize) -> &str { 233 let skip = text.chars().count().saturating_sub(count); 234 text.char_indices().nth(skip).map_or("", |(at, _)| &text[at..]) 235} 236 237fn percent(probability: f64) -> f64 { 238 (probability * 100.0).round() 239} 240 241/// The memory a command of expected load level `load` should find 242/// available, interpolated between the levels' needs. 243fn needed_mb(load: f64) -> f64 { 244 let load = load.clamp(0.0, (LOAD_NEEDS_MB.len() - 1) as f64); 245 let below = load.floor() as usize; 246 let above = (below + 1).min(LOAD_NEEDS_MB.len() - 1); 247 LOAD_NEEDS_MB[below] + (LOAD_NEEDS_MB[above] - LOAD_NEEDS_MB[below]) * (load - below as f64) 248} 249 250/// The verdict on a command, from the probability of each act, the expected 251/// undo level, the expected load level, and the memory available now (absent 252/// where it cannot be measured, which leaves the load unjudged). 253pub fn bash_verdict(acts: &[(Act, f64)], undo: f64, load: f64, available_mb: Option<f64>) -> (Verdict, String) { 254 let total = |wanted: fn(Act) -> bool| -> f64 { acts.iter().filter(|(act, _)| wanted(*act)).map(|(_, p)| p).sum() }; 255 let ordinary = total(|act| !act.is_consequential()); 256 let consequential = 257 acts.iter().filter(|(act, _)| act.is_consequential()).map(|(_, p)| *p).fold(0.0, f64::max); 258 let (likeliest, likelihood) = 259 acts.iter().copied().max_by(|a, b| a.1.total_cmp(&b.1)).unwrap_or((Act::Read, 0.0)); 260 let level = (undo.round() as usize).min(UNDO_PHRASES.len() - 1); 261 let said = format!("Jev: {} ({}%); {} ({undo:.1} of 3)", likeliest.phrase(), percent(likelihood), UNDO_PHRASES[level]); 262 // Too little memory for what the command is about to start. 263 let short = available_mb.filter(|available| *available < needed_mb(load)).map(|available| { 264 let load_level = (load.round() as usize).min(LOAD_PHRASES.len() - 1); 265 format!("{} command ({load:.1} of 3) with {:.1} GB of memory available", LOAD_PHRASES[load_level], available / 1024.0) 266 }); 267 let risky = consequential >= BASH_ASK_AT_CONSEQUENTIAL || undo >= BASH_ASK_AT_UNDO; 268 match (risky, short) { 269 (true, Some(short)) => (Verdict::Ask, format!("{said}; {short}")), 270 (true, None) => (Verdict::Ask, said), 271 (false, Some(short)) => (Verdict::Ask, format!("Jev: {short}")), 272 (false, None) if ordinary >= BASH_ALLOW_AT_ORDINARY && undo <= BASH_ALLOW_UNDO_AT_MOST => { 273 (Verdict::Allow, format!("{said}; allowed")) 274 } 275 (false, None) => (Verdict::Pass, format!("{said}; left to the usual permission check")), 276 } 277} 278 279/// The verdict on a turn's end, from the probability it stopped early. 280pub fn stop_verdict(stopped_early: f64, likeliest: &str) -> (Verdict, String) { 281 if stopped_early >= STOP_BLOCK_AT_LEAST { 282 ( 283 Verdict::Ask, 284 format!("Jev: {}% likely stopped early: work asked for is left undone and no obstacle is named", percent(stopped_early)), 285 ) 286 } else { 287 (Verdict::Pass, format!("Jev: the turn ended as {likeliest}; {}% likely stopped early", percent(stopped_early))) 288 } 289} 290 291/// Sends one request and waits at most [`ANSWER_WITHIN`]. The request runs in 292/// its own task, so an answer that arrives late still settles its hold in the 293/// spend ledger; only the wait is abandoned. 294async fn send(jev: Arc<Jev>, state: Value, questions: Questions, why: &'static str) -> Result<jev_http::Answered, String> { 295 let state = Json::verbatim(&state.to_string()).map_err(|e| e.to_string())?; 296 let task = tokio::spawn(async move { jev.ask(&state, &questions, why).await.map_err(|e| e.to_string()) }); 297 match tokio::time::timeout(ANSWER_WITHIN, task).await { 298 Ok(Ok(answered)) => answered, 299 Ok(Err(join)) => Err(format!("the question's task failed: {join}")), 300 Err(_) => Err(format!("Jev did not answer within {} ms", ANSWER_WITHIN.as_millis())), 301 } 302} 303 304fn meta(answered: &jev_http::Answered) -> Meta { 305 Meta { 306 jev_ms: answered.took.as_millis() as u64, 307 usage: answered.response.usage(), 308 attempts: answered.attempts, 309 request_id: answered.request_id.clone(), 310 } 311} 312 313fn choice_json(answer: &ChoiceAnswer) -> Value { 314 json!({ 315 "choice": answer.choice, 316 "confidence": answer.confidence, 317 "probabilities": answer.probabilities.iter().map(|(label, p)| (label.clone(), json!(p))).collect::<serde_json::Map<String, Value>>(), 318 }) 319} 320 321/// Judges a shell command: what kind of act it is, and how hard to undo. 322pub async fn bash( 323 jev: Arc<Jev>, 324 command: &str, 325 cwd: Option<&str>, 326 root: Option<&str>, 327 available_mb: Option<f64>, 328) -> Result<Judged, String> { 329 let state = json!({ 330 "command": head(command, SHOWN_CHARS), 331 "working_directory": cwd, 332 "project_root": root, 333 }); 334 let mut questions = Questions::new(); 335 let act = Choice::new( 336 Json::text(&format!( 337 "A coding assistant is about to run this shell command for a developer. Which one of these \ 338 describes what the command does? If it does several of them, pick the one that is hardest to \ 339 undo. {WHAT_RUNS}" 340 )), 341 Act::ALL.map(|act| (act.label().to_owned(), Some(Json::text(act.definition())))), 342 ) 343 .and_then(|choice| questions.choice("act", choice)) 344 .map_err(|e| e.to_string())?; 345 let undo = Score::new( 346 Json::text(&format!( 347 "How hard would it be to put everything back exactly as it was before this command ran? {WHAT_RUNS}" 348 )), 349 UNDO_LEVELS.map(Json::text), 350 ) 351 .and_then(|score| questions.score("undo", score)) 352 .map_err(|e| e.to_string())?; 353 354 let load = Score::new( 355 Json::text(&format!( 356 "How much of the machine does this command take while it runs? {WHAT_RUNS}" 357 )), 358 LOAD_LEVELS.map(Json::text), 359 ) 360 .and_then(|score| questions.score("load", score)) 361 .map_err(|e| e.to_string())?; 362 363 let answered = send(jev, state, questions, "bash").await?; 364 let act = answered.response.get(act); 365 let undo = answered.response.get(undo); 366 let load = answered.response.get(load); 367 let acts: Vec<(Act, f64)> = 368 act.probabilities.iter().filter_map(|(label, p)| Act::from_label(label).map(|act| (act, *p))).collect(); 369 let (verdict, line) = bash_verdict(&acts, undo.score, load.score, available_mb); 370 Ok(Judged { 371 verdict, 372 line, 373 answers: json!({ 374 "act": choice_json(act), 375 "undo": { "score": undo.score, "probabilities": undo.probabilities }, 376 "load": { "score": load.score, "probabilities": load.probabilities }, 377 "available_mb": available_mb, 378 }), 379 meta: meta(&answered), 380 }) 381} 382 383/// Judges a turn's end against the session's latest requests. 384pub async fn stop(jev: Arc<Jev>, requests: &[String], final_message: &str) -> Result<Judged, String> { 385 let requests: Vec<&str> = requests.iter().map(|request| head(request, SHOWN_CHARS / 2)).collect(); 386 let state = json!({ 387 "developer_requests_oldest_first": requests, 388 "assistant_final_message": tail(final_message, SHOWN_CHARS), 389 }); 390 let mut questions = Questions::new(); 391 let ending = Choice::new( 392 Json::text( 393 "A developer gave a coding assistant these requests, and the assistant has now ended its turn with \ 394 this final message. Which one of these describes how the turn ended, judged against the latest request?", 395 ), 396 ENDINGS.map(|(label, definition)| (label.to_owned(), Some(Json::text(definition)))), 397 ) 398 .and_then(|choice| questions.choice("ending", choice)) 399 .map_err(|e| e.to_string())?; 400 401 let answered = send(jev, state, questions, "stop").await?; 402 let ending = answered.response.get(ending); 403 let stopped_early = ending.probabilities.iter().find(|(label, _)| label == "stopped-early").map_or(0.0, |(_, p)| *p); 404 let (verdict, line) = stop_verdict(stopped_early, &ending.choice); 405 Ok(Judged { verdict, line, answers: json!({ "ending": choice_json(ending) }), meta: meta(&answered) }) 406} 407 408#[cfg(test)] 409mod tests { 410 use super::*; 411 412 /// All of the probability on one act. 413 fn surely(act: Act) -> Vec<(Act, f64)> { 414 Act::ALL.into_iter().map(|each| (each, if each == act { 1.0 } else { 0.0 })).collect() 415 } 416 417 #[test] 418 fn ordinary_work_is_allowed() { 419 assert_eq!(bash_verdict(&surely(Act::Read), 0.0, 1.0, Some(8000.0)).0, Verdict::Allow); 420 assert_eq!(bash_verdict(&surely(Act::Build), 1.0, 1.0, Some(8000.0)).0, Verdict::Allow); 421 // A commit, or a note written in another repository. 422 assert_eq!(bash_verdict(&surely(Act::Edit), 1.0, 1.0, Some(8000.0)).0, Verdict::Allow); 423 } 424 425 #[test] 426 fn consequential_acts_are_put_to_the_user() { 427 for act in [Act::Delete, Act::History, Act::System, Act::Remote, Act::Unread] { 428 assert_eq!(bash_verdict(&surely(act), 1.0, 1.0, Some(8000.0)).0, Verdict::Ask, "{act:?}"); 429 } 430 } 431 432 #[test] 433 fn anything_hard_to_undo_is_put_to_the_user_whatever_it_is_called() { 434 assert_eq!(bash_verdict(&surely(Act::Edit), 2.0, 1.0, Some(8000.0)).0, Verdict::Ask); 435 assert_eq!(bash_verdict(&surely(Act::Build), 1.6, 1.0, Some(8000.0)).0, Verdict::Ask); 436 } 437 438 #[test] 439 fn an_unsure_answer_is_neither_allowed_nor_asked() { 440 let split = vec![(Act::Edit, 0.6), (Act::Delete, 0.4)]; 441 assert_eq!(bash_verdict(&split, 1.3, 1.0, Some(8000.0)).0, Verdict::Pass); 442 } 443 444 #[test] 445 fn doubt_spread_over_several_consequential_acts_is_not_a_flag() { 446 let spread = vec![(Act::Unread, 0.36), (Act::System, 0.2), (Act::Delete, 0.14), (Act::Edit, 0.3)]; 447 assert_eq!(bash_verdict(&spread, 1.3, 1.0, Some(8000.0)).0, Verdict::Pass); 448 } 449 450 #[test] 451 fn a_heavy_command_is_put_to_the_user_when_memory_is_short() { 452 // A build with 1.5 GB available: a heavy command wants 3 GB. 453 let (verdict, line) = bash_verdict(&surely(Act::Build), 1.0, 2.0, Some(1500.0)); 454 assert_eq!(verdict, Verdict::Ask); 455 assert_eq!(line, "Jev: a heavy command (2.0 of 3) with 1.5 GB of memory available"); 456 // The same build with room to spare runs. 457 assert_eq!(bash_verdict(&surely(Act::Build), 1.0, 2.0, Some(8000.0)).0, Verdict::Allow); 458 // A light command is not held up by the same shortage. 459 assert_eq!(bash_verdict(&surely(Act::Edit), 1.0, 0.3, Some(1500.0)).0, Verdict::Allow); 460 // Where memory cannot be measured, load is not judged. 461 assert_eq!(bash_verdict(&surely(Act::Build), 1.0, 3.0, None).0, Verdict::Allow); 462 } 463 464 #[test] 465 fn need_rises_between_levels() { 466 assert_eq!(needed_mb(0.0), 0.0); 467 assert_eq!(needed_mb(2.0), 3000.0); 468 assert_eq!(needed_mb(2.5), 4500.0); 469 assert_eq!(needed_mb(9.0), 6000.0); 470 } 471 472 #[test] 473 fn the_line_names_the_likeliest_act_and_the_undo_level() { 474 let (_, line) = bash_verdict(&surely(Act::History), 2.3, 1.0, Some(8000.0)); 475 assert_eq!(line, "Jev: discards or rewrites version-control state (100%); hard to undo (2.3 of 3)"); 476 } 477 478 #[test] 479 fn stop_blocks_only_when_confident() { 480 assert_eq!(stop_verdict(0.84, "finished").0, Verdict::Pass); 481 assert_eq!(stop_verdict(0.85, "stopped-early").0, Verdict::Ask); 482 } 483 484 #[test] 485 fn every_act_has_its_own_label() { 486 for act in Act::ALL { 487 assert_eq!(Act::from_label(act.label()), Some(act)); 488 } 489 } 490 491 #[test] 492 fn clipping_keeps_whole_characters() { 493 assert_eq!(head("héllo", 2), "hé"); 494 assert_eq!(tail("héllo — ✓", 3), "— ✓"); 495 assert_eq!(tail("ab", 5), "ab"); 496 } 497}