Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Thirteen MCP tools and classic token scopes; agents rate their confidence and can be put on an issue in one step | 1 | //! How sure the agent is of its own change, asked for at the end of every |
| 2 | //! run that makes or revises one and reported to g1t. | |
| 3 | //! | |
| 4 | //! Like `learned`, the question rides on the prompt: the agent ends its | |
| 5 | //! closing summary with a `<g1t-confidence>` block of JSON, which is taken | |
| 6 | //! out of the summary and sent with the run's own token. g1t combines it | |
| 7 | //! with what it observes of the change (its checks, revisions, review, | |
| 8 | //! tests, size, guardrails); what it observes can only lower what the agent | |
| 9 | //! says, never raise it. | |
| 10 | ||
| 11 | use std::time::Duration; | |
| 12 | ||
| 13 | use serde_json::{Value, json}; | |
| 14 | ||
| 15 | const OPEN: &str = "<g1t-confidence>"; | |
| 16 | const CLOSE: &str = "</g1t-confidence>"; | |
| 17 | const MAX_ITEMS: usize = 5; | |
| 18 | const MAX_ITEM_CHARS: usize = 160; | |
| 19 | ||
| 20 | /// What the agent is asked, after its task. | |
| Agents no longer count not having run CI as a doubt: the workflows run as checks after the push | 21 | pub const ASK: &str = "Also end your final message with how sure you are that your change is right and complete, as JSON inside <g1t-confidence></g1t-confidence> tags: {\"confidence\": \"high\" | \"medium\" | \"low\", \"uncertain_about\": [\"a few words for each thing you could not verify or had to guess\"]}. Say high only if you ran the project's tests (and any other check you can run here) and they passed and nothing was guessed. The repository's CI workflows run as checks on the pull request after you push, and count on their own, so not having run them is not something to be unsure about. The block is taken out of your summary."; |
| Thirteen MCP tools and classic token scopes; agents rate their confidence and can be put on an issue in one step | 22 | |
| 23 | /// The prompt with the question added. | |
| 24 | pub fn ask(prompt: &str) -> String { | |
| 25 | format!("{prompt}\n\n{ASK}") | |
| 26 | } | |
| 27 | ||
| 28 | /// What the agent said: its level and what it was unsure about. | |
| 29 | #[derive(Debug, PartialEq)] | |
| 30 | pub struct SelfReport { | |
| 31 | pub confidence: &'static str, | |
| 32 | pub uncertain_about: Vec<String>, | |
| 33 | } | |
| 34 | ||
| 35 | fn clip(text: &str) -> String { | |
| 36 | let text = text.split_whitespace().collect::<Vec<_>>().join(" "); | |
| 37 | text.chars().take(MAX_ITEM_CHARS).collect() | |
| 38 | } | |
| 39 | ||
| 40 | /// The summary without its `<g1t-confidence>` block, and what the block | |
| 41 | /// said. A missing or unreadable block reports nothing. | |
| 42 | pub fn split(summary: &str) -> (String, Option<SelfReport>) { | |
| 43 | let Some(start) = summary.rfind(OPEN) else { | |
| 44 | return (summary.trim().to_owned(), None); | |
| 45 | }; | |
| 46 | let body_start = start + OPEN.len(); | |
| 47 | let end = summary[body_start..].find(CLOSE).map(|at| body_start + at); | |
| 48 | let body = &summary[body_start..end.unwrap_or(summary.len())]; | |
| 49 | let rest = end.map_or("", |end| &summary[end + CLOSE.len()..]); | |
| 50 | let clean = format!("{}{}", summary[..start].trim_end(), rest.trim_end()).trim().to_owned(); | |
| 51 | let body = body.trim().trim_start_matches("```json").trim_start_matches("```").trim_end_matches("```").trim(); | |
| 52 | let Ok(value) = serde_json::from_str::<Value>(body) else { | |
| 53 | return (clean, None); | |
| 54 | }; | |
| 55 | let confidence = match value["confidence"].as_str().map(|level| level.trim().to_ascii_lowercase()) { | |
| 56 | Some(level) if level == "high" => "high", | |
| 57 | Some(level) if level == "medium" => "medium", | |
| 58 | Some(level) if level == "low" => "low", | |
| 59 | _ => return (clean, None), | |
| 60 | }; | |
| 61 | let uncertain_about = value["uncertain_about"] | |
| 62 | .as_array() | |
| 63 | .map(|items| { | |
| 64 | items | |
| 65 | .iter() | |
| 66 | .filter_map(Value::as_str) | |
| 67 | .map(clip) | |
| 68 | .filter(|item| !item.is_empty()) | |
| 69 | .take(MAX_ITEMS) | |
| 70 | .collect() | |
| 71 | }) | |
| 72 | .unwrap_or_default(); | |
| 73 | (clean, Some(SelfReport { confidence, uncertain_about })) | |
| 74 | } | |
| 75 | ||
| 76 | /// Sends what the agent said with the run's token. Never fails the run. | |
| 77 | pub fn report(said: &SelfReport) { | |
| 78 | let (Ok(run), Ok(token)) = (std::env::var("AGENT_RUN"), std::env::var("AGENT_RUN_TOKEN")) else { | |
| 79 | return; | |
| 80 | }; | |
| 81 | let api = std::env::var("G1T_API").unwrap_or_else(|_| "https://api.g1t.sh".to_owned()); | |
| 82 | let sent = ureq::post(&format!("{}/agent-runs/{run}/confidence", api.trim_end_matches('/'))) | |
| 83 | .timeout(Duration::from_secs(10)) | |
| 84 | .send_json(json!({ | |
| 85 | "token": token, | |
| 86 | "confidence": said.confidence, | |
| 87 | "uncertain_about": said.uncertain_about, | |
| 88 | })); | |
| 89 | if let Err(error) = sent { | |
| 90 | eprintln!("g1t-runner: could not report how sure the agent is: {error}"); | |
| 91 | } | |
| 92 | } | |
| 93 | ||
| 94 | /// The summary with its block taken out, after reporting what it said. | |
| 95 | pub fn finish(summary: String) -> String { | |
| 96 | let (clean, said) = split(&summary); | |
| 97 | if let Some(said) = &said { | |
| 98 | report(said); | |
| 99 | } | |
| 100 | clean | |
| 101 | } | |
| 102 | ||
| 103 | #[cfg(test)] | |
| 104 | mod tests { | |
| 105 | use super::*; | |
| 106 | ||
| 107 | #[test] | |
| 108 | fn the_block_is_taken_out_and_read() { | |
| 109 | let summary = "Added retries with backoff.\n\n<g1t-confidence>{\"confidence\": \"Medium\", \"uncertain_about\": [\" the retry limit \", \"\", 3]}</g1t-confidence>"; | |
| 110 | let (clean, said) = split(summary); | |
| 111 | assert_eq!(clean, "Added retries with backoff."); | |
| 112 | assert_eq!( | |
| 113 | said, | |
| 114 | Some(SelfReport { confidence: "medium", uncertain_about: vec!["the retry limit".to_owned()] }) | |
| 115 | ); | |
| 116 | } | |
| 117 | ||
| 118 | #[test] | |
| 119 | fn a_missing_or_unknown_level_reports_nothing() { | |
| 120 | assert_eq!(split("Done."), ("Done.".to_owned(), None)); | |
| 121 | assert_eq!(split("Done.\n<g1t-confidence>not json</g1t-confidence>").1, None); | |
| 122 | assert_eq!(split("Done.\n<g1t-confidence>{\"confidence\": \"very\"}</g1t-confidence>").1, None); | |
| 123 | let (clean, said) = split("Done.\n<g1t-confidence>\n```json\n{\"confidence\": \"high\"}\n```\n</g1t-confidence>"); | |
| 124 | assert_eq!(clean, "Done."); | |
| 125 | assert_eq!(said.map(|said| said.confidence), Some("high")); | |
| 126 | } | |
| 127 | ||
| 128 | #[test] | |
| 129 | fn it_leaves_the_learned_block_alone() { | |
| 130 | let summary = "Fixed it.\n<g1t-learned>[]</g1t-learned>\n<g1t-confidence>{\"confidence\": \"low\"}</g1t-confidence>"; | |
| 131 | let (clean, said) = split(summary); | |
| 132 | assert_eq!(clean, "Fixed it.\n<g1t-learned>[]</g1t-learned>"); | |
| 133 | assert_eq!(said.map(|said| said.confidence), Some("low")); | |
| 134 | assert!(ask("Fix it.").starts_with("Fix it.\n\n")); | |
| 135 | } | |
| 136 | } |