| 1 | //! How sure g1t is of a change an agent made. |
| 2 | //! |
| 3 | //! Worked out from what g1t can observe, never from how the agent sounds: |
| 4 | //! whether the acceptance checks pass (and passed only on a retry), how |
| 5 | //! many times the agent was sent back, the reviewer agent's verdict and how |
| 6 | //! much it had to say, whether tests were added or changed, how large the |
| 7 | //! change is and whether it reached outside the files its plan expected, |
| 8 | //! whether it touched paths that run or configure things (CI, secrets, |
| 9 | //! infrastructure), how close its runs came to their guardrails, and what |
| 10 | //! it asked other agents without an answer. |
| 11 | //! |
| 12 | //! The agent can say how sure it is too, at the end of its run |
| 13 | //! (`report_confidence`). That is combined with the signals by taking the |
| 14 | //! lower of the two: what g1t observes can lower what the agent says, never |
| 15 | //! raise it. |
| 16 | //! |
| 17 | //! [`score`] is pure and tested on its own; [`Work::assess_confidence`] |
| 18 | //! gathers the signals for a pull request, and records the result on it |
| 19 | //! and on the run that left it so. Where a repository asks for it |
| 20 | //! (`hold_low_confidence`), a low-confidence change waits for a person |
| 21 | //! instead of merging by itself (lifecycle.rs). |
| 22 | |
| 23 | use futures_util::future::try_join; |
| 24 | use g1t_contracts::time::rfc3339; |
| 25 | use g1t_contracts::work::{ |
| 26 | ChangedFile, CheckStatus, Confidence, ConfidenceLevel, Pull, ReportConfidenceArgs, Verdict, |
| 27 | }; |
| 28 | use g1t_contracts::{FailureCode, Outcome}; |
| 29 | use g1t_kit::now_ms; |
| 30 | use serde::Deserialize; |
| 31 | use worker::Result; |
| 32 | use worker::wasm_bindgen::JsValue; |
| 33 | |
| 34 | use crate::Work; |
| 35 | use crate::checks::hash; |
| 36 | use crate::reviews::AGENT_ID; |
| 37 | use crate::rows::NumberRow; |
| 38 | |
| 39 | /// At this many points, low; at none, high; medium between. |
| 40 | const LOW_AT: u32 = 3; |
| 41 | /// The most reasons a confidence gives. |
| 42 | const MAX_REASONS: usize = 4; |
| 43 | /// The most a self-report's list of doubts keeps, and of each. |
| 44 | const MAX_UNCERTAIN: usize = 5; |
| 45 | const MAX_UNCERTAIN_CHARS: usize = 160; |
| 46 | /// A share of a run's cap past which it was close to it. |
| 47 | const NEAR_CAP: f64 = 0.8; |
| 48 | |
| 49 | /// Everything confidence is worked out from. |
| 50 | #[derive(Clone, Debug, Default)] |
| 51 | pub(crate) struct Signals { |
| 52 | /// Whether the issue has acceptance checks at all. |
| 53 | pub has_checks: bool, |
| 54 | pub check_status: Option<CheckStatus>, |
| 55 | /// A run of the checks errored, or failed and then passed on the same |
| 56 | /// commit: they pass, but not reliably. |
| 57 | pub flaky_checks: bool, |
| 58 | /// How many times the agent was sent back. |
| 59 | pub revisions: u32, |
| 60 | /// Whether a second agent reviews changes here. |
| 61 | pub agent_review: bool, |
| 62 | /// The verdict of the latest review of the change as it is now. |
| 63 | pub review: Option<Verdict>, |
| 64 | /// How many comments on lines that review left. |
| 65 | pub review_comments: u32, |
| 66 | pub files: Vec<ChangedFile>, |
| 67 | /// The files the issue's plan expected it to change; empty when it was |
| 68 | /// not planned. |
| 69 | pub expected: Vec<String>, |
| 70 | /// `budget` or `time` when a run was stopped at a cap. |
| 71 | pub halted: Option<String>, |
| 72 | /// The latest run's cost as a share of its cost cap. |
| 73 | pub budget_share: Option<f64>, |
| 74 | /// The latest run's time as a share of its time cap. |
| 75 | pub time_share: Option<f64>, |
| 76 | /// Commands and tools the guardrails refused while it worked. |
| 77 | pub denials: u32, |
| 78 | /// Questions and handoffs it sent other agents that have no answer. |
| 79 | pub unanswered: u32, |
| 80 | /// What the agent said of its own change. |
| 81 | pub self_reported: Option<ConfidenceLevel>, |
| 82 | pub uncertain_about: Vec<String>, |
| 83 | } |
| 84 | |
| 85 | /// Test files, by the names test runners look for. |
| 86 | pub(crate) fn is_test(path: &str) -> bool { |
| 87 | let lower = path.to_ascii_lowercase(); |
| 88 | let file = lower.rsplit('/').next().unwrap_or(&lower); |
| 89 | lower.split('/').any(|dir| matches!(dir, "test" | "tests" | "__tests__" | "spec" | "specs" | "testdata")) |
| 90 | || [".test.", ".spec.", "_test.", "-test.", "_spec."].iter().any(|mark| file.contains(mark)) |
| 91 | || file.starts_with("test_") |
| 92 | } |
| 93 | |
| 94 | /// Files that change nothing that runs: prose and pictures. |
| 95 | fn is_prose(path: &str) -> bool { |
| 96 | let lower = path.to_ascii_lowercase(); |
| 97 | [".md", ".mdx", ".txt", ".rst", ".png", ".jpg", ".jpeg", ".gif", ".svg", ".webp"] |
| 98 | .iter() |
| 99 | .any(|extension| lower.ends_with(extension)) |
| 100 | || lower.starts_with("docs/") |
| 101 | || lower.rsplit('/').next().is_some_and(|file| file == "license" || file == "changelog") |
| 102 | } |
| 103 | |
| 104 | /// Paths that run, configure or guard things rather than being the code |
| 105 | /// itself: CI, repository automation, secrets, infrastructure, ownership. |
| 106 | /// A change that reaches them deserves a person's eyes. |
| 107 | pub(crate) fn sensitive(path: &str) -> Option<&'static str> { |
| 108 | let lower = path.to_ascii_lowercase(); |
| 109 | let file = lower.rsplit('/').next().unwrap_or(&lower); |
| 110 | if lower.starts_with(".github/workflows/") || lower.starts_with(".gitlab-ci") || lower.starts_with(".circleci/") { |
| 111 | return Some("CI workflows"); |
| 112 | } |
| 113 | if lower.starts_with(".g1t/") || lower.starts_with(".github/") { |
| 114 | return Some("repository automation"); |
| 115 | } |
| 116 | if file == "codeowners" { |
| 117 | return Some("CODEOWNERS"); |
| 118 | } |
| 119 | if file.starts_with(".env") || file.ends_with(".pem") || file.ends_with(".key") || file.contains("secret") { |
| 120 | return Some("secrets"); |
| 121 | } |
| 122 | if file.ends_with(".tf") |
| 123 | || file.ends_with(".tfvars") |
| 124 | || file == "dockerfile" |
| 125 | || file.starts_with("docker-compose") |
| 126 | || file.starts_with("wrangler.") |
| 127 | { |
| 128 | return Some("infrastructure"); |
| 129 | } |
| 130 | None |
| 131 | } |
| 132 | |
| 133 | /// Whether `path` is where the plan said the work would be: one of its |
| 134 | /// files, or beside one, in the same directory or below it. |
| 135 | fn expected(path: &str, planned: &[String]) -> bool { |
| 136 | planned.iter().any(|file| { |
| 137 | let file = file.trim().trim_start_matches("./"); |
| 138 | if path == file { |
| 139 | return true; |
| 140 | } |
| 141 | let dir = file.rsplit_once('/').map_or("", |(dir, _)| dir); |
| 142 | // A plan that names a directory, or a file at the root, covers less. |
| 143 | let dir = if file.ends_with('/') { file.trim_end_matches('/') } else { dir }; |
| 144 | !dir.is_empty() && path.starts_with(&format!("{dir}/")) |
| 145 | }) |
| 146 | } |
| 147 | |
| 148 | fn plural(n: u32, one: &str, many: &str) -> String { |
| 149 | format!("{n} {}", if n == 1 { one } else { many }) |
| 150 | } |
| 151 | |
| 152 | /// What lowered confidence: how much, and in a few words. `None` points |
| 153 | /// makes it low on its own. |
| 154 | struct Mark { |
| 155 | points: Option<u32>, |
| 156 | reason: String, |
| 157 | } |
| 158 | |
| 159 | fn sink(reason: impl Into<String>) -> Mark { |
| 160 | Mark { points: None, reason: reason.into() } |
| 161 | } |
| 162 | |
| 163 | fn points(points: u32, reason: impl Into<String>) -> Mark { |
| 164 | Mark { points: Some(points), reason: reason.into() } |
| 165 | } |
| 166 | |
| 167 | /// How sure g1t is of a change, from `signals`. Each signal that tells |
| 168 | /// against it adds points, or makes it low outright; no points is high, |
| 169 | /// one or two medium, three or more low. The agent's own word, when it |
| 170 | /// gave one, can only make it lower. |
| 171 | pub(crate) fn score(signals: &Signals) -> (ConfidenceLevel, Vec<String>) { |
| 172 | let mut marks: Vec<Mark> = Vec::new(); |
| 173 | |
| 174 | match signals.check_status { |
| 175 | Some(CheckStatus::Failed) => marks.push(sink("checks failing")), |
| 176 | Some(CheckStatus::Errored) => marks.push(sink("checks could not run")), |
| 177 | Some(CheckStatus::Queued | CheckStatus::Running) => marks.push(points(1, "checks not finished")), |
| 178 | None if signals.has_checks => marks.push(points(1, "checks not run yet")), |
| 179 | Some(CheckStatus::Passed) | None => {} |
| 180 | } |
| 181 | if !signals.has_checks { |
| 182 | marks.push(points(1, "no acceptance checks")); |
| 183 | } |
| 184 | if signals.flaky_checks && signals.check_status == Some(CheckStatus::Passed) { |
| 185 | marks.push(points(1, "checks passed only on a retry")); |
| 186 | } |
| 187 | |
| 188 | match signals.revisions { |
| 189 | 0 => {} |
| 190 | n @ (1 | 2) => marks.push(points(n, plural(n, "revision", "revisions"))), |
| 191 | n => marks.push(points(3, plural(n, "revision", "revisions"))), |
| 192 | } |
| 193 | |
| 194 | match signals.review { |
| 195 | Some(Verdict::RequestChanges) => marks.push(sink("reviewer asked for changes")), |
| 196 | Some(Verdict::Approve) if signals.review_comments >= 3 => { |
| 197 | marks.push(points(1, format!("reviewer left {} comments", signals.review_comments))); |
| 198 | } |
| 199 | Some(Verdict::Approve) => {} |
| 200 | None if signals.agent_review => marks.push(points(1, "not reviewed yet")), |
| 201 | None => marks.push(points(1, "no review")), |
| 202 | } |
| 203 | |
| 204 | let code: Vec<&ChangedFile> = signals |
| 205 | .files |
| 206 | .iter() |
| 207 | .filter(|file| !is_test(&file.path) && !is_prose(&file.path)) |
| 208 | .collect(); |
| 209 | let tests = signals.files.iter().filter(|file| is_test(&file.path)).count(); |
| 210 | if !code.is_empty() && tests == 0 { |
| 211 | marks.push(points(1, "tests not added")); |
| 212 | } |
| 213 | |
| 214 | let lines: u32 = signals.files.iter().map(|file| file.additions + file.deletions).sum(); |
| 215 | if lines > 1000 { |
| 216 | marks.push(points(2, format!("large change ({lines} lines)"))); |
| 217 | } else if lines > 400 { |
| 218 | marks.push(points(1, format!("{lines} lines changed"))); |
| 219 | } |
| 220 | let files = signals.files.len() as u32; |
| 221 | if files > 30 { |
| 222 | marks.push(points(1, format!("{files} files changed"))); |
| 223 | } |
| 224 | if !signals.expected.is_empty() { |
| 225 | let outside = signals |
| 226 | .files |
| 227 | .iter() |
| 228 | .filter(|file| !is_test(&file.path) && !is_prose(&file.path)) |
| 229 | .filter(|file| !expected(&file.path, &signals.expected)) |
| 230 | .count() as u32; |
| 231 | if outside > 0 { |
| 232 | marks.push(points( |
| 233 | if outside >= 4 { 2 } else { 1 }, |
| 234 | format!("{} outside the planned area", plural(outside, "file", "files")), |
| 235 | )); |
| 236 | } |
| 237 | } |
| 238 | let mut touched: Vec<&str> = signals.files.iter().filter_map(|file| sensitive(&file.path)).collect(); |
| 239 | touched.dedup(); |
| 240 | if let Some(first) = touched.first() { |
| 241 | marks.push(points(2, format!("touches {first}"))); |
| 242 | } |
| 243 | |
| 244 | match signals.halted.as_deref() { |
| 245 | Some("budget") => marks.push(sink("stopped at its cost cap")), |
| 246 | Some("time") => marks.push(sink("stopped at its time cap")), |
| 247 | _ => { |
| 248 | if signals.budget_share.is_some_and(|share| share >= NEAR_CAP) { |
| 249 | let share = (signals.budget_share.unwrap_or_default() * 100.0).round() as u32; |
| 250 | marks.push(points(1, format!("used {}% of its cost cap", share.min(100)))); |
| 251 | } |
| 252 | if signals.time_share.is_some_and(|share| share >= NEAR_CAP) { |
| 253 | let share = (signals.time_share.unwrap_or_default() * 100.0).round() as u32; |
| 254 | marks.push(points(1, format!("used {}% of its time cap", share.min(100)))); |
| 255 | } |
| 256 | } |
| 257 | } |
| 258 | if signals.denials > 0 { |
| 259 | marks.push(points( |
| 260 | if signals.denials >= 3 { 2 } else { 1 }, |
| 261 | format!("{} by guardrails", plural(signals.denials, "step refused", "steps refused")), |
| 262 | )); |
| 263 | } |
| 264 | if signals.unanswered > 0 { |
| 265 | marks.push(points( |
| 266 | 2, |
| 267 | plural(signals.unanswered, "question unanswered", "questions unanswered"), |
| 268 | )); |
| 269 | } |
| 270 | if !signals.uncertain_about.is_empty() { |
| 271 | marks.push(points(1, format!("agent unsure about {}", signals.uncertain_about[0]))); |
| 272 | } |
| 273 | |
| 274 | let sunk = marks.iter().any(|mark| mark.points.is_none()); |
| 275 | let total: u32 = marks.iter().filter_map(|mark| mark.points).sum(); |
| 276 | let observed = if sunk || total >= LOW_AT { |
| 277 | ConfidenceLevel::Low |
| 278 | } else if total > 0 { |
| 279 | ConfidenceLevel::Medium |
| 280 | } else { |
| 281 | ConfidenceLevel::High |
| 282 | }; |
| 283 | let level = signals.self_reported.map_or(observed, |said| said.min(observed)); |
| 284 | |
| 285 | // Most telling first: what makes it low on its own, then by weight. |
| 286 | marks.sort_by_key(|mark| std::cmp::Reverse(mark.points.unwrap_or(u32::MAX))); |
| 287 | let mut reasons: Vec<String> = Vec::new(); |
| 288 | if let Some(said) = signals.self_reported.filter(|said| *said < observed) { |
| 289 | reasons.push(format!("agent says {}", said.as_str())); |
| 290 | } |
| 291 | reasons.extend(marks.into_iter().map(|mark| mark.reason)); |
| 292 | if reasons.is_empty() { |
| 293 | // High: what it rests on. |
| 294 | if signals.has_checks && signals.check_status == Some(CheckStatus::Passed) { |
| 295 | reasons.push("checks pass".to_owned()); |
| 296 | } |
| 297 | if signals.review == Some(Verdict::Approve) { |
| 298 | reasons.push(if signals.revisions == 0 { "approved on first review" } else { "review approved" }.to_owned()); |
| 299 | } |
| 300 | if tests > 0 { |
| 301 | reasons.push("tests added".to_owned()); |
| 302 | } |
| 303 | if lines > 0 && lines <= 100 { |
| 304 | reasons.push("small change".to_owned()); |
| 305 | } |
| 306 | } |
| 307 | reasons.truncate(MAX_REASONS); |
| 308 | (level, reasons) |
| 309 | } |
| 310 | |
| 311 | /// A self-report's doubts, tidied: short, distinct, a few. |
| 312 | fn tidy(uncertain: &[String]) -> Vec<String> { |
| 313 | let mut out: Vec<String> = Vec::new(); |
| 314 | for item in uncertain { |
| 315 | let line = item.split_whitespace().collect::<Vec<_>>().join(" "); |
| 316 | let line: String = line.chars().take(MAX_UNCERTAIN_CHARS).collect(); |
| 317 | if !line.is_empty() && !out.contains(&line) { |
| 318 | out.push(line); |
| 319 | } |
| 320 | if out.len() == MAX_UNCERTAIN { |
| 321 | break; |
| 322 | } |
| 323 | } |
| 324 | out |
| 325 | } |
| 326 | |
| 327 | #[derive(Deserialize)] |
| 328 | struct CheckRow { |
| 329 | head_commit: String, |
| 330 | status: String, |
| 331 | } |
| 332 | |
| 333 | #[derive(Deserialize)] |
| 334 | struct LatestRun { |
| 335 | id: String, |
| 336 | cost_usd: Option<f64>, |
| 337 | budget_usd: Option<f64>, |
| 338 | time_cap_minutes: Option<u32>, |
| 339 | minutes: Option<f64>, |
| 340 | self_level: Option<String>, |
| 341 | uncertain_about: Option<String>, |
| 342 | } |
| 343 | |
| 344 | #[derive(Deserialize)] |
| 345 | struct Halted { |
| 346 | halted: Option<String>, |
| 347 | } |
| 348 | |
| 349 | #[derive(Deserialize)] |
| 350 | struct PlannedFiles { |
| 351 | files: Option<String>, |
| 352 | } |
| 353 | |
| 354 | #[derive(Deserialize)] |
| 355 | struct RunTicket { |
| 356 | repo_id: String, |
| 357 | pull_id: Option<String>, |
| 358 | token_hash: String, |
| 359 | } |
| 360 | |
| 361 | /// Whether a list of check runs, oldest first, shows checks that pass but |
| 362 | /// not reliably: one errored, or failed and later passed on the same commit. |
| 363 | pub(crate) fn flaky(runs: &[(String, String)]) -> bool { |
| 364 | runs.iter().any(|(_, status)| status == "errored") |
| 365 | || runs.iter().enumerate().any(|(index, (commit, status))| { |
| 366 | status == "failed" && runs[index + 1..].iter().any(|(later, status)| later == commit && status == "passed") |
| 367 | }) |
| 368 | } |
| 369 | |
| 370 | impl Work { |
| 371 | /// Whether the repository asks a person before merging a low-confidence |
| 372 | /// change. On unless someone turned it off. |
| 373 | pub(crate) async fn holds_low_confidence(&self, repo_id: &str) -> Result<bool> { |
| 374 | Ok(self |
| 375 | .db |
| 376 | .prepare("SELECT hold_low AS n FROM confidence_rules WHERE repo_id = ?") |
| 377 | .bind(&[repo_id.into()])? |
| 378 | .first::<NumberRow>(None) |
| 379 | .await? |
| 380 | .is_none_or(|row| row.n != 0)) |
| 381 | } |
| 382 | |
| 383 | /// Records whether the repository holds low-confidence changes. |
| 384 | pub(crate) async fn set_hold_low_confidence(&self, repo_id: &str, hold: bool, by: &str, at: &str) -> Result<()> { |
| 385 | self.db |
| 386 | .prepare( |
| 387 | "INSERT INTO confidence_rules (repo_id, hold_low, updated_by, updated_at) VALUES (?, ?, ?, ?) |
| 388 | ON CONFLICT (repo_id) DO UPDATE SET |
| 389 | hold_low = excluded.hold_low, updated_by = excluded.updated_by, updated_at = excluded.updated_at", |
| 390 | ) |
| 391 | .bind(&[repo_id.into(), u32::from(hold).into(), by.into(), at.into()])? |
| 392 | .run() |
| 393 | .await?; |
| 394 | Ok(()) |
| 395 | } |
| 396 | |
| 397 | /// The signals for a pull request a g1t agent has finished, besides the |
| 398 | /// ones its lifecycle already knows (`known`). |
| 399 | async fn signals(&self, pull: &Pull, known: Signals) -> Result<(Signals, Option<String>)> { |
| 400 | let checks = async { |
| 401 | let rows = self |
| 402 | .db |
| 403 | .prepare("SELECT head_commit, status FROM check_runs WHERE pull_id = ? ORDER BY id LIMIT 50") |
| 404 | .bind(&[pull.id.as_str().into()])? |
| 405 | .all() |
| 406 | .await? |
| 407 | .results::<CheckRow>()?; |
| 408 | Ok::<_, worker::Error>(rows.into_iter().map(|row| (row.head_commit, row.status)).collect::<Vec<_>>()) |
| 409 | }; |
| 410 | let run = async { |
| 411 | // The latest run that worked on the change, and what its agent |
| 412 | // said of it. |
| 413 | let latest = self |
| 414 | .db |
| 415 | .prepare( |
| 416 | "SELECT r.id, r.cost_usd, r.budget_usd, r.time_cap_minutes, |
| 417 | (julianday(COALESCE(r.finished_at, r.updated_at)) - julianday(COALESCE(r.started_at, r.created_at))) * 1440 AS minutes, |
| 418 | c.self_level, c.uncertain_about |
| 419 | FROM agent_runs r LEFT JOIN run_confidence c ON c.run_id = r.id |
| 420 | WHERE r.pull_id = ? AND r.kind IN ('implement', 'revise') |
| 421 | ORDER BY r.created_at DESC LIMIT 1", |
| 422 | ) |
| 423 | .bind(&[pull.id.as_str().into()])? |
| 424 | .first::<LatestRun>(None) |
| 425 | .await?; |
| 426 | // Any run on it stopped at a cap. |
| 427 | let halted = self |
| 428 | .db |
| 429 | .prepare( |
| 430 | "SELECT halted FROM agent_runs WHERE pull_id = ? AND halted IS NOT NULL |
| 431 | ORDER BY created_at DESC LIMIT 1", |
| 432 | ) |
| 433 | .bind(&[pull.id.as_str().into()])? |
| 434 | .first::<Halted>(None) |
| 435 | .await? |
| 436 | .and_then(|row| row.halted); |
| 437 | Ok::<_, worker::Error>((latest, halted)) |
| 438 | }; |
| 439 | let counts = async { |
| 440 | let denials = self |
| 441 | .db |
| 442 | .prepare( |
| 443 | "SELECT count(*) AS n FROM session_entries |
| 444 | WHERE pull_id = ? AND kind = 'note' AND text LIKE 'Denied:%'", |
| 445 | ) |
| 446 | .bind(&[pull.id.as_str().into()])? |
| 447 | .first::<NumberRow>(None) |
| 448 | .await? |
| 449 | .map_or(0, |row| row.n); |
| 450 | let unanswered = self |
| 451 | .db |
| 452 | .prepare( |
| 453 | "SELECT count(*) AS n FROM agent_messages |
| 454 | WHERE repo_id = ? AND from_number = ? AND kind IN ('question', 'handoff') |
| 455 | AND answered_at IS NULL", |
| 456 | ) |
| 457 | .bind(&[pull.repo_id.as_str().into(), pull.number.into()])? |
| 458 | .first::<NumberRow>(None) |
| 459 | .await? |
| 460 | .map_or(0, |row| row.n); |
| 461 | let planned = match pull.issue { |
| 462 | Some(number) => self |
| 463 | .db |
| 464 | .prepare( |
| 465 | "SELECT json_extract(planned.value, '$.files') AS files |
| 466 | FROM plans, json_each(plans.issues) AS planned |
| 467 | WHERE plans.repo_id = ? AND plans.status = 'applied' |
| 468 | AND json_extract(planned.value, '$.number') = ? |
| 469 | LIMIT 1", |
| 470 | ) |
| 471 | .bind(&[pull.repo_id.as_str().into(), number.into()])? |
| 472 | .first::<PlannedFiles>(None) |
| 473 | .await? |
| 474 | .and_then(|row| row.files) |
| 475 | .and_then(|files| serde_json::from_str::<Vec<String>>(&files).ok()) |
| 476 | .unwrap_or_default(), |
| 477 | None => Vec::new(), |
| 478 | }; |
| 479 | Ok::<_, worker::Error>((denials, unanswered, planned)) |
| 480 | }; |
| 481 | let (runs, ((latest, halted), (denials, unanswered, expected))) = |
| 482 | try_join(checks, try_join(run, counts)).await?; |
| 483 | let run_id = latest.as_ref().map(|run| run.id.clone()); |
| 484 | let share = |used: Option<f64>, cap: Option<f64>| match (used, cap) { |
| 485 | (Some(used), Some(cap)) if cap > 0.0 => Some(used / cap), |
| 486 | _ => None, |
| 487 | }; |
| 488 | let signals = Signals { |
| 489 | flaky_checks: flaky(&runs), |
| 490 | files: pull.files.clone(), |
| 491 | expected, |
| 492 | halted, |
| 493 | budget_share: latest.as_ref().and_then(|run| share(run.cost_usd, run.budget_usd)), |
| 494 | time_share: latest |
| 495 | .as_ref() |
| 496 | .and_then(|run| share(run.minutes, run.time_cap_minutes.map(f64::from))), |
| 497 | denials, |
| 498 | unanswered, |
| 499 | self_reported: latest |
| 500 | .as_ref() |
| 501 | .and_then(|run| run.self_level.as_deref()) |
| 502 | .and_then(ConfidenceLevel::parse), |
| 503 | uncertain_about: latest |
| 504 | .as_ref() |
| 505 | .and_then(|run| run.uncertain_about.as_deref()) |
| 506 | .and_then(|items| serde_json::from_str::<Vec<String>>(items).ok()) |
| 507 | .unwrap_or_default(), |
| 508 | ..known |
| 509 | }; |
| 510 | Ok((signals, run_id)) |
| 511 | } |
| 512 | |
| 513 | /// Works out how sure g1t is of a pull request a g1t agent has finished |
| 514 | /// (`known` holds what its lifecycle already read), and records it on |
| 515 | /// the pull request and on the run that left it so when it changed. |
| 516 | pub(crate) async fn assess_confidence(&self, pull: &Pull, known: Signals) -> Result<Confidence> { |
| 517 | let (signals, run_id) = self.signals(pull, known).await?; |
| 518 | let (level, reasons) = score(&signals); |
| 519 | let unchanged = pull.confidence.as_ref().filter(|was| { |
| 520 | was.level == level |
| 521 | && was.reasons == reasons |
| 522 | && was.self_reported == signals.self_reported |
| 523 | && was.uncertain_about == signals.uncertain_about |
| 524 | && was.run_id == run_id |
| 525 | }); |
| 526 | if let Some(was) = unchanged { |
| 527 | return Ok(was.clone()); |
| 528 | } |
| 529 | let now = rfc3339(now_ms()); |
| 530 | let confidence = Confidence { |
| 531 | level, |
| 532 | reasons, |
| 533 | self_reported: signals.self_reported, |
| 534 | uncertain_about: signals.uncertain_about, |
| 535 | run_id: run_id.clone(), |
| 536 | assessed_at: now.clone(), |
| 537 | }; |
| 538 | let detail = serde_json::to_string(&confidence)?; |
| 539 | self.db |
| 540 | .prepare( |
| 541 | "INSERT INTO pull_confidence (pull_id, repo_id, level, detail, updated_at) VALUES (?, ?, ?, ?, ?) |
| 542 | ON CONFLICT (pull_id) DO UPDATE SET |
| 543 | level = excluded.level, detail = excluded.detail, updated_at = excluded.updated_at", |
| 544 | ) |
| 545 | .bind(&[ |
| 546 | pull.id.as_str().into(), |
| 547 | pull.repo_id.as_str().into(), |
| 548 | level.as_str().into(), |
| 549 | detail.as_str().into(), |
| 550 | now.as_str().into(), |
| 551 | ])? |
| 552 | .run() |
| 553 | .await?; |
| 554 | if let Some(run_id) = &run_id { |
| 555 | self.db |
| 556 | .prepare( |
| 557 | "INSERT INTO run_confidence (run_id, pull_id, repo_id, detail, updated_at) VALUES (?, ?, ?, ?, ?) |
| 558 | ON CONFLICT (run_id) DO UPDATE SET detail = excluded.detail, updated_at = excluded.updated_at", |
| 559 | ) |
| 560 | .bind(&[ |
| 561 | run_id.as_str().into(), |
| 562 | pull.id.as_str().into(), |
| 563 | pull.repo_id.as_str().into(), |
| 564 | detail.as_str().into(), |
| 565 | now.as_str().into(), |
| 566 | ])? |
| 567 | .run() |
| 568 | .await?; |
| 569 | } |
| 570 | Ok(confidence) |
| 571 | } |
| 572 | |
| 573 | /// What the agent of a run said of its own change, with the run's token. |
| 574 | pub(crate) async fn report_confidence(&self, a: ReportConfidenceArgs) -> Result<Outcome<bool>> { |
| 575 | let run = self |
| 576 | .db |
| 577 | .prepare("SELECT repo_id, pull_id, token_hash FROM agent_runs WHERE id = ?") |
| 578 | .bind(&[a.run_id.as_str().into()])? |
| 579 | .first::<RunTicket>(None) |
| 580 | .await? |
| 581 | .filter(|run| !a.token.is_empty() && run.token_hash == hash(&a.token)); |
| 582 | let Some(run) = run else { |
| 583 | return Ok(Outcome::fail(FailureCode::NotFound, "Run not found.")); |
| 584 | }; |
| 585 | let Some(level) = ConfidenceLevel::parse(&a.confidence) else { |
| 586 | return Ok(Outcome::fail(FailureCode::Invalid, "confidence is high, medium or low.")); |
| 587 | }; |
| 588 | let uncertain = serde_json::to_string(&tidy(&a.uncertain_about))?; |
| 589 | self.db |
| 590 | .prepare( |
| 591 | "INSERT INTO run_confidence (run_id, pull_id, repo_id, self_level, uncertain_about, updated_at) |
| 592 | VALUES (?, ?, ?, ?, ?, ?) |
| 593 | ON CONFLICT (run_id) DO UPDATE SET |
| 594 | self_level = excluded.self_level, uncertain_about = excluded.uncertain_about, |
| 595 | updated_at = excluded.updated_at", |
| 596 | ) |
| 597 | .bind(&[ |
| 598 | a.run_id.as_str().into(), |
| 599 | run.pull_id.as_deref().map_or(JsValue::NULL, JsValue::from), |
| 600 | run.repo_id.as_str().into(), |
| 601 | level.as_str().into(), |
| 602 | uncertain.into(), |
| 603 | rfc3339(now_ms()).into(), |
| 604 | ])? |
| 605 | .run() |
| 606 | .await?; |
| 607 | Ok(Outcome::Ok(true)) |
| 608 | } |
| 609 | |
| 610 | /// How many comments on lines a review by g1t's agent left, which it |
| 611 | /// records at the moment it finished. |
| 612 | pub(crate) async fn review_comments(&self, pull: &Pull, finished_at: &str) -> Result<u32> { |
| 613 | Ok(self |
| 614 | .db |
| 615 | .prepare( |
| 616 | "SELECT count(*) AS n FROM comments |
| 617 | WHERE repo_id = ? AND number = ? AND author_id = ? AND created_at = ? AND path IS NOT NULL", |
| 618 | ) |
| 619 | .bind(&[ |
| 620 | pull.repo_id.as_str().into(), |
| 621 | pull.number.into(), |
| 622 | AGENT_ID.into(), |
| 623 | finished_at.into(), |
| 624 | ])? |
| 625 | .first::<NumberRow>(None) |
| 626 | .await? |
| 627 | .map_or(0, |row| row.n)) |
| 628 | } |
| 629 | |
| 630 | /// Whether a person other than the author approved the change since the |
| 631 | /// agent last revised it: someone has looked, so a hold is lifted. |
| 632 | pub(crate) async fn person_approved(&self, pull: &Pull, revised_at: Option<&str>) -> Result<bool> { |
| 633 | #[derive(Deserialize)] |
| 634 | struct Latest { |
| 635 | verdict: String, |
| 636 | created_at: String, |
| 637 | } |
| 638 | let rows = self |
| 639 | .db |
| 640 | .prepare( |
| 641 | "SELECT verdict, created_at FROM comments |
| 642 | WHERE repo_id = ? AND number = ? AND verdict IS NOT NULL |
| 643 | AND author_id != ? AND author_id != ? AND author_id != ? |
| 644 | ORDER BY id DESC LIMIT 20", |
| 645 | ) |
| 646 | .bind(&[ |
| 647 | pull.repo_id.as_str().into(), |
| 648 | pull.number.into(), |
| 649 | pull.author.id.as_str().into(), |
| 650 | AGENT_ID.into(), |
| 651 | crate::lifecycle::POLICY_ACTOR_ID.into(), |
| 652 | ])? |
| 653 | .all() |
| 654 | .await? |
| 655 | .results::<Latest>()?; |
| 656 | Ok(rows |
| 657 | .first() |
| 658 | .is_some_and(|latest| latest.verdict == "approve" && revised_at.is_none_or(|revised| latest.created_at.as_str() > revised))) |
| 659 | } |
| 660 | } |
| 661 | |
| 662 | #[cfg(test)] |
| 663 | mod tests { |
| 664 | use super::*; |
| 665 | |
| 666 | fn file(path: &str, lines: u32) -> ChangedFile { |
| 667 | ChangedFile { path: path.to_owned(), additions: lines, deletions: 0 } |
| 668 | } |
| 669 | |
| 670 | /// A clean change: checks pass, approved on the first review, a test |
| 671 | /// beside the code, small. |
| 672 | fn clean() -> Signals { |
| 673 | Signals { |
| 674 | has_checks: true, |
| 675 | check_status: Some(CheckStatus::Passed), |
| 676 | agent_review: true, |
| 677 | review: Some(Verdict::Approve), |
| 678 | files: vec![file("src/retry.ts", 40), file("src/retry.test.ts", 30)], |
| 679 | ..Signals::default() |
| 680 | } |
| 681 | } |
| 682 | |
| 683 | #[test] |
| 684 | fn signals_score_as_the_table_says() { |
| 685 | use ConfidenceLevel::*; |
| 686 | let cases: Vec<(&str, Signals, ConfidenceLevel, &[&str])> = vec![ |
| 687 | ("clean", clean(), High, &["checks pass", "approved on first review", "tests added", "small change"]), |
| 688 | ( |
| 689 | "failing checks", |
| 690 | Signals { check_status: Some(CheckStatus::Failed), ..clean() }, |
| 691 | Low, |
| 692 | &["checks failing"], |
| 693 | ), |
| 694 | ( |
| 695 | "checks that could not run", |
| 696 | Signals { check_status: Some(CheckStatus::Errored), ..clean() }, |
| 697 | Low, |
| 698 | &["checks could not run"], |
| 699 | ), |
| 700 | ("one revision", Signals { revisions: 1, ..clean() }, Medium, &["1 revision"]), |
| 701 | ("three revisions", Signals { revisions: 3, ..clean() }, Low, &["3 revisions"]), |
| 702 | ( |
| 703 | "no tests and three revisions", |
| 704 | Signals { revisions: 3, files: vec![file("src/retry.ts", 40)], ..clean() }, |
| 705 | Low, |
| 706 | &["3 revisions", "tests not added"], |
| 707 | ), |
| 708 | ( |
| 709 | "no tests", |
| 710 | Signals { files: vec![file("src/retry.ts", 40)], ..clean() }, |
| 711 | Medium, |
| 712 | &["tests not added"], |
| 713 | ), |
| 714 | ( |
| 715 | "docs need no tests", |
| 716 | Signals { files: vec![file("README.md", 40), file("docs/guide.md", 10)], ..clean() }, |
| 717 | High, |
| 718 | &["checks pass", "approved on first review", "small change"], |
| 719 | ), |
| 720 | ( |
| 721 | "the reviewer asks for changes", |
| 722 | Signals { review: Some(Verdict::RequestChanges), ..clean() }, |
| 723 | Low, |
| 724 | &["reviewer asked for changes"], |
| 725 | ), |
| 726 | ( |
| 727 | "a review with much to say", |
| 728 | Signals { review_comments: 4, ..clean() }, |
| 729 | Medium, |
| 730 | &["reviewer left 4 comments"], |
| 731 | ), |
| 732 | ("no review yet", Signals { review: None, ..clean() }, Medium, &["not reviewed yet"]), |
| 733 | ( |
| 734 | "no reviewer agent and no checks", |
| 735 | Signals { review: None, agent_review: false, has_checks: false, check_status: None, ..clean() }, |
| 736 | Medium, |
| 737 | &["no acceptance checks", "no review"], |
| 738 | ), |
| 739 | ( |
| 740 | "a large change", |
| 741 | Signals { files: vec![file("src/a.ts", 900), file("src/a.test.ts", 300)], ..clean() }, |
| 742 | Medium, |
| 743 | &["large change (1200 lines)"], |
| 744 | ), |
| 745 | ( |
| 746 | "outside the planned area", |
| 747 | Signals { |
| 748 | expected: vec!["src/retry.ts".to_owned()], |
| 749 | files: vec![file("src/retry.ts", 10), file("lib/billing/charge.ts", 10), file("src/retry.test.ts", 5)], |
| 750 | ..clean() |
| 751 | }, |
| 752 | Medium, |
| 753 | &["1 file outside the planned area"], |
| 754 | ), |
| 755 | ( |
| 756 | "beside the planned files is inside", |
| 757 | Signals { |
| 758 | expected: vec!["src/webhooks/retry.ts".to_owned()], |
| 759 | files: vec![file("src/webhooks/backoff.ts", 10), file("src/webhooks/retry.test.ts", 5)], |
| 760 | ..clean() |
| 761 | }, |
| 762 | High, |
| 763 | &["checks pass", "approved on first review", "tests added", "small change"], |
| 764 | ), |
| 765 | ( |
| 766 | "a CI workflow", |
| 767 | Signals { files: vec![file(".github/workflows/ci.yml", 5), file("src/a.test.ts", 5)], ..clean() }, |
| 768 | Medium, |
| 769 | &["touches CI workflows"], |
| 770 | ), |
| 771 | ( |
| 772 | "a CI workflow and a revision", |
| 773 | Signals { revisions: 1, files: vec![file(".github/workflows/ci.yml", 5), file("src/a.test.ts", 5)], ..clean() }, |
| 774 | Low, |
| 775 | &["touches CI workflows", "1 revision"], |
| 776 | ), |
| 777 | ("stopped at a cap", Signals { halted: Some("budget".to_owned()), ..clean() }, Low, &["stopped at its cost cap"]), |
| 778 | ( |
| 779 | "near its cost cap", |
| 780 | Signals { budget_share: Some(0.92), ..clean() }, |
| 781 | Medium, |
| 782 | &["used 92% of its cost cap"], |
| 783 | ), |
| 784 | ( |
| 785 | "flaky checks", |
| 786 | Signals { flaky_checks: true, ..clean() }, |
| 787 | Medium, |
| 788 | &["checks passed only on a retry"], |
| 789 | ), |
| 790 | ( |
| 791 | "unanswered questions", |
| 792 | Signals { unanswered: 2, ..clean() }, |
| 793 | Medium, |
| 794 | &["2 questions unanswered"], |
| 795 | ), |
| 796 | ( |
| 797 | "guardrails refused a lot", |
| 798 | Signals { denials: 3, revisions: 1, ..clean() }, |
| 799 | Low, |
| 800 | &["3 steps refused by guardrails", "1 revision"], |
| 801 | ), |
| 802 | ( |
| 803 | "the agent says low", |
| 804 | Signals { self_reported: Some(Low), ..clean() }, |
| 805 | Low, |
| 806 | &["agent says low"], |
| 807 | ), |
| 808 | ( |
| 809 | "the agent cannot raise it", |
| 810 | Signals { self_reported: Some(High), revisions: 1, ..clean() }, |
| 811 | Medium, |
| 812 | &["1 revision"], |
| 813 | ), |
| 814 | ( |
| 815 | "the agent's doubts count", |
| 816 | Signals { self_reported: Some(High), uncertain_about: vec!["the retry limit".to_owned()], ..clean() }, |
| 817 | Medium, |
| 818 | &["agent unsure about the retry limit"], |
| 819 | ), |
| 820 | ]; |
| 821 | for (name, signals, level, reasons) in cases { |
| 822 | let (got, why) = score(&signals); |
| 823 | assert_eq!(got, level, "{name}: {why:?}"); |
| 824 | assert_eq!(why, reasons.iter().map(|r| (*r).to_owned()).collect::<Vec<_>>(), "{name}"); |
| 825 | } |
| 826 | } |
| 827 | |
| 828 | #[test] |
| 829 | fn reasons_are_few_and_the_worst_come_first() { |
| 830 | let signals = Signals { |
| 831 | check_status: Some(CheckStatus::Failed), |
| 832 | revisions: 2, |
| 833 | files: vec![file("src/a.ts", 600), file(".env.example", 1)], |
| 834 | unanswered: 1, |
| 835 | ..clean() |
| 836 | }; |
| 837 | let (level, reasons) = score(&signals); |
| 838 | assert_eq!(level, ConfidenceLevel::Low); |
| 839 | assert_eq!(reasons.len(), MAX_REASONS); |
| 840 | assert_eq!(reasons[0], "checks failing"); |
| 841 | } |
| 842 | |
| 843 | #[test] |
| 844 | fn checks_that_pass_on_a_retry_are_flaky() { |
| 845 | let runs = |list: &[(&str, &str)]| list.iter().map(|(c, s)| ((*c).to_owned(), (*s).to_owned())).collect::<Vec<_>>(); |
| 846 | assert!(!flaky(&runs(&[("a", "failed"), ("b", "passed")]))); |
| 847 | assert!(flaky(&runs(&[("a", "failed"), ("a", "passed")]))); |
| 848 | assert!(flaky(&runs(&[("a", "errored"), ("b", "passed")]))); |
| 849 | assert!(!flaky(&runs(&[("a", "passed")]))); |
| 850 | } |
| 851 | |
| 852 | #[test] |
| 853 | fn tests_prose_and_sensitive_paths_are_told_apart() { |
| 854 | for path in ["src/__tests__/a.ts", "tests/test_api.py", "pkg/api_test.go", "src/a.spec.tsx", "crates/x/tests/it.rs"] { |
| 855 | assert!(is_test(path), "{path}"); |
| 856 | } |
| 857 | for path in ["src/testing.ts", "src/contest.rs", "attest/a.rs"] { |
| 858 | assert!(!is_test(path), "{path}"); |
| 859 | } |
| 860 | assert!(is_prose("docs/setup.md") && is_prose("README.md") && !is_prose("src/readme.ts")); |
| 861 | assert_eq!(sensitive(".github/workflows/ci.yml"), Some("CI workflows")); |
| 862 | assert_eq!(sensitive("apps/web/wrangler.jsonc"), Some("infrastructure")); |
| 863 | assert_eq!(sensitive("config/.env.production"), Some("secrets")); |
| 864 | assert_eq!(sensitive("src/env.ts"), None); |
| 865 | } |
| 866 | |
| 867 | #[test] |
| 868 | fn doubts_are_tidied() { |
| 869 | let items = vec![" the retry limit ".to_owned(), "the retry limit".to_owned(), String::new(), "x".repeat(400)]; |
| 870 | let tidied = tidy(&items); |
| 871 | assert_eq!(tidied.len(), 2); |
| 872 | assert_eq!(tidied[0], "the retry limit"); |
| 873 | assert_eq!(tidied[1].chars().count(), MAX_UNCERTAIN_CHARS); |
| 874 | } |
| 875 | } |