| 1 | //! Memory that fills itself. Kept by the work service, beside memory. |
| 2 | //! |
| 3 | //! What agents and people learn arrives as a **candidate**: from what an |
| 4 | //! agent reports learning at the end of a run, a person's correction in a |
| 5 | //! review of an agent's pull request, a merged pull request's decision, or |
| 6 | //! a project's own docs and manifests. A candidate is given to no agent |
| 7 | //! until it is **kept**: by a member in the Review queue, or by g1t when |
| 8 | //! two independent sources say the same thing, or a doc says it with high |
| 9 | //! confidence ([`promotes`]). Nothing that looks like a secret is stored. |
| 10 | //! |
| 11 | //! Each `*Args` struct is the argument of the method of the same name, |
| 12 | //! served at `POST /rpc/<method>`. |
| 13 | |
| 14 | use serde::{Deserialize, Serialize}; |
| 15 | |
| 16 | use crate::agents::{Memory, MemoryKind, MemoryScope, MemoryStatus}; |
| 17 | use crate::repos::RepoPath; |
| 18 | use crate::{User, Viewer}; |
| 19 | |
| 20 | /// The confidence at or above which a doc's word is kept without review. |
| 21 | pub const DOC_CONFIDENCE: f64 = 0.85; |
| 22 | /// How many independent sources keep a candidate without review. |
| 23 | pub const INDEPENDENT_SOURCES: usize = 2; |
| 24 | /// The most items one capture takes. |
| 25 | pub const MAX_CAPTURE: usize = 50; |
| 26 | /// The most an agent reports learning in one run. |
| 27 | pub const MAX_LEARNED: usize = 8; |
| 28 | /// The longest piece of evidence kept, in characters. |
| 29 | pub const MAX_EVIDENCE_CHARS: usize = 400; |
| 30 | |
| 31 | /// Where a captured memory came from. |
| 32 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] |
| 33 | #[serde(rename_all = "snake_case")] |
| 34 | pub enum CaptureSource { |
| 35 | /// An agent's run, reporting what it learned. |
| 36 | Run, |
| 37 | /// A person's review of an agent's pull request. |
| 38 | Review, |
| 39 | /// A merged pull request. |
| 40 | Pr, |
| 41 | /// A project's README, AGENTS.md, docs or manifests. |
| 42 | Doc, |
| 43 | /// Written by a person. |
| 44 | Manual, |
| 45 | } |
| 46 | |
| 47 | impl CaptureSource { |
| 48 | pub fn as_str(self) -> &'static str { |
| 49 | match self { |
| 50 | CaptureSource::Run => "run", |
| 51 | CaptureSource::Review => "review", |
| 52 | CaptureSource::Pr => "pr", |
| 53 | CaptureSource::Doc => "doc", |
| 54 | CaptureSource::Manual => "manual", |
| 55 | } |
| 56 | } |
| 57 | } |
| 58 | |
| 59 | /// Whether a candidate is kept without anyone reviewing it: said by |
| 60 | /// `sources` independent sources, or by a doc with `confidence` at or above |
| 61 | /// [`DOC_CONFIDENCE`]. |
| 62 | pub fn promotes(sources: usize, kind: CaptureSource, confidence: Option<f64>) -> bool { |
| 63 | sources >= INDEPENDENT_SOURCES |
| 64 | || (kind == CaptureSource::Doc && confidence.is_some_and(|c| c >= DOC_CONFIDENCE)) |
| 65 | } |
| 66 | |
| 67 | /// A memory's text folded for comparing: lowercase words and digits, one |
| 68 | /// space apart, so wording that differs only in case, punctuation or |
| 69 | /// spacing is the same memory. |
| 70 | pub fn fingerprint(text: &str) -> String { |
| 71 | text.to_lowercase() |
| 72 | .split(|c: char| !c.is_alphanumeric()) |
| 73 | .filter(|word| !word.is_empty()) |
| 74 | .collect::<Vec<_>>() |
| 75 | .join(" ") |
| 76 | } |
| 77 | |
| 78 | /// How alike two memories' words must be (shared over all, as sets) to be |
| 79 | /// one memory. |
| 80 | pub const SAME_WORDS: f64 = 0.85; |
| 81 | /// The fewest distinct words a memory needs before it can be found inside |
| 82 | /// another: shorter ones are too general to be the same thing. |
| 83 | const CONTAINED_MIN_WORDS: usize = 6; |
| 84 | |
| 85 | /// Whether two memories say the same thing: the same words once folded |
| 86 | /// ([`fingerprint`]); one's words, in order, inside the other's; all of a |
| 87 | /// memory's words (at least six of them) among the other's; or most of |
| 88 | /// their words shared (Jaccard at or above [`SAME_WORDS`]). "g1t is a |
| 89 | /// Cargo workspace (apps/*, crates/*)" and the same sentence listing every |
| 90 | /// crate are one memory. |
| 91 | pub fn same_memory(a: &str, b: &str) -> bool { |
| 92 | let (a, b) = (fingerprint(a), fingerprint(b)); |
| 93 | if a.is_empty() || b.is_empty() { |
| 94 | return false; |
| 95 | } |
| 96 | if a == b { |
| 97 | return true; |
| 98 | } |
| 99 | let (short, long) = if a.len() <= b.len() { (&a, &b) } else { (&b, &a) }; |
| 100 | let short_words: std::collections::HashSet<&str> = short.split(' ').collect(); |
| 101 | let long_words: std::collections::HashSet<&str> = long.split(' ').collect(); |
| 102 | // Contiguous: the shorter one's words, in order, inside the longer one's. |
| 103 | if short.split(' ').count() >= 4 && format!(" {long} ").contains(&format!(" {short} ")) { |
| 104 | return true; |
| 105 | } |
| 106 | if short_words.len() >= CONTAINED_MIN_WORDS && short_words.is_subset(&long_words) { |
| 107 | return true; |
| 108 | } |
| 109 | let shared = short_words.intersection(&long_words).count(); |
| 110 | let all = short_words.union(&long_words).count(); |
| 111 | all > 0 && shared as f64 / all as f64 >= SAME_WORDS |
| 112 | } |
| 113 | |
| 114 | /// One thing learned, as a source reports it. |
| 115 | #[derive(Clone, Debug, Serialize, Deserialize)] |
| 116 | #[serde(rename_all = "camelCase")] |
| 117 | pub struct CaptureItem { |
| 118 | pub scope: MemoryScope, |
| 119 | /// For a project's memory: its repository's id. |
| 120 | #[serde(default)] |
| 121 | pub repo_id: Option<String>, |
| 122 | #[serde(default)] |
| 123 | pub kind: MemoryKind, |
| 124 | pub text: String, |
| 125 | /// 0 to 1. |
| 126 | #[serde(default)] |
| 127 | pub confidence: Option<f64>, |
| 128 | pub source: CaptureSource, |
| 129 | /// What it came from: `run:<id>`, `comment:<id>`, `pull:<repo id>#<n>`, |
| 130 | /// `doc:<repo id>:<path>`. Two items with the same reference are one |
| 131 | /// source, however often they arrive. |
| 132 | pub reference: String, |
| 133 | /// What it was learned from, quoted. |
| 134 | #[serde(default)] |
| 135 | pub evidence: Option<String>, |
| 136 | /// The pull request it was learned on. |
| 137 | #[serde(default)] |
| 138 | pub number: Option<u32>, |
| 139 | /// The agent run it was learned in. |
| 140 | #[serde(default)] |
| 141 | pub run_id: Option<String>, |
| 142 | } |
| 143 | |
| 144 | /// `capture_memories`: candidates from a service (the context service's |
| 145 | /// backfill and doc reading). Called by services only. Returns `Captured`. |
| 146 | #[derive(Debug, Serialize, Deserialize)] |
| 147 | #[serde(rename_all = "camelCase")] |
| 148 | pub struct CaptureMemoriesArgs { |
| 149 | /// The workspace's slug. |
| 150 | pub workspace: String, |
| 151 | pub items: Vec<CaptureItem>, |
| 152 | /// Who it is recorded as written by, such as `g1t`. |
| 153 | #[serde(default)] |
| 154 | pub by: Option<String>, |
| 155 | } |
| 156 | |
| 157 | #[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] |
| 158 | #[serde(rename_all = "camelCase")] |
| 159 | pub struct Captured { |
| 160 | /// New candidates. |
| 161 | pub added: u32, |
| 162 | /// Ones that matched a memory already there, as another sighting. |
| 163 | pub merged: u32, |
| 164 | /// Ones kept by the promotion rule, new or merged. |
| 165 | pub kept: u32, |
| 166 | /// Ones refused: empty, too long, or holding something like a secret. |
| 167 | pub refused: u32, |
| 168 | } |
| 169 | |
| 170 | /// What an agent says it learned, as the harness reports it. |
| 171 | #[derive(Clone, Debug, Serialize, Deserialize)] |
| 172 | #[serde(rename_all = "camelCase")] |
| 173 | pub struct LearnedItem { |
| 174 | /// `fact`, `convention`, `decision` or `gotcha`. |
| 175 | #[serde(default)] |
| 176 | pub kind: Option<String>, |
| 177 | /// `project` or `workspace`. |
| 178 | #[serde(default)] |
| 179 | pub scope: Option<String>, |
| 180 | pub text: String, |
| 181 | /// What showed it: a command's output, a file, a failing test. |
| 182 | #[serde(default)] |
| 183 | pub evidence: Option<String>, |
| 184 | } |
| 185 | |
| 186 | /// `report_learned`: a sandbox reporting what its agent learned, with the |
| 187 | /// run's token. Each item arrives as a candidate from the run. Returns |
| 188 | /// `Outcome<Captured>`. |
| 189 | #[derive(Debug, Default, Serialize, Deserialize)] |
| 190 | #[serde(rename_all = "camelCase", default)] |
| 191 | pub struct ReportLearnedArgs { |
| 192 | pub run_id: String, |
| 193 | pub token: String, |
| 194 | pub items: Vec<LearnedItem>, |
| 195 | } |
| 196 | |
| 197 | /// `list_candidates`: memory waiting for review in a workspace and, with |
| 198 | /// `repo`, only that project's. Members only. Returns `Outcome<Vec<Memory>>`. |
| 199 | #[derive(Debug, Serialize, Deserialize)] |
| 200 | pub struct ListCandidatesArgs { |
| 201 | pub viewer: Viewer, |
| 202 | pub workspace: String, |
| 203 | #[serde(default)] |
| 204 | pub repo: Option<RepoPath>, |
| 205 | } |
| 206 | |
| 207 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] |
| 208 | #[serde(rename_all = "snake_case")] |
| 209 | pub enum ReviewDecision { |
| 210 | Keep, |
| 211 | Dismiss, |
| 212 | } |
| 213 | |
| 214 | /// `review_memory`: a member keeps a candidate, edited or as it is, or |
| 215 | /// dismisses it. A dismissed memory is not suggested again from the same |
| 216 | /// wording. Returns `Outcome<Memory>`. |
| 217 | #[derive(Debug, Serialize, Deserialize)] |
| 218 | pub struct ReviewMemoryArgs { |
| 219 | pub actor: User, |
| 220 | pub workspace: String, |
| 221 | pub id: String, |
| 222 | pub decision: ReviewDecision, |
| 223 | #[serde(default)] |
| 224 | pub text: Option<String>, |
| 225 | #[serde(default)] |
| 226 | pub kind: Option<MemoryKind>, |
| 227 | } |
| 228 | |
| 229 | /// `memories_by_id`: memories as the context service indexes them, in any |
| 230 | /// status, so a dismissed one can be taken out of search. Services only. |
| 231 | /// Returns `Vec<Memory>`. |
| 232 | #[derive(Debug, Serialize, Deserialize)] |
| 233 | pub struct MemoriesByIdArgs { |
| 234 | pub workspace: String, |
| 235 | pub ids: Vec<String>, |
| 236 | } |
| 237 | |
| 238 | /// `search_memories`: kept memories in a workspace whose text has every |
| 239 | /// word of `query`, pinned first; with `repo_ids`, only the workspace's own |
| 240 | /// and those projects'. Services only: the caller has decided the viewer |
| 241 | /// may read the workspace's memory. Returns `Vec<Memory>`. |
| 242 | #[derive(Debug, Serialize, Deserialize)] |
| 243 | #[serde(rename_all = "camelCase")] |
| 244 | pub struct SearchMemoriesArgs { |
| 245 | pub workspace: String, |
| 246 | #[serde(default)] |
| 247 | pub query: Option<String>, |
| 248 | #[serde(default)] |
| 249 | pub repo_ids: Option<Vec<String>>, |
| 250 | #[serde(default)] |
| 251 | pub status: Option<MemoryStatus>, |
| 252 | #[serde(default)] |
| 253 | pub limit: Option<u32>, |
| 254 | } |
| 255 | |
| 256 | /// `seed_from_pulls`: decision candidates from a repository's last merged |
| 257 | /// pull requests, and convention candidates from people's reviews of its |
| 258 | /// agents' ones. Services only (the backfill). Returns `Captured`. |
| 259 | #[derive(Debug, Serialize, Deserialize)] |
| 260 | #[serde(rename_all = "camelCase")] |
| 261 | pub struct SeedFromPullsArgs { |
| 262 | pub repo_id: String, |
| 263 | /// At most 50; 20 if not given. |
| 264 | #[serde(default)] |
| 265 | pub limit: Option<u32>, |
| 266 | } |
| 267 | |
| 268 | /// `prune_doc_candidates`: after the context service has read every file |
| 269 | /// of a project, removes the project's candidates that came only from its |
| 270 | /// docs and manifests and are still waiting, unless one of `texts` (what |
| 271 | /// they suggest now) is the same memory ([`same_memory`]). Kept and |
| 272 | /// dismissed memory is never touched. Services only. Returns `Pruned`. |
| 273 | #[derive(Debug, Serialize, Deserialize)] |
| 274 | #[serde(rename_all = "camelCase")] |
| 275 | pub struct PruneDocCandidatesArgs { |
| 276 | pub workspace: String, |
| 277 | pub repo_id: String, |
| 278 | #[serde(default)] |
| 279 | pub texts: Vec<String>, |
| 280 | } |
| 281 | |
| 282 | #[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] |
| 283 | pub struct Pruned { |
| 284 | /// Candidates removed. |
| 285 | pub removed: u32, |
| 286 | } |
| 287 | |
| 288 | /// Memories, newest first, for listing a review queue. |
| 289 | pub type Candidates = Vec<Memory>; |
| 290 | |
| 291 | /// The `memory.changed` event: a memory was added, changed, reviewed or |
| 292 | /// forgotten. Carries no text; a subscriber asks for the memory by id. Its |
| 293 | /// `repo_id` is the project's for a project's memory, none for the |
| 294 | /// workspace's. |
| 295 | #[derive(Debug, Serialize, Deserialize)] |
| 296 | #[serde(rename_all = "camelCase")] |
| 297 | pub struct MemoryChanged { |
| 298 | pub memory_id: String, |
| 299 | pub workspace: String, |
| 300 | /// `candidate`, `kept` or `dismissed`; `deleted` once forgotten. |
| 301 | pub status: String, |
| 302 | } |
| 303 | |
| 304 | #[cfg(test)] |
| 305 | mod tests { |
| 306 | use super::*; |
| 307 | |
| 308 | #[test] |
| 309 | fn two_independent_sources_keep_a_candidate() { |
| 310 | assert!(!promotes(1, CaptureSource::Run, Some(0.9))); |
| 311 | assert!(promotes(2, CaptureSource::Run, None)); |
| 312 | assert!(promotes(3, CaptureSource::Review, Some(0.1))); |
| 313 | } |
| 314 | |
| 315 | #[test] |
| 316 | fn a_confident_doc_is_kept_and_a_doubtful_one_waits() { |
| 317 | assert!(promotes(1, CaptureSource::Doc, Some(0.9))); |
| 318 | assert!(promotes(1, CaptureSource::Doc, Some(DOC_CONFIDENCE))); |
| 319 | assert!(!promotes(1, CaptureSource::Doc, Some(0.6))); |
| 320 | assert!(!promotes(1, CaptureSource::Doc, None)); |
| 321 | // Only a doc's confidence counts on its own. |
| 322 | assert!(!promotes(1, CaptureSource::Pr, Some(1.0))); |
| 323 | } |
| 324 | |
| 325 | #[test] |
| 326 | fn fingerprints_ignore_case_punctuation_and_spacing() { |
| 327 | assert_eq!(fingerprint("Use pnpm, never npm!"), "use pnpm never npm"); |
| 328 | assert_eq!(fingerprint("use pnpm never NPM"), fingerprint("Use pnpm; never npm.")); |
| 329 | assert_ne!(fingerprint("use pnpm"), fingerprint("use npm")); |
| 330 | } |
| 331 | |
| 332 | #[test] |
| 333 | fn near_duplicates_are_one_memory() { |
| 334 | assert!(same_memory("Use pnpm, never npm.", "use pnpm never NPM")); |
| 335 | // A crate list that grew, and the same fact without the list. |
| 336 | let before = "g1t is a Cargo workspace (apps/api, crates/*, services/actions, services/billing); `cargo test` runs its tests."; |
| 337 | let after = "g1t is a Cargo workspace (apps/api, crates/*, services/actions, services/billing, services/work); `cargo test` runs its tests."; |
| 338 | assert!(same_memory(before, after)); |
| 339 | assert!(same_memory(before, "g1t is a Cargo workspace (apps/*, crates/*, services/*); `cargo test` runs its tests.")); |
| 340 | // One cut short, inside the whole of it. |
| 341 | assert!(same_memory( |
| 342 | "Roles. Viewer, commenter, planner and approver map onto", |
| 343 | "Roles. Viewer, commenter, planner and approver map onto the five repository roles." |
| 344 | )); |
| 345 | // Different facts that share words are not. |
| 346 | assert!(!same_memory("Use pnpm to install.", "Use npm to install.")); |
| 347 | assert!(!same_memory("Run cargo test in the crate you changed.", "Run npm test in the app you changed.")); |
| 348 | assert!(!same_memory("use pnpm", "use pnpm in the web app and npm in the docs, which predates it")); |
| 349 | assert!(!same_memory("", "anything")); |
| 350 | } |
| 351 | } |