| 1 | /** |
| 2 | * Spend less, keep quality (docs.g1t.sh/guides/spend/#spend-less-keep-quality): |
| 3 | * a weekly check, per workspace, of whether each agent could run at a |
| 4 | * cheaper effort level without its work getting worse, judged only on the |
| 5 | * agent's own finished sessions. |
| 6 | * |
| 7 | * - **What counts as good work.** A session is accepted when it finished |
| 8 | * (`done`) with nobody having to step in: no one steered it, and it was |
| 9 | * not stopped or failed. Only root sessions count (a helper's or |
| 10 | * subagent's work is part of its root's), and only those that recorded |
| 11 | * the level they ran at. |
| 12 | * - **What is compared.** The level the agent runs at now (its setting, or |
| 13 | * for Auto the level most of its sessions ran at) against the level |
| 14 | * below, over the last `WINDOW_DAYS`. |
| 15 | * - **When it recommends.** Both levels have at least `MIN_SESSIONS` |
| 16 | * sessions; the cheaper level's acceptance is no more than |
| 17 | * `ACCEPT_TOLERANCE` below the current one's, and even its pessimistic |
| 18 | * estimate (a Wilson lower bound) is within `ACCEPT_FLOOR`; and a typical |
| 19 | * (median) session at the cheaper level costs at most `COST_SHARE` of |
| 20 | * one at the current level. |
| 21 | * - **When it can't tell.** Too few sessions on either side: a `thin` |
| 22 | * note that says how many it has and how many it needs, never a |
| 23 | * suggestion. When the cheaper level is measured and does worse, or |
| 24 | * saves too little, nothing is said. |
| 25 | * |
| 26 | * Every figure shown comes from those sessions' recorded charges, as |
| 27 | * budgets count them: the model and the agent rate where it applies. Pure |
| 28 | * functions first (tested in recommend.test.ts), then the database. |
| 29 | */ |
| 30 | import type { AgentEffort, AgentEffortCosts, AgentRecommendation, AgentRecommendations, EffortCost, EffortEvidence, EffortLevel } from "@g1t/contracts"; |
| 31 | |
| 32 | import { dollars } from "./money.ts"; |
| 33 | import { effortOf, isLevel, lowerEffort } from "./routing.ts"; |
| 34 | import { type Row, definitionOf } from "./store.ts"; |
| 35 | |
| 36 | export const WINDOW_DAYS = 28; |
| 37 | export const MIN_SESSIONS = 10; |
| 38 | export const ACCEPT_TOLERANCE = 0.05; |
| 39 | export const ACCEPT_FLOOR = 0.15; |
| 40 | export const COST_SHARE = 0.85; |
| 41 | /** How often a workspace is checked. */ |
| 42 | export const CHECK_EVERY_DAYS = 7; |
| 43 | /** Workspaces checked per run, so one run stays short. */ |
| 44 | const PER_RUN = 25; |
| 45 | |
| 46 | const LEVELS: EffortLevel[] = ["low", "medium", "high", "max"]; |
| 47 | const DAY_MS = 86_400_000; |
| 48 | |
| 49 | /** One finished session, as the check reads it. */ |
| 50 | export type SessionOutcome = { effort: EffortLevel; charged_micros: number; accepted: boolean }; |
| 51 | |
| 52 | export function median(values: number[]): number { |
| 53 | if (!values.length) return 0; |
| 54 | const sorted = [...values].sort((a, b) => a - b); |
| 55 | const mid = Math.floor(sorted.length / 2); |
| 56 | return sorted.length % 2 ? sorted[mid] : Math.round((sorted[mid - 1] + sorted[mid]) / 2); |
| 57 | } |
| 58 | |
| 59 | /** The lower end of a Wilson score interval at about 90%: how low a share could plausibly be. */ |
| 60 | export function wilsonLower(successes: number, trials: number, z = 1.2816): number { |
| 61 | if (trials <= 0) return 0; |
| 62 | const p = successes / trials; |
| 63 | const z2 = z * z; |
| 64 | const centre = p + z2 / (2 * trials); |
| 65 | const margin = z * Math.sqrt((p * (1 - p)) / trials + z2 / (4 * trials * trials)); |
| 66 | return Math.max(0, (centre - margin) / (1 + z2 / trials)); |
| 67 | } |
| 68 | |
| 69 | export function evidenceAt(outcomes: SessionOutcome[], effort: EffortLevel): EffortEvidence | null { |
| 70 | const at = outcomes.filter((o) => o.effort === effort); |
| 71 | if (!at.length) return null; |
| 72 | const costs = at.map((o) => Math.max(0, o.charged_micros)); |
| 73 | return { |
| 74 | effort, |
| 75 | sessions: at.length, |
| 76 | accepted: at.filter((o) => o.accepted).length, |
| 77 | typical_micros: median(costs), |
| 78 | mean_micros: Math.round(costs.reduce((n, c) => n + c, 0) / costs.length), |
| 79 | }; |
| 80 | } |
| 81 | |
| 82 | /** The level an agent runs at now: its setting, or for Auto the level most of its sessions ran at (the higher on a tie). */ |
| 83 | export function currentLevel(setting: AgentEffort, outcomes: SessionOutcome[]): EffortLevel | null { |
| 84 | if (setting !== "auto") return setting; |
| 85 | let best: EffortLevel | null = null; |
| 86 | let most = 0; |
| 87 | for (const level of LEVELS) { |
| 88 | const n = outcomes.filter((o) => o.effort === level).length; |
| 89 | if (n > 0 && n >= most) { |
| 90 | best = level; |
| 91 | most = n; |
| 92 | } |
| 93 | } |
| 94 | return best; |
| 95 | } |
| 96 | |
| 97 | export type Judgement = |
| 98 | | { kind: "none" } |
| 99 | | { kind: "thin"; from: AgentEffort; to: EffortLevel; current: EffortEvidence | null; cheaper: EffortEvidence | null } |
| 100 | | { kind: "recommend"; from: AgentEffort; to: EffortLevel; current: EffortEvidence; cheaper: EffortEvidence; saving_month_micros: number }; |
| 101 | |
| 102 | /** Whether to suggest a cheaper level for one agent, from its sessions in the window. */ |
| 103 | export function judge(setting: AgentEffort, outcomes: SessionOutcome[], windowDays = WINDOW_DAYS): Judgement { |
| 104 | if (!outcomes.length) return { kind: "none" }; |
| 105 | const level = currentLevel(setting, outcomes); |
| 106 | if (!level) return { kind: "none" }; |
| 107 | const to = lowerEffort(level); |
| 108 | if (!to) return { kind: "none" }; |
| 109 | const current = evidenceAt(outcomes, level); |
| 110 | const cheaper = evidenceAt(outcomes, to); |
| 111 | if (!current || !cheaper || current.sessions < MIN_SESSIONS || cheaper.sessions < MIN_SESSIONS) { |
| 112 | return { kind: "thin", from: setting, to, current, cheaper }; |
| 113 | } |
| 114 | const rateNow = current.accepted / current.sessions; |
| 115 | const rateCheaper = cheaper.accepted / cheaper.sessions; |
| 116 | if (rateCheaper < rateNow - ACCEPT_TOLERANCE) return { kind: "none" }; |
| 117 | if (wilsonLower(cheaper.accepted, cheaper.sessions) < rateNow - ACCEPT_FLOOR) return { kind: "none" }; |
| 118 | if (current.typical_micros <= 0 || cheaper.typical_micros > current.typical_micros * COST_SHARE) return { kind: "none" }; |
| 119 | // At the pace it ran at the current level, what the difference comes to in 30 days. |
| 120 | const perDay = current.sessions / Math.max(1, windowDays); |
| 121 | const saving = Math.max(0, Math.round((current.mean_micros - cheaper.mean_micros) * perDay * 30)); |
| 122 | if (saving <= 0) return { kind: "none" }; |
| 123 | return { kind: "recommend", from: setting, to, current, cheaper, saving_month_micros: saving }; |
| 124 | } |
| 125 | |
| 126 | export const EFFORT_NAMES: Record<AgentEffort, string> = { auto: "Auto", low: "Low", medium: "Medium", high: "High", max: "Max" }; |
| 127 | |
| 128 | /** What a judgement says, in words built only from its numbers. */ |
| 129 | export function wording(judgement: Exclude<Judgement, { kind: "none" }>, agentName: string, windowDays = WINDOW_DAYS): { title: string; reason: string } { |
| 130 | const to = EFFORT_NAMES[judgement.to]; |
| 131 | const current = judgement.current; |
| 132 | const cheaper = judgement.cheaper; |
| 133 | const levelNow = current ? EFFORT_NAMES[current.effort] : judgement.from === "auto" ? "its usual level" : EFFORT_NAMES[judgement.from]; |
| 134 | if (judgement.kind === "thin") { |
| 135 | const have = `${current?.sessions ?? 0} at ${levelNow} and ${cheaper?.sessions ?? 0} at ${to}`; |
| 136 | return { |
| 137 | title: `Not enough history to say whether ${agentName} can run at ${to}`, |
| 138 | reason: `Its finished sessions in the last ${windowDays} days: ${have}. A suggestion needs at least ${MIN_SESSIONS} at each.`, |
| 139 | }; |
| 140 | } |
| 141 | const c = judgement.current; |
| 142 | const k = judgement.cheaper; |
| 143 | return { |
| 144 | title: `Run ${agentName} at ${to} effort`, |
| 145 | reason: |
| 146 | `At ${to}, ${k.accepted} of its ${k.sessions} sessions finished with nobody stepping in, against ${c.accepted} of ${c.sessions} at ${EFFORT_NAMES[c.effort]}; ` + |
| 147 | `a typical one cost ${dollars(k.typical_micros)} instead of ${dollars(c.typical_micros)}.`, |
| 148 | }; |
| 149 | } |
| 150 | |
| 151 | // --- The database ---------------------------------------------------------- |
| 152 | |
| 153 | type OutcomeRow = { agent_id: string; effort: string | null; charged_micros: number; status: string; steered: number }; |
| 154 | |
| 155 | /** Finished root sessions with a recorded level since `since`, by agent: the check's input. */ |
| 156 | export async function outcomesSince(db: D1Database, workspaceId: string, since: string, agentId?: string): Promise<Map<string, SessionOutcome[]>> { |
| 157 | const rows = await db |
| 158 | .prepare( |
| 159 | `SELECT s.agent_id, s.effort, s.charged_micros, s.status, |
| 160 | EXISTS (SELECT 1 FROM agent_session_events e WHERE e.session_id = s.id AND e.kind = 'steer') AS steered |
| 161 | FROM agent_sessions s |
| 162 | WHERE s.workspace_id = ?1 AND s.parent_id IS NULL AND s.effort IS NOT NULL AND s.finished_at >= ?2 |
| 163 | AND s.status IN ('done', 'failed', 'stopped') AND (?3 IS NULL OR s.agent_id = ?3)`, |
| 164 | ) |
| 165 | .bind(workspaceId, since, agentId ?? null) |
| 166 | .all<OutcomeRow>(); |
| 167 | const by = new Map<string, SessionOutcome[]>(); |
| 168 | for (const row of rows.results) { |
| 169 | if (!isLevel(row.effort)) continue; |
| 170 | const list = by.get(row.agent_id) ?? []; |
| 171 | list.push({ effort: row.effort, charged_micros: row.charged_micros, accepted: row.status === "done" && !row.steered }); |
| 172 | by.set(row.agent_id, list); |
| 173 | } |
| 174 | return by; |
| 175 | } |
| 176 | |
| 177 | /** What each level has cost one agent, from its own sessions. */ |
| 178 | export function effortCostsOf(handle: string, setting: AgentEffort, outcomes: SessionOutcome[], windowDays = WINDOW_DAYS): AgentEffortCosts { |
| 179 | const levels: EffortCost[] = LEVELS.map((effort) => { |
| 180 | const at = evidenceAt(outcomes, effort); |
| 181 | return { |
| 182 | effort, |
| 183 | sessions: at?.sessions ?? 0, |
| 184 | typical_micros: at ? at.typical_micros : null, |
| 185 | accepted_share: at ? at.accepted / at.sessions : null, |
| 186 | }; |
| 187 | }); |
| 188 | return { handle, effort: setting, window_days: windowDays, levels }; |
| 189 | } |
| 190 | |
| 191 | type RecommendationRow = { |
| 192 | id: string; |
| 193 | workspace_id: string; |
| 194 | agent_id: string; |
| 195 | kind: string; |
| 196 | status: string; |
| 197 | from_effort: string; |
| 198 | to_effort: string; |
| 199 | title: string; |
| 200 | reason: string; |
| 201 | evidence: string; |
| 202 | saving_month_micros: number | null; |
| 203 | checked_at: string; |
| 204 | resolved_by: string | null; |
| 205 | resolved_at: string | null; |
| 206 | handle?: string; |
| 207 | display_name?: string; |
| 208 | avatar_seed?: string; |
| 209 | }; |
| 210 | |
| 211 | function parse<T>(raw: string, fallback: T): T { |
| 212 | try { |
| 213 | return JSON.parse(raw) as T; |
| 214 | } catch { |
| 215 | return fallback; |
| 216 | } |
| 217 | } |
| 218 | |
| 219 | export function toRecommendation(row: RecommendationRow): AgentRecommendation { |
| 220 | const evidence = parse<AgentRecommendation["evidence"]>(row.evidence, { window_days: WINDOW_DAYS, current: null, cheaper: null, needed: MIN_SESSIONS }); |
| 221 | return { |
| 222 | id: row.id, |
| 223 | agent_id: row.agent_id, |
| 224 | agent_handle: row.handle ?? "agent", |
| 225 | agent_name: row.display_name ?? "An agent", |
| 226 | agent_avatar_seed: row.avatar_seed || row.handle || row.agent_id, |
| 227 | kind: "effort", |
| 228 | status: (["open", "applied", "dismissed", "thin"].includes(row.status) ? row.status : "open") as AgentRecommendation["status"], |
| 229 | from_effort: (row.from_effort as AgentEffort) ?? "auto", |
| 230 | to_effort: isLevel(row.to_effort) ? row.to_effort : "medium", |
| 231 | title: row.title, |
| 232 | reason: row.reason, |
| 233 | evidence, |
| 234 | saving_month_micros: row.saving_month_micros, |
| 235 | checked_at: row.checked_at, |
| 236 | resolved_by: row.resolved_by, |
| 237 | resolved_at: row.resolved_at, |
| 238 | }; |
| 239 | } |
| 240 | |
| 241 | const iso = (ms: number) => new Date(ms).toISOString(); |
| 242 | |
| 243 | /** |
| 244 | * Checks one workspace: judges every agent that has finished sessions in |
| 245 | * the window, and writes what it found. Open and thin suggestions are |
| 246 | * refreshed in place; one that was applied or dismissed is left alone; |
| 247 | * open or thin ones that no longer hold (the setting changed, or the |
| 248 | * evidence moved) become `stale` and are no longer shown. |
| 249 | */ |
| 250 | export async function checkWorkspace(db: D1Database, workspaceId: string, now = Date.now()): Promise<{ open: number; thin: number }> { |
| 251 | const since = iso(now - WINDOW_DAYS * DAY_MS); |
| 252 | const checkedAt = iso(now); |
| 253 | const [outcomes, agents] = await Promise.all([ |
| 254 | outcomesSince(db, workspaceId, since), |
| 255 | db.prepare("SELECT * FROM agents WHERE workspace_id = ? AND archived_at IS NULL").bind(workspaceId).all<Row>(), |
| 256 | ]); |
| 257 | const statements: D1PreparedStatement[] = []; |
| 258 | const keep: string[] = []; |
| 259 | let open = 0; |
| 260 | let thin = 0; |
| 261 | for (const agent of agents.results) { |
| 262 | const setting = effortOf(definitionOf(agent).routing); |
| 263 | const judgement = judge(setting, outcomes.get(agent.id) ?? []); |
| 264 | if (judgement.kind === "none") continue; |
| 265 | const words = wording(judgement, agent.display_name); |
| 266 | const status = judgement.kind === "recommend" ? "open" : "thin"; |
| 267 | if (status === "open") open++; |
| 268 | else thin++; |
| 269 | const evidence = JSON.stringify({ window_days: WINDOW_DAYS, current: judgement.current, cheaper: judgement.cheaper, needed: MIN_SESSIONS }); |
| 270 | const saving = judgement.kind === "recommend" ? judgement.saving_month_micros : null; |
| 271 | const id = `rec_${agent.id}_${judgement.from}_${judgement.to}`; |
| 272 | keep.push(id); |
| 273 | statements.push( |
| 274 | db |
| 275 | .prepare( |
| 276 | `INSERT INTO agent_recommendations (id, workspace_id, agent_id, kind, status, from_effort, to_effort, title, reason, evidence, saving_month_micros, checked_at, created_at) |
| 277 | VALUES (?1, ?2, ?3, 'effort', ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11, ?11) |
| 278 | ON CONFLICT (agent_id, kind, from_effort, to_effort) DO UPDATE SET |
| 279 | status = CASE WHEN agent_recommendations.status IN ('applied', 'dismissed') THEN agent_recommendations.status ELSE excluded.status END, |
| 280 | title = CASE WHEN agent_recommendations.status IN ('applied', 'dismissed') THEN agent_recommendations.title ELSE excluded.title END, |
| 281 | reason = CASE WHEN agent_recommendations.status IN ('applied', 'dismissed') THEN agent_recommendations.reason ELSE excluded.reason END, |
| 282 | evidence = CASE WHEN agent_recommendations.status IN ('applied', 'dismissed') THEN agent_recommendations.evidence ELSE excluded.evidence END, |
| 283 | saving_month_micros = CASE WHEN agent_recommendations.status IN ('applied', 'dismissed') THEN agent_recommendations.saving_month_micros ELSE excluded.saving_month_micros END, |
| 284 | checked_at = excluded.checked_at`, |
| 285 | ) |
| 286 | .bind(id, workspaceId, agent.id, status, judgement.from, judgement.to, words.title, words.reason, evidence, saving, checkedAt), |
| 287 | ); |
| 288 | } |
| 289 | // What no longer holds is put away; the record stays. |
| 290 | statements.push( |
| 291 | db |
| 292 | .prepare( |
| 293 | `UPDATE agent_recommendations SET status = 'stale', checked_at = ?2 |
| 294 | WHERE workspace_id = ?1 AND status IN ('open', 'thin') AND id NOT IN (SELECT value FROM json_each(?3))`, |
| 295 | ) |
| 296 | .bind(workspaceId, checkedAt, JSON.stringify(keep)), |
| 297 | ); |
| 298 | statements.push( |
| 299 | db |
| 300 | .prepare("INSERT INTO agent_recommendation_checks (workspace_id, checked_at) VALUES (?, ?) ON CONFLICT (workspace_id) DO UPDATE SET checked_at = excluded.checked_at") |
| 301 | .bind(workspaceId, checkedAt), |
| 302 | ); |
| 303 | await db.batch(statements); |
| 304 | return { open, thin }; |
| 305 | } |
| 306 | |
| 307 | /** |
| 308 | * The scheduled run: the workspaces with sessions finished in the window |
| 309 | * whose last check is a week old or more (or never), a few at a time. |
| 310 | * Called from the cron; does its work only in the first five minutes of an |
| 311 | * hour, so the query runs hourly, and each workspace weekly. |
| 312 | */ |
| 313 | export async function checkDue(db: D1Database, now = Date.now()): Promise<number> { |
| 314 | if (new Date(now).getUTCMinutes() >= 5) return 0; |
| 315 | const due = await db |
| 316 | .prepare( |
| 317 | `SELECT DISTINCT s.workspace_id FROM agent_sessions s |
| 318 | LEFT JOIN agent_recommendation_checks c ON c.workspace_id = s.workspace_id |
| 319 | WHERE s.finished_at >= ?1 AND s.effort IS NOT NULL AND (c.checked_at IS NULL OR c.checked_at < ?2) |
| 320 | LIMIT ?3`, |
| 321 | ) |
| 322 | .bind(iso(now - WINDOW_DAYS * DAY_MS), iso(now - CHECK_EVERY_DAYS * DAY_MS), PER_RUN) |
| 323 | .all<{ workspace_id: string }>(); |
| 324 | let checked = 0; |
| 325 | for (const { workspace_id } of due.results) { |
| 326 | try { |
| 327 | await checkWorkspace(db, workspace_id, now); |
| 328 | checked++; |
| 329 | } catch (error) { |
| 330 | console.error("agents: the spend check failed for a workspace", workspace_id, String(error)); |
| 331 | } |
| 332 | } |
| 333 | return checked; |
| 334 | } |
| 335 | |
| 336 | /** What the Spend page shows: open suggestions, thin notes, and what was decided lately. */ |
| 337 | export async function readRecommendations(db: D1Database, workspaceId: string, agentId: string | null, now = Date.now()): Promise<AgentRecommendations> { |
| 338 | const [rows, check] = await Promise.all([ |
| 339 | db |
| 340 | .prepare( |
| 341 | `SELECT r.*, a.handle, a.display_name, a.avatar_seed FROM agent_recommendations r JOIN agents a ON a.id = r.agent_id |
| 342 | WHERE r.workspace_id = ?1 AND a.archived_at IS NULL AND (?2 IS NULL OR r.agent_id = ?2) |
| 343 | AND (r.status IN ('open', 'thin') OR (r.status IN ('applied', 'dismissed') AND r.resolved_at >= ?3)) |
| 344 | ORDER BY r.saving_month_micros DESC, a.handle`, |
| 345 | ) |
| 346 | .bind(workspaceId, agentId, iso(now - 30 * DAY_MS)) |
| 347 | .all<RecommendationRow>(), |
| 348 | db.prepare("SELECT checked_at FROM agent_recommendation_checks WHERE workspace_id = ?").bind(workspaceId).first<{ checked_at: string }>(), |
| 349 | ]); |
| 350 | const all = rows.results.map(toRecommendation); |
| 351 | return { |
| 352 | checked_at: check?.checked_at ?? null, |
| 353 | window_days: WINDOW_DAYS, |
| 354 | open: all.filter((r) => r.status === "open"), |
| 355 | thin: all.filter((r) => r.status === "thin"), |
| 356 | resolved: all.filter((r) => r.status === "applied" || r.status === "dismissed").sort((a, b) => (b.resolved_at ?? "").localeCompare(a.resolved_at ?? "")), |
| 357 | }; |
| 358 | } |
| 359 | |
| 360 | export async function recommendationRow(db: D1Database, workspaceId: string, id: string): Promise<RecommendationRow | null> { |
| 361 | return db |
| 362 | .prepare( |
| 363 | `SELECT r.*, a.handle, a.display_name, a.avatar_seed FROM agent_recommendations r JOIN agents a ON a.id = r.agent_id |
| 364 | WHERE r.workspace_id = ? AND r.id = ?`, |
| 365 | ) |
| 366 | .bind(workspaceId, id) |
| 367 | .first<RecommendationRow>(); |
| 368 | } |
| 369 | |
| 370 | /** Marks one decided, only from open: two owners pressing at once decide it once. */ |
| 371 | export async function markResolved(db: D1Database, id: string, status: "applied" | "dismissed", by: string, now = Date.now()): Promise<boolean> { |
| 372 | const result = await db |
| 373 | .prepare("UPDATE agent_recommendations SET status = ?, resolved_by = ?, resolved_at = ? WHERE id = ? AND status = 'open'") |
| 374 | .bind(status, by, iso(now), id) |
| 375 | .run(); |
| 376 | return (result.meta.changes ?? 0) > 0; |
| 377 | } |
| 378 | |
| 379 | export function sinceWindow(now = Date.now()): string { |
| 380 | return iso(now - WINDOW_DAYS * DAY_MS); |
| 381 | } |