Skip to content
385 linesCodeBlameRaw
1/**
2 * Spend less, keep quality (docs.g1t.sh/guides/spend/#spend-less-keep-quality):
3 * a weekly check, per workspace, of whether each agent could run at a
4 * cheaper effort level without its work getting worse, judged only on the
5 * agent's own finished sessions.
6 *
7 * - **What counts as good work.** A session is accepted when it finished
8 * (`done`) with nobody having to step in: no one steered it, and it was
9 * not stopped or failed. Only root sessions count (a helper's or
10 * subagent's work is part of its root's), and only those that recorded
11 * the level they ran at.
12 * - **What is compared.** The level the agent runs at now (its setting, or
13 * for Auto the level most of its sessions ran at) against the level
14 * below, over the last `WINDOW_DAYS`.
15 * - **When it recommends.** Both levels have at least `MIN_SESSIONS`
16 * sessions; the cheaper level's acceptance is no more than
17 * `ACCEPT_TOLERANCE` below the current one's, and even its pessimistic
18 * estimate (a Wilson lower bound) is within `ACCEPT_FLOOR`; and a typical
19 * (median) session at the cheaper level costs at most `COST_SHARE` of
20 * one at the current level.
21 * - **When it can't tell.** Too few sessions on either side: a `thin`
22 * note that says how many it has and how many it needs, never a
23 * suggestion. When the cheaper level is measured and does worse, or
24 * saves too little, nothing is said.
25 *
26 * Every figure shown comes from those sessions' recorded charges, as
27 * budgets count them: the model and the agent rate where it applies. Pure
28 * functions first (tested in recommend.test.ts), then the database.
29 */
30import type { AgentEffort, AgentEffortCosts, AgentRecommendation, AgentRecommendations, EffortCost, EffortEvidence, EffortLevel } from "@g1t/contracts";
31
32import { readLook } from "../../../packages/contracts/src/agent-look.ts";
33
34import { dollars } from "./money.ts";
35import { effortOf, isLevel, lowerEffort } from "./routing.ts";
36import { type Row, definitionOf } from "./store.ts";
37
38export const WINDOW_DAYS = 28;
39export const MIN_SESSIONS = 10;
40export const ACCEPT_TOLERANCE = 0.05;
41export const ACCEPT_FLOOR = 0.15;
42export const COST_SHARE = 0.85;
43/** How often a workspace is checked. */
44export const CHECK_EVERY_DAYS = 7;
45/** Workspaces checked per run, so one run stays short. */
46const PER_RUN = 25;
47
48const LEVELS: EffortLevel[] = ["low", "medium", "high", "max"];
49const DAY_MS = 86_400_000;
50
51/** One finished session, as the check reads it. */
52export type SessionOutcome = { effort: EffortLevel; charged_micros: number; accepted: boolean };
53
54export function median(values: number[]): number {
55 if (!values.length) return 0;
56 const sorted = [...values].sort((a, b) => a - b);
57 const mid = Math.floor(sorted.length / 2);
58 return sorted.length % 2 ? sorted[mid] : Math.round((sorted[mid - 1] + sorted[mid]) / 2);
59}
60
61/** The lower end of a Wilson score interval at about 90%: how low a share could plausibly be. */
62export function wilsonLower(successes: number, trials: number, z = 1.2816): number {
63 if (trials <= 0) return 0;
64 const p = successes / trials;
65 const z2 = z * z;
66 const centre = p + z2 / (2 * trials);
67 const margin = z * Math.sqrt((p * (1 - p)) / trials + z2 / (4 * trials * trials));
68 return Math.max(0, (centre - margin) / (1 + z2 / trials));
69}
70
71export function evidenceAt(outcomes: SessionOutcome[], effort: EffortLevel): EffortEvidence | null {
72 const at = outcomes.filter((o) => o.effort === effort);
73 if (!at.length) return null;
74 const costs = at.map((o) => Math.max(0, o.charged_micros));
75 return {
76 effort,
77 sessions: at.length,
78 accepted: at.filter((o) => o.accepted).length,
79 typical_micros: median(costs),
80 mean_micros: Math.round(costs.reduce((n, c) => n + c, 0) / costs.length),
81 };
82}
83
84/** The level an agent runs at now: its setting, or for Auto the level most of its sessions ran at (the higher on a tie). */
85export function currentLevel(setting: AgentEffort, outcomes: SessionOutcome[]): EffortLevel | null {
86 if (setting !== "auto") return setting;
87 let best: EffortLevel | null = null;
88 let most = 0;
89 for (const level of LEVELS) {
90 const n = outcomes.filter((o) => o.effort === level).length;
91 if (n > 0 && n >= most) {
92 best = level;
93 most = n;
94 }
95 }
96 return best;
97}
98
99export type Judgement =
100 | { kind: "none" }
101 | { kind: "thin"; from: AgentEffort; to: EffortLevel; current: EffortEvidence | null; cheaper: EffortEvidence | null }
102 | { kind: "recommend"; from: AgentEffort; to: EffortLevel; current: EffortEvidence; cheaper: EffortEvidence; saving_month_micros: number };
103
104/** Whether to suggest a cheaper level for one agent, from its sessions in the window. */
105export function judge(setting: AgentEffort, outcomes: SessionOutcome[], windowDays = WINDOW_DAYS): Judgement {
106 if (!outcomes.length) return { kind: "none" };
107 const level = currentLevel(setting, outcomes);
108 if (!level) return { kind: "none" };
109 const to = lowerEffort(level);
110 if (!to) return { kind: "none" };
111 const current = evidenceAt(outcomes, level);
112 const cheaper = evidenceAt(outcomes, to);
113 if (!current || !cheaper || current.sessions < MIN_SESSIONS || cheaper.sessions < MIN_SESSIONS) {
114 return { kind: "thin", from: setting, to, current, cheaper };
115 }
116 const rateNow = current.accepted / current.sessions;
117 const rateCheaper = cheaper.accepted / cheaper.sessions;
118 if (rateCheaper < rateNow - ACCEPT_TOLERANCE) return { kind: "none" };
119 if (wilsonLower(cheaper.accepted, cheaper.sessions) < rateNow - ACCEPT_FLOOR) return { kind: "none" };
120 if (current.typical_micros <= 0 || cheaper.typical_micros > current.typical_micros * COST_SHARE) return { kind: "none" };
121 // At the pace it ran at the current level, what the difference comes to in 30 days.
122 const perDay = current.sessions / Math.max(1, windowDays);
123 const saving = Math.max(0, Math.round((current.mean_micros - cheaper.mean_micros) * perDay * 30));
124 if (saving <= 0) return { kind: "none" };
125 return { kind: "recommend", from: setting, to, current, cheaper, saving_month_micros: saving };
126}
127
128export const EFFORT_NAMES: Record<AgentEffort, string> = { auto: "Auto", low: "Low", medium: "Medium", high: "High", max: "Max" };
129
130/** What a judgement says, in words built only from its numbers. */
131export function wording(judgement: Exclude<Judgement, { kind: "none" }>, agentName: string, windowDays = WINDOW_DAYS): { title: string; reason: string } {
132 const to = EFFORT_NAMES[judgement.to];
133 const current = judgement.current;
134 const cheaper = judgement.cheaper;
135 const levelNow = current ? EFFORT_NAMES[current.effort] : judgement.from === "auto" ? "its usual level" : EFFORT_NAMES[judgement.from];
136 if (judgement.kind === "thin") {
137 const have = `${current?.sessions ?? 0} at ${levelNow} and ${cheaper?.sessions ?? 0} at ${to}`;
138 return {
139 title: `Not enough history to say whether ${agentName} can run at ${to}`,
140 reason: `Its finished sessions in the last ${windowDays} days: ${have}. A suggestion needs at least ${MIN_SESSIONS} at each.`,
141 };
142 }
143 const c = judgement.current;
144 const k = judgement.cheaper;
145 return {
146 title: `Run ${agentName} at ${to} effort`,
147 reason:
148 `At ${to}, ${k.accepted} of its ${k.sessions} sessions finished with nobody stepping in, against ${c.accepted} of ${c.sessions} at ${EFFORT_NAMES[c.effort]}; ` +
149 `a typical one cost ${dollars(k.typical_micros)} instead of ${dollars(c.typical_micros)}.`,
150 };
151}
152
153// --- The database ----------------------------------------------------------
154
155type OutcomeRow = { agent_id: string; effort: string | null; charged_micros: number; status: string; steered: number };
156
157/** Finished root sessions with a recorded level since `since`, by agent: the check's input. */
158export async function outcomesSince(db: D1Database, workspaceId: string, since: string, agentId?: string): Promise<Map<string, SessionOutcome[]>> {
159 const rows = await db
160 .prepare(
161 `SELECT s.agent_id, s.effort, s.charged_micros, s.status,
162 EXISTS (SELECT 1 FROM agent_session_events e WHERE e.session_id = s.id AND e.kind = 'steer') AS steered
163 FROM agent_sessions s
164 WHERE s.workspace_id = ?1 AND s.parent_id IS NULL AND s.effort IS NOT NULL AND s.finished_at >= ?2
165 AND s.status IN ('done', 'failed', 'stopped') AND (?3 IS NULL OR s.agent_id = ?3)`,
166 )
167 .bind(workspaceId, since, agentId ?? null)
168 .all<OutcomeRow>();
169 const by = new Map<string, SessionOutcome[]>();
170 for (const row of rows.results) {
171 if (!isLevel(row.effort)) continue;
172 const list = by.get(row.agent_id) ?? [];
173 list.push({ effort: row.effort, charged_micros: row.charged_micros, accepted: row.status === "done" && !row.steered });
174 by.set(row.agent_id, list);
175 }
176 return by;
177}
178
179/** What each level has cost one agent, from its own sessions. */
180export function effortCostsOf(handle: string, setting: AgentEffort, outcomes: SessionOutcome[], windowDays = WINDOW_DAYS): AgentEffortCosts {
181 const levels: EffortCost[] = LEVELS.map((effort) => {
182 const at = evidenceAt(outcomes, effort);
183 return {
184 effort,
185 sessions: at?.sessions ?? 0,
186 typical_micros: at ? at.typical_micros : null,
187 accepted_share: at ? at.accepted / at.sessions : null,
188 };
189 });
190 return { handle, effort: setting, window_days: windowDays, levels };
191}
192
193type RecommendationRow = {
194 id: string;
195 workspace_id: string;
196 agent_id: string;
197 kind: string;
198 status: string;
199 from_effort: string;
200 to_effort: string;
201 title: string;
202 reason: string;
203 evidence: string;
204 saving_month_micros: number | null;
205 checked_at: string;
206 resolved_by: string | null;
207 resolved_at: string | null;
208 handle?: string;
209 display_name?: string;
210 avatar_seed?: string;
211 look?: string | null;
212};
213
214function parse<T>(raw: string, fallback: T): T {
215 try {
216 return JSON.parse(raw) as T;
217 } catch {
218 return fallback;
219 }
220}
221
222export function toRecommendation(row: RecommendationRow): AgentRecommendation {
223 const evidence = parse<AgentRecommendation["evidence"]>(row.evidence, { window_days: WINDOW_DAYS, current: null, cheaper: null, needed: MIN_SESSIONS });
224 return {
225 id: row.id,
226 agent_id: row.agent_id,
227 agent_handle: row.handle ?? "agent",
228 agent_name: row.display_name ?? "An agent",
229 agent_avatar_seed: row.avatar_seed || row.handle || row.agent_id,
230 agent_look: readLook(row.look ?? null),
231 kind: "effort",
232 status: (["open", "applied", "dismissed", "thin"].includes(row.status) ? row.status : "open") as AgentRecommendation["status"],
233 from_effort: (row.from_effort as AgentEffort) ?? "auto",
234 to_effort: isLevel(row.to_effort) ? row.to_effort : "medium",
235 title: row.title,
236 reason: row.reason,
237 evidence,
238 saving_month_micros: row.saving_month_micros,
239 checked_at: row.checked_at,
240 resolved_by: row.resolved_by,
241 resolved_at: row.resolved_at,
242 };
243}
244
245const iso = (ms: number) => new Date(ms).toISOString();
246
247/**
248 * Checks one workspace: judges every agent that has finished sessions in
249 * the window, and writes what it found. Open and thin suggestions are
250 * refreshed in place; one that was applied or dismissed is left alone;
251 * open or thin ones that no longer hold (the setting changed, or the
252 * evidence moved) become `stale` and are no longer shown.
253 */
254export async function checkWorkspace(db: D1Database, workspaceId: string, now = Date.now()): Promise<{ open: number; thin: number }> {
255 const since = iso(now - WINDOW_DAYS * DAY_MS);
256 const checkedAt = iso(now);
257 const [outcomes, agents] = await Promise.all([
258 outcomesSince(db, workspaceId, since),
259 db.prepare("SELECT * FROM agents WHERE workspace_id = ? AND archived_at IS NULL").bind(workspaceId).all<Row>(),
260 ]);
261 const statements: D1PreparedStatement[] = [];
262 const keep: string[] = [];
263 let open = 0;
264 let thin = 0;
265 for (const agent of agents.results) {
266 const setting = effortOf(definitionOf(agent).routing);
267 const judgement = judge(setting, outcomes.get(agent.id) ?? []);
268 if (judgement.kind === "none") continue;
269 const words = wording(judgement, agent.display_name);
270 const status = judgement.kind === "recommend" ? "open" : "thin";
271 if (status === "open") open++;
272 else thin++;
273 const evidence = JSON.stringify({ window_days: WINDOW_DAYS, current: judgement.current, cheaper: judgement.cheaper, needed: MIN_SESSIONS });
274 const saving = judgement.kind === "recommend" ? judgement.saving_month_micros : null;
275 const id = `rec_${agent.id}_${judgement.from}_${judgement.to}`;
276 keep.push(id);
277 statements.push(
278 db
279 .prepare(
280 `INSERT INTO agent_recommendations (id, workspace_id, agent_id, kind, status, from_effort, to_effort, title, reason, evidence, saving_month_micros, checked_at, created_at)
281 VALUES (?1, ?2, ?3, 'effort', ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11, ?11)
282 ON CONFLICT (agent_id, kind, from_effort, to_effort) DO UPDATE SET
283 status = CASE WHEN agent_recommendations.status IN ('applied', 'dismissed') THEN agent_recommendations.status ELSE excluded.status END,
284 title = CASE WHEN agent_recommendations.status IN ('applied', 'dismissed') THEN agent_recommendations.title ELSE excluded.title END,
285 reason = CASE WHEN agent_recommendations.status IN ('applied', 'dismissed') THEN agent_recommendations.reason ELSE excluded.reason END,
286 evidence = CASE WHEN agent_recommendations.status IN ('applied', 'dismissed') THEN agent_recommendations.evidence ELSE excluded.evidence END,
287 saving_month_micros = CASE WHEN agent_recommendations.status IN ('applied', 'dismissed') THEN agent_recommendations.saving_month_micros ELSE excluded.saving_month_micros END,
288 checked_at = excluded.checked_at`,
289 )
290 .bind(id, workspaceId, agent.id, status, judgement.from, judgement.to, words.title, words.reason, evidence, saving, checkedAt),
291 );
292 }
293 // What no longer holds is put away; the record stays.
294 statements.push(
295 db
296 .prepare(
297 `UPDATE agent_recommendations SET status = 'stale', checked_at = ?2
298 WHERE workspace_id = ?1 AND status IN ('open', 'thin') AND id NOT IN (SELECT value FROM json_each(?3))`,
299 )
300 .bind(workspaceId, checkedAt, JSON.stringify(keep)),
301 );
302 statements.push(
303 db
304 .prepare("INSERT INTO agent_recommendation_checks (workspace_id, checked_at) VALUES (?, ?) ON CONFLICT (workspace_id) DO UPDATE SET checked_at = excluded.checked_at")
305 .bind(workspaceId, checkedAt),
306 );
307 await db.batch(statements);
308 return { open, thin };
309}
310
311/**
312 * The scheduled run: the workspaces with sessions finished in the window
313 * whose last check is a week old or more (or never), a few at a time.
314 * Called from the cron; does its work only in the first five minutes of an
315 * hour, so the query runs hourly, and each workspace weekly.
316 */
317export async function checkDue(db: D1Database, now = Date.now()): Promise<number> {
318 if (new Date(now).getUTCMinutes() >= 5) return 0;
319 const due = await db
320 .prepare(
321 `SELECT DISTINCT s.workspace_id FROM agent_sessions s
322 LEFT JOIN agent_recommendation_checks c ON c.workspace_id = s.workspace_id
323 WHERE s.finished_at >= ?1 AND s.effort IS NOT NULL AND (c.checked_at IS NULL OR c.checked_at < ?2)
324 LIMIT ?3`,
325 )
326 .bind(iso(now - WINDOW_DAYS * DAY_MS), iso(now - CHECK_EVERY_DAYS * DAY_MS), PER_RUN)
327 .all<{ workspace_id: string }>();
328 let checked = 0;
329 for (const { workspace_id } of due.results) {
330 try {
331 await checkWorkspace(db, workspace_id, now);
332 checked++;
333 } catch (error) {
334 console.error("agents: the spend check failed for a workspace", workspace_id, String(error));
335 }
336 }
337 return checked;
338}
339
340/** What the Spend page shows: open suggestions, thin notes, and what was decided lately. */
341export async function readRecommendations(db: D1Database, workspaceId: string, agentId: string | null, now = Date.now()): Promise<AgentRecommendations> {
342 const [rows, check] = await Promise.all([
343 db
344 .prepare(
345 `SELECT r.*, a.handle, a.display_name, a.avatar_seed, a.look FROM agent_recommendations r JOIN agents a ON a.id = r.agent_id
346 WHERE r.workspace_id = ?1 AND a.archived_at IS NULL AND (?2 IS NULL OR r.agent_id = ?2)
347 AND (r.status IN ('open', 'thin') OR (r.status IN ('applied', 'dismissed') AND r.resolved_at >= ?3))
348 ORDER BY r.saving_month_micros DESC, a.handle`,
349 )
350 .bind(workspaceId, agentId, iso(now - 30 * DAY_MS))
351 .all<RecommendationRow>(),
352 db.prepare("SELECT checked_at FROM agent_recommendation_checks WHERE workspace_id = ?").bind(workspaceId).first<{ checked_at: string }>(),
353 ]);
354 const all = rows.results.map(toRecommendation);
355 return {
356 checked_at: check?.checked_at ?? null,
357 window_days: WINDOW_DAYS,
358 open: all.filter((r) => r.status === "open"),
359 thin: all.filter((r) => r.status === "thin"),
360 resolved: all.filter((r) => r.status === "applied" || r.status === "dismissed").sort((a, b) => (b.resolved_at ?? "").localeCompare(a.resolved_at ?? "")),
361 };
362}
363
364export async function recommendationRow(db: D1Database, workspaceId: string, id: string): Promise<RecommendationRow | null> {
365 return db
366 .prepare(
367 `SELECT r.*, a.handle, a.display_name, a.avatar_seed, a.look FROM agent_recommendations r JOIN agents a ON a.id = r.agent_id
368 WHERE r.workspace_id = ? AND r.id = ?`,
369 )
370 .bind(workspaceId, id)
371 .first<RecommendationRow>();
372}
373
374/** Marks one decided, only from open: two owners pressing at once decide it once. */
375export async function markResolved(db: D1Database, id: string, status: "applied" | "dismissed", by: string, now = Date.now()): Promise<boolean> {
376 const result = await db
377 .prepare("UPDATE agent_recommendations SET status = ?, resolved_by = ?, resolved_at = ? WHERE id = ? AND status = 'open'")
378 .bind(status, by, iso(now), id)
379 .run();
380 return (result.meta.changes ?? 0) > 0;
381}
382
383export function sinceWindow(now = Date.now()): string {
384 return iso(now - WINDOW_DAYS * DAY_MS);
385}