g1t/bench/reviewbench/run.mjs

278 lines12,729 bytesCodeBlame

Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.

Research: shipping CSS, and a ReviewBench harness for g1t's reviewer1#!/usr/bin/env node
2// Runs g1t's pull request reviewer over ReviewBench and scores it with the
3// benchmark's own judge. docs/research/reviewbench.md explains the design.
4//
5// node bench/reviewbench/run.mjs setup clone ReviewBench (pinned) and install its judge
6// node bench/reviewbench/run.mjs build build the agent image from the runner base image
7// node bench/reviewbench/run.mjs estimate --set sample:50
8// node bench/reviewbench/run.mjs review --set sample:50 [--seed 1] [--variant production] [--model claude-sonnet-5-5] --yes
9// node bench/reviewbench/run.mjs judge --run <id> [--judge-model claude-sonnet-5] --yes
10// node bench/reviewbench/run.mjs report --run <id> [--json]
11//
12// --set is test (the benchmark's 25), full (all 219) or sample:N (N of the
13// 219, stratified by change size, reproducible with --seed).
14//
15// Model access, from the environment, passed into each container:
16// ANTHROPIC_API_KEY a provider key, or
17// ANTHROPIC_BASE_URL + ANTHROPIC_API_KEY g1t's model proxy (MODELS_URL/anthropic and a session token)
18// The judge reads ANTHROPIC_API_KEY itself (through ReviewBench's pi registry).
19//
20// Nothing that spends money runs without --yes. Each review is capped by
21// --budget-usd (default 5), and a run stops once --max-total-usd is spent.
22//
23// Needs docker, git, jq and bash (Git Bash on Windows) and Node 20+.
24
25import { execFileSync, spawnSync } from "node:child_process";
26import { existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from "node:fs";
27import { dirname, join } from "node:path";
28import { fileURLToPath } from "node:url";
29
30const HERE = dirname(fileURLToPath(import.meta.url));
31const ROOT = join(HERE, "..", "..");
32const CACHE = join(HERE, ".cache");
33const RB = join(CACHE, "ReviewBench");
34const RUNS = join(CACHE, "runs");
35// The ReviewBench commit these numbers are comparable against.
36const RB_COMMIT = "ceb0794a3768da6ef4a56e5311dfb4afd29e5dee";
37const IMAGE = "g1t-reviewbench:dev";
38// The model the official leaderboard is judged with.
39const DEFAULT_JUDGE = "claude-sonnet-5";
40// services/runner/wrangler.jsonc AGENT_ROUTES.review
41const DEFAULT_MODEL = "claude-sonnet-5-5";
42
43// $ per million tokens: input, output, cache read, cache write (5 minute).
44const PRICES = {
45 "claude-sonnet-5-5": [2, 10, 0.2, 2.5],
46 "claude-sonnet-5": [2, 10, 0.2, 2.5],
47 "claude-opus-5-5": [4, 20, 0.2, 5],
48 "claude-haiku-4-5": [1, 5, 0.1, 1.25],
49};
50
51function args() {
52 const [command, ...rest] = process.argv.slice(2);
53 const flags = {};
54 for (let i = 0; i < rest.length; i++) {
55 const name = rest[i].replace(/^--/, "");
56 const next = rest[i + 1];
57 if (next === undefined || next.startsWith("--")) flags[name] = true;
58 else flags[name] = rest[++i];
59 }
60 return { command, flags };
61}
62
63const run = (cmd, argv, opts = {}) => {
64 const result = spawnSync(cmd, argv, { stdio: "inherit", ...opts });
65 if (result.status !== 0) throw new Error(`${cmd} ${argv.join(" ")} exited ${result.status}`);
66};
67
68function manifest() {
69 return JSON.parse(readFileSync(join(RB, "corpus", "manifest.json"), "utf8"));
70}
71
72/** A seeded shuffle, so a sample is the same sample next week. */
73function shuffle(items, seed) {
74 let state = seed >>> 0 || 1;
75 const random = () => ((state = (state * 1664525 + 1013904223) >>> 0) / 2 ** 32);
76 const copy = [...items];
77 for (let i = copy.length - 1; i > 0; i--) {
78 const j = Math.floor(random() * (i + 1));
79 [copy[i], copy[j]] = [copy[j], copy[i]];
80 }
81 return copy;
82}
83
84/** Indices into the full manifest for --set. */
85function select(set, seed) {
86 const all = manifest();
87 if (set === "full") return all.map((_, i) => i);
88 if (set === "test") {
89 const test = JSON.parse(readFileSync(join(RB, "corpus", "test", "test.json"), "utf8"));
90 const keys = new Set(test.map((p) => `${p.nwo}#${p.pr_number}`));
91 return all.flatMap((p, i) => (keys.has(`${p.nwo}#${p.pr_number}`) ? [i] : []));
92 }
93 const m = /^sample:(\d+)$/.exec(set);
94 if (!m) throw new Error(`--set is test, full or sample:N, not ${set}`);
95 const n = Number(m[1]);
96 // Stratified by change size, in the corpus's proportions, so a sample is
97 // not all small or all huge.
98 const bucket = (p) => {
99 const lines = p.lines_added + p.lines_removed;
100 return lines <= 200 ? 0 : lines <= 1000 ? 1 : 2;
101 };
102 const strata = [[], [], []];
103 all.forEach((p, i) => strata[bucket(p)].push(i));
104 const picked = [];
105 for (const stratum of strata) {
106 const share = Math.round((stratum.length / all.length) * n);
107 picked.push(...shuffle(stratum, seed).slice(0, share));
108 }
109 return picked.slice(0, n).sort((a, b) => a - b);
110}
111
112/**
113 * What one review is expected to cost. An agentic review re-reads its
114 * growing context every turn, mostly from cache: turns and context grow
115 * with the size of the change. Calibrate against real review runs (the
116 * billing ledger records each run's cost) before trusting the absolute
117 * numbers; see docs/research/reviewbench.md.
118 */
119function estimateOne(p, model) {
120 const [, output, cacheRead, cacheWrite] = PRICES[model] ?? PRICES[DEFAULT_MODEL];
121 const lines = p.lines_added + p.lines_removed;
122 const turns = Math.min(80, 12 + Math.round(Math.sqrt(lines) * 0.8) + Math.min(p.files_changed, 40) * 0.4);
123 const startContext = 18_000 + Math.min(lines, 6_000) * 12; // prompt, tools, the diff when read
124 const endContext = Math.min(startContext + turns * 2_500, 180_000); // files read along the way
125 const meanContext = (startContext + endContext) / 2;
126 // Each turn re-reads the context from cache and writes only what it added.
127 const fresh = endContext;
128 const cachedReads = Math.max(0, turns * meanContext - fresh);
129 const outTokens = turns * 450 + 2_500;
130 return (cachedReads * cacheRead + fresh * cacheWrite + outTokens * output) / 1e6;
131}
132
133/**
134 * The judge: matching per file chunk plus classifying every unmatched
135 * finding with read access to the repository (a few tool calls each).
136 */
137function estimateJudge(p, model, findingsPerPr = 6) {
138 const [input, output, cacheRead] = PRICES[model] ?? PRICES[DEFAULT_JUDGE];
139 const matchCalls = Math.max(1, Math.ceil(findingsPerPr / 3));
140 const unmatched = findingsPerPr * 0.5;
141 const tokensIn = matchCalls * 6_000 + unmatched * 5 * 12_000;
142 const tokensOut = matchCalls * 1_500 + unmatched * 5 * 600;
143 return (tokensIn * 0.5 * input + tokensIn * 0.5 * cacheRead + tokensOut * output) / 1e6;
144}
145
146function estimate(indices, model, judgeModel, rounds = 1) {
147 const all = manifest();
148 let review = 0;
149 let judge = 0;
150 for (const i of indices) {
151 review += estimateOne(all[i], model);
152 judge += estimateJudge(all[i], judgeModel);
153 }
154 return { prs: indices.length, rounds, review_usd: review * rounds, judge_usd: judge * rounds, total_usd: (review + judge) * rounds };
155}
156
157function setup() {
158 mkdirSync(CACHE, { recursive: true });
159 if (!existsSync(join(RB, ".git"))) run("git", ["clone", "https://github.com/review-bench/ReviewBench.git", RB]);
160 run("git", ["-C", RB, "fetch", "-q", "origin", RB_COMMIT]);
161 run("git", ["-C", RB, "checkout", "-q", "--detach", RB_COMMIT]);
162 run("npm", ["ci", "--no-audit", "--no-fund"], { cwd: RB, shell: process.platform === "win32" });
163}
164
165function build() {
166 run(process.execPath, [join(HERE, "prompt.mjs")]);
167 const base = JSON.parse(readFileSync(join(ROOT, "services", "runner", "base.json"), "utf8"));
168 run("docker", ["build", "--platform", "linux/amd64", "--build-arg", `BASE=${base.image}@${base.digest}`, "-t", IMAGE, HERE]);
169}
170
171function review(flags) {
172 if (!flags.yes) throw new Error("review spends model credit; run estimate first, then pass --yes");
173 const set = flags.set ?? "sample:50";
174 const seed = Number(flags.seed ?? 1);
175 const model = flags.model ?? DEFAULT_MODEL;
176 const variant = flags.variant ?? "production";
177 const budget = Number(flags["budget-usd"] ?? 5);
178 const maxTotal = Number(flags["max-total-usd"] ?? 150);
179 const indices = select(set, seed);
180 const id = flags.run ?? `${new Date().toISOString().slice(0, 10)}-${set.replace(":", "")}-${variant}-${model}`;
181 const dir = join(RUNS, id);
182 mkdirSync(dir, { recursive: true });
183 writeFileSync(join(dir, "run.json"), JSON.stringify({ id, set, seed, model, variant, budget_usd: budget, indices, reviewbench: RB_COMMIT, g1t: execFileSync("git", ["-C", ROOT, "rev-parse", "HEAD"]).toString().trim(), started_at: new Date().toISOString() }, null, 2));
184 const env = { ...process.env, RB_CONFIG_MODEL: model, RB_CONFIG_VARIANT: variant, RB_CONFIG_BUDGET_USD: String(budget), TRY_AGENT_WORK: join(dir, ".work") };
185 const pass = ["-e", "RB_CONFIG_MODEL", "-e", "RB_CONFIG_VARIANT", "-e", "RB_CONFIG_BUDGET_USD", "-e", "ANTHROPIC_API_KEY"];
186 if (process.env.ANTHROPIC_BASE_URL) pass.push("-e", "ANTHROPIC_BASE_URL");
187 let spent = 0;
188 for (const i of indices) {
189 if (spent >= maxTotal) {
190 console.error(`stopped: spent $${spent.toFixed(2)} of --max-total-usd ${maxTotal}`);
191 break;
192 }
193 spawnSync("bash", [join(RB, "scripts", "try-agent.sh"), IMAGE, "--set", "full", "--pr", String(i), ...pass], { cwd: dir, env, stdio: "inherit" });
194 spent = sidecars(dir).reduce((sum, s) => sum + (s.cost_usd ?? 0), 0);
195 }
196 console.log(`run ${id}: ${readdirSync(join(dir, "findings")).length} of ${indices.length} reviewed, $${spent.toFixed(2)} spent`);
197}
198
199/** What production would have posted beside the findings, and the cost. */
200function sidecars(dir) {
201 const out = join(dir, ".work", "out");
202 if (!existsSync(out)) return [];
203 return readdirSync(out).flatMap((key) => {
204 const file = join(out, key, "findings.g1t.json");
205 return existsSync(file) ? [{ key, ...JSON.parse(readFileSync(file, "utf8")) }] : [];
206 });
207}
208
209function judge(flags) {
210 if (!flags.yes) throw new Error("judging spends model credit; pass --yes");
211 const dir = join(RUNS, flags.run);
212 const judgeModel = flags["judge-model"] ?? DEFAULT_JUDGE;
213 run("npm", ["run", "judge", "--", "--candidate", join(dir, "findings"), "--golden", join(RB, "golden"), "--manifest", join(RB, "corpus", "manifest.json"), "--provider", "anthropic", "--model", judgeModel, "--output", join(dir, "scoring", "results.json"), "--repo-dir", join(CACHE, "judge-repos"), "--concurrency", String(flags.concurrency ?? 4)], { cwd: RB, shell: process.platform === "win32" });
214}
215
216function report(flags) {
217 const dir = join(RUNS, flags.run);
218 const meta = JSON.parse(readFileSync(join(dir, "run.json"), "utf8"));
219 const results = JSON.parse(readFileSync(join(dir, "scoring", "results.json"), "utf8"));
220 const side = sidecars(dir);
221 const cost = side.reduce((sum, s) => sum + (s.cost_usd ?? 0), 0);
222 const summary = {
223 run: meta.id,
224 date: meta.started_at.slice(0, 10),
225 g1t: meta.g1t,
226 model: meta.model,
227 variant: meta.variant,
228 set: meta.set,
229 prs: side.length,
230 findings: readdirSync(join(dir, "findings")).reduce((sum, f) => sum + JSON.parse(readFileSync(join(dir, "findings", f), "utf8")).findings.length, 0),
231 review_cost_usd: Number(cost.toFixed(2)),
232 request_changes: side.filter((s) => s.verdict === "request_changes").length,
233 dropped_outside_change: side.reduce((sum, s) => sum + (s.dropped ?? 0), 0),
234 // The judge's own aggregate block, as ReviewBench writes it.
235 metrics: results.metrics ?? results.summary ?? results,
236 };
237 if (flags.json) {
238 console.log(JSON.stringify(summary));
239 return;
240 }
241 console.log(`## ReviewBench: ${summary.run}\n`);
242 console.log(`g1t ${summary.g1t.slice(0, 8)}, ${summary.model}, prompt ${summary.variant}, ${summary.set} (${summary.prs} PRs), review cost $${summary.review_cost_usd}\n`);
243 console.log("```json\n" + JSON.stringify(summary.metrics, null, 2) + "\n```");
244}
245
246const { command, flags } = args();
247try {
248 switch (command) {
249 case "setup":
250 setup();
251 break;
252 case "build":
253 build();
254 break;
255 case "estimate": {
256 const indices = select(flags.set ?? "sample:50", Number(flags.seed ?? 1));
257 const model = flags.model ?? DEFAULT_MODEL;
258 const judgeModel = flags["judge-model"] ?? DEFAULT_JUDGE;
259 console.log(JSON.stringify({ set: flags.set ?? "sample:50", model, judge: judgeModel, ...estimate(indices, model, judgeModel, Number(flags.rounds ?? 1)) }, (k, v) => (typeof v === "number" ? Number(v.toFixed(2)) : v), 2));
260 break;
261 }
262 case "review":
263 review(flags);
264 break;
265 case "judge":
266 judge(flags);
267 break;
268 case "report":
269 report(flags);
270 break;
271 default:
272 console.error(readFileSync(fileURLToPath(import.meta.url), "utf8").split("\n").slice(1, 22).join("\n"));
273 process.exit(2);
274 }
275} catch (error) {
276 console.error(String(error.message ?? error));
277 process.exit(1);
278}