g1t/bench/reviewbench/run.mjs
Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Research: shipping CSS, and a ReviewBench harness for g1t's reviewer | 1 | #!/usr/bin/env node |
| 2 | // Runs g1t's pull request reviewer over ReviewBench and scores it with the | |
| 3 | // benchmark's own judge. docs/research/reviewbench.md explains the design. | |
| 4 | // | |
| 5 | // node bench/reviewbench/run.mjs setup clone ReviewBench (pinned) and install its judge | |
| 6 | // node bench/reviewbench/run.mjs build build the agent image from the runner base image | |
| 7 | // node bench/reviewbench/run.mjs estimate --set sample:50 | |
| 8 | // node bench/reviewbench/run.mjs review --set sample:50 [--seed 1] [--variant production] [--model claude-sonnet-5-5] --yes | |
| 9 | // node bench/reviewbench/run.mjs judge --run <id> [--judge-model claude-sonnet-5] --yes | |
| 10 | // node bench/reviewbench/run.mjs report --run <id> [--json] | |
| 11 | // | |
| 12 | // --set is test (the benchmark's 25), full (all 219) or sample:N (N of the | |
| 13 | // 219, stratified by change size, reproducible with --seed). | |
| 14 | // | |
| 15 | // Model access, from the environment, passed into each container: | |
| 16 | // ANTHROPIC_API_KEY a provider key, or | |
| 17 | // ANTHROPIC_BASE_URL + ANTHROPIC_API_KEY g1t's model proxy (MODELS_URL/anthropic and a session token) | |
| 18 | // The judge reads ANTHROPIC_API_KEY itself (through ReviewBench's pi registry). | |
| 19 | // | |
| 20 | // Nothing that spends money runs without --yes. Each review is capped by | |
| 21 | // --budget-usd (default 5), and a run stops once --max-total-usd is spent. | |
| 22 | // | |
| 23 | // Needs docker, git, jq and bash (Git Bash on Windows) and Node 20+. | |
| 24 | ||
| 25 | import { execFileSync, spawnSync } from "node:child_process"; | |
| 26 | import { existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from "node:fs"; | |
| 27 | import { dirname, join } from "node:path"; | |
| 28 | import { fileURLToPath } from "node:url"; | |
| 29 | ||
| 30 | const HERE = dirname(fileURLToPath(import.meta.url)); | |
| 31 | const ROOT = join(HERE, "..", ".."); | |
| 32 | const CACHE = join(HERE, ".cache"); | |
| 33 | const RB = join(CACHE, "ReviewBench"); | |
| 34 | const RUNS = join(CACHE, "runs"); | |
| 35 | // The ReviewBench commit these numbers are comparable against. | |
| 36 | const RB_COMMIT = "ceb0794a3768da6ef4a56e5311dfb4afd29e5dee"; | |
| 37 | const IMAGE = "g1t-reviewbench:dev"; | |
| 38 | // The model the official leaderboard is judged with. | |
| 39 | const DEFAULT_JUDGE = "claude-sonnet-5"; | |
| 40 | // services/runner/wrangler.jsonc AGENT_ROUTES.review | |
| 41 | const DEFAULT_MODEL = "claude-sonnet-5-5"; | |
| 42 | ||
| 43 | // $ per million tokens: input, output, cache read, cache write (5 minute). | |
| 44 | const PRICES = { | |
| 45 | "claude-sonnet-5-5": [2, 10, 0.2, 2.5], | |
| 46 | "claude-sonnet-5": [2, 10, 0.2, 2.5], | |
| 47 | "claude-opus-5-5": [4, 20, 0.2, 5], | |
| 48 | "claude-haiku-4-5": [1, 5, 0.1, 1.25], | |
| 49 | }; | |
| 50 | ||
| 51 | function args() { | |
| 52 | const [command, ...rest] = process.argv.slice(2); | |
| 53 | const flags = {}; | |
| 54 | for (let i = 0; i < rest.length; i++) { | |
| 55 | const name = rest[i].replace(/^--/, ""); | |
| 56 | const next = rest[i + 1]; | |
| 57 | if (next === undefined || next.startsWith("--")) flags[name] = true; | |
| 58 | else flags[name] = rest[++i]; | |
| 59 | } | |
| 60 | return { command, flags }; | |
| 61 | } | |
| 62 | ||
| 63 | const run = (cmd, argv, opts = {}) => { | |
| 64 | const result = spawnSync(cmd, argv, { stdio: "inherit", ...opts }); | |
| 65 | if (result.status !== 0) throw new Error(`${cmd} ${argv.join(" ")} exited ${result.status}`); | |
| 66 | }; | |
| 67 | ||
| 68 | function manifest() { | |
| 69 | return JSON.parse(readFileSync(join(RB, "corpus", "manifest.json"), "utf8")); | |
| 70 | } | |
| 71 | ||
| 72 | /** A seeded shuffle, so a sample is the same sample next week. */ | |
| 73 | function shuffle(items, seed) { | |
| 74 | let state = seed >>> 0 || 1; | |
| 75 | const random = () => ((state = (state * 1664525 + 1013904223) >>> 0) / 2 ** 32); | |
| 76 | const copy = [...items]; | |
| 77 | for (let i = copy.length - 1; i > 0; i--) { | |
| 78 | const j = Math.floor(random() * (i + 1)); | |
| 79 | [copy[i], copy[j]] = [copy[j], copy[i]]; | |
| 80 | } | |
| 81 | return copy; | |
| 82 | } | |
| 83 | ||
| 84 | /** Indices into the full manifest for --set. */ | |
| 85 | function select(set, seed) { | |
| 86 | const all = manifest(); | |
| 87 | if (set === "full") return all.map((_, i) => i); | |
| 88 | if (set === "test") { | |
| 89 | const test = JSON.parse(readFileSync(join(RB, "corpus", "test", "test.json"), "utf8")); | |
| 90 | const keys = new Set(test.map((p) => `${p.nwo}#${p.pr_number}`)); | |
| 91 | return all.flatMap((p, i) => (keys.has(`${p.nwo}#${p.pr_number}`) ? [i] : [])); | |
| 92 | } | |
| 93 | const m = /^sample:(\d+)$/.exec(set); | |
| 94 | if (!m) throw new Error(`--set is test, full or sample:N, not ${set}`); | |
| 95 | const n = Number(m[1]); | |
| 96 | // Stratified by change size, in the corpus's proportions, so a sample is | |
| 97 | // not all small or all huge. | |
| 98 | const bucket = (p) => { | |
| 99 | const lines = p.lines_added + p.lines_removed; | |
| 100 | return lines <= 200 ? 0 : lines <= 1000 ? 1 : 2; | |
| 101 | }; | |
| 102 | const strata = [[], [], []]; | |
| 103 | all.forEach((p, i) => strata[bucket(p)].push(i)); | |
| 104 | const picked = []; | |
| 105 | for (const stratum of strata) { | |
| 106 | const share = Math.round((stratum.length / all.length) * n); | |
| 107 | picked.push(...shuffle(stratum, seed).slice(0, share)); | |
| 108 | } | |
| 109 | return picked.slice(0, n).sort((a, b) => a - b); | |
| 110 | } | |
| 111 | ||
| 112 | /** | |
| 113 | * What one review is expected to cost. An agentic review re-reads its | |
| 114 | * growing context every turn, mostly from cache: turns and context grow | |
| 115 | * with the size of the change. Calibrate against real review runs (the | |
| 116 | * billing ledger records each run's cost) before trusting the absolute | |
| 117 | * numbers; see docs/research/reviewbench.md. | |
| 118 | */ | |
| 119 | function estimateOne(p, model) { | |
| 120 | const [, output, cacheRead, cacheWrite] = PRICES[model] ?? PRICES[DEFAULT_MODEL]; | |
| 121 | const lines = p.lines_added + p.lines_removed; | |
| 122 | const turns = Math.min(80, 12 + Math.round(Math.sqrt(lines) * 0.8) + Math.min(p.files_changed, 40) * 0.4); | |
| 123 | const startContext = 18_000 + Math.min(lines, 6_000) * 12; // prompt, tools, the diff when read | |
| 124 | const endContext = Math.min(startContext + turns * 2_500, 180_000); // files read along the way | |
| 125 | const meanContext = (startContext + endContext) / 2; | |
| 126 | // Each turn re-reads the context from cache and writes only what it added. | |
| 127 | const fresh = endContext; | |
| 128 | const cachedReads = Math.max(0, turns * meanContext - fresh); | |
| 129 | const outTokens = turns * 450 + 2_500; | |
| 130 | return (cachedReads * cacheRead + fresh * cacheWrite + outTokens * output) / 1e6; | |
| 131 | } | |
| 132 | ||
| 133 | /** | |
| 134 | * The judge: matching per file chunk plus classifying every unmatched | |
| 135 | * finding with read access to the repository (a few tool calls each). | |
| 136 | */ | |
| 137 | function estimateJudge(p, model, findingsPerPr = 6) { | |
| 138 | const [input, output, cacheRead] = PRICES[model] ?? PRICES[DEFAULT_JUDGE]; | |
| 139 | const matchCalls = Math.max(1, Math.ceil(findingsPerPr / 3)); | |
| 140 | const unmatched = findingsPerPr * 0.5; | |
| 141 | const tokensIn = matchCalls * 6_000 + unmatched * 5 * 12_000; | |
| 142 | const tokensOut = matchCalls * 1_500 + unmatched * 5 * 600; | |
| 143 | return (tokensIn * 0.5 * input + tokensIn * 0.5 * cacheRead + tokensOut * output) / 1e6; | |
| 144 | } | |
| 145 | ||
| 146 | function estimate(indices, model, judgeModel, rounds = 1) { | |
| 147 | const all = manifest(); | |
| 148 | let review = 0; | |
| 149 | let judge = 0; | |
| 150 | for (const i of indices) { | |
| 151 | review += estimateOne(all[i], model); | |
| 152 | judge += estimateJudge(all[i], judgeModel); | |
| 153 | } | |
| 154 | return { prs: indices.length, rounds, review_usd: review * rounds, judge_usd: judge * rounds, total_usd: (review + judge) * rounds }; | |
| 155 | } | |
| 156 | ||
| 157 | function setup() { | |
| 158 | mkdirSync(CACHE, { recursive: true }); | |
| 159 | if (!existsSync(join(RB, ".git"))) run("git", ["clone", "https://github.com/review-bench/ReviewBench.git", RB]); | |
| 160 | run("git", ["-C", RB, "fetch", "-q", "origin", RB_COMMIT]); | |
| 161 | run("git", ["-C", RB, "checkout", "-q", "--detach", RB_COMMIT]); | |
| 162 | run("npm", ["ci", "--no-audit", "--no-fund"], { cwd: RB, shell: process.platform === "win32" }); | |
| 163 | } | |
| 164 | ||
| 165 | function build() { | |
| 166 | run(process.execPath, [join(HERE, "prompt.mjs")]); | |
| 167 | const base = JSON.parse(readFileSync(join(ROOT, "services", "runner", "base.json"), "utf8")); | |
| 168 | run("docker", ["build", "--platform", "linux/amd64", "--build-arg", `BASE=${base.image}@${base.digest}`, "-t", IMAGE, HERE]); | |
| 169 | } | |
| 170 | ||
| 171 | function review(flags) { | |
| 172 | if (!flags.yes) throw new Error("review spends model credit; run estimate first, then pass --yes"); | |
| 173 | const set = flags.set ?? "sample:50"; | |
| 174 | const seed = Number(flags.seed ?? 1); | |
| 175 | const model = flags.model ?? DEFAULT_MODEL; | |
| 176 | const variant = flags.variant ?? "production"; | |
| 177 | const budget = Number(flags["budget-usd"] ?? 5); | |
| 178 | const maxTotal = Number(flags["max-total-usd"] ?? 150); | |
| 179 | const indices = select(set, seed); | |
| 180 | const id = flags.run ?? `${new Date().toISOString().slice(0, 10)}-${set.replace(":", "")}-${variant}-${model}`; | |
| 181 | const dir = join(RUNS, id); | |
| 182 | mkdirSync(dir, { recursive: true }); | |
| 183 | writeFileSync(join(dir, "run.json"), JSON.stringify({ id, set, seed, model, variant, budget_usd: budget, indices, reviewbench: RB_COMMIT, g1t: execFileSync("git", ["-C", ROOT, "rev-parse", "HEAD"]).toString().trim(), started_at: new Date().toISOString() }, null, 2)); | |
| 184 | const env = { ...process.env, RB_CONFIG_MODEL: model, RB_CONFIG_VARIANT: variant, RB_CONFIG_BUDGET_USD: String(budget), TRY_AGENT_WORK: join(dir, ".work") }; | |
| 185 | const pass = ["-e", "RB_CONFIG_MODEL", "-e", "RB_CONFIG_VARIANT", "-e", "RB_CONFIG_BUDGET_USD", "-e", "ANTHROPIC_API_KEY"]; | |
| 186 | if (process.env.ANTHROPIC_BASE_URL) pass.push("-e", "ANTHROPIC_BASE_URL"); | |
| 187 | let spent = 0; | |
| 188 | for (const i of indices) { | |
| 189 | if (spent >= maxTotal) { | |
| 190 | console.error(`stopped: spent $${spent.toFixed(2)} of --max-total-usd ${maxTotal}`); | |
| 191 | break; | |
| 192 | } | |
| 193 | spawnSync("bash", [join(RB, "scripts", "try-agent.sh"), IMAGE, "--set", "full", "--pr", String(i), ...pass], { cwd: dir, env, stdio: "inherit" }); | |
| 194 | spent = sidecars(dir).reduce((sum, s) => sum + (s.cost_usd ?? 0), 0); | |
| 195 | } | |
| 196 | console.log(`run ${id}: ${readdirSync(join(dir, "findings")).length} of ${indices.length} reviewed, $${spent.toFixed(2)} spent`); | |
| 197 | } | |
| 198 | ||
| 199 | /** What production would have posted beside the findings, and the cost. */ | |
| 200 | function sidecars(dir) { | |
| 201 | const out = join(dir, ".work", "out"); | |
| 202 | if (!existsSync(out)) return []; | |
| 203 | return readdirSync(out).flatMap((key) => { | |
| 204 | const file = join(out, key, "findings.g1t.json"); | |
| 205 | return existsSync(file) ? [{ key, ...JSON.parse(readFileSync(file, "utf8")) }] : []; | |
| 206 | }); | |
| 207 | } | |
| 208 | ||
| 209 | function judge(flags) { | |
| 210 | if (!flags.yes) throw new Error("judging spends model credit; pass --yes"); | |
| 211 | const dir = join(RUNS, flags.run); | |
| 212 | const judgeModel = flags["judge-model"] ?? DEFAULT_JUDGE; | |
| 213 | run("npm", ["run", "judge", "--", "--candidate", join(dir, "findings"), "--golden", join(RB, "golden"), "--manifest", join(RB, "corpus", "manifest.json"), "--provider", "anthropic", "--model", judgeModel, "--output", join(dir, "scoring", "results.json"), "--repo-dir", join(CACHE, "judge-repos"), "--concurrency", String(flags.concurrency ?? 4)], { cwd: RB, shell: process.platform === "win32" }); | |
| 214 | } | |
| 215 | ||
| 216 | function report(flags) { | |
| 217 | const dir = join(RUNS, flags.run); | |
| 218 | const meta = JSON.parse(readFileSync(join(dir, "run.json"), "utf8")); | |
| 219 | const results = JSON.parse(readFileSync(join(dir, "scoring", "results.json"), "utf8")); | |
| 220 | const side = sidecars(dir); | |
| 221 | const cost = side.reduce((sum, s) => sum + (s.cost_usd ?? 0), 0); | |
| 222 | const summary = { | |
| 223 | run: meta.id, | |
| 224 | date: meta.started_at.slice(0, 10), | |
| 225 | g1t: meta.g1t, | |
| 226 | model: meta.model, | |
| 227 | variant: meta.variant, | |
| 228 | set: meta.set, | |
| 229 | prs: side.length, | |
| 230 | findings: readdirSync(join(dir, "findings")).reduce((sum, f) => sum + JSON.parse(readFileSync(join(dir, "findings", f), "utf8")).findings.length, 0), | |
| 231 | review_cost_usd: Number(cost.toFixed(2)), | |
| 232 | request_changes: side.filter((s) => s.verdict === "request_changes").length, | |
| 233 | dropped_outside_change: side.reduce((sum, s) => sum + (s.dropped ?? 0), 0), | |
| 234 | // The judge's own aggregate block, as ReviewBench writes it. | |
| 235 | metrics: results.metrics ?? results.summary ?? results, | |
| 236 | }; | |
| 237 | if (flags.json) { | |
| 238 | console.log(JSON.stringify(summary)); | |
| 239 | return; | |
| 240 | } | |
| 241 | console.log(`## ReviewBench: ${summary.run}\n`); | |
| 242 | console.log(`g1t ${summary.g1t.slice(0, 8)}, ${summary.model}, prompt ${summary.variant}, ${summary.set} (${summary.prs} PRs), review cost $${summary.review_cost_usd}\n`); | |
| 243 | console.log("```json\n" + JSON.stringify(summary.metrics, null, 2) + "\n```"); | |
| 244 | } | |
| 245 | ||
| 246 | const { command, flags } = args(); | |
| 247 | try { | |
| 248 | switch (command) { | |
| 249 | case "setup": | |
| 250 | setup(); | |
| 251 | break; | |
| 252 | case "build": | |
| 253 | build(); | |
| 254 | break; | |
| 255 | case "estimate": { | |
| 256 | const indices = select(flags.set ?? "sample:50", Number(flags.seed ?? 1)); | |
| 257 | const model = flags.model ?? DEFAULT_MODEL; | |
| 258 | const judgeModel = flags["judge-model"] ?? DEFAULT_JUDGE; | |
| 259 | console.log(JSON.stringify({ set: flags.set ?? "sample:50", model, judge: judgeModel, ...estimate(indices, model, judgeModel, Number(flags.rounds ?? 1)) }, (k, v) => (typeof v === "number" ? Number(v.toFixed(2)) : v), 2)); | |
| 260 | break; | |
| 261 | } | |
| 262 | case "review": | |
| 263 | review(flags); | |
| 264 | break; | |
| 265 | case "judge": | |
| 266 | judge(flags); | |
| 267 | break; | |
| 268 | case "report": | |
| 269 | report(flags); | |
| 270 | break; | |
| 271 | default: | |
| 272 | console.error(readFileSync(fileURLToPath(import.meta.url), "utf8").split("\n").slice(1, 22).join("\n")); | |
| 273 | process.exit(2); | |
| 274 | } | |
| 275 | } catch (error) { | |
| 276 | console.error(String(error.message ?? error)); | |
| 277 | process.exit(1); | |
| 278 | } |