g1t/bench/reviewbench/run.mjs
| 1 | #!/usr/bin/env node |
| 2 | // Runs g1t's pull request reviewer over ReviewBench and scores it with the |
| 3 | // benchmark's own judge. docs/research/reviewbench.md explains the design. |
| 4 | // |
| 5 | // node bench/reviewbench/run.mjs setup clone ReviewBench (pinned) and install its judge |
| 6 | // node bench/reviewbench/run.mjs build build the agent image from the runner base image |
| 7 | // node bench/reviewbench/run.mjs estimate --set sample:50 |
| 8 | // node bench/reviewbench/run.mjs review --set sample:50 [--seed 1] [--variant production] [--model claude-sonnet-5-5] --yes |
| 9 | // node bench/reviewbench/run.mjs judge --run <id> [--judge-model claude-sonnet-5] --yes |
| 10 | // node bench/reviewbench/run.mjs report --run <id> [--json] |
| 11 | // |
| 12 | // --set is test (the benchmark's 25), full (all 219) or sample:N (N of the |
| 13 | // 219, stratified by change size, reproducible with --seed). |
| 14 | // |
| 15 | // Model access, from the environment, passed into each container: |
| 16 | // ANTHROPIC_API_KEY a provider key, or |
| 17 | // ANTHROPIC_BASE_URL + ANTHROPIC_API_KEY g1t's model proxy (MODELS_URL/anthropic and a session token) |
| 18 | // The judge reads ANTHROPIC_API_KEY itself (through ReviewBench's pi registry). |
| 19 | // |
| 20 | // Nothing that spends money runs without --yes. Each review is capped by |
| 21 | // --budget-usd (default 5), and a run stops once --max-total-usd is spent. |
| 22 | // |
| 23 | // Needs docker, git, jq and bash (Git Bash on Windows) and Node 20+. |
| 24 | |
| 25 | import { execFileSync, spawnSync } from "node:child_process"; |
| 26 | import { existsSync, mkdirSync, readFileSync, readdirSync, writeFileSync } from "node:fs"; |
| 27 | import { dirname, join } from "node:path"; |
| 28 | import { fileURLToPath } from "node:url"; |
| 29 | |
| 30 | const HERE = dirname(fileURLToPath(import.meta.url)); |
| 31 | const ROOT = join(HERE, "..", ".."); |
| 32 | const CACHE = join(HERE, ".cache"); |
| 33 | const RB = join(CACHE, "ReviewBench"); |
| 34 | const RUNS = join(CACHE, "runs"); |
| 35 | // The ReviewBench commit these numbers are comparable against. |
| 36 | const RB_COMMIT = "ceb0794a3768da6ef4a56e5311dfb4afd29e5dee"; |
| 37 | const IMAGE = "g1t-reviewbench:dev"; |
| 38 | // The model the official leaderboard is judged with. |
| 39 | const DEFAULT_JUDGE = "claude-sonnet-5"; |
| 40 | // services/runner/wrangler.jsonc AGENT_ROUTES.review |
| 41 | const DEFAULT_MODEL = "claude-sonnet-5-5"; |
| 42 | |
| 43 | // $ per million tokens: input, output, cache read, cache write (5 minute). |
| 44 | const PRICES = { |
| 45 | "claude-sonnet-5-5": [2, 10, 0.2, 2.5], |
| 46 | "claude-sonnet-5": [2, 10, 0.2, 2.5], |
| 47 | "claude-opus-5-5": [4, 20, 0.2, 5], |
| 48 | "claude-haiku-4-5": [1, 5, 0.1, 1.25], |
| 49 | }; |
| 50 | |
| 51 | function args() { |
| 52 | const [command, ...rest] = process.argv.slice(2); |
| 53 | const flags = {}; |
| 54 | for (let i = 0; i < rest.length; i++) { |
| 55 | const name = rest[i].replace(/^--/, ""); |
| 56 | const next = rest[i + 1]; |
| 57 | if (next === undefined || next.startsWith("--")) flags[name] = true; |
| 58 | else flags[name] = rest[++i]; |
| 59 | } |
| 60 | return { command, flags }; |
| 61 | } |
| 62 | |
| 63 | const run = (cmd, argv, opts = {}) => { |
| 64 | const result = spawnSync(cmd, argv, { stdio: "inherit", ...opts }); |
| 65 | if (result.status !== 0) throw new Error(`${cmd} ${argv.join(" ")} exited ${result.status}`); |
| 66 | }; |
| 67 | |
| 68 | function manifest() { |
| 69 | return JSON.parse(readFileSync(join(RB, "corpus", "manifest.json"), "utf8")); |
| 70 | } |
| 71 | |
| 72 | /** A seeded shuffle, so a sample is the same sample next week. */ |
| 73 | function shuffle(items, seed) { |
| 74 | let state = seed >>> 0 || 1; |
| 75 | const random = () => ((state = (state * 1664525 + 1013904223) >>> 0) / 2 ** 32); |
| 76 | const copy = [...items]; |
| 77 | for (let i = copy.length - 1; i > 0; i--) { |
| 78 | const j = Math.floor(random() * (i + 1)); |
| 79 | [copy[i], copy[j]] = [copy[j], copy[i]]; |
| 80 | } |
| 81 | return copy; |
| 82 | } |
| 83 | |
| 84 | /** Indices into the full manifest for --set. */ |
| 85 | function select(set, seed) { |
| 86 | const all = manifest(); |
| 87 | if (set === "full") return all.map((_, i) => i); |
| 88 | if (set === "test") { |
| 89 | const test = JSON.parse(readFileSync(join(RB, "corpus", "test", "test.json"), "utf8")); |
| 90 | const keys = new Set(test.map((p) => `${p.nwo}#${p.pr_number}`)); |
| 91 | return all.flatMap((p, i) => (keys.has(`${p.nwo}#${p.pr_number}`) ? [i] : [])); |
| 92 | } |
| 93 | const m = /^sample:(\d+)$/.exec(set); |
| 94 | if (!m) throw new Error(`--set is test, full or sample:N, not ${set}`); |
| 95 | const n = Number(m[1]); |
| 96 | // Stratified by change size, in the corpus's proportions, so a sample is |
| 97 | // not all small or all huge. |
| 98 | const bucket = (p) => { |
| 99 | const lines = p.lines_added + p.lines_removed; |
| 100 | return lines <= 200 ? 0 : lines <= 1000 ? 1 : 2; |
| 101 | }; |
| 102 | const strata = [[], [], []]; |
| 103 | all.forEach((p, i) => strata[bucket(p)].push(i)); |
| 104 | const picked = []; |
| 105 | for (const stratum of strata) { |
| 106 | const share = Math.round((stratum.length / all.length) * n); |
| 107 | picked.push(...shuffle(stratum, seed).slice(0, share)); |
| 108 | } |
| 109 | return picked.slice(0, n).sort((a, b) => a - b); |
| 110 | } |
| 111 | |
| 112 | /** |
| 113 | * What one review is expected to cost. An agentic review re-reads its |
| 114 | * growing context every turn, mostly from cache: turns and context grow |
| 115 | * with the size of the change. Calibrate against real review runs (the |
| 116 | * billing ledger records each run's cost) before trusting the absolute |
| 117 | * numbers; see docs/research/reviewbench.md. |
| 118 | */ |
| 119 | function estimateOne(p, model) { |
| 120 | const [, output, cacheRead, cacheWrite] = PRICES[model] ?? PRICES[DEFAULT_MODEL]; |
| 121 | const lines = p.lines_added + p.lines_removed; |
| 122 | const turns = Math.min(80, 12 + Math.round(Math.sqrt(lines) * 0.8) + Math.min(p.files_changed, 40) * 0.4); |
| 123 | const startContext = 18_000 + Math.min(lines, 6_000) * 12; // prompt, tools, the diff when read |
| 124 | const endContext = Math.min(startContext + turns * 2_500, 180_000); // files read along the way |
| 125 | const meanContext = (startContext + endContext) / 2; |
| 126 | // Each turn re-reads the context from cache and writes only what it added. |
| 127 | const fresh = endContext; |
| 128 | const cachedReads = Math.max(0, turns * meanContext - fresh); |
| 129 | const outTokens = turns * 450 + 2_500; |
| 130 | return (cachedReads * cacheRead + fresh * cacheWrite + outTokens * output) / 1e6; |
| 131 | } |
| 132 | |
| 133 | /** |
| 134 | * The judge: matching per file chunk plus classifying every unmatched |
| 135 | * finding with read access to the repository (a few tool calls each). |
| 136 | */ |
| 137 | function estimateJudge(p, model, findingsPerPr = 6) { |
| 138 | const [input, output, cacheRead] = PRICES[model] ?? PRICES[DEFAULT_JUDGE]; |
| 139 | const matchCalls = Math.max(1, Math.ceil(findingsPerPr / 3)); |
| 140 | const unmatched = findingsPerPr * 0.5; |
| 141 | const tokensIn = matchCalls * 6_000 + unmatched * 5 * 12_000; |
| 142 | const tokensOut = matchCalls * 1_500 + unmatched * 5 * 600; |
| 143 | return (tokensIn * 0.5 * input + tokensIn * 0.5 * cacheRead + tokensOut * output) / 1e6; |
| 144 | } |
| 145 | |
| 146 | function estimate(indices, model, judgeModel, rounds = 1) { |
| 147 | const all = manifest(); |
| 148 | let review = 0; |
| 149 | let judge = 0; |
| 150 | for (const i of indices) { |
| 151 | review += estimateOne(all[i], model); |
| 152 | judge += estimateJudge(all[i], judgeModel); |
| 153 | } |
| 154 | return { prs: indices.length, rounds, review_usd: review * rounds, judge_usd: judge * rounds, total_usd: (review + judge) * rounds }; |
| 155 | } |
| 156 | |
| 157 | function setup() { |
| 158 | mkdirSync(CACHE, { recursive: true }); |
| 159 | if (!existsSync(join(RB, ".git"))) run("git", ["clone", "https://github.com/review-bench/ReviewBench.git", RB]); |
| 160 | run("git", ["-C", RB, "fetch", "-q", "origin", RB_COMMIT]); |
| 161 | run("git", ["-C", RB, "checkout", "-q", "--detach", RB_COMMIT]); |
| 162 | run("npm", ["ci", "--no-audit", "--no-fund"], { cwd: RB, shell: process.platform === "win32" }); |
| 163 | } |
| 164 | |
| 165 | function build() { |
| 166 | run(process.execPath, [join(HERE, "prompt.mjs")]); |
| 167 | const base = JSON.parse(readFileSync(join(ROOT, "services", "runner", "base.json"), "utf8")); |
| 168 | run("docker", ["build", "--platform", "linux/amd64", "--build-arg", `BASE=${base.image}@${base.digest}`, "-t", IMAGE, HERE]); |
| 169 | } |
| 170 | |
| 171 | function review(flags) { |
| 172 | if (!flags.yes) throw new Error("review spends model credit; run estimate first, then pass --yes"); |
| 173 | const set = flags.set ?? "sample:50"; |
| 174 | const seed = Number(flags.seed ?? 1); |
| 175 | const model = flags.model ?? DEFAULT_MODEL; |
| 176 | const variant = flags.variant ?? "production"; |
| 177 | const budget = Number(flags["budget-usd"] ?? 5); |
| 178 | const maxTotal = Number(flags["max-total-usd"] ?? 150); |
| 179 | const indices = select(set, seed); |
| 180 | const id = flags.run ?? `${new Date().toISOString().slice(0, 10)}-${set.replace(":", "")}-${variant}-${model}`; |
| 181 | const dir = join(RUNS, id); |
| 182 | mkdirSync(dir, { recursive: true }); |
| 183 | writeFileSync(join(dir, "run.json"), JSON.stringify({ id, set, seed, model, variant, budget_usd: budget, indices, reviewbench: RB_COMMIT, g1t: execFileSync("git", ["-C", ROOT, "rev-parse", "HEAD"]).toString().trim(), started_at: new Date().toISOString() }, null, 2)); |
| 184 | const env = { ...process.env, RB_CONFIG_MODEL: model, RB_CONFIG_VARIANT: variant, RB_CONFIG_BUDGET_USD: String(budget), TRY_AGENT_WORK: join(dir, ".work") }; |
| 185 | const pass = ["-e", "RB_CONFIG_MODEL", "-e", "RB_CONFIG_VARIANT", "-e", "RB_CONFIG_BUDGET_USD", "-e", "ANTHROPIC_API_KEY"]; |
| 186 | if (process.env.ANTHROPIC_BASE_URL) pass.push("-e", "ANTHROPIC_BASE_URL"); |
| 187 | let spent = 0; |
| 188 | for (const i of indices) { |
| 189 | if (spent >= maxTotal) { |
| 190 | console.error(`stopped: spent $${spent.toFixed(2)} of --max-total-usd ${maxTotal}`); |
| 191 | break; |
| 192 | } |
| 193 | spawnSync("bash", [join(RB, "scripts", "try-agent.sh"), IMAGE, "--set", "full", "--pr", String(i), ...pass], { cwd: dir, env, stdio: "inherit" }); |
| 194 | spent = sidecars(dir).reduce((sum, s) => sum + (s.cost_usd ?? 0), 0); |
| 195 | } |
| 196 | console.log(`run ${id}: ${readdirSync(join(dir, "findings")).length} of ${indices.length} reviewed, $${spent.toFixed(2)} spent`); |
| 197 | } |
| 198 | |
| 199 | /** What production would have posted beside the findings, and the cost. */ |
| 200 | function sidecars(dir) { |
| 201 | const out = join(dir, ".work", "out"); |
| 202 | if (!existsSync(out)) return []; |
| 203 | return readdirSync(out).flatMap((key) => { |
| 204 | const file = join(out, key, "findings.g1t.json"); |
| 205 | return existsSync(file) ? [{ key, ...JSON.parse(readFileSync(file, "utf8")) }] : []; |
| 206 | }); |
| 207 | } |
| 208 | |
| 209 | function judge(flags) { |
| 210 | if (!flags.yes) throw new Error("judging spends model credit; pass --yes"); |
| 211 | const dir = join(RUNS, flags.run); |
| 212 | const judgeModel = flags["judge-model"] ?? DEFAULT_JUDGE; |
| 213 | run("npm", ["run", "judge", "--", "--candidate", join(dir, "findings"), "--golden", join(RB, "golden"), "--manifest", join(RB, "corpus", "manifest.json"), "--provider", "anthropic", "--model", judgeModel, "--output", join(dir, "scoring", "results.json"), "--repo-dir", join(CACHE, "judge-repos"), "--concurrency", String(flags.concurrency ?? 4)], { cwd: RB, shell: process.platform === "win32" }); |
| 214 | } |
| 215 | |
| 216 | function report(flags) { |
| 217 | const dir = join(RUNS, flags.run); |
| 218 | const meta = JSON.parse(readFileSync(join(dir, "run.json"), "utf8")); |
| 219 | const results = JSON.parse(readFileSync(join(dir, "scoring", "results.json"), "utf8")); |
| 220 | const side = sidecars(dir); |
| 221 | const cost = side.reduce((sum, s) => sum + (s.cost_usd ?? 0), 0); |
| 222 | const summary = { |
| 223 | run: meta.id, |
| 224 | date: meta.started_at.slice(0, 10), |
| 225 | g1t: meta.g1t, |
| 226 | model: meta.model, |
| 227 | variant: meta.variant, |
| 228 | set: meta.set, |
| 229 | prs: side.length, |
| 230 | findings: readdirSync(join(dir, "findings")).reduce((sum, f) => sum + JSON.parse(readFileSync(join(dir, "findings", f), "utf8")).findings.length, 0), |
| 231 | review_cost_usd: Number(cost.toFixed(2)), |
| 232 | request_changes: side.filter((s) => s.verdict === "request_changes").length, |
| 233 | dropped_outside_change: side.reduce((sum, s) => sum + (s.dropped ?? 0), 0), |
| 234 | // The judge's own aggregate block, as ReviewBench writes it. |
| 235 | metrics: results.metrics ?? results.summary ?? results, |
| 236 | }; |
| 237 | if (flags.json) { |
| 238 | console.log(JSON.stringify(summary)); |
| 239 | return; |
| 240 | } |
| 241 | console.log(`## ReviewBench: ${summary.run}\n`); |
| 242 | console.log(`g1t ${summary.g1t.slice(0, 8)}, ${summary.model}, prompt ${summary.variant}, ${summary.set} (${summary.prs} PRs), review cost $${summary.review_cost_usd}\n`); |
| 243 | console.log("```json\n" + JSON.stringify(summary.metrics, null, 2) + "\n```"); |
| 244 | } |
| 245 | |
| 246 | const { command, flags } = args(); |
| 247 | try { |
| 248 | switch (command) { |
| 249 | case "setup": |
| 250 | setup(); |
| 251 | break; |
| 252 | case "build": |
| 253 | build(); |
| 254 | break; |
| 255 | case "estimate": { |
| 256 | const indices = select(flags.set ?? "sample:50", Number(flags.seed ?? 1)); |
| 257 | const model = flags.model ?? DEFAULT_MODEL; |
| 258 | const judgeModel = flags["judge-model"] ?? DEFAULT_JUDGE; |
| 259 | console.log(JSON.stringify({ set: flags.set ?? "sample:50", model, judge: judgeModel, ...estimate(indices, model, judgeModel, Number(flags.rounds ?? 1)) }, (k, v) => (typeof v === "number" ? Number(v.toFixed(2)) : v), 2)); |
| 260 | break; |
| 261 | } |
| 262 | case "review": |
| 263 | review(flags); |
| 264 | break; |
| 265 | case "judge": |
| 266 | judge(flags); |
| 267 | break; |
| 268 | case "report": |
| 269 | report(flags); |
| 270 | break; |
| 271 | default: |
| 272 | console.error(readFileSync(fileURLToPath(import.meta.url), "utf8").split("\n").slice(1, 22).join("\n")); |
| 273 | process.exit(2); |
| 274 | } |
| 275 | } catch (error) { |
| 276 | console.error(String(error.message ?? error)); |
| 277 | process.exit(1); |
| 278 | } |