Skip to content
77 linesCodeBlameRaw
1import assert from "node:assert/strict";
2import { test } from "node:test";
3
4import { MIN_SESSIONS, type SessionOutcome, currentLevel, effortCostsOf, judge, median, wilsonLower, wording } from "./recommend.ts";
5
6/** `n` sessions at a level, `accepted` of them accepted, each costing `micros`. */
7function at(effort: SessionOutcome["effort"], n: number, accepted: number, micros: number | ((i: number) => number)): SessionOutcome[] {
8 return Array.from({ length: n }, (_, i) => ({ effort, accepted: i < accepted, charged_micros: typeof micros === "function" ? micros(i) : micros }));
9}
10
11test("a cheaper level that holds up on the agent's own work is recommended, with the saving from measured costs", () => {
12 const outcomes = [...at("high", 20, 19, 480_000), ...at("medium", 12, 12, 210_000)];
13 const judged = judge("high", outcomes, 28);
14 assert.equal(judged.kind, "recommend");
15 if (judged.kind !== "recommend") return;
16 assert.equal(judged.to, "medium");
17 assert.equal(judged.current.sessions, 20);
18 assert.equal(judged.cheaper.accepted, 12);
19 // (480,000 - 210,000) × 20 sessions / 28 days × 30 days.
20 assert.equal(judged.saving_month_micros, Math.round(270_000 * (20 / 28) * 30));
21 const words = wording(judged, "Reviewer");
22 assert.equal(words.title, "Run Reviewer at Medium effort");
23 assert.match(words.reason, /12 of its 12 sessions .* against 19 of 20 at High; a typical one cost \$0\.21 instead of \$0\.48/);
24});
25
26test("too little history says so, and recommends nothing", () => {
27 const judged = judge("high", [...at("high", 20, 20, 400_000), ...at("medium", 3, 3, 100_000)]);
28 assert.equal(judged.kind, "thin");
29 if (judged.kind !== "thin") return;
30 const words = wording(judged, "Reviewer");
31 assert.match(words.reason, new RegExp(`20 at High and 3 at Medium.*at least ${MIN_SESSIONS} at each`));
32 assert.equal(judge("high", at("high", 25, 25, 400_000)).kind, "thin", "never ran cheaper: nothing measured to compare");
33});
34
35test("a cheaper level that does worse, or saves too little, is not suggested", () => {
36 // Acceptance falls from 95% to 70%.
37 assert.equal(judge("high", [...at("high", 20, 19, 480_000), ...at("medium", 20, 14, 200_000)]).kind, "none");
38 // Same quality, but a typical session costs 95% as much.
39 assert.equal(judge("high", [...at("high", 20, 20, 400_000), ...at("medium", 20, 20, 380_000)]).kind, "none");
40});
41
42test("nothing below low, and nothing with no history", () => {
43 assert.equal(judge("low", at("low", 30, 30, 10_000)).kind, "none");
44 assert.equal(judge("high", []).kind, "none");
45});
46
47test("auto is judged at the level most of its sessions ran at", () => {
48 const outcomes = [...at("medium", 15, 15, 300_000), ...at("high", 4, 3, 600_000), ...at("low", 11, 11, 90_000)];
49 assert.equal(currentLevel("auto", outcomes), "medium");
50 const judged = judge("auto", outcomes);
51 assert.equal(judged.kind, "recommend");
52 if (judged.kind === "recommend") {
53 assert.equal(judged.from, "auto");
54 assert.equal(judged.to, "low");
55 }
56});
57
58test("the pessimistic estimate stops a lucky small sample", () => {
59 assert.ok(wilsonLower(10, 10) < 1);
60 assert.ok(wilsonLower(10, 10) > 0.8);
61 assert.equal(wilsonLower(0, 0), 0);
62});
63
64test("what each level cost is measured, and empty where it never ran", () => {
65 const costs = effortCostsOf("reviewer", "high", [...at("high", 3, 2, (i) => [100, 300, 200][i]), ...at("low", 1, 1, 40)]);
66 assert.deepEqual(
67 costs.levels.map((l) => [l.effort, l.sessions, l.typical_micros]),
68 [
69 ["low", 1, 40],
70 ["medium", 0, null],
71 ["high", 3, 200],
72 ["max", 0, null],
73 ],
74 );
75 assert.equal(costs.levels[2].accepted_share, 2 / 3);
76 assert.equal(median([1, 2, 3, 4]), 3);
77});