| 1 | import assert from "node:assert/strict"; |
| 2 | import { test } from "node:test"; |
| 3 | |
| 4 | import type { CatalogueModel } from "@g1t/contracts"; |
| 5 | |
| 6 | import { choicesFor, describeDefault, impact, parseApproval, parseDefault, parsePerMillion, perMillion, tokens } from "./models.ts"; |
| 7 | |
| 8 | function model(id: string, name: string, typical: number, extra: Partial<CatalogueModel> = {}): CatalogueModel { |
| 9 | return { |
| 10 | model: id, |
| 11 | name, |
| 12 | provider: "anthropic", |
| 13 | kind: "chat", |
| 14 | inputMicros: 1_000_000, |
| 15 | outputMicros: 5_000_000, |
| 16 | cacheReadMicros: 100_000, |
| 17 | cacheWriteMicros: 1_250_000, |
| 18 | aliases: [], |
| 19 | family: "haiku", |
| 20 | tierHint: "small", |
| 21 | contextWindow: 0, |
| 22 | maxOutput: 0, |
| 23 | capabilities: [], |
| 24 | dimensions: 0, |
| 25 | status: "available", |
| 26 | priced: true, |
| 27 | source: "staff", |
| 28 | firstSeenAt: null, |
| 29 | lastSeenAt: null, |
| 30 | missingSince: null, |
| 31 | approvedBy: null, |
| 32 | approvedAt: null, |
| 33 | note: "", |
| 34 | typicalRunMicros: typical, |
| 35 | ...extra, |
| 36 | }; |
| 37 | } |
| 38 | |
| 39 | const catalogue = [ |
| 40 | model("claude-haiku-5-5", "Claude Haiku 5.5", 76_000), |
| 41 | model("claude-sonnet-5-5", "Claude Sonnet 5.5", 1_340_000, { tierHint: "large" }), |
| 42 | model("claude-haiku-6", "Claude Haiku 6", 0, { status: "new", priced: false }), |
| 43 | model("@cf/openai/gpt-oss-120b", "gpt-oss-120b", 30_000, { provider: "workers-ai" }), |
| 44 | model("claude-haiku-4-5", "Claude Haiku 4.5", 600_000, { status: "retired" }), |
| 45 | ]; |
| 46 | |
| 47 | function form(fields: Record<string, string>): FormData { |
| 48 | const data = new FormData(); |
| 49 | for (const [name, value] of Object.entries(fields)) data.set(name, value); |
| 50 | return data; |
| 51 | } |
| 52 | |
| 53 | test("prices per million are typed in dollars to a millionth", () => { |
| 54 | assert.equal(parsePerMillion("0.125"), 125_000); |
| 55 | assert.equal(parsePerMillion("$2"), 2_000_000); |
| 56 | assert.equal(parsePerMillion("12.50"), 12_500_000); |
| 57 | assert.equal(parsePerMillion("0.000001"), 1); |
| 58 | assert.equal(parsePerMillion(""), 0); |
| 59 | assert.equal(parsePerMillion("-1"), null); |
| 60 | assert.equal(parsePerMillion("0.0000001"), null); |
| 61 | assert.equal(parsePerMillion("abc"), null); |
| 62 | assert.equal(perMillion(125_000), "0.125"); |
| 63 | assert.equal(perMillion(100_000), "0.10"); |
| 64 | assert.equal(perMillion(2_000_000), "2"); |
| 65 | }); |
| 66 | |
| 67 | test("approving a model takes its name, tier, prices and why", () => { |
| 68 | const approved = parseApproval( |
| 69 | form({ name: "Claude Haiku 6", tier_hint: "small", input: "0.08", output: "0.40", cache_read: "0.008", cache_write: "0.10", reason: "Anthropic's price page" }), |
| 70 | "chat", |
| 71 | ); |
| 72 | assert.ok(approved.ok); |
| 73 | assert.equal(approved.value.prices.inputMicros, 80_000); |
| 74 | assert.equal(approved.value.prices.outputMicros, 400_000); |
| 75 | // No hour-long price: what a five-minute write costs. |
| 76 | assert.equal(approved.value.prices.cacheWrite1hMicros, 100_000); |
| 77 | assert.equal(approved.value.prices.threshold, 0); |
| 78 | assert.equal(parseApproval(form({ name: "X", input: "1", output: "", reason: "r" }), "chat").ok, false); |
| 79 | assert.equal(parseApproval(form({ name: "X", input: "0.012", reason: "r" }), "embeddings").ok, true); |
| 80 | assert.deepEqual(parseApproval(form({ name: "X", input: "1", output: "5", reason: "" }), "chat"), { ok: false, error: "Say why, for whoever looks next." }); |
| 81 | assert.equal(parseApproval(form({ name: "X", input: "1", output: "5", over_input: "2", reason: "r" }), "chat").ok, false); |
| 82 | const long = parseApproval(form({ name: "X", input: "1", output: "5", threshold: "200,000", over_input: "2", over_output: "10", reason: "r" }), "chat"); |
| 83 | assert.ok(long.ok); |
| 84 | assert.equal(long.value.prices.threshold, 200_000); |
| 85 | assert.equal(parseApproval(form({ name: "X", tier_hint: "huge", input: "1", output: "5", reason: "r" }), "chat").ok, false); |
| 86 | assert.equal(parseApproval(form({ name: "", input: "1", output: "5", reason: "r" }), "chat").ok, false); |
| 87 | }); |
| 88 | |
| 89 | test("a model purpose takes only an available, priced Claude", () => { |
| 90 | const choices = choicesFor(catalogue); |
| 91 | assert.deepEqual(choices.map((m) => m.model), ["claude-haiku-5-5", "claude-sonnet-5-5"]); |
| 92 | const set = parseDefault(form({ purpose: "tier_small", model: "claude-sonnet-5-5", reason: "testing" }), choices); |
| 93 | assert.deepEqual(set, { ok: true, value: { purpose: "tier_small", model: "claude-sonnet-5-5", tier: null, effort: null, reason: "testing" } }); |
| 94 | assert.equal(parseDefault(form({ purpose: "tier_small", model: "claude-haiku-6", reason: "x" }), choices).ok, false); |
| 95 | assert.equal(parseDefault(form({ purpose: "background", model: "@cf/openai/gpt-oss-120b", reason: "x" }), choices).ok, false); |
| 96 | assert.equal(parseDefault(form({ purpose: "tier_small", model: "claude-haiku-5-5", reason: "" }), choices).ok, false); |
| 97 | assert.equal(parseDefault(form({ purpose: "nonsense", model: "claude-haiku-5-5", reason: "x" }), choices).ok, false); |
| 98 | }); |
| 99 | |
| 100 | test("a job takes a tier and an effort, or the harness's own", () => { |
| 101 | const choices = choicesFor(catalogue); |
| 102 | assert.deepEqual(parseDefault(form({ purpose: "job_plan", tier: "small", effort: "xhigh", reason: "x" }), choices), { |
| 103 | ok: true, |
| 104 | value: { purpose: "job_plan", model: null, tier: "small", effort: "xhigh", reason: "x" }, |
| 105 | }); |
| 106 | const own = parseDefault(form({ purpose: "job_update", tier: "small", effort: "", reason: "x" }), choices); |
| 107 | assert.ok(own.ok); |
| 108 | assert.equal(own.value.effort, null); |
| 109 | // Only a review is sized by the change. |
| 110 | assert.equal(parseDefault(form({ purpose: "job_review", tier: "change", reason: "x" }), choices).ok, true); |
| 111 | assert.equal(parseDefault(form({ purpose: "job_plan", tier: "change", reason: "x" }), choices).ok, false); |
| 112 | assert.equal(parseDefault(form({ purpose: "job_plan", tier: "small", effort: "huge", reason: "x" }), choices).ok, false); |
| 113 | }); |
| 114 | |
| 115 | test("a change shows what a typical run would cost before it is saved", () => { |
| 116 | const before = { purpose: "tier_small", chosen: "claude-haiku-5-5", model: catalogue[0], capabilities: [], note: null }; |
| 117 | assert.equal( |
| 118 | impact(before, catalogue[1], catalogue), |
| 119 | "A typical run: about $1.34 on Claude Sonnet 5.5, 17.6 times $0.076 on Claude Haiku 5.5.", |
| 120 | ); |
| 121 | assert.equal(impact({ ...before, model: catalogue[1] }, catalogue[0], catalogue), "A typical run: about $0.076 on Claude Haiku 5.5, 94% less than $1.34 on Claude Sonnet 5.5."); |
| 122 | assert.equal(impact(before, catalogue[0], catalogue), "No change: Claude Haiku 5.5 already runs it, at about $0.076 a typical run."); |
| 123 | assert.equal(impact(undefined, catalogue[1], catalogue), "A typical run would cost about $1.34 on Claude Sonnet 5.5."); |
| 124 | assert.equal(impact(before, undefined, catalogue), null); |
| 125 | }); |
| 126 | |
| 127 | test("defaults and sizes read plainly", () => { |
| 128 | assert.equal(describeDefault({ model: "claude-haiku-5-5", tier: null, effort: null }, catalogue), "Claude Haiku 5.5"); |
| 129 | assert.equal(describeDefault({ model: null, tier: "small", effort: "high" }, catalogue), "Fast, high effort"); |
| 130 | assert.equal(describeDefault({ model: null, tier: "change", effort: null }, catalogue), "By the change's size, the harness's own effort"); |
| 131 | assert.equal(tokens(1_000_000), "1M"); |
| 132 | assert.equal(tokens(200_000), "200k"); |
| 133 | assert.equal(tokens(0), "—"); |
| 134 | }); |