Merge platform pause and the hourly usage watcher: staff can pause compute, schedules, indexing or renders for everyone, the watcher emails on a breach and is never blind quietly, and the models proxy holds each run to its cap (billing 0051, integrations 0006)
The platform watcher is never blind quietly: each run records the queries it could not read, sudo names them, three failed runs in a row (or every dataset empty) email staff, and platform-usage.mjs exits 1 on any (billing 0050 renamed 0051).
| 275 | 275 | spend the same way it reports it for billing and stops the agent once | |
| 276 | 276 | the spend reaches the cap. The step in flight when it does can take the | |
| 277 | 277 | run a little past it. | |
| 278 | + | - g1t's model proxy holds the run to the same cap, whatever happens in the | |
| 279 | + | sandbox. It adds up what each of the run's model answers cost, and once | |
| 280 | + | the run has spent its cap it refuses the run's model requests with | |
| 281 | + | `402` and the error code `run_cap_reached`, which shows in the run's log. | |
| 282 | + | The proxy also takes at most 16 of a run's model requests at a time, | |
| 283 | + | and a run's model token reaches only `/v1/messages` (with | |
| 284 | + | `/v1/messages/count_tokens`) and `/v1/models`. The token stops working | |
| 285 | + | when the run ends. | |
| 278 | 286 | - The time cap is enforced twice: the harness stops the agent when it | |
| 279 | 287 | passes, and the sandbox itself is stopped three minutes after, whatever | |
| 280 | 288 | is running in it. |
| 852 | 852 | workflow job is recorded as failed with "Not started:" and the reason. | |
| 853 | 853 | Runs already under way finish, so usage can go slightly past a limit. | |
| 854 | 854 | ||
| 855 | + | ### When g1t pauses work for everyone | |
| 856 | + | ||
| 857 | + | If usage across all of g1t climbs far past normal, g1t can pause some | |
| 858 | + | kinds of work for every workspace while it looks into it. Each pause is | |
| 859 | + | separate, and runs already under way finish: | |
| 860 | + | ||
| 861 | + | | Paused | What you see | | |
| 862 | + | | --- | --- | | |
| 863 | + | | Compute | New agent runs, checks, workflow jobs and deploy builds are refused with `paused` and a message that g1t has paused them across the platform. Try again later. | | |
| 864 | + | | Schedules | Workflows on `schedule:` skip the minutes while it lasts; they are not run late. Issues waiting for an agent stay in the queue. | | |
| 865 | + | | Indexing | **Rebuild** in the [context hub](/guides/context-hub/) is refused, and new semantic search embeddings wait. Text search still answers, and search keeps up with new pushes. | | |
| 866 | + | | Renders | A link to g1t shows g1t's logo instead of the page's own card. | | |
| 867 | + | ||
| 868 | + | Nothing in your workspace changes, and nothing is charged while work | |
| 869 | + | waits. | |
| 870 | + | ||
| 855 | 871 | ## Security on every plan | |
| 856 | 872 | ||
| 857 | 873 | Every workspace, free or on the plan, has the [audit log](/guides/audit-log/), |
| 4 | 4 | */ | |
| 5 | 5 | import { data, redirect } from "react-router"; | |
| 6 | 6 | ||
| 7 | − | import { parseCostSettings, parseMapping, parseRange } from "./costs"; | |
| 7 | + | import { parseCostSettings, parseMapping, parsePauseLevel, parseRange } from "./costs"; | |
| 8 | 8 | import { admin } from "./services.server"; | |
| 9 | 9 | import { settle } from "./settle"; | |
| 10 | 10 | import { requireStaff } from "./staff"; | |
| ⋯ | |||
| 17 | 17 | mapping: "Mapping saved. It applies from the next run; read the bill now to see it.", | |
| 18 | 18 | removed: "Mapping removed.", | |
| 19 | 19 | lifted: "Breaker lifted for the rest of today (UTC). Hosted-model runs start again; it is recorded in the audit log.", | |
| 20 | + | paused: "Paused across g1t. Every service sees it within 30 seconds; it is recorded in the audit log.", | |
| 21 | + | resumed: "Resumed across g1t. Every service sees it within 30 seconds; it is recorded in the audit log.", | |
| 20 | 22 | }; | |
| 21 | 23 | ||
| 22 | 24 | export type CostsActionResult = { error: string; section: string; values?: Record<string, string> }; | |
| ⋯ | |||
| 25 | 27 | requireStaff(context as Parameters<typeof requireStaff>[0]); | |
| 26 | 28 | const url = new URL(request.url); | |
| 27 | 29 | const range = parseRange(url.searchParams.get("days")); | |
| 28 | − | const report = await settle(admin.costs(range)); | |
| 30 | + | const [report, guard] = await Promise.all([settle(admin.costs(range)), settle(admin.platformGuard())]); | |
| 29 | 31 | const done = url.searchParams.get("done"); | |
| 30 | 32 | return { | |
| 31 | 33 | range, | |
| 32 | 34 | bucket: url.searchParams.get("product"), | |
| 33 | 35 | report: report.ok ? report.value : null, | |
| 34 | 36 | error: report.ok ? null : report.error, | |
| 37 | + | // The platform pause and usage watch (billing's platform.rs). | |
| 38 | + | guard: guard.ok ? guard.value : null, | |
| 39 | + | guardError: guard.ok ? null : guard.error, | |
| 35 | 40 | done: done ? (DONE[done] ?? null) : null, | |
| 36 | 41 | }; | |
| 37 | 42 | } | |
| ⋯ | |||
| 62 | 67 | if (!result.value.ok) return fail("lift", result.value.error.message); | |
| 63 | 68 | throw back("lifted", "#spend"); | |
| 64 | 69 | } | |
| 70 | + | if (intent === "pause" || intent === "resume") { | |
| 71 | + | const level = parsePauseLevel(form.get("level")); | |
| 72 | + | if (!level) return fail("platform", "Pick compute, schedules, indexing or renders."); | |
| 73 | + | const note = String(form.get("note") ?? "").trim().slice(0, 500); | |
| 74 | + | if (note.length < 5) return fail(`pause-${level}`, "Say why, for whoever looks next."); | |
| 75 | + | const result = await settle(admin.setPause(level, intent === "pause", note, staff.email)); | |
| 76 | + | if (!result.ok) return fail(`pause-${level}`, `Billing did not answer: ${result.error}`); | |
| 77 | + | if (!result.value.ok) return fail(`pause-${level}`, result.value.error.message); | |
| 78 | + | throw back(intent === "pause" ? "paused" : "resumed", "#platform"); | |
| 79 | + | } | |
| 65 | 80 | if (intent === "decide") { | |
| 66 | 81 | const id = String(form.get("id") ?? ""); | |
| 67 | 82 | const decision = form.get("decision") === "approve" ? "approve" : "reject"; | |
| 1 | 1 | import assert from "node:assert/strict"; | |
| 2 | 2 | import { test } from "node:test"; | |
| 3 | 3 | ||
| 4 | − | import type { CostDay, SpendCaps } from "@g1t/contracts"; | |
| 4 | + | import type { CostDay, PlatformGuard, SpendCaps } from "@g1t/contracts"; | |
| 5 | 5 | ||
| 6 | 6 | import { | |
| 7 | + | count, | |
| 7 | 8 | daySeries, | |
| 8 | 9 | daysBetween, | |
| 9 | 10 | marginOnPrice, | |
| ⋯ | |||
| 12 | 13 | parseBucket, | |
| 13 | 14 | parseCostSettings, | |
| 14 | 15 | parseMapping, | |
| 16 | + | parsePauseLevel, | |
| 17 | + | pauseBanner, | |
| 15 | 18 | parseRange, | |
| 16 | 19 | percentLabel, | |
| 17 | 20 | proposalOutcome, | |
| 18 | 21 | spendBanner, | |
| 19 | 22 | spendRows, | |
| 20 | 23 | subscriptionsOver, | |
| 24 | + | thresholdShare, | |
| 21 | 25 | unitDollars, | |
| 22 | 26 | versionCells, | |
| 23 | 27 | whoPaid, | |
| ⋯ | |||
| 200 | 204 | assert.equal(proposalOutcome({ status: "superseded", decidedBy: null, effectiveAt: null }), "replaced by a later measurement"); | |
| 201 | 205 | assert.equal(proposalOutcome({ status: "open", decidedBy: null, effectiveAt: null }), null); | |
| 202 | 206 | }); | |
| 207 | + | ||
| 208 | + | test("the pause banner names every paused level, and who paused it when it was not a person", () => { | |
| 209 | + | const level = (name: "compute" | "schedules" | "indexing" | "renders", paused: boolean, auto = false) => ({ | |
| 210 | + | level: name, | |
| 211 | + | paused, | |
| 212 | + | note: null, | |
| 213 | + | set_by: null, | |
| 214 | + | set_at: null, | |
| 215 | + | auto, | |
| 216 | + | }); | |
| 217 | + | const guard = (levels: ReturnType<typeof level>[]): PlatformGuard => ({ | |
| 218 | + | levels, | |
| 219 | + | hour: null, | |
| 220 | + | last_hour: [], | |
| 221 | + | month: "2026-10", | |
| 222 | + | month_to_date: [], | |
| 223 | + | breaches: [], | |
| 224 | + | can_read: true, | |
| 225 | + | auto_pause: ["schedules", "indexing"], | |
| 226 | + | }); | |
| 227 | + | assert.equal(pauseBanner(guard([level("compute", false), level("renders", false)])), null); | |
| 228 | + | assert.equal(pauseBanner(guard([level("schedules", true, true)])), "Paused across g1t: schedules (by the usage watcher)."); | |
| 229 | + | assert.equal( | |
| 230 | + | pauseBanner(guard([level("compute", true), level("schedules", true), level("indexing", true, true)])), | |
| 231 | + | "Paused across g1t: compute, schedules and indexing (by the usage watcher).", | |
| 232 | + | ); | |
| 233 | + | assert.equal(parsePauseLevel("renders"), "renders"); | |
| 234 | + | assert.equal(parsePauseLevel("everything"), null); | |
| 235 | + | assert.equal(count(240_000), "240k"); | |
| 236 | + | assert.equal(count(2_000_000_000), "2.0B"); | |
| 237 | + | assert.equal(thresholdShare(240_000, 200_000), 120); | |
| 238 | + | assert.equal(thresholdShare(5, 0), null); | |
| 239 | + | }); | |
| 2 | 2 | * Costs & margin: the arithmetic behind the page, apart from the SVG and | |
| 3 | 3 | * the Workers runtime so it can be tested under Node. Money is in micros. | |
| 4 | 4 | */ | |
| 5 | − | import type { CostDay, CostMappingInput, CostSettings, SpendCaps } from "@g1t/contracts"; | |
| 5 | + | import type { CostDay, CostMappingInput, CostSettings, PauseLevel, PlatformGuard, SpendCaps } from "@g1t/contracts"; | |
| 6 | 6 | ||
| 7 | 7 | import { parseDollars, usd } from "./money.ts"; | |
| 8 | 8 | ||
| ⋯ | |||
| 266 | 266 | if (capMicros <= 0) return 0; | |
| 267 | 267 | return Math.max(0, Math.min(100, (usedMicros / capMicros) * 100)); | |
| 268 | 268 | } | |
| 269 | + | ||
| 270 | + | // --- Platform pause and usage watch (billing's platform.rs) ----------------- | |
| 271 | + | ||
| 272 | + | /** What each level of the platform pause stops, for the page and the banner. */ | |
| 273 | + | export const PAUSE_LEVELS: { level: PauseLevel; title: string; stops: string }[] = [ | |
| 274 | + | { level: "compute", title: "Compute", stops: "New agent runs, checks, workflow jobs and deploy builds, for every workspace. Runs already going finish." }, | |
| 275 | + | { level: "schedules", title: "Schedules", stops: "Actions' cron-triggered runs, and the runner's sweep that starts queued agents." }, | |
| 276 | + | { level: "indexing", title: "Indexing", stops: "Context embeddings and backfills, and search's backfills (they go on from where they were when resumed)." }, | |
| 277 | + | { level: "renders", title: "Renders", stops: "Social card images: a cache miss gets the brand card or the static logo." }, | |
| 278 | + | ]; | |
| 279 | + | ||
| 280 | + | /** Whether a form's level is one of the four. */ | |
| 281 | + | export function parsePauseLevel(value: unknown): PauseLevel | null { | |
| 282 | + | const level = String(value ?? ""); | |
| 283 | + | return PAUSE_LEVELS.some((l) => l.level === level) ? (level as PauseLevel) : null; | |
| 284 | + | } | |
| 285 | + | ||
| 286 | + | /** The red bar on every sudo page while any level is paused. Null when none is. */ | |
| 287 | + | export function pauseBanner(guard: PlatformGuard): string | null { | |
| 288 | + | const paused = guard.levels.filter((l) => l.paused); | |
| 289 | + | if (paused.length === 0) return null; | |
| 290 | + | const names = paused.map((l) => (l.auto ? `${l.level} (by the usage watcher)` : l.level)); | |
| 291 | + | const list = names.length === 1 ? names[0] : `${names.slice(0, -1).join(", ")} and ${names[names.length - 1]}`; | |
| 292 | + | return `Paused across g1t: ${list}.`; | |
| 293 | + | } | |
| 294 | + | ||
| 295 | + | /** `1.2M`, `240k`, `2.0B`: a count in a few characters. */ | |
| 296 | + | export function count(value: number): string { | |
| 297 | + | const n = Math.abs(value); | |
| 298 | + | if (n >= 1e9) return `${(value / 1e9).toFixed(1)}B`; | |
| 299 | + | if (n >= 1e6) return `${(value / 1e6).toFixed(1)}M`; | |
| 300 | + | if (n >= 1e3) return `${Math.round(value / 1e3)}k`; | |
| 301 | + | return `${Math.round(value)}`; | |
| 302 | + | } | |
| 303 | + | ||
| 304 | + | /** An hour's value as a share of its threshold, rounded; null with no threshold. */ | |
| 305 | + | export function thresholdShare(value: number, threshold: number): number | null { | |
| 306 | + | return threshold > 0 ? Math.round((value / threshold) * 100) : null; | |
| 307 | + | } | |
| 7 | 7 | import { MobileBar, Sidebar } from "./components/shell"; | |
| 8 | 8 | import { ButtonLink } from "./components/ui"; | |
| 9 | 9 | import type { NavCounts } from "./lib/nav"; | |
| 10 | − | import { spendBanner } from "./lib/costs"; | |
| 10 | + | import { pauseBanner, spendBanner } from "./lib/costs"; | |
| 11 | 11 | import { admin, identity, statusAdmin } from "./lib/services.server"; | |
| 12 | 12 | import { settle } from "./lib/settle"; | |
| 13 | 13 | import { requireStaff, zoneContext } from "./lib/staff"; | |
| ⋯ | |||
| 27 | 27 | export async function loader({ context }: Route.LoaderArgs) { | |
| 28 | 28 | const { email } = requireStaff(context); | |
| 29 | 29 | // The sidebar's counts: a service that does not answer shows none. | |
| 30 | − | const [waitlist, incidents, alerts, caps] = await Promise.all([ | |
| 30 | + | const [waitlist, incidents, alerts, caps, guard] = await Promise.all([ | |
| 31 | 31 | settle(identity.waitlistPending()), | |
| 32 | 32 | settle(statusAdmin.openCount()), | |
| 33 | 33 | settle(admin.costAlerts()), | |
| 34 | 34 | settle(admin.spendCaps()), | |
| 35 | + | settle(admin.platformGuard()), | |
| 35 | 36 | ]); | |
| 36 | 37 | const counts: NavCounts = { waitlist: waitlist.ok ? waitlist.value : 0, incidents: incidents.ok ? incidents.value : 0 }; | |
| 37 | 38 | // Every page says times in this zone (components/ui.tsx `When`). | |
| ⋯ | |||
| 44 | 45 | // g1t's own spend (billing's budget): the daily breaker open, or a comped | |
| 45 | 46 | // account's monthly budget used up. Red until it clears or staff act. | |
| 46 | 47 | const spend = caps.ok ? spendBanner(caps.value) : null; | |
| 47 | − | return { email, counts, zone, zoneChosen: chosen, margin, spend }; | |
| 48 | + | // A platform pause (billing's platform.rs): red on every page while any | |
| 49 | + | // level is paused, by staff or by the usage watcher. | |
| 50 | + | const paused = guard.ok ? pauseBanner(guard.value) : null; | |
| 51 | + | return { email, counts, zone, zoneChosen: chosen, margin, spend, paused }; | |
| 48 | 52 | } | |
| 49 | 53 | ||
| 50 | 54 | export function Layout({ children }: { children: React.ReactNode }) { | |
| ⋯ | |||
| 79 | 83 | </a> | |
| 80 | 84 | </div> | |
| 81 | 85 | )} | |
| 86 | + | {root?.paused && ( | |
| 87 | + | <div role="alert" className="border-b border-danger/40 bg-danger/12 px-4 py-2 text-sm text-danger"> | |
| 88 | + | <span className="font-medium">Platform pause:</span> {root.paused}{" "} | |
| 89 | + | <a href="/costs#platform" className="underline underline-offset-2"> | |
| 90 | + | Platform pause | |
| 91 | + | </a> | |
| 92 | + | </div> | |
| 93 | + | )} | |
| 82 | 94 | {children} | |
| 83 | 95 | </div> | |
| 84 | 96 | {/* No <Scripts />: sudo ships no JavaScript, and its policy allows none. */} | |
| 1 | 1 | import type { ReactNode } from "react"; | |
| 2 | 2 | import { Link } from "react-router"; | |
| 3 | 3 | ||
| 4 | − | import type { CostsReport } from "@g1t/contracts"; | |
| 4 | + | import type { CostsReport, PlatformGuard, PlatformMetric } from "@g1t/contracts"; | |
| 5 | 5 | ||
| 6 | 6 | import type { Route } from "./+types/costs"; | |
| 7 | 7 | import { DaysChart } from "~/components/costs"; | |
| 8 | 8 | import { CostsHeader, chip, costsHref } from "~/components/costs-header"; | |
| 9 | 9 | import { Badge, Button, Field, Input, Notice, Section, Stat, When } from "~/components/ui"; | |
| 10 | − | import { capPercent, daySeries, marginOnPrice, marginTone, parseBucket, percentLabel, spendRows, subscriptionsOver, whoPaid } from "~/lib/costs"; | |
| 10 | + | import { PAUSE_LEVELS, capPercent, count, daySeries, marginOnPrice, marginTone, parseBucket, percentLabel, spendRows, subscriptionsOver, thresholdShare, whoPaid } from "~/lib/costs"; | |
| 11 | 11 | import { type CostsActionResult, costsAction, costsLoader } from "~/lib/costs-route.server"; | |
| 12 | 12 | import { usd } from "~/lib/money"; | |
| 13 | 13 | ||
| ⋯ | |||
| 34 | 34 | <div className="mt-5"> | |
| 35 | 35 | <Notice tone="warn">Billing did not answer for costs: {error}</Notice> | |
| 36 | 36 | </div> | |
| 37 | + | <PlatformSection guard={loaderData.guard} unavailable={loaderData.guardError} failed={failed} /> | |
| 37 | 38 | </main> | |
| 38 | 39 | ); | |
| 39 | 40 | } | |
| ⋯ | |||
| 50 | 51 | ||
| 51 | 52 | <SpendSection caps={report.caps} range={range} error={failed?.section === "lift" ? failed.error : null} /> | |
| 52 | 53 | ||
| 54 | + | <PlatformSection guard={loaderData.guard} unavailable={loaderData.guardError} failed={failed} /> | |
| 55 | + | ||
| 53 | 56 | <Section | |
| 54 | 57 | className="mt-6" | |
| 55 | 58 | title={product ? `${product.title}, by day` : "By day"} | |
| ⋯ | |||
| 504 | 507 | ); | |
| 505 | 508 | } | |
| 506 | 509 | ||
| 510 | + | ||
| 511 | + | /** | |
| 512 | + | * The platform pause and the hourly usage watch (billing's platform.rs, | |
| 513 | + | * docs/SPEND-GUARDRAILS.md): four levels staff can pause across g1t, what | |
| 514 | + | * Cloudflare counted in the last hour against each threshold, and the | |
| 515 | + | * last day's breaches. | |
| 516 | + | */ | |
| 517 | + | function PlatformSection({ | |
| 518 | + | guard, | |
| 519 | + | unavailable, | |
| 520 | + | failed, | |
| 521 | + | }: { | |
| 522 | + | guard: PlatformGuard | null; | |
| 523 | + | unavailable: string | null; | |
| 524 | + | failed: CostsActionResult | null; | |
| 525 | + | }) { | |
| 526 | + | return ( | |
| 527 | + | <Section | |
| 528 | + | className="mt-6" | |
| 529 | + | id="platform" | |
| 530 | + | title="Platform pause" | |
| 531 | + | description="What Cloudflare counted for all of g1t, read at a quarter past each hour: Workers, D1, Queues, Durable Objects, KV and Artifacts, each against an hourly threshold (PLATFORM_HOURLY_* in billing's wrangler.jsonc). A breach emails staff once per metric every 6 hours. Five times a threshold pauses what that metric feeds, of the levels AUTO_PAUSE names. Any level can be paused here by hand; every service sees a change within 30 seconds." | |
| 532 | + | > | |
| 533 | + | {!guard ? ( | |
| 534 | + | <Notice tone="warn">Billing did not answer for the platform pause: {unavailable}</Notice> | |
| 535 | + | ) : ( | |
| 536 | + | <> | |
| 537 | + | {!guard.can_read && ( | |
| 538 | + | <div className="mb-4"> | |
| 539 | + | <Notice tone="warn">Billing has no Cloudflare token with Account Analytics Read, so the hourly watch reads nothing. The pause still works.</Notice> | |
| 540 | + | </div> | |
| 541 | + | )} | |
| 542 | + | {(guard.blind ?? []).length > 0 && ( | |
| 543 | + | <div className="mb-4"> | |
| 544 | + | <Notice tone="warn"> | |
| 545 | + | The watcher can't see: {(guard.blind ?? []).map((b) => b.key).join(", ")} | |
| 546 | + | {guard.last_run ? ` (the run for ${guard.last_run})` : ""}. Their metrics are not watched until it is fixed; three runs in a row email | |
| 547 | + | staff. | |
| 548 | + | <ul className="mt-2 space-y-1 text-xs"> | |
| 549 | + | {(guard.blind ?? []).map((b) => ( | |
| 550 | + | <li key={b.key}> | |
| 551 | + | <code>{b.dataset}</code> ({b.key}): {b.error} | |
| 552 | + | </li> | |
| 553 | + | ))} | |
| 554 | + | </ul> | |
| 555 | + | </Notice> | |
| 556 | + | </div> | |
| 557 | + | )} | |
| 558 | + | {guard.empty && ( | |
| 559 | + | <div className="mb-4"> | |
| 560 | + | <Notice tone="warn"> | |
| 561 | + | Every dataset answered with no rows{guard.last_run ? ` in the run for ${guard.last_run}` : ""}, the month so far included: likely the wrong | |
| 562 | + | account or a token that cannot see its analytics. The watcher sees nothing. | |
| 563 | + | </Notice> | |
| 564 | + | </div> | |
| 565 | + | )} | |
| 566 | + | {failed?.section === "platform" && ( | |
| 567 | + | <div className="mb-4"> | |
| 568 | + | <Notice tone="error">{failed.error}</Notice> | |
| 569 | + | </div> | |
| 570 | + | )} | |
| 571 | + | <div className="grid gap-3 lg:grid-cols-2"> | |
| 572 | + | {PAUSE_LEVELS.map(({ level, title, stops }) => { | |
| 573 | + | const state = guard.levels.find((l) => l.level === level); | |
| 574 | + | const paused = state?.paused ?? false; | |
| 575 | + | const error = failed?.section === `pause-${level}` ? failed.error : null; | |
| 576 | + | return ( | |
| 577 | + | <div key={level} className={`rounded-lg border p-4 ${paused ? "border-danger/50 bg-danger/8" : "border-line"}`}> | |
| 578 | + | <div className="flex items-center justify-between gap-2"> | |
| 579 | + | <span className="font-medium">{title}</span> | |
| 580 | + | {paused ? <Badge tone="danger">Paused</Badge> : <Badge tone="mint">Running</Badge>} | |
| 581 | + | </div> | |
| 582 | + | <p className="mt-1 text-xs text-muted">{stops}</p> | |
| 583 | + | {state?.set_at && ( | |
| 584 | + | <p className="mt-2 text-xs text-faint"> | |
| 585 | + | {paused ? "Paused" : "Resumed"} <When at={state.set_at} time /> by {state.auto ? "the usage watcher" : state.set_by} | |
| 586 | + | {state.note ? `: “${state.note}”` : ""} | |
| 587 | + | </p> | |
| 588 | + | )} | |
| 589 | + | <form method="post" action="#platform" className="mt-3 flex flex-col gap-2 sm:flex-row sm:items-end"> | |
| 590 | + | <input type="hidden" name="intent" value={paused ? "resume" : "pause"} /> | |
| 591 | + | <input type="hidden" name="level" value={level} /> | |
| 592 | + | <Field label={paused ? "Why resume it" : "Why pause it"} hint="Recorded in the audit log."> | |
| 593 | + | <Input name="note" required minLength={5} maxLength={500} placeholder={paused ? "e.g. Fixed the loop in search" : "e.g. KV lists runaway"} /> | |
| 594 | + | </Field> | |
| 595 | + | <Button type="submit" variant={paused ? "primary" : "danger"}> | |
| 596 | + | {paused ? "Resume" : "Pause"} | |
| 597 | + | </Button> | |
| 598 | + | </form> | |
| 599 | + | {error && ( | |
| 600 | + | <div className="mt-2"> | |
| 601 | + | <Notice tone="error">{error}</Notice> | |
| 602 | + | </div> | |
| 603 | + | )} | |
| 604 | + | </div> | |
| 605 | + | ); | |
| 606 | + | })} | |
| 607 | + | </div> | |
| 608 | + | <p className="mt-3 text-xs text-muted"> | |
| 609 | + | A severe breach may pause: {guard.auto_pause.length ? guard.auto_pause.join(", ") : "nothing (AUTO_PAUSE is empty)"}. | |
| 610 | + | </p> | |
| 611 | + | ||
| 612 | + | <h3 className="mt-6 text-sm font-medium">Breaches in the last 24 hours</h3> | |
| 613 | + | {guard.breaches.length === 0 ? ( | |
| 614 | + | <p className="mt-2 text-sm text-muted">None.</p> | |
| 615 | + | ) : ( | |
| 616 | + | <ul className="mt-2 space-y-2 text-sm"> | |
| 617 | + | {guard.breaches.map((b) => ( | |
| 618 | + | <li key={b.id} className="rounded-md border border-line px-3 py-2"> | |
| 619 | + | <div className="flex flex-wrap items-center gap-2"> | |
| 620 | + | <Badge tone={b.severe ? "danger" : "warn"}>{b.severe ? "Severe" : b.rule === "spike" ? "Spike" : "Over threshold"}</Badge> | |
| 621 | + | <span className="text-xs text-faint"> | |
| 622 | + | <When at={b.opened_at} time /> | |
| 623 | + | {b.emailed_at ? ", emailed" : ""} | |
| 624 | + | </span> | |
| 625 | + | </div> | |
| 626 | + | <p className="mt-1 text-fg-soft">{b.detail}</p> | |
| 627 | + | </li> | |
| 628 | + | ))} | |
| 629 | + | </ul> | |
| 630 | + | )} | |
| 631 | + | ||
| 632 | + | <UsageTable title={guard.hour ? `The hour from ${guard.hour}` : "The last hour"} metrics={guard.last_hour} hourly /> | |
| 633 | + | <UsageTable title={`${guard.month}, so far`} metrics={guard.month_to_date} hourly={false} /> | |
| 634 | + | <p className="mt-3 text-xs text-muted"> | |
| 635 | + | Ids are Cloudflare's. <code>node scripts/ops/platform-usage.mjs</code> names them and shows the last 24 hours per script, queue, database and | |
| 636 | + | namespace. | |
| 637 | + | </p> | |
| 638 | + | </> | |
| 639 | + | )} | |
| 640 | + | </Section> | |
| 641 | + | ); | |
| 642 | + | } | |
| 643 | + | ||
| 644 | + | function UsageTable({ title, metrics, hourly }: { title: string; metrics: PlatformMetric[]; hourly: boolean }) { | |
| 645 | + | return ( | |
| 646 | + | <> | |
| 647 | + | <h3 className="mt-6 text-sm font-medium">{title}</h3> | |
| 648 | + | {metrics.length === 0 ? ( | |
| 649 | + | <p className="mt-2 text-sm text-muted">Nothing read yet.</p> | |
| 650 | + | ) : ( | |
| 651 | + | <div className="-mx-4 mt-2 overflow-x-auto sm:-mx-5"> | |
| 652 | + | <table className="w-full min-w-[36rem] text-sm"> | |
| 653 | + | <thead> | |
| 654 | + | <tr className="border-b border-line text-left text-xs text-muted"> | |
| 655 | + | <th className="px-4 py-2 font-medium sm:px-5">Metric</th> | |
| 656 | + | <th className="px-4 py-2 text-right font-medium">Count</th> | |
| 657 | + | {hourly && <th className="px-4 py-2 text-right font-medium">Of threshold</th>} | |
| 658 | + | <th className="px-4 py-2 font-medium sm:pr-5">Most from</th> | |
| 659 | + | </tr> | |
| 660 | + | </thead> | |
| 661 | + | <tbody> | |
| 662 | + | {metrics.map((m) => { | |
| 663 | + | const share = thresholdShare(m.value, m.threshold); | |
| 664 | + | return ( | |
| 665 | + | <tr key={m.metric} className="border-b border-line last:border-0"> | |
| 666 | + | <td className="px-4 py-2.5 sm:px-5">{m.title}</td> | |
| 667 | + | <td className="tabular px-4 py-2.5 text-right">{count(m.value)}</td> | |
| 668 | + | {hourly && ( | |
| 669 | + | <td className={`tabular px-4 py-2.5 text-right ${share != null && share > 100 ? "text-danger" : "text-muted"}`}> | |
| 670 | + | {share == null ? "—" : `${share}% of ${count(m.threshold)}`} | |
| 671 | + | </td> | |
| 672 | + | )} | |
| 673 | + | <td className="max-w-[16rem] truncate px-4 py-2.5 text-xs text-faint sm:pr-5"> | |
| 674 | + | {m.top_name ? `${m.top_name} (${count(m.top_value ?? 0)})` : "—"} | |
| 675 | + | </td> | |
| 676 | + | </tr> | |
| 677 | + | ); | |
| 678 | + | })} | |
| 679 | + | </tbody> | |
| 680 | + | </table> | |
| 681 | + | </div> | |
| 682 | + | )} | |
| 683 | + | </> | |
| 684 | + | ); | |
| 685 | + | } | |
| 3644 | 3644 | pub by: String, | |
| 3645 | 3645 | } | |
| 3646 | 3646 | ||
| 3647 | + | // ---- Platform pauses and the usage watcher (docs/SPEND-GUARDRAILS.md) ---- | |
| 3648 | + | ||
| 3649 | + | /// A g1t-wide pause, set by staff in sudo or by billing's hourly usage | |
| 3650 | + | /// watcher on a severe breach. Each level is independent. | |
| 3651 | + | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 3652 | + | #[serde(rename_all = "snake_case")] | |
| 3653 | + | pub enum PauseLevel { | |
| 3654 | + | /// Agents, sandboxes, Actions hosted jobs, deploy builds: every | |
| 3655 | + | /// reservation through billing's `reserve` but embeddings. | |
| 3656 | + | Compute, | |
| 3657 | + | /// Actions' cron-triggered runs, and the runner's sweep that starts | |
| 3658 | + | /// queued agents. | |
| 3659 | + | Schedules, | |
| 3660 | + | /// Context embeddings and backfills, and search's backfills. | |
| 3661 | + | Indexing, | |
| 3662 | + | /// Social card rendering, which falls back to a static image. | |
| 3663 | + | Renders, | |
| 3664 | + | } | |
| 3665 | + | ||
| 3666 | + | impl PauseLevel { | |
| 3667 | + | pub const ALL: [PauseLevel; 4] = [PauseLevel::Compute, PauseLevel::Schedules, PauseLevel::Indexing, PauseLevel::Renders]; | |
| 3668 | + | ||
| 3669 | + | pub fn as_str(self) -> &'static str { | |
| 3670 | + | match self { | |
| 3671 | + | PauseLevel::Compute => "compute", | |
| 3672 | + | PauseLevel::Schedules => "schedules", | |
| 3673 | + | PauseLevel::Indexing => "indexing", | |
| 3674 | + | PauseLevel::Renders => "renders", | |
| 3675 | + | } | |
| 3676 | + | } | |
| 3677 | + | ||
| 3678 | + | pub fn parse(text: &str) -> Option<PauseLevel> { | |
| 3679 | + | PauseLevel::ALL.into_iter().find(|level| level.as_str() == text.trim()) | |
| 3680 | + | } | |
| 3681 | + | } | |
| 3682 | + | ||
| 3683 | + | /// `platform_pause`: which levels are paused now. Takes nothing. Callers | |
| 3684 | + | /// keep the answer about 30 seconds; one that cannot read it runs (fails | |
| 3685 | + | /// open). | |
| 3686 | + | #[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] | |
| 3687 | + | pub struct PlatformPause { | |
| 3688 | + | #[serde(default)] | |
| 3689 | + | pub compute: bool, | |
| 3690 | + | #[serde(default)] | |
| 3691 | + | pub schedules: bool, | |
| 3692 | + | #[serde(default)] | |
| 3693 | + | pub indexing: bool, | |
| 3694 | + | #[serde(default)] | |
| 3695 | + | pub renders: bool, | |
| 3696 | + | } | |
| 3697 | + | ||
| 3698 | + | impl PlatformPause { | |
| 3699 | + | pub fn is(&self, level: PauseLevel) -> bool { | |
| 3700 | + | match level { | |
| 3701 | + | PauseLevel::Compute => self.compute, | |
| 3702 | + | PauseLevel::Schedules => self.schedules, | |
| 3703 | + | PauseLevel::Indexing => self.indexing, | |
| 3704 | + | PauseLevel::Renders => self.renders, | |
| 3705 | + | } | |
| 3706 | + | } | |
| 3707 | + | ||
| 3708 | + | pub fn set(&mut self, level: PauseLevel, paused: bool) { | |
| 3709 | + | match level { | |
| 3710 | + | PauseLevel::Compute => self.compute = paused, | |
| 3711 | + | PauseLevel::Schedules => self.schedules = paused, | |
| 3712 | + | PauseLevel::Indexing => self.indexing = paused, | |
| 3713 | + | PauseLevel::Renders => self.renders = paused, | |
| 3714 | + | } | |
| 3715 | + | } | |
| 3716 | + | ||
| 3717 | + | pub fn any(&self) -> bool { | |
| 3718 | + | PauseLevel::ALL.into_iter().any(|level| self.is(level)) | |
| 3719 | + | } | |
| 3720 | + | } | |
| 3721 | + | ||
| 3722 | + | /// One level as sudo shows it: who paused it, when and why. | |
| 3723 | + | #[derive(Clone, Debug, Default, Serialize, Deserialize)] | |
| 3724 | + | pub struct PauseState { | |
| 3725 | + | pub level: String, | |
| 3726 | + | pub paused: bool, | |
| 3727 | + | pub note: Option<String>, | |
| 3728 | + | pub set_by: Option<String>, | |
| 3729 | + | pub set_at: Option<String>, | |
| 3730 | + | /// Set by the usage watcher, not a person. | |
| 3731 | + | pub auto: bool, | |
| 3732 | + | } | |
| 3733 | + | ||
| 3734 | + | /// One metric's usage over an hour or the month so far. | |
| 3735 | + | #[derive(Clone, Debug, Default, Serialize, Deserialize)] | |
| 3736 | + | pub struct PlatformMetric { | |
| 3737 | + | pub metric: String, | |
| 3738 | + | pub title: String, | |
| 3739 | + | pub value: f64, | |
| 3740 | + | /// Its hourly threshold (`PLATFORM_HOURLY_*`); zero: none. | |
| 3741 | + | pub threshold: f64, | |
| 3742 | + | /// The script, queue, database or namespace that counted most. | |
| 3743 | + | pub top_name: Option<String>, | |
| 3744 | + | pub top_value: Option<f64>, | |
| 3745 | + | } | |
| 3746 | + | ||
| 3747 | + | /// A breach the watcher found. | |
| 3748 | + | #[derive(Clone, Debug, Default, Serialize, Deserialize)] | |
| 3749 | + | pub struct PlatformBreach { | |
| 3750 | + | pub id: String, | |
| 3751 | + | pub metric: String, | |
| 3752 | + | pub hour: String, | |
| 3753 | + | /// `threshold` or `spike`. | |
| 3754 | + | pub rule: String, | |
| 3755 | + | pub value: f64, | |
| 3756 | + | pub threshold: f64, | |
| 3757 | + | pub severe: bool, | |
| 3758 | + | pub top_name: Option<String>, | |
| 3759 | + | pub detail: String, | |
| 3760 | + | /// Levels it paused. | |
| 3761 | + | pub paused: Vec<String>, | |
| 3762 | + | pub opened_at: String, | |
| 3763 | + | pub emailed_at: Option<String>, | |
| 3764 | + | } | |
| 3765 | + | ||
| 3766 | + | /// `admin_platform_guard`: the pauses, the last hour read, the month so | |
| 3767 | + | /// far and the last day's breaches, for sudo's Costs page and its banner. | |
| 3768 | + | #[derive(Clone, Debug, Default, Serialize, Deserialize)] | |
| 3769 | + | pub struct PlatformGuard { | |
| 3770 | + | pub levels: Vec<PauseState>, | |
| 3771 | + | /// The last hour the watcher read (`YYYY-MM-DDTHH:00:00Z`), if any. | |
| 3772 | + | pub hour: Option<String>, | |
| 3773 | + | pub last_hour: Vec<PlatformMetric>, | |
| 3774 | + | pub month: String, | |
| 3775 | + | pub month_to_date: Vec<PlatformMetric>, | |
| 3776 | + | /// The last 24 hours' breaches, newest first. | |
| 3777 | + | pub breaches: Vec<PlatformBreach>, | |
| 3778 | + | /// Whether the watcher can read Cloudflare's analytics at all. | |
| 3779 | + | pub can_read: bool, | |
| 3780 | + | /// `AUTO_PAUSE`: the levels a severe breach may pause. | |
| 3781 | + | pub auto_pause: Vec<String>, | |
| 3782 | + | /// What the latest run could not see: each query that failed. | |
| 3783 | + | #[serde(default)] | |
| 3784 | + | pub blind: Vec<BlindQuery>, | |
| 3785 | + | /// The latest run found every dataset empty (the wrong account, or a | |
| 3786 | + | /// token that cannot see its analytics). | |
| 3787 | + | #[serde(default)] | |
| 3788 | + | pub empty: bool, | |
| 3789 | + | /// The hour the latest run read. | |
| 3790 | + | #[serde(default)] | |
| 3791 | + | pub last_run: Option<String>, | |
| 3792 | + | } | |
| 3793 | + | ||
| 3794 | + | /// A query the watcher's latest run could not read. | |
| 3795 | + | #[derive(Clone, Debug, Default, Serialize, Deserialize)] | |
| 3796 | + | pub struct BlindQuery { | |
| 3797 | + | pub key: String, | |
| 3798 | + | pub dataset: String, | |
| 3799 | + | pub error: String, | |
| 3800 | + | } | |
| 3801 | + | ||
| 3802 | + | /// `admin_platform_guard`. Returns `PlatformGuard`. | |
| 3803 | + | #[derive(Debug, Default, Serialize, Deserialize)] | |
| 3804 | + | pub struct AdminPlatformGuardArgs {} | |
| 3805 | + | ||
| 3806 | + | /// `admin_set_pause`: pauses or resumes one level, with why. Recorded in | |
| 3807 | + | /// the audit log. Returns `Outcome<PlatformGuard>`. | |
| 3808 | + | #[derive(Debug, Serialize, Deserialize)] | |
| 3809 | + | pub struct AdminSetPauseArgs { | |
| 3810 | + | pub level: String, | |
| 3811 | + | pub paused: bool, | |
| 3812 | + | pub note: String, | |
| 3813 | + | pub by: String, | |
| 3814 | + | } | |
| 3815 | + | ||
| 3647 | 3816 | /// `admin_cost_alerts`: the open margin alerts, for sudo's banner. | |
| 3648 | 3817 | /// Returns `Vec<MarginAlert>`. | |
| 3649 | 3818 | #[derive(Debug, Default, Serialize, Deserialize)] |
| 464 | 464 | /// gateway's own token, sent as `cf-aig-authorization`. | |
| 465 | 465 | #[serde(default)] | |
| 466 | 466 | pub gateway_token: Option<String>, | |
| 467 | + | /// The most the run may spend on models, in millionths of a dollar: the | |
| 468 | + | /// lower of its project's cost cap and its plan's. The proxy refuses | |
| 469 | + | /// the run's requests once it has spent this. `None` until the sandbox | |
| 470 | + | /// sets it (`cap_model_sessions`), and for the AI Gateway. | |
| 471 | + | #[serde(default)] | |
| 472 | + | pub cap_micros: Option<i64>, | |
| 467 | 473 | } | |
| 468 | 474 | ||
| 469 | 475 | // --- Methods ----------------------------------------------------------------- | |
| ⋯ | |||
| 722 | 728 | pub token_hashes: Vec<String>, | |
| 723 | 729 | } | |
| 724 | 730 | ||
| 731 | + | /// `cap_model_sessions`: sets the most the runs whose model tokens hash to | |
| 732 | + | /// these (SHA-256, lowercase hex) may spend on models, in millionths of a | |
| 733 | + | /// dollar, which the model proxy holds them to. The sandbox sets it once | |
| 734 | + | /// it knows the run's guardrails and plan; zero or less clears it. Returns | |
| 735 | + | /// how many open sessions it set. | |
| 736 | + | #[derive(Debug, Serialize, Deserialize)] | |
| 737 | + | pub struct CapModelSessionsArgs { | |
| 738 | + | #[serde(alias = "tokenHashes")] | |
| 739 | + | pub token_hashes: Vec<String>, | |
| 740 | + | #[serde(alias = "capMicros")] | |
| 741 | + | pub cap_micros: i64, | |
| 742 | + | } | |
| 743 | + | ||
| 725 | 744 | /// `model_provider`: the workspace's own model connection, if it has one. | |
| 726 | 745 | /// Returns `Option<Connection>`. | |
| 727 | 746 | #[derive(Debug, Serialize, Deserialize)] | |
| 62 | 62 | ||
| 63 | 63 | pub mod d1; | |
| 64 | 64 | pub mod limits; | |
| 65 | + | pub mod pause; | |
| 65 | 66 | pub mod wire; | |
| 66 | 67 | ||
| 67 | 68 | /// Helpers for bindings that workers-rs has no typed wrapper for, such as |
| 1 | + | //! g1t-wide pauses (billing's `platform_pause`; docs/SPEND-GUARDRAILS.md), | |
| 2 | + | //! read cheaply: one call to billing per isolate every 30 seconds at most, | |
| 3 | + | //! never a database read per request. | |
| 4 | + | //! | |
| 5 | + | //! When billing cannot say, nothing is paused: a pause is a brake staff | |
| 6 | + | //! (or billing's usage watcher) pull on purpose, and a billing outage must | |
| 7 | + | //! not stop schedules and indexing everywhere. The failure is kept for the | |
| 8 | + | //! same 30 seconds, so an outage is not asked about on every request. | |
| 9 | + | ||
| 10 | + | use std::cell::RefCell; | |
| 11 | + | ||
| 12 | + | use g1t_contracts::billing::{PauseLevel, PlatformPause}; | |
| 13 | + | use worker::Fetcher; | |
| 14 | + | ||
| 15 | + | /// How long an answer is kept in the isolate. | |
| 16 | + | pub const KEEP_MS: u64 = 30_000; | |
| 17 | + | ||
| 18 | + | thread_local! { | |
| 19 | + | static KEPT: RefCell<Option<(PlatformPause, u64)>> = const { RefCell::new(None) }; | |
| 20 | + | } | |
| 21 | + | ||
| 22 | + | /// A kept answer, while it is fresh. | |
| 23 | + | fn fresh(kept: Option<(PlatformPause, u64)>, now: u64) -> Option<PlatformPause> { | |
| 24 | + | kept.filter(|(_, until)| *until > now).map(|(pause, _)| pause) | |
| 25 | + | } | |
| 26 | + | ||
| 27 | + | /// Whether `level` is paused, through the billing service's binding. | |
| 28 | + | pub async fn paused(billing: &Fetcher, level: PauseLevel) -> bool { | |
| 29 | + | current(billing).await.is(level) | |
| 30 | + | } | |
| 31 | + | ||
| 32 | + | /// Every level, kept for `KEEP_MS`. Nothing paused when billing cannot say. | |
| 33 | + | pub async fn current(billing: &Fetcher) -> PlatformPause { | |
| 34 | + | let now = crate::now_ms(); | |
| 35 | + | if let Some(pause) = KEPT.with(|kept| fresh(*kept.borrow(), now)) { | |
| 36 | + | return pause; | |
| 37 | + | } | |
| 38 | + | let pause = match crate::call::<_, PlatformPause>(billing, "platform_pause", &serde_json::json!({})).await { | |
| 39 | + | Ok(pause) => pause, | |
| 40 | + | Err(error) => { | |
| 41 | + | worker::console_error!("platform pause unreadable, so nothing is paused: {error}"); | |
| 42 | + | PlatformPause::default() | |
| 43 | + | } | |
| 44 | + | }; | |
| 45 | + | KEPT.with(|kept| *kept.borrow_mut() = Some((pause, now + KEEP_MS))); | |
| 46 | + | pause | |
| 47 | + | } | |
| 48 | + | ||
| 49 | + | #[cfg(test)] | |
| 50 | + | mod tests { | |
| 51 | + | use super::*; | |
| 52 | + | ||
| 53 | + | #[test] | |
| 54 | + | fn an_answer_is_kept_thirty_seconds() { | |
| 55 | + | let paused = PlatformPause { schedules: true, ..PlatformPause::default() }; | |
| 56 | + | assert_eq!(fresh(Some((paused, 1_000 + KEEP_MS)), 1_000), Some(paused)); | |
| 57 | + | assert_eq!(fresh(Some((paused, 1_000 + KEEP_MS)), 1_000 + KEEP_MS), None); | |
| 58 | + | assert_eq!(fresh(None, 0), None); | |
| 59 | + | } | |
| 60 | + | ||
| 61 | + | #[test] | |
| 62 | + | fn an_unread_pause_pauses_nothing() { | |
| 63 | + | let none = PlatformPause::default(); | |
| 64 | + | assert!(PauseLevel::ALL.into_iter().all(|level| !none.is(level))); | |
| 65 | + | assert!(!none.any()); | |
| 66 | + | } | |
| 67 | + | } |
| 13 | 13 | | The runner's images | `services/runner/base/Dockerfile`, `services/runner/Dockerfile`, `services/runner/base.json`, `scripts/build-runner.mjs`, `scripts/deploy/image.mjs` | | |
| 14 | 14 | | The workflows | `.g1t/workflows/deploy.yml`, `.g1t/workflows/runner-base.yml` | | |
| 15 | 15 | | The old entry point | `scripts/deploy.sh`, now a wrapper | | |
| 16 | + | | Spend guardrails (platform pause, hourly usage watch) | [SPEND-GUARDRAILS.md](SPEND-GUARDRAILS.md) | | |
| 16 | 17 | ||
| 17 | 18 | ## The manifest | |
| 18 | 19 |
| 1 | + | # Spend guardrails | |
| 2 | + | ||
| 3 | + | Cloudflare has no hard spending cap. A loop in a Worker, a queue that | |
| 4 | + | retries forever or a cron that lists KV every second shows up on the bill | |
| 5 | + | weeks later unless something here catches it. This is what catches it, | |
| 6 | + | how to stop it, and what the owner has to set up by hand in Cloudflare's | |
| 7 | + | dashboard. Internal: the public side is the "When g1t pauses work for | |
| 8 | + | everyone" section of `apps/docs/src/content/docs/guides/usage-and-billing.md` | |
| 9 | + | and "How they are enforced" in `guides/guardrails.md`. | |
| 10 | + | ||
| 11 | + | ## What catches what | |
| 12 | + | ||
| 13 | + | | Guardrail | Catches | Where | | |
| 14 | + | | --- | --- | --- | | |
| 15 | + | | A workspace's limits, caps and spike pause | One workspace's agents, sandboxes and builds running away | `services/billing/src/limits.rs`, `compute.rs`; reserved through `ComputeGate.admit` (`packages/contracts/src/compute.ts`) | | |
| 16 | + | | The comped budget and the daily breaker | g1t's own spend on agents: comped accounts, trials, pools; $75 a day pauses hosted-model agent runs g1t pays for | `services/billing/src/budget.rs`; [BILLING_OPERATIONS.md](BILLING_OPERATIONS.md) | | |
| 17 | + | | The daily reconciliation | What Cloudflare billed against what g1t counted, a day later | `services/billing/src/costs.rs`, `margin.rs` | | |
| 18 | + | | **The hourly platform watch** | Platform cost no workspace's limit covers: Workers requests and CPU, D1 rows, Queue operations, Durable Objects, KV, Artifacts. Within the hour. | `services/billing/src/platform.rs` | | |
| 19 | + | | **The platform pause** | A staff (or automatic) brake on whole kinds of work across g1t | `platform.rs`; read by every service through `g1t_kit::pause` (Rust) or `platformPaused` (`packages/contracts/src/platform.ts`) | | |
| 20 | + | | **The models proxy's run cap** | An agent spending past its run's cap by calling the model proxy itself (a prompt-injected `curl`), around the sandbox's `--max-budget-usd` | `services/models/src/spend.ts`, `run-spend.ts` | | |
| 21 | + | | Cloudflare's own notifications | Anything else, by product, as a last line | Set up by hand: [the checklist](#manual-steps-in-cloudflares-dashboard) | | |
| 22 | + | ||
| 23 | + | ## The hourly platform watch | |
| 24 | + | ||
| 25 | + | At a quarter past each hour (billing's `*/15` cron, gated to the tick at | |
| 26 | + | `:15`), billing reads two windows from Cloudflare's GraphQL Analytics API: | |
| 27 | + | the hour before, and the month so far. It uses the costs reconciliation's | |
| 28 | + | token, `CLOUDFLARE_BILLING_TOKEN` (else `CLOUDFLARE_USAGE_TOKEN`), which | |
| 29 | + | needs **Account Analytics Read**. Without either it reads nothing and sudo | |
| 30 | + | says so; the pause still works. | |
| 31 | + | ||
| 32 | + | | Metric | Dataset and field | Named by | | |
| 33 | + | | --- | --- | --- | | |
| 34 | + | | `workers_requests` | `workersInvocationsAdaptive` `sum.requests` | script | | |
| 35 | + | | `workers_cpu_ms` | `workersInvocationsAdaptive` `sum.cpuTimeUs` / 1000 | script | | |
| 36 | + | | `d1_rows_read`, `d1_rows_written` | `d1AnalyticsAdaptiveGroups` `sum.rowsRead`, `sum.rowsWritten` | database id | | |
| 37 | + | | `queue_operations` | `queueMessageOperationsAdaptiveGroups` `sum.billableOperations` | queue id | | |
| 38 | + | | `do_requests` | `durableObjectsInvocationsAdaptiveGroups` `sum.requests` | script | | |
| 39 | + | | `do_active_seconds`, `do_storage_write_units` | `durableObjectsPeriodicGroups` `sum.activeTime` (µs), `sum.storageWriteUnits` | namespace id | | |
| 40 | + | | `do_rows_written` | `durableObjectsPeriodicGroups` `sum.rowsWritten` | namespace id | | |
| 41 | + | | `kv_reads`, `kv_writes`, `kv_deletes`, `kv_lists` | `kvOperationsAdaptiveGroups` `sum.requests` by `actionType` | namespace id | | |
| 42 | + | | `artifacts_events` | `artifactsEventsAdaptiveGroups` `count` | repository | | |
| 43 | + | ||
| 44 | + | Each metric whose field is less certain has a query of its own, so a | |
| 45 | + | dataset or field GraphQL refuses is skipped and the others still count. | |
| 46 | + | Workers Logs has no dataset here; its volume follows Workers requests, and | |
| 47 | + | log sampling is set per Worker. | |
| 48 | + | ||
| 49 | + | Each hour is kept in billing's D1 (`platform_usage`, migration 0051) with | |
| 50 | + | the script, queue, database or namespace that counted most; the month so | |
| 51 | + | far in `platform_usage_month`; breaches in `platform_alerts`. | |
| 52 | + | ||
| 53 | + | ### When the watcher cannot see | |
| 54 | + | ||
| 55 | + | A skipped query leaves the watcher blind on its metrics, which is the | |
| 56 | + | failure this guards against, so it is never quiet: | |
| 57 | + | ||
| 58 | + | - Every run records the queries that failed, with Cloudflare's error, and | |
| 59 | + | whether every dataset answered with no rows (`platform_watch_runs`). | |
| 60 | + | - sudo's **Platform pause** shows **The watcher can't see: …** with each | |
| 61 | + | error whenever the latest run skipped anything, and a warning when every | |
| 62 | + | dataset was empty. | |
| 63 | + | - A query failing **3 hourly runs in a row** emails staff through the same | |
| 64 | + | alert path, at most once a day per query (`platform_watch_alerts`). | |
| 65 | + | - **Every dataset answering with no rows** for 3 runs in a row, the month so | |
| 66 | + | far included, is emailed the same way (key `all_empty`): almost certainly | |
| 67 | + | the wrong `CLOUDFLARE_ACCOUNT_ID` or a token without Account Analytics Read. | |
| 68 | + | ||
| 69 | + | These field names are from Cloudflare's documentation and were **not | |
| 70 | + | checked against the live schema** when the watcher was written: | |
| 71 | + | ||
| 72 | + | | Query | Unverified | | |
| 73 | + | | --- | --- | | |
| 74 | + | | `workers_cpu` | `workersInvocationsAdaptive` `sum.cpuTimeUs` | | |
| 75 | + | | `do_periodic`, `do_sql` | `durableObjectsPeriodicGroups` with a `datetime_geq` / `datetime_lt` filter, and its `sum.activeTime`, `sum.storageWriteUnits` and `sum.rowsWritten` | | |
| 76 | + | | `d1` | `d1AnalyticsAdaptiveGroups` with a `datetimeHour_geq` / `datetimeHour_lt` filter | | |
| 77 | + | ||
| 78 | + | **Check them with one command** after any deploy that touches the queries: | |
| 79 | + | ||
| 80 | + | ```sh | |
| 81 | + | CLOUDFLARE_API_TOKEN=<Account Analytics Read> node scripts/ops/platform-usage.mjs | |
| 82 | + | ``` | |
| 83 | + | ||
| 84 | + | It runs billing's own queries (`WATCHER_QUERIES`, kept identical to | |
| 85 | + | `QUERIES` in `platform.rs` by its test) over the last full hour. Every | |
| 86 | + | dataset should report rows. It exits 1 and names every dataset that | |
| 87 | + | errored, fell back to fewer fields, or answered empty. A renamed field is | |
| 88 | + | fixed in `QUERIES` in `services/billing/src/platform.rs` and in | |
| 89 | + | `WATCHER_QUERIES` together. | |
| 90 | + | ||
| 91 | + | ### Thresholds | |
| 92 | + | ||
| 93 | + | Hourly, in `services/billing/wrangler.jsonc` `vars`. Each is about a dollar | |
| 94 | + | to a few dollars an hour at list prices, far above a small alpha's normal | |
| 95 | + | hour. `0` turns a metric's threshold (and its spike rule) off. | |
| 96 | + | ||
| 97 | + | | Variable | Default | About | | |
| 98 | + | | --- | --- | --- | | |
| 99 | + | | `PLATFORM_HOURLY_WORKERS_REQUESTS` | 20,000,000 | $6/hour | | |
| 100 | + | | `PLATFORM_HOURLY_WORKERS_CPU_MS` | 100,000,000 | $2/hour | | |
| 101 | + | | `PLATFORM_HOURLY_D1_ROWS_READ` | 2,000,000,000 | $2/hour | | |
| 102 | + | | `PLATFORM_HOURLY_D1_ROWS_WRITTEN` | 5,000,000 | $5/hour | | |
| 103 | + | | `PLATFORM_HOURLY_QUEUE_OPERATIONS` | 5,000,000 | $2/hour | | |
| 104 | + | | `PLATFORM_HOURLY_DO_REQUESTS` | 20,000,000 | $3/hour | | |
| 105 | + | | `PLATFORM_HOURLY_DO_ROWS_WRITTEN` | 5,000,000 | $5/hour | | |
| 106 | + | | `PLATFORM_HOURLY_DO_STORAGE_WRITE_UNITS` | 5,000,000 | $5/hour | | |
| 107 | + | | `PLATFORM_HOURLY_DO_ACTIVE_SECONDS` | 3,000,000 | about $5/hour at 128 MB | | |
| 108 | + | | `PLATFORM_HOURLY_KV_READS` | 10,000,000 | $5/hour | | |
| 109 | + | | `PLATFORM_HOURLY_KV_WRITES` | 200,000 | $1/hour | | |
| 110 | + | | `PLATFORM_HOURLY_KV_DELETES` | 200,000 | $1/hour | | |
| 111 | + | | `PLATFORM_HOURLY_KV_LISTS` | 200,000 | $1/hour | | |
| 112 | + | | `PLATFORM_HOURLY_ARTIFACTS_EVENTS` | 1,000,000 | | | |
| 113 | + | ||
| 114 | + | The rules, in order: | |
| 115 | + | ||
| 116 | + | 1. **Threshold**: an hour over its threshold is a breach. | |
| 117 | + | 2. **Spike**: otherwise, an hour over `PLATFORM_SPIKE_FACTOR` (10) times the | |
| 118 | + | median hour of the week before, and at least `PLATFORM_SPIKE_FLOOR_PERCENT` | |
| 119 | + | (10%) of its threshold, is a breach. It needs a day of history first. | |
| 120 | + | 3. **Severe**: a threshold breach at `PLATFORM_SEVERE_FACTOR` (5) times the | |
| 121 | + | threshold or more pauses the levels that metric feeds, of those | |
| 122 | + | `AUTO_PAUSE` names (default `schedules,indexing`; `compute` and `renders` | |
| 123 | + | are off unless added). Spikes never pause. | |
| 124 | + | ||
| 125 | + | | Metric | Feeds | | |
| 126 | + | | --- | --- | | |
| 127 | + | | Workers | schedules, indexing, renders | | |
| 128 | + | | D1, Queues | schedules, indexing | | |
| 129 | + | | Durable Objects, Artifacts | compute, schedules | | |
| 130 | + | | KV | indexing, renders | | |
| 131 | + | ||
| 132 | + | On a breach, staff are emailed at `COSTS_ALERT_EMAIL` through billing's | |
| 133 | + | `EMAIL` binding, once per metric every 6 hours (a new automatic pause is | |
| 134 | + | always emailed), and sudo's Costs & margin shows it under **Platform | |
| 135 | + | pause**. The alert names the top script, queue, database or namespace. | |
| 136 | + | Ids are Cloudflare's; `node scripts/ops/platform-usage.mjs` names them. | |
| 137 | + | ||
| 138 | + | To change a threshold: edit the variable, push, and billing redeploys. The | |
| 139 | + | next hour reads with it. | |
| 140 | + | ||
| 141 | + | ### Looking by hand | |
| 142 | + | ||
| 143 | + | ```sh | |
| 144 | + | node scripts/ops/platform-usage.mjs # month so far and the last 24 hours | |
| 145 | + | node scripts/ops/platform-usage.mjs --json # the same, as JSON | |
| 146 | + | ``` | |
| 147 | + | ||
| 148 | + | It needs `CLOUDFLARE_API_TOKEN` (or `CLOUDFLARE_API_KEY` and `CLOUDFLARE_EMAIL`) with Account | |
| 149 | + | Analytics Read, as `scripts/deploy/cloudflare.mjs` `cloudflareAuth` reads them, and names ids | |
| 150 | + | with the D1, Queues, KV and Durable Objects listings when the token can | |
| 151 | + | read them. | |
| 152 | + | ||
| 153 | + | ## The platform pause | |
| 154 | + | ||
| 155 | + | Four levels, each independent, kept in billing's `platform_pause` table. | |
| 156 | + | ||
| 157 | + | | Level | Stops | Where it is checked | | |
| 158 | + | | --- | --- | --- | | |
| 159 | + | | `compute` | Every reservation through billing's `reserve` except embeddings: agent runs, checks, the merge queue, workflow jobs on g1t's machines, deploy builds. For every workspace and plan, and where payments are not set up. Refused with `paused`. | `services/billing/src/compute.rs` `reserve` | | |
| 160 | + | | `schedules` | Actions' cron-triggered runs (skipped, not run late), and the runner's sweep that starts queued agents (they stay queued) | `services/actions/src/plan.rs` `on_minute`; `services/runner/src/index.ts` `scheduled` | | |
| 161 | + | | `indexing` | Embeddings (refused in `reserve`), context backfills (**Rebuild** refused, queued jobs closed with a note), search backfills (pages parked in search's `meta` and resumed where they were) | `reserve`; `services/context/src/index.ts`; `services/search/src/lib.rs` | | |
| 162 | + | | `renders` | Social cards: a cache miss gets the cached brand card or a redirect to `https://g1t.sh/brand/g1t-logo-on-dark.png`, kept a minute | `services/og/src/index.ts`, `paused.ts` | | |
| 163 | + | ||
| 164 | + | Work already running finishes at every level. | |
| 165 | + | ||
| 166 | + | **Reads are cheap.** Every caller keeps the flags 30 seconds in its isolate: | |
| 167 | + | billing itself (`pause_now`), Rust services through `g1t_kit::pause`, and | |
| 168 | + | TypeScript services through `platformPaused`. A change reaches everything | |
| 169 | + | within about 30 seconds, and there is never a D1 read per request. | |
| 170 | + | ||
| 171 | + | **When the flag cannot be read, nothing is paused** (fail open), and the | |
| 172 | + | failure is kept for the same 30 seconds. A pause is a brake someone pulls | |
| 173 | + | on purpose; failing closed would turn a billing outage into a platform | |
| 174 | + | outage. The other guardrails (workspace limits, the breaker, the model | |
| 175 | + | proxy's run cap) do not depend on it. | |
| 176 | + | ||
| 177 | + | ### Pausing and resuming | |
| 178 | + | ||
| 179 | + | 1. Open sudo, **Costs & margin**, **Platform pause** (`https://sudo.g1t.sh/costs#platform`). | |
| 180 | + | 2. On the level's card, write why, and choose **Pause** or **Resume**. | |
| 181 | + | ||
| 182 | + | Every change is in the audit log (`platform_paused`, `platform_resumed`), | |
| 183 | + | with who and why. While any level is paused, every sudo page shows a red | |
| 184 | + | **Platform pause** bar. A level the usage watcher paused says so and stays | |
| 185 | + | paused until staff resume it: fix or understand the cause first. | |
| 186 | + | ||
| 187 | + | Without sudo (billing's RPC, through a service binding): | |
| 188 | + | `admin_set_pause` with `{ "level": "schedules", "paused": false, "note": "…", "by": "you@flagon.io" }`. | |
| 189 | + | ||
| 190 | + | ## The models proxy's run cap | |
| 191 | + | ||
| 192 | + | A run's model cap was only enforced inside the sandbox | |
| 193 | + | (`--max-budget-usd`). Now the proxy holds it too. When a run starts, the | |
| 194 | + | runner gives its model session token (`g1tm_`) the run's cap | |
| 195 | + | (`cap_model_sessions` on integrations). The proxy counts each answer's cost | |
| 196 | + | from its token usage in one `RunSpend` Durable Object per session, so | |
| 197 | + | requests fanned out across isolates cannot each spend the cap. At the cap it | |
| 198 | + | answers `402` with `run_cap_reached`. At most 16 answers are counted in | |
| 199 | + | flight at once, which bounds the overshoot. A session without a cap gets a | |
| 200 | + | $100 backstop. A run token reaches only the message and model routes, and is | |
| 201 | + | closed when the run ends. | |
| 202 | + | ||
| 203 | + | ## Manual steps in Cloudflare's dashboard | |
| 204 | + | ||
| 205 | + | These are the owner's, once per account. None can be set from code. | |
| 206 | + | ||
| 207 | + | - [ ] **Billing → Billable Usage notifications** (Manage Account → Billing → | |
| 208 | + | Notifications, or Notifications → Add → "Usage Based Billing"). Add | |
| 209 | + | one per product g1t uses: Workers (requests and CPU), Workers KV, D1, | |
| 210 | + | Queues, Durable Objects, R2, Workers Logs, Containers, Browser | |
| 211 | + | Rendering, Workers AI and Vectorize. Set each threshold near this | |
| 212 | + | doc's hourly threshold times about 24 times 3 (a day at a third of the | |
| 213 | + | watch's line), and send them to hey@flagon.io. | |
| 214 | + | - [ ] **Budget alerts** (Manage Account → Billing → Budget alerts, where the | |
| 215 | + | account has them): one for the whole account's monthly usage, at the | |
| 216 | + | month's expected bill and at twice it. | |
| 217 | + | - [ ] **Notifications → Destinations**: add hey@flagon.io (and a webhook, | |
| 218 | + | if one is set up for paging) so the alerts above reach someone. | |
| 219 | + | - [ ] **Account API token for billing**: check `CLOUDFLARE_BILLING_TOKEN` | |
| 220 | + | (or `CLOUDFLARE_USAGE_TOKEN`) has Account Analytics Read. sudo's | |
| 221 | + | Platform pause says when the watch cannot read. | |
| 222 | + | - [ ] Where to see them: Notifications → History for what fired; Billing → | |
| 223 | + | Billable Usage for the month so far by product. | |
| 224 | + | ||
| 225 | + | ## Deploy order for these changes | |
| 226 | + | ||
| 227 | + | 1. Billing (migration 0051 runs first, then the Worker): the pause, the | |
| 228 | + | watch, `platform_pause`, `admin_platform_guard`, `admin_set_pause`. | |
| 229 | + | 2. Search and og, with their new `BILLING` binding; actions, context and | |
| 230 | + | runner. Before billing has `platform_pause`, they read nothing paused. | |
| 231 | + | 3. Sudo. | |
| 232 | + | 4. For the run cap: integrations (migration 0006), then models (its | |
| 233 | + | `RunSpend` Durable Object migration), then runner. Any order works; a | |
| 234 | + | session without a cap gets the $100 backstop. |
| 1 | 1 | import type { ComputeKind, Reservation } from "./compute"; | |
| 2 | + | import type { PauseLevel } from "./platform"; | |
| 2 | 3 | import type { User, Viewer } from "./identity"; | |
| 3 | 4 | import type { RepoPath } from "./repos"; | |
| 4 | 5 | import type { Result } from "./result"; | |
| ⋯ | |||
| 675 | 676 | spendCaps(): Promise<SpendCaps>; | |
| 676 | 677 | /** Lets hosted-model runs start again for the rest of today (UTC); needs a note. */ | |
| 677 | 678 | liftBreaker(note: string, by: string): Promise<Result<SpendCaps>>; | |
| 679 | + | /** Platform pauses, the last hour of platform usage, the month so far and the last day's breaches (billing's platform.rs). */ | |
| 680 | + | platformGuard(): Promise<PlatformGuard>; | |
| 681 | + | /** Pauses or resumes one level across g1t; needs a note, recorded in the audit log. */ | |
| 682 | + | setPause(level: PauseLevel, paused: boolean, note: string, by: string): Promise<Result<PlatformGuard>>; | |
| 678 | 683 | /** Approve or reject a price proposal; a rejection needs a note. An approved rise waits out the notice period. */ | |
| 679 | 684 | decideProposal(id: string, decision: "approve" | "reject", note: string, by: string): Promise<Result<PriceProposal>>; | |
| 680 | 685 | setCostSettings(settings: CostSettings, by: string): Promise<Result<CostSettings>>; | |
| ⋯ | |||
| 1677 | 1682 | caps: SpendCaps; | |
| 1678 | 1683 | }; | |
| 1679 | 1684 | ||
| 1685 | + | /** One level of the platform pause, as sudo shows it. Snake case, as billing sends it. */ | |
| 1686 | + | export type PauseState = { | |
| 1687 | + | level: PauseLevel; | |
| 1688 | + | paused: boolean; | |
| 1689 | + | note: string | null; | |
| 1690 | + | set_by: string | null; | |
| 1691 | + | set_at: string | null; | |
| 1692 | + | /** Set by billing's usage watcher, not a person. */ | |
| 1693 | + | auto: boolean; | |
| 1694 | + | }; | |
| 1695 | + | ||
| 1696 | + | /** One platform metric over an hour or the month so far. */ | |
| 1697 | + | export type PlatformMetric = { | |
| 1698 | + | metric: string; | |
| 1699 | + | title: string; | |
| 1700 | + | value: number; | |
| 1701 | + | /** Its hourly threshold (`PLATFORM_HOURLY_*`); 0: none. */ | |
| 1702 | + | threshold: number; | |
| 1703 | + | /** The script, queue, database or namespace that counted most. */ | |
| 1704 | + | top_name: string | null; | |
| 1705 | + | top_value: number | null; | |
| 1706 | + | }; | |
| 1707 | + | ||
| 1708 | + | /** A breach billing's usage watcher found. */ | |
| 1709 | + | export type PlatformBreach = { | |
| 1710 | + | id: string; | |
| 1711 | + | metric: string; | |
| 1712 | + | hour: string; | |
| 1713 | + | rule: "threshold" | "spike"; | |
| 1714 | + | value: number; | |
| 1715 | + | threshold: number; | |
| 1716 | + | severe: boolean; | |
| 1717 | + | top_name: string | null; | |
| 1718 | + | detail: string; | |
| 1719 | + | /** Levels it paused. */ | |
| 1720 | + | paused: PauseLevel[]; | |
| 1721 | + | opened_at: string; | |
| 1722 | + | emailed_at: string | null; | |
| 1723 | + | }; | |
| 1724 | + | ||
| 1725 | + | /** Billing's `admin_platform_guard`: the platform pause and usage watcher (docs/SPEND-GUARDRAILS.md). */ | |
| 1726 | + | export type PlatformGuard = { | |
| 1727 | + | levels: PauseState[]; | |
| 1728 | + | /** The last hour read, `YYYY-MM-DDTHH:00:00Z`. */ | |
| 1729 | + | hour: string | null; | |
| 1730 | + | last_hour: PlatformMetric[]; | |
| 1731 | + | month: string; | |
| 1732 | + | month_to_date: PlatformMetric[]; | |
| 1733 | + | breaches: PlatformBreach[]; | |
| 1734 | + | /** Whether billing can read Cloudflare's analytics. */ | |
| 1735 | + | can_read: boolean; | |
| 1736 | + | /** `AUTO_PAUSE`: the levels a severe breach may pause. */ | |
| 1737 | + | auto_pause: PauseLevel[]; | |
| 1738 | + | /** What the latest run could not see: each query that failed, with its error. */ | |
| 1739 | + | blind?: { key: string; dataset: string; error: string }[]; | |
| 1740 | + | /** The latest run found every dataset empty: the wrong account, or a token that cannot see it. */ | |
| 1741 | + | empty?: boolean; | |
| 1742 | + | /** The hour the latest run read. */ | |
| 1743 | + | last_run?: string | null; | |
| 1744 | + | }; | |
| 1745 | + | ||
| 1680 | 1746 | /** What g1t pays for itself, at cost, against its caps (billing's `budget`). */ | |
| 1681 | 1747 | export type SpendCaps = { | |
| 1682 | 1748 | /** Today (UTC), YYYY-MM-DD, and this month, YYYY-MM. */ | |
| 673 | 673 | costAlerts: () => call("admin_cost_alerts", {}), | |
| 674 | 674 | spendCaps: () => call("admin_spend_caps", {}), | |
| 675 | 675 | liftBreaker: (note, by) => call("admin_lift_breaker", { note, by }), | |
| 676 | + | platformGuard: () => call("admin_platform_guard", {}), | |
| 677 | + | setPause: (level, paused, note, by) => call("admin_set_pause", { level, paused, note, by }), | |
| 676 | 678 | decideProposal: (id, decision, note, by) => call("admin_decide_proposal", { id, decision, note, by }), | |
| 677 | 679 | setCostSettings: (settings, by) => call("admin_set_cost_settings", { settings, by }), | |
| 678 | 680 | setCostMapping: (mapping, by) => | |
| ⋯ | |||
| 753 | 755 | gatewayUpstream: (workspace) => call("gateway_upstream", { workspace }), | |
| 754 | 756 | gatewayProviders: (workspace) => call("gateway_providers", { workspace }), | |
| 755 | 757 | closeModelSessions: (tokenHashes) => call("close_model_sessions", { token_hashes: tokenHashes }), | |
| 758 | + | capModelSessions: (tokenHashes, capMicros) => call("cap_model_sessions", { token_hashes: tokenHashes, cap_micros: capMicros }), | |
| 756 | 759 | routes: (workspace, viewer) => call("routes", { workspace, viewer }), | |
| 757 | 760 | setRoutes: (actor, workspace, routes) => call("set_routes", { actor, workspace, routes }), | |
| 758 | 761 | }; | |
| 29 | 29 | export * from "./oauth"; | |
| 30 | 30 | export * from "./og"; | |
| 31 | 31 | export * from "./packages"; | |
| 32 | + | export * from "./platform"; | |
| 32 | 33 | export * from "./projects"; | |
| 33 | 34 | export * from "./rate-limits"; | |
| 34 | 35 | export * from "./repos"; |
| 203 | 203 | authHeader: string | null; | |
| 204 | 204 | /** For an endpoint behind an authenticated Cloudflare AI Gateway: its token. */ | |
| 205 | 205 | gatewayToken?: string | null; | |
| 206 | + | /** | |
| 207 | + | * The most the run may spend on models, in millionths of a dollar: the | |
| 208 | + | * lower of its project's cost cap and its plan's. The proxy refuses the | |
| 209 | + | * run's requests once it has spent this. Null until the sandbox sets it. | |
| 210 | + | */ | |
| 211 | + | capMicros?: number | null; | |
| 206 | 212 | }; | |
| 207 | 213 | ||
| 208 | 214 | export type ConnectInput = { | |
| ⋯ | |||
| 265 | 271 | * how many were open. | |
| 266 | 272 | */ | |
| 267 | 273 | closeModelSessions(tokenHashes: string[]): Promise<number>; | |
| 274 | + | /** | |
| 275 | + | * Sets the most the runs whose model tokens hash to these may spend on | |
| 276 | + | * models, in millionths of a dollar, which the model proxy holds them | |
| 277 | + | * to. Zero clears it. Returns how many open sessions it set. | |
| 278 | + | */ | |
| 279 | + | capModelSessions(tokenHashes: string[], capMicros: number): Promise<number>; | |
| 268 | 280 | } | |
| 269 | 281 | ||
| 270 | 282 | /** What each provider is for, as people choose between them. */ | |
| 1 | + | /** | |
| 2 | + | * g1t-wide pauses: staff (or billing's hourly usage watcher) can stop | |
| 3 | + | * whole kinds of work across the platform while unusual usage is looked | |
| 4 | + | * into. See docs/SPEND-GUARDRAILS.md and the billing service's | |
| 5 | + | * `platform.rs`. | |
| 6 | + | * | |
| 7 | + | * - `compute`: agents, sandboxes, Actions hosted jobs and builds. Billing's | |
| 8 | + | * `reserve` refuses them itself, so every `ComputeGate.admit` caller gets | |
| 9 | + | * it without asking here. | |
| 10 | + | * - `schedules`: Actions' cron-triggered runs, and the runner's sweep that | |
| 11 | + | * starts queued agents. | |
| 12 | + | * - `indexing`: context embeddings and backfills, search backfills. | |
| 13 | + | * - `renders`: social card rendering, which falls back to a static image. | |
| 14 | + | * | |
| 15 | + | * Read through billing's `platform_pause`, kept for 30 seconds in the | |
| 16 | + | * isolate: never a database read per request. When billing cannot say, | |
| 17 | + | * nothing is paused (fails open): a pause is pulled on purpose, and a | |
| 18 | + | * billing outage must not stop the platform with it. The failure is kept | |
| 19 | + | * for the same 30 seconds. | |
| 20 | + | * | |
| 21 | + | * Only type imports, so services' unit tests can load it on its own. | |
| 22 | + | */ | |
| 23 | + | import type { ServiceBinding } from "./clients"; | |
| 24 | + | ||
| 25 | + | export type PauseLevel = "compute" | "schedules" | "indexing" | "renders"; | |
| 26 | + | ||
| 27 | + | export const PAUSE_LEVELS: readonly PauseLevel[] = ["compute", "schedules", "indexing", "renders"]; | |
| 28 | + | ||
| 29 | + | /** Billing's `platform_pause`: which levels are paused now. */ | |
| 30 | + | export type PlatformPause = Record<PauseLevel, boolean>; | |
| 31 | + | ||
| 32 | + | /** How long an answer is kept in the isolate. */ | |
| 33 | + | export const PAUSE_KEPT_MS = 30_000; | |
| 34 | + | ||
| 35 | + | const NOTHING_PAUSED: PlatformPause = { compute: false, schedules: false, indexing: false, renders: false }; | |
| 36 | + | ||
| 37 | + | let kept: { value: PlatformPause; until: number } | null = null; | |
| 38 | + | ||
| 39 | + | /** Billing's answer as a pause, anything it does not say as not paused. */ | |
| 40 | + | export function readPause(answer: unknown): PlatformPause { | |
| 41 | + | const raw = answer && typeof answer === "object" ? (answer as Record<string, unknown>) : {}; | |
| 42 | + | return { | |
| 43 | + | compute: raw.compute === true, | |
| 44 | + | schedules: raw.schedules === true, | |
| 45 | + | indexing: raw.indexing === true, | |
| 46 | + | renders: raw.renders === true, | |
| 47 | + | }; | |
| 48 | + | } | |
| 49 | + | ||
| 50 | + | /** Every level, kept for 30 seconds. Nothing paused when billing cannot say. Never throws. */ | |
| 51 | + | export async function platformPause(billing: ServiceBinding | undefined, now = Date.now()): Promise<PlatformPause> { | |
| 52 | + | if (!billing) return NOTHING_PAUSED; | |
| 53 | + | if (kept && kept.until > now) return kept.value; | |
| 54 | + | let value = NOTHING_PAUSED; | |
| 55 | + | try { | |
| 56 | + | const response = await billing.fetch("https://service/rpc/platform_pause", { | |
| 57 | + | method: "POST", | |
| 58 | + | headers: { "content-type": "application/json" }, | |
| 59 | + | body: "{}", | |
| 60 | + | }); | |
| 61 | + | if (!response.ok) throw new Error(`platform_pause failed with status ${response.status}`); | |
| 62 | + | value = readPause(await response.json()); | |
| 63 | + | } catch (error) { | |
| 64 | + | console.error("platform pause unreadable, so nothing is paused", String(error)); | |
| 65 | + | } | |
| 66 | + | kept = { value, until: now + PAUSE_KEPT_MS }; | |
| 67 | + | return value; | |
| 68 | + | } | |
| 69 | + | ||
| 70 | + | /** Whether `level` is paused across g1t. */ | |
| 71 | + | export async function platformPaused(billing: ServiceBinding | undefined, level: PauseLevel): Promise<boolean> { | |
| 72 | + | return (await platformPause(billing))[level]; | |
| 73 | + | } | |
| 74 | + | ||
| 75 | + | /** For tests: forget the kept answer. */ | |
| 76 | + | export function forgetPlatformPause(): void { | |
| 77 | + | kept = null; | |
| 78 | + | } |
| 1 | + | #!/usr/bin/env node | |
| 2 | + | // What is the platform using on Cloudflare? Asks Cloudflare's GraphQL | |
| 3 | + | // Analytics API, for the month so far (UTC, from the 1st) and the last 24 | |
| 4 | + | // hours, how much each Worker, D1 database, queue, Durable Object namespace, | |
| 5 | + | // KV namespace and Artifacts namespace did, and prints the top of each with | |
| 6 | + | // totals. Workers Logs are added when the account's schema has a dataset for | |
| 7 | + | // them. | |
| 8 | + | // | |
| 9 | + | // Read-only: one GraphQL query per dataset and window, all at once, and a few | |
| 10 | + | // REST listings to put names on database, queue and namespace ids. A dataset | |
| 11 | + | // or field Cloudflare refuses is a note in the report, never the end of it. | |
| 12 | + | // | |
| 13 | + | // CLOUDFLARE_API_TOKEN=<token with Account Analytics: Read> \ | |
| 14 | + | // node scripts/ops/platform-usage.mjs [--top 10] [--json] [--account <id>] | |
| 15 | + | // | |
| 16 | + | // --account (or CLOUDFLARE_ACCOUNT_ID) reads another account than g1t's. | |
| 17 | + | // --json prints one object with snake_case keys and every row, not just the top. | |
| 18 | + | // Names come from the D1, Queues, KV and Durable Objects listings when the | |
| 19 | + | // token may read them (D1: Read, Queues: Read, Workers KV Storage: Read, | |
| 20 | + | // Workers Scripts: Read); otherwise the ids are printed. | |
| 21 | + | // | |
| 22 | + | // It is also the check of billing's hourly watcher (services/billing/src/ | |
| 23 | + | // platform.rs): it asks Cloudflare billing's own queries, field for field, | |
| 24 | + | // over the last full hour, and exits 1 naming every dataset that errored, | |
| 25 | + | // fell back to fewer fields, or (all of them) answered with no rows. Run it | |
| 26 | + | // once after deploying a change to the watcher's queries. | |
| 27 | + | ||
| 28 | + | import { ACCOUNT_ID, cloudflareAuth } from "../deploy/cloudflare.mjs"; | |
| 29 | + | ||
| 30 | + | const API = "https://api.cloudflare.com/client/v4"; | |
| 31 | + | ||
| 32 | + | const HELP = `node scripts/ops/platform-usage.mjs [--top N] [--json] [--account <id>] | |
| 33 | + | ||
| 34 | + | Cloudflare usage per Worker, D1 database, queue, Durable Object namespace, | |
| 35 | + | KV namespace and Artifacts namespace, month to date (UTC) and the last 24 hours. | |
| 36 | + | ||
| 37 | + | --top N rows shown per dataset in the tables (default 10) | |
| 38 | + | --json one JSON object, snake_case keys, every row | |
| 39 | + | --account <id> the Cloudflare account (default CLOUDFLARE_ACCOUNT_ID, else g1t's) | |
| 40 | + | --help this text | |
| 41 | + | ||
| 42 | + | Needs CLOUDFLARE_API_TOKEN (Account Analytics: Read), or CLOUDFLARE_API_KEY | |
| 43 | + | with CLOUDFLARE_EMAIL. | |
| 44 | + | ||
| 45 | + | Exits 1 when any dataset errored or fell back to fewer fields, when any of | |
| 46 | + | billing's watcher queries errored, or when every dataset answered with no | |
| 47 | + | rows (the wrong account, or a token that cannot see it).`; | |
| 48 | + | ||
| 49 | + | /** | |
| 50 | + | * Billing's hourly watcher queries, as `QUERIES` in | |
| 51 | + | * services/billing/src/platform.rs has them (a test keeps the two the same): | |
| 52 | + | * the dataset, what it selects, what it groups by, and the filter field for | |
| 53 | + | * an hour. | |
| 54 | + | */ | |
| 55 | + | export const WATCHER_QUERIES = [ | |
| 56 | + | { key: "workers", dataset: "workersInvocationsAdaptive", select: "sum { requests }", dimensions: "scriptName", hourFilter: "datetime" }, | |
| 57 | + | { key: "workers_cpu", dataset: "workersInvocationsAdaptive", select: "sum { cpuTimeUs }", dimensions: "scriptName", hourFilter: "datetime" }, | |
| 58 | + | { key: "d1", dataset: "d1AnalyticsAdaptiveGroups", select: "sum { rowsRead rowsWritten }", dimensions: "databaseId", hourFilter: "datetimeHour" }, | |
| 59 | + | { key: "queues", dataset: "queueMessageOperationsAdaptiveGroups", select: "sum { billableOperations }", dimensions: "queueId", hourFilter: "datetime" }, | |
| 60 | + | { key: "do_invocations", dataset: "durableObjectsInvocationsAdaptiveGroups", select: "sum { requests }", dimensions: "scriptName", hourFilter: "datetime" }, | |
| 61 | + | { key: "do_periodic", dataset: "durableObjectsPeriodicGroups", select: "sum { activeTime storageWriteUnits }", dimensions: "namespaceId", hourFilter: "datetime" }, | |
| 62 | + | { key: "do_sql", dataset: "durableObjectsPeriodicGroups", select: "sum { rowsWritten }", dimensions: "namespaceId", hourFilter: "datetime" }, | |
| 63 | + | { key: "kv", dataset: "kvOperationsAdaptiveGroups", select: "sum { requests }", dimensions: "namespaceId actionType", hourFilter: "datetime" }, | |
| 64 | + | { key: "artifacts", dataset: "artifactsEventsAdaptiveGroups", select: "count", dimensions: "repositoryName", hourFilter: "datetime" }, | |
| 65 | + | ]; | |
| 66 | + | ||
| 67 | + | /** The last full hour (UTC) before `now`, as the watcher reads it. */ | |
| 68 | + | export function lastFullHour(now) { | |
| 69 | + | const until = new Date(Math.floor(now.getTime() / 3_600_000) * 3_600_000); | |
| 70 | + | const since = new Date(until.getTime() - 3_600_000); | |
| 71 | + | const label = (date) => `${date.toISOString().slice(0, 13)}:00:00Z`; | |
| 72 | + | return { since: label(since), until: label(until) }; | |
| 73 | + | } | |
| 74 | + | ||
| 75 | + | /** One watcher query over an hour, as billing sends it (its account variable named as this script names it). */ | |
| 76 | + | export function watcherQuery(query) { | |
| 77 | + | return `query ($accountTag: String!, $since: Time!, $until: Time!) { | |
| 78 | + | viewer { accounts(filter: { accountTag: $accountTag }) { | |
| 79 | + | rows: ${query.dataset}(limit: 10000, filter: { ${query.hourFilter}_geq: $since, ${query.hourFilter}_lt: $until }) { | |
| 80 | + | ${query.select} | |
| 81 | + | dimensions { ${query.dimensions} } | |
| 82 | + | } | |
| 83 | + | } } | |
| 84 | + | }`; | |
| 85 | + | } | |
| 86 | + | ||
| 87 | + | /** | |
| 88 | + | * Whether the schema checks out: every problem in the report's datasets and | |
| 89 | + | * in the watcher's queries (`watcher`: key to `{ error }` or `{ rows }`). | |
| 90 | + | * An empty list is a pass. | |
| 91 | + | */ | |
| 92 | + | export function verdict(report, watcher = {}) { | |
| 93 | + | const problems = []; | |
| 94 | + | for (const window of report.windows) { | |
| 95 | + | for (const error of window.errors ?? []) problems.push(`${window.label}: ${error}`); | |
| 96 | + | } | |
| 97 | + | let answered = 0; | |
| 98 | + | let rows = 0; | |
| 99 | + | for (const query of WATCHER_QUERIES) { | |
| 100 | + | const outcome = watcher[query.key]; | |
| 101 | + | if (!outcome) continue; | |
| 102 | + | if (outcome.error) problems.push(`watcher query ${query.key} (${query.dataset}): ${outcome.error}`); | |
| 103 | + | else { | |
| 104 | + | answered += 1; | |
| 105 | + | rows += outcome.rows; | |
| 106 | + | } | |
| 107 | + | } | |
| 108 | + | const reportRows = report.windows.flatMap((w) => w.datasets).reduce((sum, set) => sum + set.rows.length, 0); | |
| 109 | + | const reportAnswered = report.windows.flatMap((w) => w.datasets).length; | |
| 110 | + | if (answered + reportAnswered > 0 && rows + reportRows === 0) { | |
| 111 | + | problems.push("every dataset answered with no rows: likely the wrong account, or a token that cannot see its analytics"); | |
| 112 | + | } | |
| 113 | + | return problems; | |
| 114 | + | } | |
| 115 | + | ||
| 116 | + | /** | |
| 117 | + | * The datasets the report asks for. Each has one or more variants, tried in | |
| 118 | + | * order: when Cloudflare refuses a field, the next variant asks for less. | |
| 119 | + | * `names` says which REST listing puts a name on the first dimension's ids. | |
| 120 | + | */ | |
| 121 | + | export const DATASETS = [ | |
| 122 | + | { | |
| 123 | + | key: "workers", | |
| 124 | + | label: "Workers invocations, by script", | |
| 125 | + | dataset: "workersInvocationsAdaptive", | |
| 126 | + | variants: [ | |
| 127 | + | { sum: ["requests", "errors", "cpuTimeUs"], dims: ["scriptName"] }, | |
| 128 | + | { sum: ["requests", "errors"], dims: ["scriptName"] }, | |
| 129 | + | ], | |
| 130 | + | }, | |
| 131 | + | { | |
| 132 | + | key: "d1", | |
| 133 | + | label: "D1 rows, by database", | |
| 134 | + | dataset: "d1AnalyticsAdaptiveGroups", | |
| 135 | + | names: "d1", | |
| 136 | + | variants: [ | |
| 137 | + | { sum: ["rowsRead", "rowsWritten", "readQueries", "writeQueries"], dims: ["databaseId"] }, | |
| 138 | + | { sum: ["rowsRead", "rowsWritten"], dims: ["databaseId"] }, | |
| 139 | + | ], | |
| 140 | + | }, | |
| 141 | + | { | |
| 142 | + | key: "queues", | |
| 143 | + | label: "Queue operations, by queue", | |
| 144 | + | dataset: "queueMessageOperationsAdaptiveGroups", | |
| 145 | + | names: "queues", | |
| 146 | + | variants: [{ sum: ["billableOperations"], dims: ["queueId"] }], | |
| 147 | + | }, | |
| 148 | + | { | |
| 149 | + | key: "durable_objects", | |
| 150 | + | label: "Durable Object requests, by script", | |
| 151 | + | dataset: "durableObjectsInvocationsAdaptiveGroups", | |
| 152 | + | variants: [ | |
| 153 | + | { sum: ["requests", "errors"], dims: ["scriptName"] }, | |
| 154 | + | { sum: ["requests"], dims: ["scriptName"] }, | |
| 155 | + | ], | |
| 156 | + | }, | |
| 157 | + | { | |
| 158 | + | key: "durable_objects_periodic", | |
| 159 | + | label: "Durable Object time and storage, by namespace", | |
| 160 | + | dataset: "durableObjectsPeriodicGroups", | |
| 161 | + | names: "durable_objects", | |
| 162 | + | variants: [ | |
| 163 | + | { sum: ["activeTime", "cpuTime", "storageReadUnits", "storageWriteUnits"], dims: ["namespaceId"] }, | |
| 164 | + | { sum: ["activeTime", "storageWriteUnits"], dims: ["namespaceId"] }, | |
| 165 | + | // Periodic groups may filter by the minute rather than by datetime. | |
| 166 | + | { sum: ["activeTime"], dims: ["namespaceId"], time: "datetimeMinute" }, | |
| 167 | + | ], | |
| 168 | + | }, | |
| 169 | + | { | |
| 170 | + | key: "kv", | |
| 171 | + | label: "KV operations, by namespace and action", | |
| 172 | + | dataset: "kvOperationsAdaptiveGroups", | |
| 173 | + | names: "kv", | |
| 174 | + | variants: [{ sum: ["requests"], dims: ["namespaceId", "actionType"] }], | |
| 175 | + | }, | |
| 176 | + | { | |
| 177 | + | key: "artifacts", | |
| 178 | + | label: "Artifacts events, by namespace and type", | |
| 179 | + | dataset: "artifactsEventsAdaptiveGroups", | |
| 180 | + | variants: [{ count: true, sum: ["durationMs"], dims: ["repositoryNamespace", "eventType"] }], | |
| 181 | + | }, | |
| 182 | + | { | |
| 183 | + | key: "workers_logs", | |
| 184 | + | label: "Workers Logs events, by script", | |
| 185 | + | // Which dataset holds Workers Logs is read from the schema (logsDataset). | |
| 186 | + | dataset: null, | |
| 187 | + | optional: true, | |
| 188 | + | variants: [ | |
| 189 | + | { count: true, dims: ["scriptName"] }, | |
| 190 | + | { count: true, dims: [] }, | |
| 191 | + | ], | |
| 192 | + | }, | |
| 193 | + | ]; | |
| 194 | + | ||
| 195 | + | /** The two windows: the UTC month so far, and the last 24 hours. */ | |
| 196 | + | export function windows(now = new Date()) { | |
| 197 | + | const end = new Date(now.getTime()); | |
| 198 | + | const monthStart = new Date(Date.UTC(end.getUTCFullYear(), end.getUTCMonth(), 1)); | |
| 199 | + | return [ | |
| 200 | + | { key: "month_to_date", label: "Month to date (UTC)", start: monthStart.toISOString(), end: end.toISOString() }, | |
| 201 | + | { key: "last_24h", label: "Last 24 hours", start: new Date(end.getTime() - 24 * 3600 * 1000).toISOString(), end: end.toISOString() }, | |
| 202 | + | ]; | |
| 203 | + | } | |
| 204 | + | ||
| 205 | + | /** Which account field holds Workers Logs, from the account type's field names; null when none does. */ | |
| 206 | + | export function logsDataset(fieldNames) { | |
| 207 | + | const known = ["workersObservabilityEventsAdaptiveGroups", "workersLogsEventsAdaptiveGroups", "workersLogsAdaptiveGroups"]; | |
| 208 | + | return known.find((name) => fieldNames.includes(name)) ?? fieldNames.find((name) => /^workers.*(logs|observability).*groups$/i.test(name)) ?? null; | |
| 209 | + | } | |
| 210 | + | ||
| 211 | + | /** One dataset's query, for one variant. The rows come back under `rows`. */ | |
| 212 | + | export function buildQuery(dataset, variant) { | |
| 213 | + | const time = variant.time ?? "datetime"; | |
| 214 | + | const fields = [ | |
| 215 | + | variant.count ? "count" : "", | |
| 216 | + | variant.sum?.length ? `sum { ${variant.sum.join(" ")} }` : "", | |
| 217 | + | variant.dims.length ? `dimensions { ${variant.dims.join(" ")} }` : "", | |
| 218 | + | ].filter(Boolean); | |
| 219 | + | return `query PlatformUsage($accountTag: String!, $start: Time!, $end: Time!) { | |
| 220 | + | viewer { | |
| 221 | + | accounts(filter: { accountTag: $accountTag }) { | |
| 222 | + | rows: ${dataset}(limit: 10000, filter: { ${time}_geq: $start, ${time}_leq: $end }) { | |
| 223 | + | ${fields.join("\n ")} | |
| 224 | + | } | |
| 225 | + | } | |
| 226 | + | } | |
| 227 | + | }`; | |
| 228 | + | } | |
| 229 | + | ||
| 230 | + | /** camelCase to snake_case, for the JSON report's keys. */ | |
| 231 | + | export const snake = (name) => name.replace(/[A-Z]/g, (letter) => `_${letter.toLowerCase()}`); | |
| 232 | + | ||
| 233 | + | /** The metric names a variant reports, snake_case: `count` first, then its sums. */ | |
| 234 | + | export const metricsOf = (variant) => [...(variant.count ? ["count"] : []), ...(variant.sum ?? []).map(snake)]; | |
| 235 | + | ||
| 236 | + | /** | |
| 237 | + | * A GraphQL answer to one dataset's query as rows, one per name, summed and | |
| 238 | + | * sorted by the first metric (largest first), with totals per metric. Throws | |
| 239 | + | * with Cloudflare's message when the answer has errors or no such dataset. | |
| 240 | + | */ | |
| 241 | + | export function parseGroups(body, variant, names = {}) { | |
| 242 | + | if (body?.errors?.length) throw new Error(body.errors.map((error) => error.message).join("; ").slice(0, 400)); | |
| 243 | + | const groups = body?.data?.viewer?.accounts?.[0]?.rows; | |
| 244 | + | if (!Array.isArray(groups)) throw new Error("no rows in the answer"); | |
| 245 | + | const metrics = metricsOf(variant); | |
| 246 | + | const byName = new Map(); | |
| 247 | + | for (const group of groups) { | |
| 248 | + | const parts = variant.dims.map((dim, at) => { | |
| 249 | + | const value = group.dimensions?.[dim]; | |
| 250 | + | const text = value == null || value === "" ? "(none)" : String(value); | |
| 251 | + | return at === 0 ? (names[text] ?? text) : text; | |
| 252 | + | }); | |
| 253 | + | const name = parts.join(" / ") || "(all)"; | |
| 254 | + | const row = byName.get(name) ?? { name, ...Object.fromEntries(metrics.map((metric) => [metric, 0])) }; | |
| 255 | + | if (variant.count) row.count += Number(group.count ?? 0); | |
| 256 | + | for (const field of variant.sum ?? []) row[snake(field)] += Number(group.sum?.[field] ?? 0); | |
| 257 | + | byName.set(name, row); | |
| 258 | + | } | |
| 259 | + | const rows = sortRows([...byName.values()], metrics[0]); | |
| 260 | + | return { metrics, rows, totals: totalsOf(rows, metrics) }; | |
| 261 | + | } | |
| 262 | + | ||
| 263 | + | /** Rows largest first by one metric, ties by name. */ | |
| 264 | + | export function sortRows(rows, metric) { | |
| 265 | + | return [...rows].sort((a, b) => (b[metric] ?? 0) - (a[metric] ?? 0) || a.name.localeCompare(b.name)); | |
| 266 | + | } | |
| 267 | + | ||
| 268 | + | /** Each metric summed over all rows. */ | |
| 269 | + | export function totalsOf(rows, metrics) { | |
| 270 | + | return Object.fromEntries(metrics.map((metric) => [metric, rows.reduce((total, row) => total + (row[metric] ?? 0), 0)])); | |
| 271 | + | } | |
| 272 | + | ||
| 273 | + | /** | |
| 274 | + | * The report from every dataset's outcome in every window. An outcome is | |
| 275 | + | * `{ body, variant }` (a GraphQL answer) or `{ error }` or `{ skipped }`; | |
| 276 | + | * whatever cannot be read becomes a note and the other datasets still count. | |
| 277 | + | */ | |
| 278 | + | export function assemble({ account, now, windowList, outcomes, names = {} }) { | |
| 279 | + | return { | |
| 280 | + | account, | |
| 281 | + | generated_at: now.toISOString(), | |
| 282 | + | windows: windowList.map((window) => { | |
| 283 | + | const notes = []; | |
| 284 | + | const errors = []; | |
| 285 | + | const datasets = []; | |
| 286 | + | for (const spec of DATASETS) { | |
| 287 | + | const outcome = outcomes[window.key]?.[spec.key]; | |
| 288 | + | if (!outcome) continue; | |
| 289 | + | if (outcome.skipped) { | |
| 290 | + | notes.push(`${spec.label}: ${outcome.skipped}`); | |
| 291 | + | continue; | |
| 292 | + | } | |
| 293 | + | if (outcome.fellBack) { | |
| 294 | + | const note = `${spec.label} (${outcome.dataset ?? spec.dataset}): fell back to fewer fields: ${outcome.fellBack}`; | |
| 295 | + | notes.push(note); | |
| 296 | + | errors.push(note); | |
| 297 | + | } | |
| 298 | + | try { | |
| 299 | + | if (outcome.error) throw outcome.error; | |
| 300 | + | const parsed = parseGroups(outcome.body, outcome.variant, names[spec.names] ?? {}); | |
| 301 | + | datasets.push({ key: spec.key, label: spec.label, dataset: outcome.dataset ?? spec.dataset, ...parsed }); | |
| 302 | + | } catch (error) { | |
| 303 | + | const note = `${spec.label} (${outcome.dataset ?? spec.dataset ?? "no dataset"}): ${String(error.message ?? error).split("\n")[0]}`; | |
| 304 | + | notes.push(note); | |
| 305 | + | errors.push(note); | |
| 306 | + | } | |
| 307 | + | } | |
| 308 | + | return { key: window.key, label: window.label, start: window.start, end: window.end, datasets, notes, errors }; | |
| 309 | + | }), | |
| 310 | + | }; | |
| 311 | + | } | |
| 312 | + | ||
| 313 | + | const number = (value) => (Number.isInteger(value) ? value.toLocaleString("en-US") : value.toLocaleString("en-US", { maximumFractionDigits: 2 })); | |
| 314 | + | ||
| 315 | + | /** The report as plain tables: the top rows of each dataset, and a totals line. */ | |
| 316 | + | export function format(report, top = 10) { | |
| 317 | + | const lines = [`Cloudflare usage for account ${report.account}, ${report.generated_at}`]; | |
| 318 | + | for (const window of report.windows) { | |
| 319 | + | lines.push("", `== ${window.label}: ${window.start} to ${window.end}`); | |
| 320 | + | for (const set of window.datasets) { | |
| 321 | + | lines.push("", `${set.label} (${set.dataset})`); | |
| 322 | + | const shown = set.rows.slice(0, top); | |
| 323 | + | const cells = [["name", ...set.metrics], ...shown.map((row) => [row.name, ...set.metrics.map((m) => number(row[m]))]), ["total", ...set.metrics.map((m) => number(set.totals[m]))]]; | |
| 324 | + | const widths = cells[0].map((_, at) => Math.max(...cells.map((row) => String(row[at]).length))); | |
| 325 | + | const render = (row) => " " + row.map((cell, at) => (at === 0 ? String(cell).padEnd(widths[at]) : String(cell).padStart(widths[at]))).join(" "); | |
| 326 | + | lines.push(render(cells[0])); | |
| 327 | + | if (!shown.length) lines.push(" (nothing in this window)"); | |
| 328 | + | for (const row of cells.slice(1, -1)) lines.push(render(row)); | |
| 329 | + | if (set.rows.length > shown.length) lines.push(` ... and ${set.rows.length - shown.length} more`); | |
| 330 | + | lines.push(render(cells.at(-1))); | |
| 331 | + | } | |
| 332 | + | if (window.notes.length) { | |
| 333 | + | lines.push("", "Notes:"); | |
| 334 | + | for (const note of window.notes) lines.push(` ${note}`); | |
| 335 | + | } | |
| 336 | + | } | |
| 337 | + | return lines.join("\n"); | |
| 338 | + | } | |
| 339 | + | ||
| 340 | + | /** The command line: flags and the account. */ | |
| 341 | + | export function parseArgs(argv, env = process.env) { | |
| 342 | + | const option = (name) => { | |
| 343 | + | const at = argv.indexOf(name); | |
| 344 | + | return at >= 0 && argv[at + 1] && !argv[at + 1].startsWith("--") ? argv[at + 1] : null; | |
| 345 | + | }; | |
| 346 | + | return { | |
| 347 | + | help: argv.includes("--help") || argv.includes("-h"), | |
| 348 | + | json: argv.includes("--json"), | |
| 349 | + | top: Math.max(1, Number(option("--top")) || 10), | |
| 350 | + | account: option("--account") || env.CLOUDFLARE_ACCOUNT_ID || ACCOUNT_ID, | |
| 351 | + | }; | |
| 352 | + | } | |
| 353 | + | ||
| 354 | + | async function graphql(auth, account, query, variables = {}) { | |
| 355 | + | const response = await fetch(`${API}/graphql`, { | |
| 356 | + | method: "POST", | |
| 357 | + | headers: { ...auth, "content-type": "application/json", "user-agent": "g1t-ops" }, | |
| 358 | + | body: JSON.stringify({ query, variables: { accountTag: account, ...variables } }), | |
| 359 | + | }); | |
| 360 | + | const body = await response.json().catch(() => ({ errors: [{ message: `HTTP ${response.status}, not JSON` }] })); | |
| 361 | + | if (!response.ok && !body.errors?.length) body.errors = [{ message: `HTTP ${response.status}` }]; | |
| 362 | + | return body; | |
| 363 | + | } | |
| 364 | + | ||
| 365 | + | /** The account type's field names, to skip datasets the schema lacks; null when it cannot be read. */ | |
| 366 | + | async function schemaFields(auth, account) { | |
| 367 | + | try { | |
| 368 | + | const body = await graphql(auth, account, `{ __type(name: "account") { fields { name } } }`); | |
| 369 | + | const fields = body.data?.__type?.fields; | |
| 370 | + | return Array.isArray(fields) ? fields.map((field) => field.name) : null; | |
| 371 | + | } catch { | |
| 372 | + | return null; | |
| 373 | + | } | |
| 374 | + | } | |
| 375 | + | ||
| 376 | + | /** One dataset in one window: each variant in turn until one is answered. */ | |
| 377 | + | async function ask(auth, account, dataset, spec, window) { | |
| 378 | + | let last = null; | |
| 379 | + | let first = null; | |
| 380 | + | for (const variant of spec.variants) { | |
| 381 | + | try { | |
| 382 | + | const body = await graphql(auth, account, buildQuery(dataset, variant), { start: window.start, end: window.end }); | |
| 383 | + | if (!body.errors?.length) return first ? { body, variant, dataset, fellBack: first } : { body, variant, dataset }; | |
| 384 | + | first ??= body.errors.map((e) => e.message).join("; "); | |
| 385 | + | last = { body, variant, dataset }; | |
| 386 | + | } catch (error) { | |
| 387 | + | last = { error, dataset }; | |
| 388 | + | } | |
| 389 | + | } | |
| 390 | + | return last; | |
| 391 | + | } | |
| 392 | + | ||
| 393 | + | /** A REST listing as id to name; empty when the token may not read it. */ | |
| 394 | + | async function listing(auth, path, id, name) { | |
| 395 | + | const out = {}; | |
| 396 | + | try { | |
| 397 | + | for (let page = 1; page <= 10; page++) { | |
| 398 | + | const response = await fetch(`${API}${path}${path.includes("?") ? "&" : "?"}per_page=100&page=${page}`, { headers: { ...auth, "user-agent": "g1t-ops" } }); | |
| 399 | + | const body = await response.json(); | |
| 400 | + | if (!response.ok || !Array.isArray(body.result)) break; | |
| 401 | + | for (const item of body.result) if (item[id]) out[item[id]] = item[name] ?? item[id]; | |
| 402 | + | if (body.result.length < 100) break; | |
| 403 | + | } | |
| 404 | + | } catch { | |
| 405 | + | // Ids stand in for names. | |
| 406 | + | } | |
| 407 | + | return out; | |
| 408 | + | } | |
| 409 | + | ||
| 410 | + | async function main() { | |
| 411 | + | const options = parseArgs(process.argv.slice(2)); | |
| 412 | + | if (options.help) { | |
| 413 | + | console.log(HELP); | |
| 414 | + | return 0; | |
| 415 | + | } | |
| 416 | + | const auth = cloudflareAuth(); | |
| 417 | + | if (!auth) { | |
| 418 | + | console.error(`Set CLOUDFLARE_API_TOKEN to a token with Account Analytics: Read on account ${options.account}, or CLOUDFLARE_API_KEY and CLOUDFLARE_EMAIL.`); | |
| 419 | + | return 2; | |
| 420 | + | } | |
| 421 | + | const now = new Date(); | |
| 422 | + | const windowList = windows(now); | |
| 423 | + | const account = options.account; | |
| 424 | + | const base = `/accounts/${account}`; | |
| 425 | + | const [fields, d1, queues, kv, durable] = await Promise.all([ | |
| 426 | + | schemaFields(auth, account), | |
| 427 | + | listing(auth, `${base}/d1/database`, "uuid", "name"), | |
| 428 | + | listing(auth, `${base}/queues`, "queue_id", "queue_name"), | |
| 429 | + | listing(auth, `${base}/storage/kv/namespaces`, "id", "title"), | |
| 430 | + | listing(auth, `${base}/workers/durable_objects/namespaces`, "id", "name"), | |
| 431 | + | ]); | |
| 432 | + | const names = { d1, queues, kv, durable_objects: durable }; | |
| 433 | + | const outcomes = {}; | |
| 434 | + | const jobs = []; | |
| 435 | + | for (const window of windowList) { | |
| 436 | + | outcomes[window.key] = {}; | |
| 437 | + | for (const spec of DATASETS) { | |
| 438 | + | const dataset = spec.dataset ?? (fields ? logsDataset(fields) : null); | |
| 439 | + | if (!dataset) { | |
| 440 | + | if (!spec.optional) outcomes[window.key][spec.key] = { skipped: "no dataset" }; | |
| 441 | + | continue; | |
| 442 | + | } | |
| 443 | + | if (fields && !fields.includes(dataset)) { | |
| 444 | + | if (!spec.optional) outcomes[window.key][spec.key] = { skipped: `${dataset} is not in this account's schema` }; | |
| 445 | + | continue; | |
| 446 | + | } | |
| 447 | + | jobs.push(ask(auth, account, dataset, spec, window).then((outcome) => (outcomes[window.key][spec.key] = outcome))); | |
| 448 | + | } | |
| 449 | + | } | |
| 450 | + | // Billing's own watcher queries, over the last full hour. | |
| 451 | + | const hour = lastFullHour(now); | |
| 452 | + | const watcher = {}; | |
| 453 | + | jobs.push( | |
| 454 | + | ...WATCHER_QUERIES.map(async (query) => { | |
| 455 | + | try { | |
| 456 | + | const body = await graphql(auth, account, watcherQuery(query), { since: hour.since, until: hour.until }); | |
| 457 | + | watcher[query.key] = body.errors?.length | |
| 458 | + | ? { error: body.errors.map((e) => e.message).join("; ") } | |
| 459 | + | : { rows: body.data?.viewer?.accounts?.[0]?.rows?.length ?? 0 }; | |
| 460 | + | } catch (error) { | |
| 461 | + | watcher[query.key] = { error: String(error.message ?? error) }; | |
| 462 | + | } | |
| 463 | + | }), | |
| 464 | + | ); | |
| 465 | + | await Promise.all(jobs); | |
| 466 | + | const report = assemble({ account, now, windowList, outcomes, names }); | |
| 467 | + | const problems = verdict(report, watcher); | |
| 468 | + | if (options.json) { | |
| 469 | + | console.log(JSON.stringify({ ...report, watcher_hour: hour, watcher, problems }, null, 2)); | |
| 470 | + | } else { | |
| 471 | + | console.log(format(report, options.top)); | |
| 472 | + | console.log("", `== Billing's watcher queries, ${hour.since} to ${hour.until}`); | |
| 473 | + | for (const query of WATCHER_QUERIES) { | |
| 474 | + | const outcome = watcher[query.key]; | |
| 475 | + | console.log(` ${query.key.padEnd(15)} ${outcome?.error ? `ERROR ${outcome.error}` : `${outcome?.rows ?? 0} rows`}`); | |
| 476 | + | } | |
| 477 | + | } | |
| 478 | + | if (problems.length) { | |
| 479 | + | console.error("", `platform-usage: ${problems.length} problem(s); the watcher may be blind on these:`); | |
| 480 | + | for (const problem of problems) console.error(` ${problem}`); | |
| 481 | + | return 1; | |
| 482 | + | } | |
| 483 | + | return 0; | |
| 484 | + | } | |
| 485 | + | ||
| 486 | + | if (process.argv[1]?.replaceAll("\\", "/").endsWith("scripts/ops/platform-usage.mjs")) { | |
| 487 | + | main().then( | |
| 488 | + | (code) => process.exit(code), | |
| 489 | + | (error) => { | |
| 490 | + | console.error(`platform-usage: ${error.message}`); | |
| 491 | + | process.exit(1); | |
| 492 | + | }, | |
| 493 | + | ); | |
| 494 | + | } |
| 1 | + | // The platform usage report, from GraphQL answers as Cloudflare gives them, | |
| 2 | + | // without the network. | |
| 3 | + | ||
| 4 | + | import assert from "node:assert/strict"; | |
| 5 | + | import { test } from "node:test"; | |
| 6 | + | ||
| 7 | + | import { DATASETS, WATCHER_QUERIES, assemble, buildQuery, format, lastFullHour, logsDataset, parseArgs, parseGroups, snake, verdict, watcherQuery, windows } from "./platform-usage.mjs"; | |
| 8 | + | ||
| 9 | + | const NOW = new Date("2026-10-08T15:30:00.000Z"); | |
| 10 | + | const answer = (rows) => ({ data: { viewer: { accounts: [{ rows }] } } }); | |
| 11 | + | const variantOf = (key, at = 0) => DATASETS.find((spec) => spec.key === key).variants[at]; | |
| 12 | + | ||
| 13 | + | test("the windows are the UTC month so far and the last 24 hours", () => { | |
| 14 | + | const [month, day] = windows(NOW); | |
| 15 | + | assert.deepEqual(month, { key: "month_to_date", label: "Month to date (UTC)", start: "2026-10-01T00:00:00.000Z", end: "2026-10-08T15:30:00.000Z" }); | |
| 16 | + | assert.equal(day.key, "last_24h"); | |
| 17 | + | assert.equal(day.start, "2026-10-07T15:30:00.000Z"); | |
| 18 | + | assert.equal(day.end, NOW.toISOString()); | |
| 19 | + | // Just after midnight on the 1st, the month has only begun. | |
| 20 | + | assert.equal(windows(new Date("2026-11-01T00:05:00Z"))[0].start, "2026-11-01T00:00:00.000Z"); | |
| 21 | + | }); | |
| 22 | + | ||
| 23 | + | test("a query asks for the variant's sums and dimensions under one alias", () => { | |
| 24 | + | const query = buildQuery("kvOperationsAdaptiveGroups", variantOf("kv")); | |
| 25 | + | assert.match(query, /rows: kvOperationsAdaptiveGroups\(limit: 10000, filter: \{ datetime_geq: \$start, datetime_leq: \$end \}\)/); | |
| 26 | + | assert.match(query, /sum \{ requests \}/); | |
| 27 | + | assert.match(query, /dimensions \{ namespaceId actionType \}/); | |
| 28 | + | assert.match(query, /accounts\(filter: \{ accountTag: \$accountTag \}\)/); | |
| 29 | + | const artifacts = buildQuery("artifactsEventsAdaptiveGroups", variantOf("artifacts")); | |
| 30 | + | assert.match(artifacts, /\bcount\b/); | |
| 31 | + | const minute = buildQuery("durableObjectsPeriodicGroups", variantOf("durable_objects_periodic", 2)); | |
| 32 | + | assert.match(minute, /datetimeMinute_geq: \$start, datetimeMinute_leq: \$end/); | |
| 33 | + | }); | |
| 34 | + | ||
| 35 | + | test("Workers Logs are read from whichever dataset the schema has, or skipped", () => { | |
| 36 | + | assert.equal(logsDataset(["workersInvocationsAdaptive", "workersObservabilityEventsAdaptiveGroups"]), "workersObservabilityEventsAdaptiveGroups"); | |
| 37 | + | assert.equal(logsDataset(["workersLogsSomethingGroups"]), "workersLogsSomethingGroups"); | |
| 38 | + | assert.equal(logsDataset(["workersInvocationsAdaptive", "kvOperationsAdaptiveGroups"]), null); | |
| 39 | + | }); | |
| 40 | + | ||
| 41 | + | test("groups are summed per name, named from the listing, sorted largest first, with totals", () => { | |
| 42 | + | const body = answer([ | |
| 43 | + | { sum: { rowsRead: 10, rowsWritten: 1, readQueries: 2, writeQueries: 1 }, dimensions: { databaseId: "aaa" } }, | |
| 44 | + | { sum: { rowsRead: 500, rowsWritten: 20, readQueries: 9, writeQueries: 3 }, dimensions: { databaseId: "bbb" } }, | |
| 45 | + | { sum: { rowsRead: 5, rowsWritten: 0, readQueries: 1, writeQueries: 0 }, dimensions: { databaseId: "aaa" } }, | |
| 46 | + | ]); | |
| 47 | + | const parsed = parseGroups(body, variantOf("d1"), { bbb: "g1t-repos" }); | |
| 48 | + | assert.deepEqual(parsed.metrics, ["rows_read", "rows_written", "read_queries", "write_queries"]); | |
| 49 | + | assert.deepEqual(parsed.rows.map((row) => [row.name, row.rows_read]), [["g1t-repos", 500], ["aaa", 15]]); | |
| 50 | + | assert.deepEqual(parsed.totals, { rows_read: 515, rows_written: 21, read_queries: 12, write_queries: 4 }); | |
| 51 | + | const kv = parseGroups( | |
| 52 | + | answer([ | |
| 53 | + | { sum: { requests: 3 }, dimensions: { namespaceId: "n1", actionType: "write" } }, | |
| 54 | + | { sum: { requests: 40 }, dimensions: { namespaceId: "n1", actionType: "read" } }, | |
| 55 | + | ]), | |
| 56 | + | variantOf("kv"), | |
| 57 | + | { n1: "SESSIONS" }, | |
| 58 | + | ); | |
| 59 | + | assert.deepEqual(kv.rows.map((row) => row.name), ["SESSIONS / read", "SESSIONS / write"]); | |
| 60 | + | assert.throws(() => parseGroups({ errors: [{ message: "unknown field cpuTimeUs" }] }, variantOf("workers")), /cpuTimeUs/); | |
| 61 | + | assert.equal(snake("billableOperations"), "billable_operations"); | |
| 62 | + | }); | |
| 63 | + | ||
| 64 | + | test("a dataset that errors or is missing is a note, and the others still report", () => { | |
| 65 | + | const windowList = windows(NOW); | |
| 66 | + | const outcomes = { | |
| 67 | + | month_to_date: { | |
| 68 | + | workers: { body: answer([{ sum: { requests: 7, errors: 0, cpuTimeUs: 1200 }, dimensions: { scriptName: "web" } }]), variant: variantOf("workers"), dataset: "workersInvocationsAdaptive" }, | |
| 69 | + | d1: { body: { errors: [{ message: "unknown field \"readQueries\"" }] }, variant: variantOf("d1"), dataset: "d1AnalyticsAdaptiveGroups" }, | |
| 70 | + | queues: { skipped: "queueMessageOperationsAdaptiveGroups is not in this account's schema" }, | |
| 71 | + | kv: { error: new Error("fetch failed") }, | |
| 72 | + | artifacts: { body: answer([{ count: 4, sum: { durationMs: 80 }, dimensions: { repositoryNamespace: "g1t", eventType: "pull" } }]), variant: variantOf("artifacts"), dataset: "artifactsEventsAdaptiveGroups" }, | |
| 73 | + | }, | |
| 74 | + | last_24h: {}, | |
| 75 | + | }; | |
| 76 | + | const report = assemble({ account: "acct", now: NOW, windowList, outcomes }); | |
| 77 | + | const month = report.windows[0]; | |
| 78 | + | assert.deepEqual(month.datasets.map((set) => set.key), ["workers", "artifacts"]); | |
| 79 | + | assert.equal(month.notes.length, 3); | |
| 80 | + | assert.match(month.notes.join("\n"), /D1 rows.*readQueries/); | |
| 81 | + | assert.match(month.notes.join("\n"), /not in this account's schema/); | |
| 82 | + | assert.match(month.notes.join("\n"), /fetch failed/); | |
| 83 | + | assert.deepEqual(report.windows[1].datasets, []); | |
| 84 | + | const text = format(report, 10); | |
| 85 | + | assert.match(text, /Workers invocations, by script/); | |
| 86 | + | assert.match(text, /web\s+7\s+0\s+1,200/); | |
| 87 | + | assert.match(text, /Notes:/); | |
| 88 | + | }); | |
| 89 | + | ||
| 90 | + | test("the JSON report is one snake_case object with every row", () => { | |
| 91 | + | const windowList = windows(NOW); | |
| 92 | + | const rows = Array.from({ length: 15 }, (_, at) => ({ sum: { requests: at + 1 }, dimensions: { scriptName: `s${at}` } })); | |
| 93 | + | const outcomes = { month_to_date: { durable_objects: { body: answer(rows), variant: variantOf("durable_objects", 1), dataset: "durableObjectsInvocationsAdaptiveGroups" } }, last_24h: {} }; | |
| 94 | + | const report = JSON.parse(JSON.stringify(assemble({ account: "acct", now: NOW, windowList, outcomes }))); | |
| 95 | + | assert.deepEqual(Object.keys(report), ["account", "generated_at", "windows"]); | |
| 96 | + | assert.deepEqual(Object.keys(report.windows[0]), ["key", "label", "start", "end", "datasets", "notes", "errors"]); | |
| 97 | + | const set = report.windows[0].datasets[0]; | |
| 98 | + | assert.deepEqual(Object.keys(set), ["key", "label", "dataset", "metrics", "rows", "totals"]); | |
| 99 | + | assert.equal(set.rows.length, 15); | |
| 100 | + | assert.equal(set.rows[0].name, "s14"); | |
| 101 | + | assert.equal(set.totals.requests, 120); | |
| 102 | + | const keys = JSON.stringify(report).match(/"([^"]+)":/g).map((key) => key.slice(1, -2)); | |
| 103 | + | for (const key of keys) assert.match(key, /^[a-z0-9_]+$/, key); | |
| 104 | + | // The tables show only the top. | |
| 105 | + | assert.match(format(assemble({ account: "acct", now: NOW, windowList, outcomes }), 10), /\.\.\. and 5 more/); | |
| 106 | + | }); | |
| 107 | + | ||
| 108 | + | test("the command line reads --json, --top and the account", () => { | |
| 109 | + | assert.deepEqual(parseArgs(["--json", "--top", "3", "--account", "abc"], {}), { help: false, json: true, top: 3, account: "abc" }); | |
| 110 | + | assert.equal(parseArgs([], { CLOUDFLARE_ACCOUNT_ID: "env" }).account, "env"); | |
| 111 | + | assert.equal(parseArgs(["--help"], {}).help, true); | |
| 112 | + | assert.equal(parseArgs([], {}).top, 10); | |
| 113 | + | }); | |
| 114 | + | ||
| 115 | + | test("errors, fallbacks, failed watcher queries and an all-empty account fail the check", () => { | |
| 116 | + | const windowList = windows(NOW); | |
| 117 | + | const ok = { | |
| 118 | + | month_to_date: { workers: { body: answer([{ sum: { requests: 7, errors: 0, cpuTimeUs: 1 }, dimensions: { scriptName: "web" } }]), variant: variantOf("workers"), dataset: "workersInvocationsAdaptive" } }, | |
| 119 | + | last_24h: {}, | |
| 120 | + | }; | |
| 121 | + | assert.deepEqual(verdict(assemble({ account: "acct", now: NOW, windowList, outcomes: ok }), { workers: { rows: 3 } }), []); | |
| 122 | + | const broken = { | |
| 123 | + | month_to_date: { | |
| 124 | + | workers: { ...ok.month_to_date.workers, fellBack: 'unknown field "cpuTimeUs"' }, | |
| 125 | + | kv: { error: new Error("fetch failed") }, | |
| 126 | + | }, | |
| 127 | + | last_24h: {}, | |
| 128 | + | }; | |
| 129 | + | const problems = verdict(assemble({ account: "acct", now: NOW, windowList, outcomes: broken }), { do_sql: { error: 'unknown field "rowsWritten"' }, d1: { rows: 2 } }); | |
| 130 | + | assert.equal(problems.length, 3); | |
| 131 | + | assert.match(problems.join("\n"), /fell back to fewer fields: unknown field "cpuTimeUs"/); | |
| 132 | + | assert.match(problems.join("\n"), /fetch failed/); | |
| 133 | + | assert.match(problems.join("\n"), /watcher query do_sql \(durableObjectsPeriodicGroups\): unknown field "rowsWritten"/); | |
| 134 | + | const empty = { month_to_date: { workers: { body: answer([]), variant: variantOf("workers"), dataset: "workersInvocationsAdaptive" } }, last_24h: {} }; | |
| 135 | + | assert.match(verdict(assemble({ account: "acct", now: NOW, windowList, outcomes: empty }), { kv: { rows: 0 } }).join("\n"), /every dataset answered with no rows/); | |
| 136 | + | }); | |
| 137 | + | ||
| 138 | + | test("the watcher queries are billing's, field for field, over the last full hour", async () => { | |
| 139 | + | const { readFile } = await import("node:fs/promises"); | |
| 140 | + | const rust = await readFile(new URL("../../services/billing/src/platform.rs", import.meta.url), "utf8"); | |
| 141 | + | const theirs = [...rust.matchAll(/Query \{ key: "([^"]+)", dataset: "([^"]+)", select: "([^"]+)", dimensions: "([^"]+)", hour_filter: "([^"]+)" \}/g)].map( | |
| 142 | + | ([, key, dataset, select, dimensions, hourFilter]) => ({ key, dataset, select, dimensions, hourFilter }), | |
| 143 | + | ); | |
| 144 | + | assert.deepEqual(WATCHER_QUERIES, theirs); | |
| 145 | + | assert.deepEqual(lastFullHour(NOW), { since: "2026-10-08T14:00:00Z", until: "2026-10-08T15:00:00Z" }); | |
| 146 | + | const d1 = watcherQuery(WATCHER_QUERIES.find((q) => q.key === "d1")); | |
| 147 | + | assert.match(d1, /rows: d1AnalyticsAdaptiveGroups\(limit: 10000, filter: \{ datetimeHour_geq: \$since, datetimeHour_lt: \$until \}\)/); | |
| 148 | + | assert.match(d1, /\$since: Time!/); | |
| 149 | + | }); |
| 2291 | 2291 | ||
| 2292 | 2292 | pub async fn on_minute(&self, now_ms: u64) -> Result<()> { | |
| 2293 | 2293 | let minute = now_ms / 60_000 * 60_000; | |
| 2294 | − | if let Err(error) = self.run_schedules(minute).await { | |
| 2294 | + | // Staff (or billing's usage watcher) paused scheduled runs across | |
| 2295 | + | // g1t: this minute's schedules are skipped, not queued for later. | |
| 2296 | + | // Kept 30 seconds in the isolate (g1t_kit::pause). | |
| 2297 | + | if g1t_kit::pause::paused(&self.billing, g1t_contracts::billing::PauseLevel::Schedules).await { | |
| 2298 | + | worker::console_log!("actions: schedules are paused across g1t; skipped this minute's"); | |
| 2299 | + | } else if let Err(error) = self.run_schedules(minute).await { | |
| 2295 | 2300 | worker::console_error!("actions: schedules failed: {error}"); | |
| 2296 | 2301 | } | |
| 2297 | 2302 | // Jobs whose sandbox went quiet or ran past their time. |
| 1 | + | -- Platform spend guardrails (src/platform.rs, docs/SPEND-GUARDRAILS.md). | |
| 2 | + | -- | |
| 3 | + | -- platform_pause: g1t-wide pauses, one row per level (compute, schedules, | |
| 4 | + | -- indexing, renders), set by staff in sudo or by the hourly usage watcher | |
| 5 | + | -- on a severe breach. No row, or paused = 0: running. | |
| 6 | + | CREATE TABLE IF NOT EXISTS platform_pause ( | |
| 7 | + | level TEXT PRIMARY KEY, | |
| 8 | + | paused INTEGER NOT NULL DEFAULT 0, | |
| 9 | + | note TEXT, | |
| 10 | + | set_by TEXT, | |
| 11 | + | set_at TEXT, | |
| 12 | + | -- 1 when the usage watcher set it, not a person. | |
| 13 | + | auto INTEGER NOT NULL DEFAULT 0 | |
| 14 | + | ); | |
| 15 | + | ||
| 16 | + | -- platform_usage: what Cloudflare counted each hour (UTC) for each metric | |
| 17 | + | -- (workers_requests, d1_rows_read, kv_lists, …), with the script, queue, | |
| 18 | + | -- database or namespace that counted most. The spike rule reads the last | |
| 19 | + | -- week of it. | |
| 20 | + | CREATE TABLE IF NOT EXISTS platform_usage ( | |
| 21 | + | hour TEXT NOT NULL, | |
| 22 | + | metric TEXT NOT NULL, | |
| 23 | + | value REAL NOT NULL DEFAULT 0, | |
| 24 | + | top_name TEXT, | |
| 25 | + | top_value REAL, | |
| 26 | + | read_at TEXT NOT NULL, | |
| 27 | + | PRIMARY KEY (hour, metric) | |
| 28 | + | ); | |
| 29 | + | CREATE INDEX IF NOT EXISTS platform_usage_by_metric ON platform_usage (metric, hour); | |
| 30 | + | ||
| 31 | + | -- platform_usage_month: the month so far for each metric, as last read. | |
| 32 | + | CREATE TABLE IF NOT EXISTS platform_usage_month ( | |
| 33 | + | month TEXT NOT NULL, | |
| 34 | + | metric TEXT NOT NULL, | |
| 35 | + | value REAL NOT NULL DEFAULT 0, | |
| 36 | + | top_name TEXT, | |
| 37 | + | top_value REAL, | |
| 38 | + | read_at TEXT NOT NULL, | |
| 39 | + | PRIMARY KEY (month, metric) | |
| 40 | + | ); | |
| 41 | + | ||
| 42 | + | -- platform_alerts: each breach the watcher found: over its hourly | |
| 43 | + | -- threshold, or a spike over the week's usual hour. Emailed at most once | |
| 44 | + | -- per metric every 6 hours. | |
| 45 | + | CREATE TABLE IF NOT EXISTS platform_alerts ( | |
| 46 | + | id TEXT PRIMARY KEY, | |
| 47 | + | metric TEXT NOT NULL, | |
| 48 | + | hour TEXT NOT NULL, | |
| 49 | + | rule TEXT NOT NULL, | |
| 50 | + | value REAL NOT NULL, | |
| 51 | + | threshold REAL NOT NULL, | |
| 52 | + | severe INTEGER NOT NULL DEFAULT 0, | |
| 53 | + | top_name TEXT, | |
| 54 | + | top_value REAL, | |
| 55 | + | detail TEXT NOT NULL, | |
| 56 | + | -- The levels it paused, comma-separated; NULL when none. | |
| 57 | + | paused TEXT, | |
| 58 | + | opened_at TEXT NOT NULL, | |
| 59 | + | emailed_at TEXT | |
| 60 | + | ); | |
| 61 | + | CREATE INDEX IF NOT EXISTS platform_alerts_by_metric ON platform_alerts (metric, opened_at); | |
| 62 | + | ||
| 63 | + | -- platform_watch_runs: what each hourly run could not see. failed is a | |
| 64 | + | -- JSON object of query key to its error ({} when every query answered); | |
| 65 | + | -- empty is 1 when every dataset that answered had no rows, the month so | |
| 66 | + | -- far included (the wrong account, or a token that cannot see it). Sudo | |
| 67 | + | -- shows the latest; three runs in a row email staff. Kept 14 days. | |
| 68 | + | CREATE TABLE IF NOT EXISTS platform_watch_runs ( | |
| 69 | + | hour TEXT PRIMARY KEY, | |
| 70 | + | read_at TEXT NOT NULL, | |
| 71 | + | failed TEXT NOT NULL DEFAULT '{}', | |
| 72 | + | empty INTEGER NOT NULL DEFAULT 0 | |
| 73 | + | ); | |
| 74 | + | ||
| 75 | + | -- platform_watch_alerts: when staff were last emailed that a query (or | |
| 76 | + | -- all_empty) stayed blind; at most once a day per key. | |
| 77 | + | CREATE TABLE IF NOT EXISTS platform_watch_alerts ( | |
| 78 | + | key TEXT PRIMARY KEY, | |
| 79 | + | emailed_at TEXT NOT NULL | |
| 80 | + | ); |
| 532 | 532 | let now = now_ms(); | |
| 533 | 533 | let expires_at = rfc3339(now + RESERVATION_HOURS * 60 * 60 * 1000); | |
| 534 | 534 | let repo = format!("{}/{}", a.repo.namespace, a.repo.name).to_lowercase(); | |
| 535 | + | // A platform pause holds for everyone, a g1t that does not charge | |
| 536 | + | // included (platform.rs): kept 30 seconds in the isolate. | |
| 537 | + | if let Some(why) = self.platform_refuses(a.kind).await { | |
| 538 | + | return Ok(Outcome::fail(FailureCode::Paused, why)); | |
| 539 | + | } | |
| 535 | 540 | // A g1t that does not charge holds nothing. | |
| 536 | 541 | if self.stripe.is_none() { | |
| 537 | 542 | return Ok(Outcome::Ok(Reservation { id: new_id("rsv", now), paid_by: PaidBy::OnDemand, held_micros: 0, expires_at })); |
| 34 | 34 | ||
| 35 | 35 | /// The cron that also checks costs against Cloudflare's bill. | |
| 36 | 36 | pub(crate) const DAILY: &str = "17 4 * * *"; | |
| 37 | + | /// The quarter-hourly tick: settling, and once an hour the platform watch. | |
| 38 | + | pub(crate) const QUARTER_HOURLY: &str = "*/15 * * * *"; | |
| 37 | 39 | ||
| 38 | 40 | /// A run is settled once its logs have had time to land. | |
| 39 | 41 | const SETTLE_AFTER_MS: u64 = 5 * 60 * 1000; |
| 26 | 26 | mod compute; | |
| 27 | 27 | mod costs; | |
| 28 | 28 | mod margin; | |
| 29 | + | mod platform; | |
| 29 | 30 | mod pricing; | |
| 30 | 31 | mod report; | |
| 31 | 32 | mod details; | |
| ⋯ | |||
| 1226 | 1227 | worker::console_error!("watching g1t's own spend failed: {error}"); | |
| 1227 | 1228 | } | |
| 1228 | 1229 | let keeper = keeper::Keeper::from_env(&env); | |
| 1230 | + | // Once an hour, at the quarter past: what Cloudflare counted for the | |
| 1231 | + | // whole platform in the hour before, against its thresholds | |
| 1232 | + | // (platform.rs). | |
| 1233 | + | if event.cron() == keeper::QUARTER_HOURLY && platform::hourly_due(now_ms()) { | |
| 1234 | + | match billing.watch_platform(&keeper).await { | |
| 1235 | + | Ok((read, breached)) => worker::console_log!("platform watch: {read} metrics, {breached} breaches"), | |
| 1236 | + | Err(error) => worker::console_error!("watching platform usage failed: {error}"), | |
| 1237 | + | } | |
| 1238 | + | } | |
| 1229 | 1239 | let daily = event.cron() == keeper::DAILY; | |
| 1230 | 1240 | if !daily { | |
| 1231 | 1241 | tick(&billing, &env, &keeper).await; | |
| ⋯ | |||
| 1432 | 1442 | "admin_cost_alerts" => reply(&billing.admin_cost_alerts(args(body)?).await?), | |
| 1433 | 1443 | "admin_spend_caps" => reply(&billing.spend_caps().await?), | |
| 1434 | 1444 | "admin_lift_breaker" => reply(&billing.admin_lift_breaker(args(body)?).await?), | |
| 1445 | + | // Platform pauses (src/platform.rs): read by every service that | |
| 1446 | + | // honours one, kept 30 seconds in each isolate. | |
| 1447 | + | "platform_pause" => reply(&billing.pause_now().await), | |
| 1448 | + | "admin_platform_guard" => reply(&billing.platform_guard(&keeper::Keeper::from_env(&env)).await?), | |
| 1449 | + | "admin_set_pause" => reply(&billing.admin_set_pause(args(body)?, &keeper::Keeper::from_env(&env)).await?), | |
| 1435 | 1450 | "admin_decide_proposal" => reply(&billing.admin_decide_proposal(args(body)?).await?), | |
| 1436 | 1451 | "admin_set_cost_settings" => reply(&billing.admin_set_cost_settings(args(body)?).await?), | |
| 1437 | 1452 | "admin_set_cost_mapping" => reply(&billing.admin_set_cost_mapping(args(body)?).await?), | |
| ⋯ | |||
| 1542 | 1557 | include_str!("../migrations/0046_reset_costs.sql"), | |
| 1543 | 1558 | include_str!("../migrations/0047_gateway_formats.sql"), | |
| 1544 | 1559 | include_str!("../migrations/0048_model_catalogue.sql"), | |
| 1560 | + | include_str!("../migrations/0049_superseded_proposals.sql"), | |
| 1561 | + | include_str!("../migrations/0050_ledger_usage_by_time.sql"), | |
| 1562 | + | include_str!("../migrations/0051_platform_guardrails.sql"), | |
| 1545 | 1563 | ]; | |
| 1546 | 1564 | ||
| 1547 | 1565 | /// The columns of `table` after the migrations: each with whether an | |
| 1 | + | //! Platform spend guardrails: what Cloudflare counts for all of g1t, read | |
| 2 | + | //! every hour, and a g1t-wide pause to stop it. See | |
| 3 | + | //! docs/SPEND-GUARDRAILS.md. | |
| 4 | + | //! | |
| 5 | + | //! The caps in `budget` hold what g1t pays for agents and sandboxes; this | |
| 6 | + | //! is about the platform underneath, which no workspace's limit covers and | |
| 7 | + | //! Cloudflare never caps: Workers requests and CPU, D1 rows, Queue | |
| 8 | + | //! operations, Durable Objects, KV and Artifacts. A loop in any of them is | |
| 9 | + | //! otherwise found on the bill. | |
| 10 | + | //! | |
| 11 | + | //! - **The pause** (`platform_pause`): four independent levels, set by | |
| 12 | + | //! staff in sudo (Costs & margin, Platform pause) or by the watcher on a | |
| 13 | + | //! severe breach. `compute` refuses every reservation through `reserve` | |
| 14 | + | //! but embeddings; `indexing` refuses embeddings, and context's and | |
| 15 | + | //! search's backfills check it themselves; `schedules` stops Actions' | |
| 16 | + | //! cron runs and the runner's sweep; `renders` makes social cards a | |
| 17 | + | //! static image. Callers keep the answer 30 seconds in their isolate | |
| 18 | + | //! (`g1t_kit::pause`, `@g1t/contracts` `platformPaused`), and so does | |
| 19 | + | //! billing (`pause_now`): never a database read per request. When the | |
| 20 | + | //! flag cannot be read, nothing is paused: a pause is pulled on purpose, | |
| 21 | + | //! and a billing outage must not stop the platform with it. | |
| 22 | + | //! - **The watcher** (`watch_platform`): at the quarter past each hour, the | |
| 23 | + | //! hour before and the month so far, from Cloudflare's GraphQL Analytics | |
| 24 | + | //! with the costs reconciliation's token (`CLOUDFLARE_BILLING_TOKEN`, else | |
| 25 | + | //! `CLOUDFLARE_USAGE_TOKEN`; Account Analytics Read). Each hour goes in | |
| 26 | + | //! `platform_usage` with the script, queue, database or namespace that | |
| 27 | + | //! counted most. A metric over its hourly threshold (`PLATFORM_HOURLY_*`), | |
| 28 | + | //! or over `PLATFORM_SPIKE_FACTOR` times the last week's median hour (and | |
| 29 | + | //! at least `PLATFORM_SPIKE_FLOOR_PERCENT` of its threshold), is a breach: | |
| 30 | + | //! staff are emailed (`COSTS_ALERT_EMAIL`, once per metric every 6 | |
| 31 | + | //! hours) and sudo shows it. Over `PLATFORM_SEVERE_FACTOR` times its | |
| 32 | + | //! threshold, the levels that metric feeds are paused, of those | |
| 33 | + | //! `AUTO_PAUSE` allows (`schedules,indexing` unless set). | |
| 34 | + | //! | |
| 35 | + | //! A dataset GraphQL refuses (a field renamed, a dataset the token cannot | |
| 36 | + | //! read) is skipped and the others still count, but never quietly: each | |
| 37 | + | //! run records what it could not see (`platform_watch_runs`), sudo shows | |
| 38 | + | //! it, and a query failing `BLIND_RUNS` runs in a row, or every dataset | |
| 39 | + | //! answering empty that long (the wrong account or token), emails staff | |
| 40 | + | //! once a day (`watch_health`). | |
| 41 | + | ||
| 42 | + | use std::cell::RefCell; | |
| 43 | + | use std::collections::BTreeMap; | |
| 44 | + | ||
| 45 | + | use futures_util::future::join_all; | |
| 46 | + | use g1t_contracts::billing::{ | |
| 47 | + | AdminSetPauseArgs, BlindQuery, PauseLevel, PauseState, PlatformBreach, PlatformGuard, PlatformMetric, PlatformPause, | |
| 48 | + | }; | |
| 49 | + | use g1t_contracts::time::rfc3339; | |
| 50 | + | use g1t_contracts::{FailureCode, Outcome, new_id}; | |
| 51 | + | use g1t_kit::now_ms; | |
| 52 | + | use serde::Deserialize; | |
| 53 | + | use serde_json::{Value, json}; | |
| 54 | + | use worker::{Env, Result}; | |
| 55 | + | ||
| 56 | + | use crate::Billing; | |
| 57 | + | use crate::keeper::Keeper; | |
| 58 | + | ||
| 59 | + | const HOUR_MS: u64 = 60 * 60 * 1000; | |
| 60 | + | /// How long a pause read is kept in the isolate. | |
| 61 | + | const KEEP_MS: u64 = 30_000; | |
| 62 | + | /// A metric's breach is emailed at most this often. | |
| 63 | + | const ALERT_EVERY_MS: u64 = 6 * HOUR_MS; | |
| 64 | + | /// The spike rule needs this many hours of history first. | |
| 65 | + | const SPIKE_MIN_HOURS: usize = 24; | |
| 66 | + | ||
| 67 | + | thread_local! { | |
| 68 | + | static KEPT: RefCell<Option<(PlatformPause, u64)>> = const { RefCell::new(None) }; | |
| 69 | + | } | |
| 70 | + | ||
| 71 | + | /// One metric the watcher reads, with its threshold's variable and the | |
| 72 | + | /// pause levels it feeds. | |
| 73 | + | pub(crate) struct Metric { | |
| 74 | + | pub key: &'static str, | |
| 75 | + | pub title: &'static str, | |
| 76 | + | /// `PLATFORM_HOURLY_<VAR>`. | |
| 77 | + | pub var: &'static str, | |
| 78 | + | /// The hourly threshold when the variable is not set: about a dollar | |
| 79 | + | /// to a few dollars an hour at Cloudflare's list prices, far above a | |
| 80 | + | /// small alpha's normal hour. | |
| 81 | + | pub default: f64, | |
| 82 | + | /// What kind of thing `top_name` is, for the alert. | |
| 83 | + | pub of: &'static str, | |
| 84 | + | /// The levels a severe breach may pause (of those `AUTO_PAUSE` allows). | |
| 85 | + | pub levels: &'static [PauseLevel], | |
| 86 | + | } | |
| 87 | + | ||
| 88 | + | use PauseLevel::{Compute, Indexing, Renders, Schedules}; | |
| 89 | + | ||
| 90 | + | pub(crate) const METRICS: &[Metric] = &[ | |
| 91 | + | Metric { key: "workers_requests", title: "Workers requests", var: "WORKERS_REQUESTS", default: 20_000_000.0, of: "script", levels: &[Schedules, Indexing, Renders] }, | |
| 92 | + | Metric { key: "workers_cpu_ms", title: "Workers CPU (ms)", var: "WORKERS_CPU_MS", default: 100_000_000.0, of: "script", levels: &[Schedules, Indexing, Renders] }, | |
| 93 | + | Metric { key: "d1_rows_read", title: "D1 rows read", var: "D1_ROWS_READ", default: 2_000_000_000.0, of: "D1 database id", levels: &[Schedules, Indexing] }, | |
| 94 | + | Metric { key: "d1_rows_written", title: "D1 rows written", var: "D1_ROWS_WRITTEN", default: 5_000_000.0, of: "D1 database id", levels: &[Schedules, Indexing] }, | |
| 95 | + | Metric { key: "queue_operations", title: "Queue operations", var: "QUEUE_OPERATIONS", default: 5_000_000.0, of: "queue id", levels: &[Schedules, Indexing] }, | |
| 96 | + | Metric { key: "do_requests", title: "Durable Object requests", var: "DO_REQUESTS", default: 20_000_000.0, of: "script", levels: &[Compute, Schedules] }, | |
| 97 | + | Metric { key: "do_rows_written", title: "Durable Object rows written", var: "DO_ROWS_WRITTEN", default: 5_000_000.0, of: "Durable Object namespace id", levels: &[Compute, Schedules] }, | |
| 98 | + | Metric { key: "do_storage_write_units", title: "Durable Object storage writes", var: "DO_STORAGE_WRITE_UNITS", default: 5_000_000.0, of: "Durable Object namespace id", levels: &[Compute, Schedules] }, | |
| 99 | + | Metric { key: "do_active_seconds", title: "Durable Object active seconds", var: "DO_ACTIVE_SECONDS", default: 3_000_000.0, of: "Durable Object namespace id", levels: &[Compute, Schedules] }, | |
| 100 | + | Metric { key: "kv_reads", title: "KV reads", var: "KV_READS", default: 10_000_000.0, of: "KV namespace id", levels: &[Indexing, Renders] }, | |
| 101 | + | Metric { key: "kv_writes", title: "KV writes", var: "KV_WRITES", default: 200_000.0, of: "KV namespace id", levels: &[Indexing, Renders] }, | |
| 102 | + | Metric { key: "kv_deletes", title: "KV deletes", var: "KV_DELETES", default: 200_000.0, of: "KV namespace id", levels: &[Indexing, Renders] }, | |
| 103 | + | Metric { key: "kv_lists", title: "KV lists", var: "KV_LISTS", default: 200_000.0, of: "KV namespace id", levels: &[Indexing, Renders] }, | |
| 104 | + | Metric { key: "artifacts_events", title: "Artifacts events", var: "ARTIFACTS_EVENTS", default: 1_000_000.0, of: "repository", levels: &[Compute, Schedules] }, | |
| 105 | + | ]; | |
| 106 | + | ||
| 107 | + | pub(crate) fn metric(key: &str) -> Option<&'static Metric> { | |
| 108 | + | METRICS.iter().find(|m| m.key == key) | |
| 109 | + | } | |
| 110 | + | ||
| 111 | + | /// The watcher's numbers, from the billing service's variables. | |
| 112 | + | #[derive(Clone, Debug)] | |
| 113 | + | pub(crate) struct Watch { | |
| 114 | + | /// Each metric's hourly threshold; zero turns its threshold rule off. | |
| 115 | + | pub thresholds: BTreeMap<&'static str, f64>, | |
| 116 | + | /// `PLATFORM_SPIKE_FACTOR`: an hour above this many times the week's | |
| 117 | + | /// median hour is a spike. Zero: no spike rule. | |
| 118 | + | pub spike_factor: f64, | |
| 119 | + | /// `PLATFORM_SPIKE_FLOOR_PERCENT`: a spike must also be at least this | |
| 120 | + | /// share of the metric's threshold, so a quiet metric doubling is not one. | |
| 121 | + | pub spike_floor_percent: f64, | |
| 122 | + | /// `PLATFORM_SEVERE_FACTOR`: this many times the threshold pauses. | |
| 123 | + | pub severe_factor: f64, | |
| 124 | + | /// `AUTO_PAUSE`: the levels a severe breach may pause. | |
| 125 | + | pub auto_pause: Vec<PauseLevel>, | |
| 126 | + | } | |
| 127 | + | ||
| 128 | + | impl Watch { | |
| 129 | + | pub(crate) fn from_env(env: &Env) -> Self { | |
| 130 | + | let var = |name: &str| env.var(name).ok().map(|v| v.to_string()); | |
| 131 | + | Watch::from_vars(var) | |
| 132 | + | } | |
| 133 | + | ||
| 134 | + | pub(crate) fn from_vars(var: impl Fn(&str) -> Option<String>) -> Self { | |
| 135 | + | let number = |name: &str, default: f64| { | |
| 136 | + | var(name).and_then(|v| v.trim().replace('_', "").parse::<f64>().ok()).filter(|n| n.is_finite() && *n >= 0.0).unwrap_or(default) | |
| 137 | + | }; | |
| 138 | + | let thresholds = METRICS.iter().map(|m| (m.key, number(&format!("PLATFORM_HOURLY_{}", m.var), m.default))).collect(); | |
| 139 | + | let auto_pause = match var("AUTO_PAUSE") { | |
| 140 | + | Some(list) => list.split(',').filter_map(PauseLevel::parse).collect(), | |
| 141 | + | None => vec![Schedules, Indexing], | |
| 142 | + | }; | |
| 143 | + | Watch { | |
| 144 | + | thresholds, | |
| 145 | + | spike_factor: number("PLATFORM_SPIKE_FACTOR", 10.0), | |
| 146 | + | spike_floor_percent: number("PLATFORM_SPIKE_FLOOR_PERCENT", 10.0), | |
| 147 | + | severe_factor: number("PLATFORM_SEVERE_FACTOR", 5.0).max(1.0), | |
| 148 | + | auto_pause, | |
| 149 | + | } | |
| 150 | + | } | |
| 151 | + | ||
| 152 | + | pub(crate) fn threshold(&self, key: &str) -> f64 { | |
| 153 | + | self.thresholds.get(key).copied().unwrap_or(0.0) | |
| 154 | + | } | |
| 155 | + | } | |
| 156 | + | ||
| 157 | + | // ---- The pause ------------------------------------------------------------ | |
| 158 | + | ||
| 159 | + | /// What a refused reservation is told while `level` is paused. | |
| 160 | + | pub(crate) fn pause_refusal(level: PauseLevel) -> String { | |
| 161 | + | let what = match level { | |
| 162 | + | Compute => "new agent runs, checks, workflow jobs and builds", | |
| 163 | + | Indexing => "new indexing and semantic search embeddings", | |
| 164 | + | Schedules => "scheduled workflow runs and queued agents", | |
| 165 | + | Renders => "new social card images", | |
| 166 | + | }; | |
| 167 | + | format!("g1t has paused {what} across the platform while it looks into unusual usage. Work already running finishes; try again later.") | |
| 168 | + | } | |
| 169 | + | ||
| 170 | + | /// The level a reservation of `kind` waits on. | |
| 171 | + | pub(crate) fn level_for(kind: g1t_contracts::billing::ComputeKind) -> PauseLevel { | |
| 172 | + | if kind == g1t_contracts::billing::ComputeKind::Embedding { Indexing } else { Compute } | |
| 173 | + | } | |
| 174 | + | ||
| 175 | + | #[derive(Deserialize)] | |
| 176 | + | struct PauseRow { | |
| 177 | + | level: String, | |
| 178 | + | paused: i64, | |
| 179 | + | note: Option<String>, | |
| 180 | + | set_by: Option<String>, | |
| 181 | + | set_at: Option<String>, | |
| 182 | + | auto: i64, | |
| 183 | + | } | |
| 184 | + | ||
| 185 | + | #[derive(Deserialize)] | |
| 186 | + | struct UsageRow { | |
| 187 | + | metric: String, | |
| 188 | + | value: f64, | |
| 189 | + | top_name: Option<String>, | |
| 190 | + | top_value: Option<f64>, | |
| 191 | + | } | |
| 192 | + | ||
| 193 | + | #[derive(Deserialize)] | |
| 194 | + | struct AlertRow { | |
| 195 | + | id: String, | |
| 196 | + | metric: String, | |
| 197 | + | hour: String, | |
| 198 | + | rule: String, | |
| 199 | + | value: f64, | |
| 200 | + | threshold: f64, | |
| 201 | + | severe: i64, | |
| 202 | + | top_name: Option<String>, | |
| 203 | + | detail: String, | |
| 204 | + | paused: Option<String>, | |
| 205 | + | opened_at: String, | |
| 206 | + | emailed_at: Option<String>, | |
| 207 | + | } | |
| 208 | + | ||
| 209 | + | impl Billing { | |
| 210 | + | async fn pause_rows(&self) -> Result<Vec<PauseRow>> { | |
| 211 | + | self.db.prepare("SELECT level, paused, note, set_by, set_at, auto FROM platform_pause").all().await?.results::<PauseRow>() | |
| 212 | + | } | |
| 213 | + | ||
| 214 | + | /// Every level, kept in the isolate for 30 seconds. Nothing paused | |
| 215 | + | /// when the table cannot be read. | |
| 216 | + | pub(crate) async fn pause_now(&self) -> PlatformPause { | |
| 217 | + | let now = now_ms(); | |
| 218 | + | if let Some(pause) = KEPT.with(|kept| kept.borrow().filter(|(_, until)| *until > now).map(|(p, _)| p)) { | |
| 219 | + | return pause; | |
| 220 | + | } | |
| 221 | + | let pause = match self.pause_rows().await { | |
| 222 | + | Ok(rows) => { | |
| 223 | + | let mut pause = PlatformPause::default(); | |
| 224 | + | for row in rows { | |
| 225 | + | if let Some(level) = PauseLevel::parse(&row.level) { | |
| 226 | + | pause.set(level, row.paused != 0); | |
| 227 | + | } | |
| 228 | + | } | |
| 229 | + | pause | |
| 230 | + | } | |
| 231 | + | Err(error) => { | |
| 232 | + | worker::console_error!("platform pause unreadable, so nothing is paused: {error}"); | |
| 233 | + | PlatformPause::default() | |
| 234 | + | } | |
| 235 | + | }; | |
| 236 | + | KEPT.with(|kept| *kept.borrow_mut() = Some((pause, now + KEEP_MS))); | |
| 237 | + | pause | |
| 238 | + | } | |
| 239 | + | ||
| 240 | + | /// Why a reservation of `kind` is refused by a platform pause, if it is. | |
| 241 | + | pub(crate) async fn platform_refuses(&self, kind: g1t_contracts::billing::ComputeKind) -> Option<String> { | |
| 242 | + | let level = level_for(kind); | |
| 243 | + | self.pause_now().await.is(level).then(|| pause_refusal(level)) | |
| 244 | + | } | |
| 245 | + | ||
| 246 | + | /// Pauses or resumes a level, and records who and why in the audit log. | |
| 247 | + | async fn set_pause(&self, level: PauseLevel, paused: bool, note: &str, by: &str, auto: bool) -> Result<()> { | |
| 248 | + | let now = rfc3339(now_ms()); | |
| 249 | + | self.db | |
| 250 | + | .prepare( | |
| 251 | + | "INSERT INTO platform_pause (level, paused, note, set_by, set_at, auto) VALUES (?1, ?2, ?3, ?4, ?5, ?6) | |
| 252 | + | ON CONFLICT (level) DO UPDATE SET paused = ?2, note = ?3, set_by = ?4, set_at = ?5, auto = ?6", | |
| 253 | + | ) | |
| 254 | + | .bind(&[level.as_str().into(), i32::from(paused).into(), note.into(), by.into(), now.as_str().into(), i32::from(auto).into()])? | |
| 255 | + | .run() | |
| 256 | + | .await?; | |
| 257 | + | KEPT.with(|kept| *kept.borrow_mut() = None); | |
| 258 | + | let action = if paused { "platform_paused" } else { "platform_resumed" }; | |
| 259 | + | self.audit("costs", action, &format!("{}: {note}", level.as_str()), by).await | |
| 260 | + | } | |
| 261 | + | ||
| 262 | + | /// `admin_set_pause`: staff pause or resume one level, with why. | |
| 263 | + | pub(crate) async fn admin_set_pause(&self, a: AdminSetPauseArgs, keeper: &Keeper) -> Result<Outcome<PlatformGuard>> { | |
| 264 | + | let (by, note) = (a.by.trim(), a.note.trim()); | |
| 265 | + | let Some(level) = PauseLevel::parse(&a.level) else { | |
| 266 | + | return Ok(Outcome::fail(FailureCode::Invalid, "Pick compute, schedules, indexing or renders.")); | |
| 267 | + | }; | |
| 268 | + | if by.is_empty() || note.chars().count() < 5 { | |
| 269 | + | return Ok(Outcome::fail(FailureCode::Invalid, "Say who is changing it, and why, in the note.")); | |
| 270 | + | } | |
| 271 | + | let note: String = note.chars().take(500).collect(); | |
| 272 | + | self.set_pause(level, a.paused, ¬e, by, false).await?; | |
| 273 | + | Ok(Outcome::Ok(self.platform_guard(keeper).await?)) | |
| 274 | + | } | |
| 275 | + | ||
| 276 | + | /// `admin_platform_guard`: the pauses, the last hour, the month so far | |
| 277 | + | /// and the last day's breaches. | |
| 278 | + | pub(crate) async fn platform_guard(&self, keeper: &Keeper) -> Result<PlatformGuard> { | |
| 279 | + | let watch = Watch::from_env(&self.env); | |
| 280 | + | let rows = self.pause_rows().await?; | |
| 281 | + | let levels = PauseLevel::ALL | |
| 282 | + | .iter() | |
| 283 | + | .map(|level| match rows.iter().find(|r| r.level == level.as_str()) { | |
| 284 | + | Some(r) => PauseState { | |
| 285 | + | level: r.level.clone(), | |
| 286 | + | paused: r.paused != 0, | |
| 287 | + | note: r.note.clone(), | |
| 288 | + | set_by: r.set_by.clone(), | |
| 289 | + | set_at: r.set_at.clone(), | |
| 290 | + | auto: r.auto != 0, | |
| 291 | + | }, | |
| 292 | + | None => PauseState { level: level.as_str().to_owned(), ..PauseState::default() }, | |
| 293 | + | }) | |
| 294 | + | .collect(); | |
| 295 | + | #[derive(Deserialize)] | |
| 296 | + | struct Hour { | |
| 297 | + | hour: Option<String>, | |
| 298 | + | } | |
| 299 | + | let hour = self.db.prepare("SELECT MAX(hour) AS hour FROM platform_usage").first::<Hour>(None).await?.and_then(|h| h.hour); | |
| 300 | + | let to_metrics = |rows: Vec<UsageRow>| -> Vec<PlatformMetric> { | |
| 301 | + | METRICS | |
| 302 | + | .iter() | |
| 303 | + | .filter_map(|m| { | |
| 304 | + | let row = rows.iter().find(|r| r.metric == m.key)?; | |
| 305 | + | Some(PlatformMetric { | |
| 306 | + | metric: m.key.to_owned(), | |
| 307 | + | title: m.title.to_owned(), | |
| 308 | + | value: row.value, | |
| 309 | + | threshold: watch.threshold(m.key), | |
| 310 | + | top_name: row.top_name.clone(), | |
| 311 | + | top_value: row.top_value, | |
| 312 | + | }) | |
| 313 | + | }) | |
| 314 | + | .collect() | |
| 315 | + | }; | |
| 316 | + | let last_hour = match &hour { | |
| 317 | + | Some(hour) => to_metrics( | |
| 318 | + | self.db | |
| 319 | + | .prepare("SELECT metric, value, top_name, top_value FROM platform_usage WHERE hour = ?") | |
| 320 | + | .bind(&[hour.as_str().into()])? | |
| 321 | + | .all() | |
| 322 | + | .await? | |
| 323 | + | .results::<UsageRow>()?, | |
| 324 | + | ), | |
| 325 | + | None => vec![], | |
| 326 | + | }; | |
| 327 | + | let month = rfc3339(now_ms())[..7].to_owned(); | |
| 328 | + | let month_to_date = to_metrics( | |
| 329 | + | self.db | |
| 330 | + | .prepare("SELECT metric, value, top_name, top_value FROM platform_usage_month WHERE month = ?") | |
| 331 | + | .bind(&[month.as_str().into()])? | |
| 332 | + | .all() | |
| 333 | + | .await? | |
| 334 | + | .results::<UsageRow>()?, | |
| 335 | + | ); | |
| 336 | + | let breaches = self | |
| 337 | + | .db | |
| 338 | + | .prepare( | |
| 339 | + | "SELECT id, metric, hour, rule, value, threshold, severe, top_name, detail, paused, opened_at, emailed_at | |
| 340 | + | FROM platform_alerts WHERE opened_at >= ? ORDER BY opened_at DESC LIMIT 50", | |
| 341 | + | ) | |
| 342 | + | .bind(&[rfc3339(now_ms().saturating_sub(24 * HOUR_MS)).into()])? | |
| 343 | + | .all() | |
| 344 | + | .await? | |
| 345 | + | .results::<AlertRow>()? | |
| 346 | + | .into_iter() | |
| 347 | + | .map(|r| PlatformBreach { | |
| 348 | + | id: r.id, | |
| 349 | + | metric: r.metric, | |
| 350 | + | hour: r.hour, | |
| 351 | + | rule: r.rule, | |
| 352 | + | value: r.value, | |
| 353 | + | threshold: r.threshold, | |
| 354 | + | severe: r.severe != 0, | |
| 355 | + | top_name: r.top_name, | |
| 356 | + | detail: r.detail, | |
| 357 | + | paused: r.paused.map(|p| p.split(',').filter(|s| !s.is_empty()).map(str::to_owned).collect()).unwrap_or_default(), | |
| 358 | + | opened_at: r.opened_at, | |
| 359 | + | emailed_at: r.emailed_at, | |
| 360 | + | }) | |
| 361 | + | .collect(); | |
| 362 | + | // What the latest run could not see. | |
| 363 | + | let latest = self.watch_runs(1).await?.into_iter().next(); | |
| 364 | + | let blind = latest.as_ref().map_or(vec![], |(_, failed, _)| { | |
| 365 | + | failed.iter().map(|(key, error)| BlindQuery { key: key.clone(), dataset: dataset_of(key).to_owned(), error: error.clone() }).collect() | |
| 366 | + | }); | |
| 367 | + | Ok(PlatformGuard { | |
| 368 | + | blind, | |
| 369 | + | empty: latest.as_ref().is_some_and(|(_, _, empty)| *empty), | |
| 370 | + | last_run: latest.map(|(hour, _, _)| hour), | |
| 371 | + | levels, | |
| 372 | + | hour, | |
| 373 | + | last_hour, | |
| 374 | + | month, | |
| 375 | + | month_to_date, | |
| 376 | + | breaches, | |
| 377 | + | can_read: keeper.can_read_bill(), | |
| 378 | + | auto_pause: watch.auto_pause.iter().map(|l| l.as_str().to_owned()).collect(), | |
| 379 | + | }) | |
| 380 | + | } | |
| 381 | + | } | |
| 382 | + | ||
| 383 | + | // ---- Reading Cloudflare's analytics --------------------------------------- | |
| 384 | + | ||
| 385 | + | /// One GraphQL query: a dataset, what to sum, and what to group by. | |
| 386 | + | pub(crate) struct Query { | |
| 387 | + | pub key: &'static str, | |
| 388 | + | pub dataset: &'static str, | |
| 389 | + | /// What the dataset is selected with: `sum { … }`, or `count`. | |
| 390 | + | pub select: &'static str, | |
| 391 | + | /// The dimension that names what counted (and `actionType` for KV). | |
| 392 | + | pub dimensions: &'static str, | |
| 393 | + | /// The filter fields for an hour: `datetime` (Time) on most datasets, | |
| 394 | + | /// `datetimeHour` on D1's. | |
| 395 | + | pub hour_filter: &'static str, | |
| 396 | + | } | |
| 397 | + | ||
| 398 | + | /// Each metric from its own query where a field is less certain, so one | |
| 399 | + | /// GraphQL refuses does not take the others with it. Field names as | |
| 400 | + | /// Cloudflare documents them; a refused one is logged and skipped. | |
| 401 | + | pub(crate) const QUERIES: &[Query] = &[ | |
| 402 | + | Query { key: "workers", dataset: "workersInvocationsAdaptive", select: "sum { requests }", dimensions: "scriptName", hour_filter: "datetime" }, | |
| 403 | + | Query { key: "workers_cpu", dataset: "workersInvocationsAdaptive", select: "sum { cpuTimeUs }", dimensions: "scriptName", hour_filter: "datetime" }, | |
| 404 | + | Query { key: "d1", dataset: "d1AnalyticsAdaptiveGroups", select: "sum { rowsRead rowsWritten }", dimensions: "databaseId", hour_filter: "datetimeHour" }, | |
| 405 | + | Query { key: "queues", dataset: "queueMessageOperationsAdaptiveGroups", select: "sum { billableOperations }", dimensions: "queueId", hour_filter: "datetime" }, | |
| 406 | + | Query { key: "do_invocations", dataset: "durableObjectsInvocationsAdaptiveGroups", select: "sum { requests }", dimensions: "scriptName", hour_filter: "datetime" }, | |
| 407 | + | Query { key: "do_periodic", dataset: "durableObjectsPeriodicGroups", select: "sum { activeTime storageWriteUnits }", dimensions: "namespaceId", hour_filter: "datetime" }, | |
| 408 | + | Query { key: "do_sql", dataset: "durableObjectsPeriodicGroups", select: "sum { rowsWritten }", dimensions: "namespaceId", hour_filter: "datetime" }, | |
| 409 | + | Query { key: "kv", dataset: "kvOperationsAdaptiveGroups", select: "sum { requests }", dimensions: "namespaceId actionType", hour_filter: "datetime" }, | |
| 410 | + | Query { key: "artifacts", dataset: "artifactsEventsAdaptiveGroups", select: "count", dimensions: "repositoryName", hour_filter: "datetime" }, | |
| 411 | + | ]; | |
| 412 | + | ||
| 413 | + | /// A window to read: one hour (`Time` bounds) or days (`Date` bounds). | |
| 414 | + | #[derive(Clone, Debug, PartialEq, Eq)] | |
| 415 | + | pub(crate) enum Window { | |
| 416 | + | Hour { since: String, until: String }, | |
| 417 | + | Days { since: String, until: String }, | |
| 418 | + | } | |
| 419 | + | ||
| 420 | + | /// The hour before the one `now` is in, `[since, until)`, as Cloudflare's | |
| 421 | + | /// `Time` takes it. | |
| 422 | + | pub(crate) fn last_hour(now: u64) -> Window { | |
| 423 | + | let until = now / HOUR_MS * HOUR_MS; | |
| 424 | + | Window::Hour { since: hour_label(until - HOUR_MS), until: hour_label(until) } | |
| 425 | + | } | |
| 426 | + | ||
| 427 | + | /// The month so far (UTC), today included. | |
| 428 | + | pub(crate) fn month_so_far(now: u64) -> Window { | |
| 429 | + | let today = rfc3339(now)[..10].to_owned(); | |
| 430 | + | Window::Days { since: format!("{}-01", &today[..7]), until: today } | |
| 431 | + | } | |
| 432 | + | ||
| 433 | + | /// `2026-10-08T13:00:00Z`. | |
| 434 | + | pub(crate) fn hour_label(ms: u64) -> String { | |
| 435 | + | format!("{}:00:00Z", &rfc3339(ms)[..13]) | |
| 436 | + | } | |
| 437 | + | ||
| 438 | + | /// The request body for `query` over `window`, its dataset aliased `rows`. | |
| 439 | + | pub(crate) fn query_body(account: &str, query: &Query, window: &Window) -> Value { | |
| 440 | + | let (filter, kind, since, until) = match window { | |
| 441 | + | Window::Hour { since, until } => (format!("{f}_geq: $since, {f}_lt: $until", f = query.hour_filter), "Time", since, until), | |
| 442 | + | Window::Days { since, until } => ("date_geq: $since, date_leq: $until".to_owned(), "Date", since, until), | |
| 443 | + | }; | |
| 444 | + | let text = format!( | |
| 445 | + | "query ($account: String!, $since: {kind}!, $until: {kind}!) {{ | |
| 446 | + | viewer {{ accounts(filter: {{ accountTag: $account }}) {{ | |
| 447 | + | rows: {dataset}(limit: 10000, filter: {{ {filter} }}) {{ | |
| 448 | + | {select} | |
| 449 | + | dimensions {{ {dimensions} }} | |
| 450 | + | }} | |
| 451 | + | }} }} | |
| 452 | + | }}", | |
| 453 | + | dataset = query.dataset, | |
| 454 | + | select = query.select, | |
| 455 | + | dimensions = query.dimensions, | |
| 456 | + | ); | |
| 457 | + | json!({ "query": text, "variables": { "account": account, "since": since, "until": until } }) | |
| 458 | + | } | |
| 459 | + | ||
| 460 | + | /// What one query counted, per metric: the total and each name's part. | |
| 461 | + | pub(crate) type Counted = BTreeMap<&'static str, BTreeMap<String, f64>>; | |
| 462 | + | ||
| 463 | + | /// A KV operation's metric, by its `actionType`. | |
| 464 | + | fn kv_metric(action: &str) -> Option<&'static str> { | |
| 465 | + | match action.to_ascii_lowercase().as_str() { | |
| 466 | + | "read" | "get" => Some("kv_reads"), | |
| 467 | + | "write" | "put" => Some("kv_writes"), | |
| 468 | + | "delete" => Some("kv_deletes"), | |
| 469 | + | "list" => Some("kv_lists"), | |
| 470 | + | _ => None, | |
| 471 | + | } | |
| 472 | + | } | |
| 473 | + | ||
| 474 | + | /// Reads one query's answer into metrics. Errors when GraphQL does. | |
| 475 | + | pub(crate) fn counted(query: &Query, body: &Value) -> std::result::Result<Counted, String> { | |
| 476 | + | if let Some(errors) = body["errors"].as_array().filter(|e| !e.is_empty()) { | |
| 477 | + | let messages: Vec<&str> = errors.iter().filter_map(|e| e["message"].as_str()).collect(); | |
| 478 | + | return Err(format!("{} ({}): {}", query.dataset, query.key, messages.join("; "))); | |
| 479 | + | } | |
| 480 | + | let Some(groups) = body["data"]["viewer"]["accounts"][0]["rows"].as_array() else { | |
| 481 | + | return Err(format!("{} ({}): no rows in the answer", query.dataset, query.key)); | |
| 482 | + | }; | |
| 483 | + | let mut out: Counted = BTreeMap::new(); | |
| 484 | + | let mut add = |metric: &'static str, name: &str, value: f64| { | |
| 485 | + | if value.is_finite() && value > 0.0 { | |
| 486 | + | *out.entry(metric).or_default().entry(name.to_owned()).or_default() += value; | |
| 487 | + | } | |
| 488 | + | }; | |
| 489 | + | for g in groups { | |
| 490 | + | let sum = &g["sum"]; | |
| 491 | + | let number = |field: &str| sum[field].as_f64().unwrap_or(0.0); | |
| 492 | + | let dims = &g["dimensions"]; | |
| 493 | + | let name_of = |field: &str| dims[field].as_str().filter(|s| !s.is_empty()).unwrap_or("(unnamed)").to_owned(); | |
| 494 | + | match query.key { | |
| 495 | + | "workers" => add("workers_requests", &name_of("scriptName"), number("requests")), | |
| 496 | + | "workers_cpu" => add("workers_cpu_ms", &name_of("scriptName"), number("cpuTimeUs") / 1000.0), | |
| 497 | + | "d1" => { | |
| 498 | + | let name = name_of("databaseId"); | |
| 499 | + | add("d1_rows_read", &name, number("rowsRead")); | |
| 500 | + | add("d1_rows_written", &name, number("rowsWritten")); | |
| 501 | + | } | |
| 502 | + | "queues" => add("queue_operations", &name_of("queueId"), number("billableOperations")), | |
| 503 | + | "do_invocations" => add("do_requests", &name_of("scriptName"), number("requests")), | |
| 504 | + | "do_periodic" => { | |
| 505 | + | let name = name_of("namespaceId"); | |
| 506 | + | // activeTime is in microseconds. | |
| 507 | + | add("do_active_seconds", &name, number("activeTime") / 1_000_000.0); | |
| 508 | + | add("do_storage_write_units", &name, number("storageWriteUnits")); | |
| 509 | + | } | |
| 510 | + | "do_sql" => add("do_rows_written", &name_of("namespaceId"), number("rowsWritten")), | |
| 511 | + | "kv" => { | |
| 512 | + | if let Some(metric) = dims["actionType"].as_str().and_then(kv_metric) { | |
| 513 | + | add(metric, &name_of("namespaceId"), number("requests")); | |
| 514 | + | } | |
| 515 | + | } | |
| 516 | + | "artifacts" => add("artifacts_events", &name_of("repositoryName"), g["count"].as_f64().unwrap_or(0.0)), | |
| 517 | + | _ => {} | |
| 518 | + | } | |
| 519 | + | } | |
| 520 | + | Ok(out) | |
| 521 | + | } | |
| 522 | + | ||
| 523 | + | /// A metric's total, and the name that counted most. | |
| 524 | + | #[derive(Clone, Debug, PartialEq)] | |
| 525 | + | pub(crate) struct Total { | |
| 526 | + | pub value: f64, | |
| 527 | + | pub top_name: Option<String>, | |
| 528 | + | pub top_value: Option<f64>, | |
| 529 | + | } | |
| 530 | + | ||
| 531 | + | /// Every query's answers merged into one total per metric. | |
| 532 | + | pub(crate) fn totals(answers: &[Counted]) -> BTreeMap<&'static str, Total> { | |
| 533 | + | let mut merged: Counted = BTreeMap::new(); | |
| 534 | + | for answer in answers { | |
| 535 | + | for (metric, names) in answer { | |
| 536 | + | let entry = merged.entry(metric).or_default(); | |
| 537 | + | for (name, value) in names { | |
| 538 | + | *entry.entry(name.clone()).or_default() += value; | |
| 539 | + | } | |
| 540 | + | } | |
| 541 | + | } | |
| 542 | + | merged | |
| 543 | + | .into_iter() | |
| 544 | + | .map(|(metric, names)| { | |
| 545 | + | let value = names.values().sum(); | |
| 546 | + | let top = names.into_iter().max_by(|a, b| a.1.total_cmp(&b.1)); | |
| 547 | + | (metric, Total { value, top_name: top.as_ref().map(|t| t.0.clone()), top_value: top.map(|t| t.1) }) | |
| 548 | + | }) | |
| 549 | + | .collect() | |
| 550 | + | } | |
| 551 | + | ||
| 552 | + | // ---- Deciding what is a breach -------------------------------------------- | |
| 553 | + | ||
| 554 | + | /// The middle of the week's hours (the mean of the middle two when even). | |
| 555 | + | pub(crate) fn median(values: &mut [f64]) -> Option<f64> { | |
| 556 | + | if values.is_empty() { | |
| 557 | + | return None; | |
| 558 | + | } | |
| 559 | + | values.sort_by(f64::total_cmp); | |
| 560 | + | let mid = values.len() / 2; | |
| 561 | + | Some(if values.len() % 2 == 0 { (values[mid - 1] + values[mid]) / 2.0 } else { values[mid] }) | |
| 562 | + | } | |
| 563 | + | ||
| 564 | + | /// A breach of one metric, before it is recorded. | |
| 565 | + | #[derive(Clone, Debug, PartialEq)] | |
| 566 | + | pub(crate) struct Breach { | |
| 567 | + | pub metric: &'static str, | |
| 568 | + | /// `threshold` or `spike`. | |
| 569 | + | pub rule: &'static str, | |
| 570 | + | pub value: f64, | |
| 571 | + | /// What it was held to: the threshold, or the spike's line. | |
| 572 | + | pub limit: f64, | |
| 573 | + | pub severe: bool, | |
| 574 | + | } | |
| 575 | + | ||
| 576 | + | /// Whether an hour's `value` of `metric` is a breach, given the week's | |
| 577 | + | /// earlier hours (`history`). The threshold rule wins over the spike rule. | |
| 578 | + | pub(crate) fn judge(watch: &Watch, metric: &'static str, value: f64, history: &[f64]) -> Option<Breach> { | |
| 579 | + | let threshold = watch.threshold(metric); | |
| 580 | + | if threshold > 0.0 && value > threshold { | |
| 581 | + | return Some(Breach { metric, rule: "threshold", value, limit: threshold, severe: value >= threshold * watch.severe_factor }); | |
| 582 | + | } | |
| 583 | + | if watch.spike_factor <= 0.0 || history.len() < SPIKE_MIN_HOURS { | |
| 584 | + | return None; | |
| 585 | + | } | |
| 586 | + | let mut week = history.to_vec(); | |
| 587 | + | let usual = median(&mut week)?; | |
| 588 | + | let line = usual * watch.spike_factor; | |
| 589 | + | let floor = threshold * watch.spike_floor_percent / 100.0; | |
| 590 | + | // A metric with no threshold never spikes: there is no floor to hold it to. | |
| 591 | + | (threshold > 0.0 && value > line && value >= floor).then_some(Breach { metric, rule: "spike", value, limit: line, severe: false }) | |
| 592 | + | } | |
| 593 | + | ||
| 594 | + | /// The levels a severe breach of `metric` pauses: those it feeds that | |
| 595 | + | /// `AUTO_PAUSE` allows. | |
| 596 | + | pub(crate) fn levels_to_pause(watch: &Watch, breach: &Breach) -> Vec<PauseLevel> { | |
| 597 | + | if !breach.severe { | |
| 598 | + | return vec![]; | |
| 599 | + | } | |
| 600 | + | metric(breach.metric).map_or(vec![], |m| m.levels.iter().copied().filter(|l| watch.auto_pause.contains(l)).collect()) | |
| 601 | + | } | |
| 602 | + | ||
| 603 | + | /// `20,000,000`, or `1.5` for small fractions. | |
| 604 | + | pub(crate) fn amount(n: f64) -> String { | |
| 605 | + | if n.fract().abs() > 0.0 && n.abs() < 100.0 { | |
| 606 | + | return format!("{n:.1}"); | |
| 607 | + | } | |
| 608 | + | let digits = format!("{:.0}", n.abs()); | |
| 609 | + | let mut out = String::new(); | |
| 610 | + | for (i, c) in digits.chars().enumerate() { | |
| 611 | + | if i > 0 && (digits.len() - i).is_multiple_of(3) { | |
| 612 | + | out.push(','); | |
| 613 | + | } | |
| 614 | + | out.push(c); | |
| 615 | + | } | |
| 616 | + | if n < 0.0 { format!("-{out}") } else { out } | |
| 617 | + | } | |
| 618 | + | ||
| 619 | + | /// The sentence an alert and sudo carry for a breach. | |
| 620 | + | pub(crate) fn describe(breach: &Breach, hour: &str, top: Option<(&str, f64)>) -> String { | |
| 621 | + | let m = metric(breach.metric); | |
| 622 | + | let title = m.map_or(breach.metric, |m| m.title); | |
| 623 | + | let of = m.map_or("source", |m| m.of); | |
| 624 | + | let what = match breach.rule { | |
| 625 | + | "spike" => format!("more than its spike line of {} (the last week's usual hour times PLATFORM_SPIKE_FACTOR)", amount(breach.limit)), | |
| 626 | + | _ => format!( | |
| 627 | + | "over its hourly threshold of {} (PLATFORM_HOURLY_{})", | |
| 628 | + | amount(breach.limit), | |
| 629 | + | m.map_or("?", |m| m.var) | |
| 630 | + | ), | |
| 631 | + | }; | |
| 632 | + | let mut text = format!("{title}: {} in the hour from {hour}, {what}.", amount(breach.value)); | |
| 633 | + | if let Some((name, value)) = top { | |
| 634 | + | text.push_str(&format!(" Most of it from the {of} {name} ({}).", amount(value))); | |
| 635 | + | } | |
| 636 | + | text | |
| 637 | + | } | |
| 638 | + | ||
| 639 | + | /// Whether the quarter-hour tick at `now` is the hour's watch: the one at | |
| 640 | + | /// a quarter past, when the hour before is in Cloudflare's analytics. | |
| 641 | + | pub(crate) fn hourly_due(now: u64) -> bool { | |
| 642 | + | let minute = now / 60_000 % 60; | |
| 643 | + | (15..30).contains(&minute) | |
| 644 | + | } | |
| 645 | + | ||
| 646 | + | impl Billing { | |
| 647 | + | /// Each hour: the hour before and the month so far, from Cloudflare, | |
| 648 | + | /// kept and judged; staff emailed and levels paused on a breach. | |
| 649 | + | /// Returns how many metrics were read and how many breached. | |
| 650 | + | pub(crate) async fn watch_platform(&self, keeper: &Keeper) -> Result<(usize, usize)> { | |
| 651 | + | if !keeper.can_read_bill() { | |
| 652 | + | worker::console_log!("platform watch: no Cloudflare token with Account Analytics Read, so nothing is read"); | |
| 653 | + | return Ok((0, 0)); | |
| 654 | + | } | |
| 655 | + | let now = now_ms(); | |
| 656 | + | let watch = Watch::from_env(&self.env); | |
| 657 | + | let hour_window = last_hour(now); | |
| 658 | + | let month_window = month_so_far(now); | |
| 659 | + | let Window::Hour { since: hour, .. } = &hour_window else { unreachable!() }; | |
| 660 | + | let hour = hour.clone(); | |
| 661 | + | let month = rfc3339(now)[..7].to_owned(); | |
| 662 | + | let read = |window: &Window| { | |
| 663 | + | join_all(QUERIES.iter().map(|query| { | |
| 664 | + | let body = query_body(keeper.account(), query, window); | |
| 665 | + | async move { | |
| 666 | + | let answer = match keeper.graphql(body).await { | |
| 667 | + | Ok(answer) => counted(query, &answer).map(|c| (c, rows_in(&answer))), | |
| 668 | + | Err(error) => Err(format!("{} ({}): {error}", query.dataset, query.key)), | |
| 669 | + | }; | |
| 670 | + | (query.key, answer) | |
| 671 | + | } | |
| 672 | + | })) | |
| 673 | + | }; | |
| 674 | + | let (hour_answers, month_answers) = futures_util::future::join(read(&hour_window), read(&month_window)).await; | |
| 675 | + | // What the watcher could not see this run, by query: loud in sudo, | |
| 676 | + | // and emailed when it lasts (`watch_health`). | |
| 677 | + | let mut sight = Sight::default(); | |
| 678 | + | let mut keep = |answers: Vec<(&'static str, std::result::Result<(Counted, usize), String>)>| { | |
| 679 | + | let mut kept = vec![]; | |
| 680 | + | for (key, answer) in answers { | |
| 681 | + | match answer { | |
| 682 | + | Ok((counted, rows)) => { | |
| 683 | + | sight.answered += 1; | |
| 684 | + | sight.rows += rows; | |
| 685 | + | kept.push(counted); | |
| 686 | + | } | |
| 687 | + | Err(error) => { | |
| 688 | + | worker::console_error!("platform watch: skipped {error}"); | |
| 689 | + | sight.fail(key, &error); | |
| 690 | + | } | |
| 691 | + | } | |
| 692 | + | } | |
| 693 | + | kept | |
| 694 | + | }; | |
| 695 | + | let hourly = totals(&keep(hour_answers)); | |
| 696 | + | let monthly = totals(&keep(month_answers)); | |
| 697 | + | let read_at = rfc3339(now); | |
| 698 | + | if let Err(error) = self.watch_health(&hour, &read_at, &sight).await { | |
| 699 | + | worker::console_error!("platform watch: could not record what it could not see: {error}"); | |
| 700 | + | } | |
| 701 | + | let mut writes = vec![]; | |
| 702 | + | for (table, period, values) in [("platform_usage", "hour", &hourly), ("platform_usage_month", "month", &monthly)] { | |
| 703 | + | let key = if period == "hour" { hour.as_str() } else { month.as_str() }; | |
| 704 | + | for (metric, total) in values.iter() { | |
| 705 | + | writes.push( | |
| 706 | + | self.db | |
| 707 | + | .prepare(format!( | |
| 708 | + | "INSERT INTO {table} ({period}, metric, value, top_name, top_value, read_at) VALUES (?1, ?2, ?3, ?4, ?5, ?6) | |
| 709 | + | ON CONFLICT ({period}, metric) DO UPDATE SET value = ?3, top_name = ?4, top_value = ?5, read_at = ?6" | |
| 710 | + | )) | |
| 711 | + | .bind(&[ | |
| 712 | + | key.into(), | |
| 713 | + | (*metric).into(), | |
| 714 | + | total.value.into(), | |
| 715 | + | crate::optional(total.top_name.as_deref()), | |
| 716 | + | total.top_value.map_or(worker::wasm_bindgen::JsValue::NULL, Into::into), | |
| 717 | + | read_at.as_str().into(), | |
| 718 | + | ])?, | |
| 719 | + | ); | |
| 720 | + | } | |
| 721 | + | } | |
| 722 | + | if !writes.is_empty() { | |
| 723 | + | self.db.batch(writes).await?; | |
| 724 | + | } | |
| 725 | + | // The week before this hour, for the spike rule. | |
| 726 | + | #[derive(Deserialize)] | |
| 727 | + | struct Past { | |
| 728 | + | metric: String, | |
| 729 | + | value: f64, | |
| 730 | + | } | |
| 731 | + | let past = self | |
| 732 | + | .db | |
| 733 | + | .prepare("SELECT metric, value FROM platform_usage WHERE hour >= ? AND hour < ?") | |
| 734 | + | .bind(&[hour_label(now / HOUR_MS * HOUR_MS - 8 * 24 * HOUR_MS).into(), hour.as_str().into()])? | |
| 735 | + | .all() | |
| 736 | + | .await? | |
| 737 | + | .results::<Past>()?; | |
| 738 | + | let mut breaches = vec![]; | |
| 739 | + | for (metric, total) in &hourly { | |
| 740 | + | let history: Vec<f64> = past.iter().filter(|p| p.metric == *metric).map(|p| p.value).collect(); | |
| 741 | + | if let Some(breach) = judge(&watch, metric, total.value, &history) { | |
| 742 | + | breaches.push((breach, total.clone())); | |
| 743 | + | } | |
| 744 | + | } | |
| 745 | + | let found = breaches.len(); | |
| 746 | + | if found > 0 { | |
| 747 | + | self.on_breaches(&watch, &hour, breaches).await?; | |
| 748 | + | } | |
| 749 | + | Ok((hourly.len(), found)) | |
| 750 | + | } | |
| 751 | + | ||
| 752 | + | /// Records each breach, pauses what a severe one should, and emails | |
| 753 | + | /// staff about the metrics not emailed in the last 6 hours. | |
| 754 | + | async fn on_breaches(&self, watch: &Watch, hour: &str, breaches: Vec<(Breach, Total)>) -> Result<()> { | |
| 755 | + | let now = now_ms(); | |
| 756 | + | let opened_at = rfc3339(now); | |
| 757 | + | let current = self.pause_now().await; | |
| 758 | + | let mut lines = vec![]; | |
| 759 | + | let mut ids = vec![]; | |
| 760 | + | let mut paused_now: Vec<PauseLevel> = vec![]; | |
| 761 | + | for (breach, total) in breaches { | |
| 762 | + | let top = total.top_name.as_deref().zip(total.top_value); | |
| 763 | + | let mut detail = describe(&breach, hour, top); | |
| 764 | + | let pause: Vec<PauseLevel> = | |
| 765 | + | levels_to_pause(watch, &breach).into_iter().filter(|l| !current.is(*l) && !paused_now.contains(l)).collect(); | |
| 766 | + | for level in &pause { | |
| 767 | + | self.set_pause(*level, true, &format!("Automatic: {detail}"), "g1t-billing's usage watcher", true).await?; | |
| 768 | + | paused_now.push(*level); | |
| 769 | + | } | |
| 770 | + | if !pause.is_empty() { | |
| 771 | + | let names: Vec<&str> = pause.iter().map(|l| l.as_str()).collect(); | |
| 772 | + | detail.push_str(&format!(" Severe (over {}× the threshold): paused {}.", amount(watch.severe_factor), names.join(" and "))); | |
| 773 | + | } | |
| 774 | + | let id = new_id("pal", now); | |
| 775 | + | let paused_text = pause.iter().map(|l| l.as_str()).collect::<Vec<_>>().join(","); | |
| 776 | + | self.db | |
| 777 | + | .prepare( | |
| 778 | + | "INSERT INTO platform_alerts (id, metric, hour, rule, value, threshold, severe, top_name, top_value, detail, paused, opened_at) | |
| 779 | + | VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)", | |
| 780 | + | ) | |
| 781 | + | .bind(&[ | |
| 782 | + | id.as_str().into(), | |
| 783 | + | breach.metric.into(), | |
| 784 | + | hour.into(), | |
| 785 | + | breach.rule.into(), | |
| 786 | + | breach.value.into(), | |
| 787 | + | breach.limit.into(), | |
| 788 | + | i32::from(breach.severe).into(), | |
| 789 | + | crate::optional(total.top_name.as_deref()), | |
| 790 | + | total.top_value.map_or(worker::wasm_bindgen::JsValue::NULL, Into::into), | |
| 791 | + | detail.as_str().into(), | |
| 792 | + | crate::optional(Some(paused_text.as_str()).filter(|t| !t.is_empty())), | |
| 793 | + | opened_at.as_str().into(), | |
| 794 | + | ])? | |
| 795 | + | .run() | |
| 796 | + | .await?; | |
| 797 | + | // Emailed at most once per metric every 6 hours; a new pause is | |
| 798 | + | // always said. | |
| 799 | + | #[derive(Deserialize)] | |
| 800 | + | struct Last { | |
| 801 | + | at: Option<String>, | |
| 802 | + | } | |
| 803 | + | let last = self | |
| 804 | + | .db | |
| 805 | + | .prepare("SELECT MAX(emailed_at) AS at FROM platform_alerts WHERE metric = ? AND emailed_at >= ?") | |
| 806 | + | .bind(&[breach.metric.into(), rfc3339(now.saturating_sub(ALERT_EVERY_MS)).into()])? | |
| 807 | + | .first::<Last>(None) | |
| 808 | + | .await? | |
| 809 | + | .and_then(|l| l.at); | |
| 810 | + | if last.is_none() || !pause.is_empty() { | |
| 811 | + | lines.push(detail); | |
| 812 | + | ids.push(id); | |
| 813 | + | } | |
| 814 | + | } | |
| 815 | + | let alert_to = &self.caps.alert_to; | |
| 816 | + | if lines.is_empty() || alert_to.is_empty() { | |
| 817 | + | return Ok(()); | |
| 818 | + | } | |
| 819 | + | let subject = if paused_now.is_empty() { | |
| 820 | + | format!("g1t: platform usage breach in the hour from {hour}") | |
| 821 | + | } else { | |
| 822 | + | let names: Vec<&str> = paused_now.iter().map(|l| l.as_str()).collect(); | |
| 823 | + | format!("g1t: platform usage breach, {} paused", names.join(" and ")) | |
| 824 | + | }; | |
| 825 | + | lines.push( | |
| 826 | + | "Ids are Cloudflare's: `node scripts/ops/platform-usage.mjs` names them and shows the last 24 hours. To pause or resume a level: sudo, Costs & margin, Platform pause. Thresholds: PLATFORM_HOURLY_* in services/billing/wrangler.jsonc. See docs/SPEND-GUARDRAILS.md.".to_owned(), | |
| 827 | + | ); | |
| 828 | + | match crate::margin::email_staff_page(&self.env, alert_to, &subject, &lines, ("Platform pause", "https://sudo.g1t.sh/costs#platform"), "g1t-billing's usage watcher").await { | |
| 829 | + | Ok(()) => { | |
| 830 | + | let at = rfc3339(now_ms()); | |
| 831 | + | for id in ids { | |
| 832 | + | self.db.prepare("UPDATE platform_alerts SET emailed_at = ? WHERE id = ?").bind(&[at.as_str().into(), id.as_str().into()])?.run().await?; | |
| 833 | + | } | |
| 834 | + | } | |
| 835 | + | Err(error) => worker::console_error!("could not email the platform usage breach: {error}"), | |
| 836 | + | } | |
| 837 | + | Ok(()) | |
| 838 | + | } | |
| 839 | + | } | |
| 840 | + | ||
| 841 | + | // ---- What the watcher cannot see ------------------------------------------ | |
| 842 | + | ||
| 843 | + | /// Runs in a row a query must fail (or every dataset answer empty) before | |
| 844 | + | /// staff are emailed: one bad hour is Cloudflare's, three are ours. | |
| 845 | + | pub(crate) const BLIND_RUNS: usize = 3; | |
| 846 | + | /// A blind query is emailed at most this often. | |
| 847 | + | const BLIND_EVERY_MS: u64 = 24 * HOUR_MS; | |
| 848 | + | /// The key `platform_watch_alerts` uses for "every dataset answered empty". | |
| 849 | + | pub(crate) const ALL_EMPTY: &str = "all_empty"; | |
| 850 | + | ||
| 851 | + | /// What one run could and could not see. | |
| 852 | + | #[derive(Clone, Debug, Default, PartialEq)] | |
| 853 | + | pub(crate) struct Sight { | |
| 854 | + | /// Each query that failed, in either window, with its first error. | |
| 855 | + | pub failed: BTreeMap<String, String>, | |
| 856 | + | /// Queries that answered, and the rows they answered with in all. | |
| 857 | + | pub answered: usize, | |
| 858 | + | pub rows: usize, | |
| 859 | + | } | |
| 860 | + | ||
| 861 | + | impl Sight { | |
| 862 | + | pub(crate) fn fail(&mut self, key: &str, error: &str) { | |
| 863 | + | self.failed.entry(key.to_owned()).or_insert_with(|| error.chars().take(500).collect()); | |
| 864 | + | } | |
| 865 | + | ||
| 866 | + | /// Every dataset that answered answered with nothing: no Worker ran | |
| 867 | + | /// all month, or (likelier) the wrong account or a token that cannot | |
| 868 | + | /// see its analytics. | |
| 869 | + | pub(crate) fn empty(&self) -> bool { | |
| 870 | + | self.answered > 0 && self.rows == 0 | |
| 871 | + | } | |
| 872 | + | } | |
| 873 | + | ||
| 874 | + | /// How many groups a GraphQL answer has under `rows`. | |
| 875 | + | pub(crate) fn rows_in(body: &Value) -> usize { | |
| 876 | + | body["data"]["viewer"]["accounts"][0]["rows"].as_array().map_or(0, Vec::len) | |
| 877 | + | } | |
| 878 | + | ||
| 879 | + | /// What has been blind for the last `BLIND_RUNS` runs (newest first): the | |
| 880 | + | /// queries failing in every one, and `all_empty` when every one was empty. | |
| 881 | + | pub(crate) fn blind_for_long(runs: &[(BTreeMap<String, String>, bool)]) -> Vec<String> { | |
| 882 | + | if runs.len() < BLIND_RUNS { | |
| 883 | + | return vec![]; | |
| 884 | + | } | |
| 885 | + | let recent = &runs[..BLIND_RUNS]; | |
| 886 | + | let mut out: Vec<String> = recent[0].0.keys().filter(|key| recent.iter().all(|(failed, _)| failed.contains_key(*key))).cloned().collect(); | |
| 887 | + | if recent.iter().all(|(_, empty)| *empty) { | |
| 888 | + | out.push(ALL_EMPTY.to_owned()); | |
| 889 | + | } | |
| 890 | + | out | |
| 891 | + | } | |
| 892 | + | ||
| 893 | + | /// The dataset behind a query key, for people. | |
| 894 | + | pub(crate) fn dataset_of(key: &str) -> &str { | |
| 895 | + | QUERIES.iter().find(|q| q.key == key).map_or(key, |q| q.dataset) | |
| 896 | + | } | |
| 897 | + | ||
| 898 | + | #[derive(Deserialize)] | |
| 899 | + | struct RunRow { | |
| 900 | + | hour: String, | |
| 901 | + | failed: Option<String>, | |
| 902 | + | empty: i64, | |
| 903 | + | } | |
| 904 | + | ||
| 905 | + | impl Billing { | |
| 906 | + | /// Records what this run could not see, and emails staff about what | |
| 907 | + | /// has stayed unseen for `BLIND_RUNS` runs, once a day per query. | |
| 908 | + | async fn watch_health(&self, hour: &str, read_at: &str, sight: &Sight) -> Result<()> { | |
| 909 | + | let failed = serde_json::to_string(&sight.failed)?; | |
| 910 | + | self.db | |
| 911 | + | .batch(vec![ | |
| 912 | + | self.db | |
| 913 | + | .prepare( | |
| 914 | + | "INSERT INTO platform_watch_runs (hour, read_at, failed, empty) VALUES (?1, ?2, ?3, ?4) | |
| 915 | + | ON CONFLICT (hour) DO UPDATE SET read_at = ?2, failed = ?3, empty = ?4", | |
| 916 | + | ) | |
| 917 | + | .bind(&[hour.into(), read_at.into(), failed.as_str().into(), i32::from(sight.empty()).into()])?, | |
| 918 | + | self.db.prepare("DELETE FROM platform_watch_runs WHERE hour < ?").bind(&[rfc3339(now_ms().saturating_sub(14 * 24 * HOUR_MS)).into()])?, | |
| 919 | + | ]) | |
| 920 | + | .await?; | |
| 921 | + | let runs = self.watch_runs(BLIND_RUNS as u32).await?; | |
| 922 | + | let blind = blind_for_long(&runs.iter().map(|(_, failed, empty)| (failed.clone(), *empty)).collect::<Vec<_>>()); | |
| 923 | + | if blind.is_empty() || self.caps.alert_to.is_empty() { | |
| 924 | + | return Ok(()); | |
| 925 | + | } | |
| 926 | + | #[derive(Deserialize)] | |
| 927 | + | struct Sent { | |
| 928 | + | key: String, | |
| 929 | + | } | |
| 930 | + | let since = rfc3339(now_ms().saturating_sub(BLIND_EVERY_MS)); | |
| 931 | + | let sent: Vec<String> = self | |
| 932 | + | .db | |
| 933 | + | .prepare("SELECT key FROM platform_watch_alerts WHERE emailed_at >= ?") | |
| 934 | + | .bind(&[since.into()])? | |
| 935 | + | .all() | |
| 936 | + | .await? | |
| 937 | + | .results::<Sent>()? | |
| 938 | + | .into_iter() | |
| 939 | + | .map(|s| s.key) | |
| 940 | + | .collect(); | |
| 941 | + | let due: Vec<&String> = blind.iter().filter(|key| !sent.contains(key)).collect(); | |
| 942 | + | if due.is_empty() { | |
| 943 | + | return Ok(()); | |
| 944 | + | } | |
| 945 | + | let mut lines = vec![]; | |
| 946 | + | for key in &due { | |
| 947 | + | if key.as_str() == ALL_EMPTY { | |
| 948 | + | lines.push(format!( | |
| 949 | + | "Every dataset answered with no rows for the last {BLIND_RUNS} hourly runs, the month so far included. That is almost certainly the wrong account (CLOUDFLARE_ACCOUNT_ID) or a token without Account Analytics Read on it, so the watcher sees nothing at all." | |
| 950 | + | )); | |
| 951 | + | } else { | |
| 952 | + | let error = sight.failed.get(key.as_str()).map_or("(no error this run)", String::as_str); | |
| 953 | + | lines.push(format!( | |
| 954 | + | "The watcher can't see {} ({key}): it failed for the last {BLIND_RUNS} hourly runs. Its metrics are not watched until it is fixed. Last error: {error}", | |
| 955 | + | dataset_of(key) | |
| 956 | + | )); | |
| 957 | + | } | |
| 958 | + | } | |
| 959 | + | lines.push( | |
| 960 | + | "Check with `node scripts/ops/platform-usage.mjs` and a token: every dataset should report rows, and it exits non-zero naming any that errored. A renamed field is fixed in QUERIES in services/billing/src/platform.rs. See docs/SPEND-GUARDRAILS.md.".to_owned(), | |
| 961 | + | ); | |
| 962 | + | let subject = format!("g1t: the platform usage watcher is blind on {}", due.iter().map(|k| k.as_str()).collect::<Vec<_>>().join(", ")); | |
| 963 | + | match crate::margin::email_staff_page(&self.env, &self.caps.alert_to, &subject, &lines, ("Platform pause", "https://sudo.g1t.sh/costs#platform"), "g1t-billing's usage watcher").await { | |
| 964 | + | Ok(()) => { | |
| 965 | + | let at = rfc3339(now_ms()); | |
| 966 | + | let mut writes = vec![]; | |
| 967 | + | for key in due { | |
| 968 | + | writes.push( | |
| 969 | + | self.db | |
| 970 | + | .prepare("INSERT INTO platform_watch_alerts (key, emailed_at) VALUES (?1, ?2) ON CONFLICT (key) DO UPDATE SET emailed_at = ?2") | |
| 971 | + | .bind(&[key.as_str().into(), at.as_str().into()])?, | |
| 972 | + | ); | |
| 973 | + | } | |
| 974 | + | self.db.batch(writes).await?; | |
| 975 | + | } | |
| 976 | + | Err(error) => worker::console_error!("could not email that the watcher is blind: {error}"), | |
| 977 | + | } | |
| 978 | + | Ok(()) | |
| 979 | + | } | |
| 980 | + | ||
| 981 | + | /// The last `n` runs, newest first: their hour, what failed, and | |
| 982 | + | /// whether every dataset answered empty. | |
| 983 | + | async fn watch_runs(&self, n: u32) -> Result<Vec<(String, BTreeMap<String, String>, bool)>> { | |
| 984 | + | Ok(self | |
| 985 | + | .db | |
| 986 | + | .prepare("SELECT hour, failed, empty FROM platform_watch_runs ORDER BY hour DESC LIMIT ?") | |
| 987 | + | .bind(&[n.into()])? | |
| 988 | + | .all() | |
| 989 | + | .await? | |
| 990 | + | .results::<RunRow>()? | |
| 991 | + | .into_iter() | |
| 992 | + | .map(|r| (r.hour, r.failed.and_then(|f| serde_json::from_str(&f).ok()).unwrap_or_default(), r.empty != 0)) | |
| 993 | + | .collect()) | |
| 994 | + | } | |
| 995 | + | } | |
| 996 | + | ||
| 997 | + | #[cfg(test)] | |
| 998 | + | mod tests { | |
| 999 | + | use super::*; | |
| 1000 | + | ||
| 1001 | + | #[test] | |
| 1002 | + | fn a_query_failing_three_runs_in_a_row_is_blind_and_so_is_every_dataset_empty() { | |
| 1003 | + | let failed = |keys: &[&str]| keys.iter().map(|k| ((*k).to_owned(), "unknown field".to_owned())).collect::<BTreeMap<_, _>>(); | |
| 1004 | + | let runs = vec![ | |
| 1005 | + | (failed(&["workers_cpu", "do_sql"]), false), | |
| 1006 | + | (failed(&["workers_cpu", "do_sql"]), false), | |
| 1007 | + | (failed(&["workers_cpu"]), false), | |
| 1008 | + | (failed(&[]), false), | |
| 1009 | + | ]; | |
| 1010 | + | assert_eq!(blind_for_long(&runs), vec!["workers_cpu".to_owned()]); | |
| 1011 | + | // Two runs are not yet three. | |
| 1012 | + | assert!(blind_for_long(&runs[..2]).is_empty()); | |
| 1013 | + | let empty = vec![(failed(&[]), true); 3]; | |
| 1014 | + | assert_eq!(blind_for_long(&empty), vec![ALL_EMPTY.to_owned()]); | |
| 1015 | + | // One full hour among them clears it. | |
| 1016 | + | assert!(blind_for_long(&[(failed(&[]), true), (failed(&[]), false), (failed(&[]), true)]).is_empty()); | |
| 1017 | + | assert_eq!(dataset_of("do_sql"), "durableObjectsPeriodicGroups"); | |
| 1018 | + | } | |
| 1019 | + | ||
| 1020 | + | #[test] | |
| 1021 | + | fn a_run_is_empty_only_when_something_answered_with_nothing() { | |
| 1022 | + | let mut sight = Sight::default(); | |
| 1023 | + | // Nothing answered at all: failures, not emptiness. | |
| 1024 | + | sight.fail("kv", "403"); | |
| 1025 | + | sight.fail("kv", "a second error is not kept"); | |
| 1026 | + | assert!(!sight.empty()); | |
| 1027 | + | assert_eq!(sight.failed["kv"], "403"); | |
| 1028 | + | sight.answered = 8; | |
| 1029 | + | assert!(sight.empty()); | |
| 1030 | + | sight.rows = 1; | |
| 1031 | + | assert!(!sight.empty()); | |
| 1032 | + | assert_eq!(rows_in(&json!({ "data": { "viewer": { "accounts": [{ "rows": [{}, {}] }] } } })), 2); | |
| 1033 | + | assert_eq!(rows_in(&json!({ "errors": [{}] })), 0); | |
| 1034 | + | } | |
| 1035 | + | ||
| 1036 | + | fn watch() -> Watch { | |
| 1037 | + | Watch::from_vars(|_| None) | |
| 1038 | + | } | |
| 1039 | + | ||
| 1040 | + | #[test] | |
| 1041 | + | fn every_metric_has_a_threshold_and_a_query_that_counts_it() { | |
| 1042 | + | let w = watch(); | |
| 1043 | + | for m in METRICS { | |
| 1044 | + | assert!(w.threshold(m.key) > 0.0, "{}", m.key); | |
| 1045 | + | } | |
| 1046 | + | // The thresholds named in docs/SPEND-GUARDRAILS.md. | |
| 1047 | + | assert_eq!(w.threshold("kv_lists"), 200_000.0); | |
| 1048 | + | assert_eq!(w.threshold("queue_operations"), 5_000_000.0); | |
| 1049 | + | assert_eq!(w.threshold("d1_rows_read"), 2_000_000_000.0); | |
| 1050 | + | assert_eq!(w.threshold("do_rows_written"), 5_000_000.0); | |
| 1051 | + | assert_eq!(w.threshold("workers_requests"), 20_000_000.0); | |
| 1052 | + | assert_eq!(w.auto_pause, vec![Schedules, Indexing]); | |
| 1053 | + | } | |
| 1054 | + | ||
| 1055 | + | #[test] | |
| 1056 | + | fn variables_change_thresholds_and_auto_pause() { | |
| 1057 | + | let w = Watch::from_vars(|name| match name { | |
| 1058 | + | "PLATFORM_HOURLY_KV_LISTS" => Some("50_000".into()), | |
| 1059 | + | "PLATFORM_HOURLY_D1_ROWS_READ" => Some("0".into()), | |
| 1060 | + | "AUTO_PAUSE" => Some("compute, renders,nonsense".into()), | |
| 1061 | + | "PLATFORM_SEVERE_FACTOR" => Some("0.5".into()), | |
| 1062 | + | _ => None, | |
| 1063 | + | }); | |
| 1064 | + | assert_eq!(w.threshold("kv_lists"), 50_000.0); | |
| 1065 | + | assert_eq!(w.threshold("d1_rows_read"), 0.0); | |
| 1066 | + | assert_eq!(w.auto_pause, vec![Compute, Renders]); | |
| 1067 | + | // Never below the threshold itself. | |
| 1068 | + | assert_eq!(w.severe_factor, 1.0); | |
| 1069 | + | // Empty: nothing pauses by itself. | |
| 1070 | + | assert!(Watch::from_vars(|n| (n == "AUTO_PAUSE").then(String::new)).auto_pause.is_empty()); | |
| 1071 | + | } | |
| 1072 | + | ||
| 1073 | + | #[test] | |
| 1074 | + | fn the_hour_read_is_the_one_before_at_a_quarter_past() { | |
| 1075 | + | // 2026-10-08 13:17:00 UTC. | |
| 1076 | + | let now = 1_791_465_420_000; | |
| 1077 | + | assert_eq!(rfc3339(now), "2026-10-08T13:17:00.000Z"); | |
| 1078 | + | assert!(hourly_due(now)); | |
| 1079 | + | assert!(!hourly_due(now - 15 * 60_000)); | |
| 1080 | + | assert!(!hourly_due(now + 15 * 60_000)); | |
| 1081 | + | assert_eq!(last_hour(now), Window::Hour { since: "2026-10-08T12:00:00Z".into(), until: "2026-10-08T13:00:00Z".into() }); | |
| 1082 | + | assert_eq!(month_so_far(now), Window::Days { since: "2026-10-01".into(), until: "2026-10-08".into() }); | |
| 1083 | + | } | |
| 1084 | + | ||
| 1085 | + | #[test] | |
| 1086 | + | fn queries_alias_their_dataset_and_filter_by_the_window() { | |
| 1087 | + | let d1 = QUERIES.iter().find(|q| q.key == "d1").unwrap(); | |
| 1088 | + | let hour = query_body("acct", d1, &last_hour(1_791_465_420_000)); | |
| 1089 | + | let text = hour["query"].as_str().unwrap(); | |
| 1090 | + | assert!(text.contains("rows: d1AnalyticsAdaptiveGroups(limit: 10000, filter: { datetimeHour_geq: $since, datetimeHour_lt: $until })"), "{text}"); | |
| 1091 | + | assert!(text.contains("$since: Time!") && text.contains("sum { rowsRead rowsWritten }") && text.contains("dimensions { databaseId }")); | |
| 1092 | + | assert_eq!(hour["variables"]["since"], "2026-10-08T12:00:00Z"); | |
| 1093 | + | let month = query_body("acct", d1, &month_so_far(1_791_465_420_000)); | |
| 1094 | + | let text = month["query"].as_str().unwrap(); | |
| 1095 | + | assert!(text.contains("date_geq: $since, date_leq: $until") && text.contains("$since: Date!"), "{text}"); | |
| 1096 | + | } | |
| 1097 | + | ||
| 1098 | + | fn answer(rows: Value) -> Value { | |
| 1099 | + | json!({ "data": { "viewer": { "accounts": [{ "rows": rows }] } }, "errors": null }) | |
| 1100 | + | } | |
| 1101 | + | ||
| 1102 | + | fn query(key: &str) -> &'static Query { | |
| 1103 | + | QUERIES.iter().find(|q| q.key == key).unwrap() | |
| 1104 | + | } | |
| 1105 | + | ||
| 1106 | + | #[test] | |
| 1107 | + | fn answers_become_metrics_by_name_and_a_refused_dataset_is_skipped() { | |
| 1108 | + | let kv = counted( | |
| 1109 | + | query("kv"), | |
| 1110 | + | &answer(json!([ | |
| 1111 | + | { "sum": { "requests": 150000 }, "dimensions": { "namespaceId": "ns_a", "actionType": "list" } }, | |
| 1112 | + | { "sum": { "requests": 90000 }, "dimensions": { "namespaceId": "ns_b", "actionType": "list" } }, | |
| 1113 | + | { "sum": { "requests": 4000 }, "dimensions": { "namespaceId": "ns_a", "actionType": "read" } }, | |
| 1114 | + | { "sum": { "requests": 7 }, "dimensions": { "namespaceId": "ns_a", "actionType": "mystery" } }, | |
| 1115 | + | ])), | |
| 1116 | + | ) | |
| 1117 | + | .unwrap(); | |
| 1118 | + | let d1 = counted(query("d1"), &answer(json!([{ "sum": { "rowsRead": 10.0, "rowsWritten": 2.0 }, "dimensions": { "databaseId": "db1" } }]))).unwrap(); | |
| 1119 | + | let cpu = counted(query("workers_cpu"), &answer(json!([{ "sum": { "cpuTimeUs": 5_000_000 }, "dimensions": { "scriptName": "g1t-web" } }]))).unwrap(); | |
| 1120 | + | let refused = counted(query("do_sql"), &json!({ "data": null, "errors": [{ "message": "unknown field rowsWritten" }] })); | |
| 1121 | + | assert!(refused.unwrap_err().contains("unknown field rowsWritten")); | |
| 1122 | + | assert!(counted(query("queues"), &json!({ "data": { "viewer": { "accounts": [] } } })).is_err()); | |
| 1123 | + | let all = totals(&[kv, d1, cpu]); | |
| 1124 | + | assert_eq!(all["kv_lists"], Total { value: 240_000.0, top_name: Some("ns_a".into()), top_value: Some(150_000.0) }); | |
| 1125 | + | assert_eq!(all["kv_reads"].value, 4_000.0); | |
| 1126 | + | assert_eq!(all["d1_rows_written"].value, 2.0); | |
| 1127 | + | assert_eq!(all["workers_cpu_ms"].value, 5_000.0); | |
| 1128 | + | assert!(!all.contains_key("do_rows_written")); | |
| 1129 | + | } | |
| 1130 | + | ||
| 1131 | + | #[test] | |
| 1132 | + | fn over_the_threshold_is_a_breach_and_five_times_it_is_severe() { | |
| 1133 | + | let w = watch(); | |
| 1134 | + | assert_eq!(judge(&w, "kv_lists", 200_000.0, &[]), None); | |
| 1135 | + | let breach = judge(&w, "kv_lists", 240_000.0, &[]).unwrap(); | |
| 1136 | + | assert_eq!((breach.rule, breach.severe), ("threshold", false)); | |
| 1137 | + | assert!(levels_to_pause(&w, &breach).is_empty()); | |
| 1138 | + | let severe = judge(&w, "kv_lists", 1_000_000.0, &[]).unwrap(); | |
| 1139 | + | assert!(severe.severe); | |
| 1140 | + | // KV feeds indexing and renders; AUTO_PAUSE allows indexing only. | |
| 1141 | + | assert_eq!(levels_to_pause(&w, &severe), vec![Indexing]); | |
| 1142 | + | // Durable Objects feed compute, which is never paused by itself by default. | |
| 1143 | + | let dos = judge(&w, "do_rows_written", 30_000_000.0, &[]).unwrap(); | |
| 1144 | + | assert_eq!(levels_to_pause(&w, &dos), vec![Schedules]); | |
| 1145 | + | } | |
| 1146 | + | ||
| 1147 | + | #[test] | |
| 1148 | + | fn a_spike_is_ten_times_the_weeks_usual_hour_above_a_floor() { | |
| 1149 | + | let w = watch(); | |
| 1150 | + | let quiet = vec![1_000.0; 168]; | |
| 1151 | + | // Ten times the usual hour, but under 10% of the 5M threshold: not one. | |
| 1152 | + | assert_eq!(judge(&w, "queue_operations", 20_000.0, &quiet), None); | |
| 1153 | + | let usual = vec![60_000.0; 168]; | |
| 1154 | + | let spike = judge(&w, "queue_operations", 700_000.0, &usual).unwrap(); | |
| 1155 | + | assert_eq!((spike.rule, spike.limit, spike.severe), ("spike", 600_000.0, false)); | |
| 1156 | + | assert_eq!(judge(&w, "queue_operations", 590_000.0, &usual), None); | |
| 1157 | + | // Under a day of history: no spike rule yet. | |
| 1158 | + | assert_eq!(judge(&w, "queue_operations", 700_000.0, &usual[..23]), None); | |
| 1159 | + | assert_eq!(median(&mut [3.0, 1.0, 2.0, 10.0]), Some(2.5)); | |
| 1160 | + | assert_eq!(median(&mut []), None); | |
| 1161 | + | } | |
| 1162 | + | ||
| 1163 | + | #[test] | |
| 1164 | + | fn an_alert_names_where_to_look() { | |
| 1165 | + | let breach = Breach { metric: "kv_lists", rule: "threshold", value: 1_250_000.0, limit: 200_000.0, severe: true }; | |
| 1166 | + | let text = describe(&breach, "2026-10-08T12:00:00Z", Some(("e627b571", 1_200_000.0))); | |
| 1167 | + | assert_eq!( | |
| 1168 | + | text, | |
| 1169 | + | "KV lists: 1,250,000 in the hour from 2026-10-08T12:00:00Z, over its hourly threshold of 200,000 (PLATFORM_HOURLY_KV_LISTS). Most of it from the KV namespace id e627b571 (1,200,000)." | |
| 1170 | + | ); | |
| 1171 | + | assert_eq!(amount(1.5), "1.5"); | |
| 1172 | + | assert_eq!(amount(5.0), "5"); | |
| 1173 | + | } | |
| 1174 | + | ||
| 1175 | + | #[test] | |
| 1176 | + | fn a_pause_refuses_the_level_a_reservation_waits_on() { | |
| 1177 | + | use g1t_contracts::billing::ComputeKind; | |
| 1178 | + | assert_eq!(level_for(ComputeKind::Embedding), Indexing); | |
| 1179 | + | for kind in [ComputeKind::Agent, ComputeKind::Check, ComputeKind::Workflow, ComputeKind::Queue, ComputeKind::Deploy] { | |
| 1180 | + | assert_eq!(level_for(kind), Compute); | |
| 1181 | + | } | |
| 1182 | + | assert!(pause_refusal(Compute).contains("Work already running finishes")); | |
| 1183 | + | assert_eq!(PauseLevel::parse(" indexing "), Some(Indexing)); | |
| 1184 | + | assert_eq!(PauseLevel::parse("everything"), None); | |
| 1185 | + | } | |
| 1186 | + | } |
| 134 | 134 | // 00:00 UTC; staff are emailed and can lift it in sudo. Workspaces | |
| 135 | 135 | // paying with real money are never paused. 0 turns either cap off. | |
| 136 | 136 | "PLATFORM_DAILY_SPEND_CAP_MICROS": "75000000", | |
| 137 | + | // The platform watch (src/platform.rs, docs/SPEND-GUARDRAILS.md): at | |
| 138 | + | // a quarter past each hour, what Cloudflare counted in the hour | |
| 139 | + | // before, per metric, against these thresholds. Each is a dollar or | |
| 140 | + | // a few an hour at list prices, far above a small alpha's hour; 0 | |
| 141 | + | // turns that metric's threshold off. Over one, staff are emailed | |
| 142 | + | // (COSTS_ALERT_EMAIL, once per metric every 6 hours) and sudo shows | |
| 143 | + | // it. | |
| 144 | + | "PLATFORM_HOURLY_WORKERS_REQUESTS": "20000000", | |
| 145 | + | "PLATFORM_HOURLY_WORKERS_CPU_MS": "100000000", | |
| 146 | + | "PLATFORM_HOURLY_D1_ROWS_READ": "2000000000", | |
| 147 | + | "PLATFORM_HOURLY_D1_ROWS_WRITTEN": "5000000", | |
| 148 | + | "PLATFORM_HOURLY_QUEUE_OPERATIONS": "5000000", | |
| 149 | + | "PLATFORM_HOURLY_DO_REQUESTS": "20000000", | |
| 150 | + | "PLATFORM_HOURLY_DO_ROWS_WRITTEN": "5000000", | |
| 151 | + | "PLATFORM_HOURLY_DO_STORAGE_WRITE_UNITS": "5000000", | |
| 152 | + | "PLATFORM_HOURLY_DO_ACTIVE_SECONDS": "3000000", | |
| 153 | + | "PLATFORM_HOURLY_KV_READS": "10000000", | |
| 154 | + | "PLATFORM_HOURLY_KV_WRITES": "200000", | |
| 155 | + | "PLATFORM_HOURLY_KV_DELETES": "200000", | |
| 156 | + | "PLATFORM_HOURLY_KV_LISTS": "200000", | |
| 157 | + | "PLATFORM_HOURLY_ARTIFACTS_EVENTS": "1000000", | |
| 158 | + | // A spike: an hour over 10 times the last week's median hour, and at | |
| 159 | + | // least 10% of its threshold. Emailed, never paused by itself. | |
| 160 | + | "PLATFORM_SPIKE_FACTOR": "10", | |
| 161 | + | "PLATFORM_SPIKE_FLOOR_PERCENT": "10", | |
| 162 | + | // Severe: 5 times a threshold pauses the levels that metric feeds, | |
| 163 | + | // of those AUTO_PAUSE names (compute, schedules, indexing, renders; | |
| 164 | + | // comma-separated, empty for none). Staff resume them in sudo. | |
| 165 | + | "PLATFORM_SEVERE_FACTOR": "5", | |
| 166 | + | "AUTO_PAUSE": "schedules,indexing", | |
| 137 | 167 | // Cloudflare's fixed subscriptions a month, for sudo's figures only: | |
| 138 | 168 | // Workers Paid ($5) and Workers for Platforms ($25). | |
| 139 | 169 | "CLOUDFLARE_FIXED_MONTHLY_MICROS": "30000000" | |
| 140 | 170 | }, | |
| 141 | − | // Settling runs every 15 minutes; checking costs daily (keeper::DAILY). | |
| 171 | + | // Settling runs every 15 minutes, and the tick at a quarter past each | |
| 172 | + | // hour also runs the platform watch; checking costs daily (keeper::DAILY). | |
| 142 | 173 | "triggers": { "crons": ["*/15 * * * *", "17 4 * * *"] }, | |
| 143 | 174 | // Secrets: STRIPE_SECRET_KEY. Without it nothing is charged and the | |
| 144 | 175 | // runner decides who may start agents some other way. |
| 48 | 48 | billingClient, | |
| 49 | 49 | embeddingEstimateMicros, | |
| 50 | 50 | localRefusal, | |
| 51 | + | platformPaused, | |
| 51 | 52 | currentMovedPath, | |
| 52 | 53 | currentWorkspaceSlug, | |
| 53 | 54 | repoMove, | |
| ⋯ | |||
| 125 | 126 | const SHARED: Set<EntityKind> = new Set(["owner", "language", "integration"]); | |
| 126 | 127 | ||
| 127 | 128 | /** One compute gate per isolate, so entitlements are kept between calls. */ | |
| 129 | + | /** What a backfill says while indexing is paused across g1t (billing's `platform_pause`). */ | |
| 130 | + | const INDEXING_PAUSED = "g1t has paused indexing across the platform for now. Rebuild the context again later."; | |
| 131 | + | ||
| 128 | 132 | let computeGate: ComputeGate | null = null; | |
| 129 | 133 | function gateFor(billing: ServiceBinding): ComputeGate { | |
| 130 | 134 | computeGate ??= new ComputeGate(billing); | |
| ⋯ | |||
| 965 | 969 | async backfill(a: { actor: User; workspace: string }): Promise<Result<Backfill>> { | |
| 966 | 970 | const workspace = a.workspace.toLowerCase(); | |
| 967 | 971 | if (!isMember(a.actor, workspace)) return fail("forbidden", "Only members can rebuild a workspace's context."); | |
| 972 | + | if (await platformPaused(this.env.BILLING, "indexing")) return fail("paused", INDEXING_PAUSED); | |
| 968 | 973 | const running = await this.backfillRow(workspace); | |
| 969 | 974 | if (running?.status === "running" && Date.now() - Date.parse(running.startedAt) < BACKFILL_STALE_MS) return ok(running); | |
| 970 | 975 | const actor = (await this.workspaceActor(workspace)) ?? a.actor; | |
| ⋯ | |||
| 990 | 995 | async runJob(job: Job): Promise<void> { | |
| 991 | 996 | const actor = await this.workspaceActor(job.workspace); | |
| 992 | 997 | if (!actor) return; | |
| 998 | + | // Indexing paused across g1t: the job is counted done with the pause | |
| 999 | + | // as its note, so the backfill finishes and can be run again later. | |
| 1000 | + | if (await platformPaused(this.env.BILLING, "indexing")) { | |
| 1001 | + | if (job.type === "backfill_project") { | |
| 1002 | + | await this.db | |
| 1003 | + | .prepare( | |
| 1004 | + | `UPDATE backfills SET done = done + 1, error = ?, | |
| 1005 | + | status = CASE WHEN done + 1 >= projects THEN 'done' ELSE status END, | |
| 1006 | + | finished_at = CASE WHEN done + 1 >= projects THEN ? ELSE finished_at END | |
| 1007 | + | WHERE workspace = ?`, | |
| 1008 | + | ) | |
| 1009 | + | .bind(INDEXING_PAUSED, now(), job.workspace) | |
| 1010 | + | .run(); | |
| 1011 | + | } | |
| 1012 | + | return; | |
| 1013 | + | } | |
| 993 | 1014 | if (job.type === "backfill_memory") { | |
| 994 | 1015 | const memories = await memoryReviewClient(this.env.WORK).searchMemories(job.workspace, null, { limit: 100 }); | |
| 995 | 1016 | const indexed = await this.indexMemories(job.workspace, memories); | |
| 1 | + | -- The most a run may spend on models, in millionths of a dollar: the lower | |
| 2 | + | -- of its project's cost cap and its plan's, set by the sandbox once it | |
| 3 | + | -- knows them. The model proxy refuses the run's requests past it. Null | |
| 4 | + | -- until set; the proxy then holds the run to the most any run may cost. | |
| 5 | + | ALTER TABLE model_sessions ADD COLUMN cap_micros INTEGER; |
| 159 | 159 | tier: Option<String>, | |
| 160 | 160 | #[serde(default)] | |
| 161 | 161 | requested_by: Option<String>, | |
| 162 | + | #[serde(default)] | |
| 163 | + | cap_micros: Option<i64>, | |
| 164 | + | } | |
| 165 | + | ||
| 166 | + | /// A run's model spend cap as a session keeps it: null for none (zero or | |
| 167 | + | /// less), and never more than $1,000, which no run is allowed. D1 takes | |
| 168 | + | /// numbers as doubles, which hold any such cap exactly. | |
| 169 | + | fn session_cap(cap_micros: i64) -> JsValue { | |
| 170 | + | if cap_micros > 0 { JsValue::from_f64(cap_micros.min(1_000_000_000) as f64) } else { JsValue::NULL } | |
| 162 | 171 | } | |
| 163 | 172 | ||
| 164 | 173 | #[derive(Deserialize)] | |
| ⋯ | |||
| 1324 | 1333 | api_key: None, | |
| 1325 | 1334 | auth_header: None, | |
| 1326 | 1335 | gateway_token: None, | |
| 1336 | + | cap_micros: session.cap_micros.filter(|cap| *cap > 0), | |
| 1327 | 1337 | }; | |
| 1328 | 1338 | let Some(connection_id) = session.connection_id else { | |
| 1329 | 1339 | return Ok(Some(base)); | |
| ⋯ | |||
| 1381 | 1391 | api_key: secrets.secret.clone(), | |
| 1382 | 1392 | auth_header: Some(models::auth_header(provider, &config)), | |
| 1383 | 1393 | gateway_token: (provider == Provider::AnthropicEndpoint).then_some(secrets.signing_secret).flatten(), | |
| 1394 | + | cap_micros: None, | |
| 1384 | 1395 | })) | |
| 1385 | 1396 | } | |
| 1386 | 1397 | ||
| ⋯ | |||
| 1418 | 1429 | Ok(result.meta()?.and_then(|meta| meta.changes).unwrap_or(0) as u32) | |
| 1419 | 1430 | } | |
| 1420 | 1431 | ||
| 1432 | + | /// Sets the most a run may spend on models, which the model proxy holds | |
| 1433 | + | /// its token to. The sandbox calls it once it knows the run's caps. | |
| 1434 | + | async fn cap_model_sessions(&self, a: CapModelSessionsArgs) -> Result<u32> { | |
| 1435 | + | let hashes = closable_hashes(&a.token_hashes); | |
| 1436 | + | if hashes.is_empty() { | |
| 1437 | + | return Ok(0); | |
| 1438 | + | } | |
| 1439 | + | let marks = vec!["?"; hashes.len()].join(", "); | |
| 1440 | + | let mut values: Vec<JsValue> = vec![session_cap(a.cap_micros)]; | |
| 1441 | + | values.extend(hashes.iter().map(|hash| JsValue::from(hash.as_str()))); | |
| 1442 | + | values.push(rfc3339(now_ms()).into()); | |
| 1443 | + | let result = self | |
| 1444 | + | .db | |
| 1445 | + | .prepare(format!("UPDATE model_sessions SET cap_micros = ? WHERE token_hash IN ({marks}) AND expires_at > ?")) | |
| 1446 | + | .bind(&values)? | |
| 1447 | + | .run() | |
| 1448 | + | .await?; | |
| 1449 | + | Ok(result.meta()?.and_then(|meta| meta.changes).unwrap_or(0) as u32) | |
| 1450 | + | } | |
| 1451 | + | ||
| 1421 | 1452 | // --- Writing back ----------------------------------------------------------- | |
| 1422 | 1453 | ||
| 1423 | 1454 | async fn on_event(&self, event: &Event) -> Result<()> { | |
| ⋯ | |||
| 1569 | 1600 | "gateway_upstream" => reply(&service.gateway_upstream(args(body)?).await?), | |
| 1570 | 1601 | "gateway_providers" => reply(&service.gateway_providers(args(body)?).await?), | |
| 1571 | 1602 | "close_model_sessions" => reply(&service.close_model_sessions(args(body)?).await?), | |
| 1603 | + | "cap_model_sessions" => reply(&service.cap_model_sessions(args(body)?).await?), | |
| 1572 | 1604 | "routes" => reply(&service.routes(args(body)?).await?), | |
| 1573 | 1605 | "set_routes" => reply(&service.set_routes(args(body)?).await?), | |
| 1574 | 1606 | _ => Response::error("Unknown method", 404), | |
| 15 | 15 | * it passes and reported to billing afterwards, counted per run for usage | |
| 16 | 16 | * views. | |
| 17 | 17 | * | |
| 18 | + | * Each run is held to its cost cap here too, not only by the harness in the | |
| 19 | + | * sandbox: every answer's cost is added to the run's count (a Durable | |
| 20 | + | * Object per run, `run-spend.ts`), and once the run has spent its cap its | |
| 21 | + | * requests are refused with a 402 (`spend.ts`). | |
| 22 | + | * | |
| 18 | 23 | * The same address is the AI Gateway for a workspace's own code: a request | |
| 19 | 24 | * with one of the workspace's access tokens (`g1t_…`) instead of a run's, | |
| 20 | 25 | * at `/anthropic` in Anthropic's format or `/openai/v1` in OpenAI's, goes | |
| ⋯ | |||
| 37 | 42 | ||
| 38 | 43 | import { openaiError } from "./chat"; | |
| 39 | 44 | import { discover } from "./discover"; | |
| 45 | + | import { anthropicErrorType } from "./gateway"; | |
| 40 | 46 | import { type AnthropicRequest, StreamTranslator, errorFromChat, estimateTokens, fromChat, toChat } from "./openai"; | |
| 41 | − | import { isAnswer, tokenReport } from "./report"; | |
| 47 | + | import { isAnswer, runMayCall, tokenReport } from "./report"; | |
| 42 | 48 | import { type HostedRouting, presentedToken, upstreamRequest } from "./route"; | |
| 49 | + | import type { RunSpend } from "./run-spend"; | |
| 43 | 50 | import { type GatewayDeps, isOpenAiPath, serveGateway } from "./serve"; | |
| 51 | + | import { capOf, capReached, ceilingMicros, chargeFor, pricesFor, tooBusy } from "./spend"; | |
| 44 | 52 | import { measure } from "./usage"; | |
| 45 | 53 | ||
| 54 | + | export { RunSpend } from "./run-spend"; | |
| 55 | + | ||
| 46 | 56 | interface Env extends HostedRouting { | |
| 47 | 57 | INTEGRATIONS: ServiceBinding; | |
| 48 | 58 | BILLING: ServiceBinding; | |
| 49 | 59 | IDENTITY: ServiceBinding; | |
| 60 | + | /** Each run's model spend, one object per model session. */ | |
| 61 | + | RUN_SPEND: DurableObjectNamespace<RunSpend>; | |
| 50 | 62 | } | |
| 51 | 63 | ||
| 52 | 64 | /** | |
| ⋯ | |||
| 69 | 81 | ||
| 70 | 82 | /** An error in the shape Anthropic's API uses, which the harness understands. */ | |
| 71 | 83 | function refuse(status: number, message: string): Response { | |
| 72 | − | return Response.json( | |
| 73 | − | { type: "error", error: { type: status === 401 ? "authentication_error" : "not_found_error", message } }, | |
| 74 | − | { status }, | |
| 75 | − | ); | |
| 84 | + | return Response.json({ type: "error", error: { type: anthropicErrorType(status), message } }, { status }); | |
| 85 | + | } | |
| 86 | + | ||
| 87 | + | /** The run's spend count, by its session's id. */ | |
| 88 | + | function runSpend(env: Env, upstream: ModelUpstream): DurableObjectStub<RunSpend> { | |
| 89 | + | return env.RUN_SPEND.get(env.RUN_SPEND.idFromName(upstream.session || `${upstream.workspace}/${upstream.repo}#${upstream.number}`)); | |
| 76 | 90 | } | |
| 77 | 91 | ||
| 92 | + | /** One answer's place in its run's count, settled once the answer has gone by. */ | |
| 93 | + | type Held = { spend: DurableObjectStub<RunSpend>; ticket: string; requested: Partial<AnthropicRequest> | null; bodyLength: number }; | |
| 94 | + | ||
| 78 | 95 | /** | |
| 79 | − | * Passes an answer through and, once it has all gone by, tells billing what | |
| 80 | − | * it used. Reporting happens after the answer, and a report that fails is | |
| 81 | − | * dropped: the answer never waits on it or breaks for it. | |
| 96 | + | * Passes an answer through and, once it has all gone by, adds its cost to | |
| 97 | + | * the run's count and tells billing what it used. Both happen after the | |
| 98 | + | * answer, and a report that fails is dropped: the answer never waits on | |
| 99 | + | * it or breaks for it. | |
| 82 | 100 | */ | |
| 83 | − | function counted(answer: Response, upstream: ModelUpstream, env: Env, ctx: ExecutionContext): Response { | |
| 101 | + | function counted(answer: Response, upstream: ModelUpstream, env: Env, ctx: ExecutionContext, held: Held): Response { | |
| 84 | 102 | const { response, tokens, model } = measure(answer); | |
| 85 | 103 | ctx.waitUntil( | |
| 86 | 104 | (async () => { | |
| 87 | − | const report = tokenReport(upstream, await model, await tokens); | |
| 88 | − | if (report) await billingClient(env.BILLING).recordTokens(report); | |
| 105 | + | const used = await tokens; | |
| 106 | + | const answeredBy = await model; | |
| 107 | + | const settle = (async () => { | |
| 108 | + | const prices = pricesFor(upstream.model ?? answeredBy ?? held.requested?.model, upstream.route, await offered(env).catch(() => [])); | |
| 109 | + | const charge = chargeFor(prices, used, answer.ok, ceilingMicros(prices, held.bodyLength, held.requested?.max_tokens)); | |
| 110 | + | await held.spend.settle(held.ticket, charge); | |
| 111 | + | })().catch((error: unknown) => console.error("models: a run's spend was not counted", upstream.session, String(error))); | |
| 112 | + | const report = tokenReport(upstream, answeredBy, used); | |
| 113 | + | const reported = report ? billingClient(env.BILLING).recordTokens(report).catch(() => undefined) : Promise.resolve(); | |
| 114 | + | await Promise.all([settle, reported]); | |
| 89 | 115 | })().catch(() => undefined), | |
| 90 | 116 | ); | |
| 91 | 117 | return response; | |
| 92 | 118 | } | |
| 93 | 119 | ||
| 120 | + | /** The fields of a request body the proxy reads, or null when it is not a JSON object. */ | |
| 121 | + | function parsedBody(text: string): Record<string, unknown> | null { | |
| 122 | + | try { | |
| 123 | + | const parsed = JSON.parse(text) as unknown; | |
| 124 | + | return parsed && typeof parsed === "object" && !Array.isArray(parsed) ? (parsed as Record<string, unknown>) : null; | |
| 125 | + | } catch { | |
| 126 | + | return null; | |
| 127 | + | } | |
| 128 | + | } | |
| 129 | + | ||
| 94 | 130 | /** Sends an Anthropic request to a provider that speaks OpenAI's API. */ | |
| 95 | − | async function viaChat(upstream: ModelUpstream, path: string, request: Request): Promise<Response> { | |
| 96 | − | const body = (await request.json()) as AnthropicRequest; | |
| 131 | + | async function viaChat(upstream: ModelUpstream, path: string, body: AnthropicRequest | null): Promise<Response> { | |
| 132 | + | if (!path.startsWith("/v1/messages")) return refuse(404, `${path} has no counterpart at this provider.`); | |
| 133 | + | if (!body) return refuse(400, "The request body is not a JSON object."); | |
| 97 | 134 | const model = upstream.model ?? body.model ?? ""; | |
| 98 | 135 | if (path.startsWith("/v1/messages/count_tokens")) { | |
| 99 | 136 | return Response.json({ input_tokens: estimateTokens(body) }); | |
| 100 | 137 | } | |
| 101 | − | if (!path.startsWith("/v1/messages")) return refuse(404, `${path} has no counterpart at this provider.`); | |
| 102 | 138 | ||
| 103 | 139 | const headers = new Headers({ "content-type": "application/json" }); | |
| 104 | 140 | if (upstream.gatewayToken) headers.set("cf-aig-authorization", `Bearer ${upstream.gatewayToken}`); | |
| ⋯ | |||
| 237 | 273 | const upstream = await lookUp(env, token); | |
| 238 | 274 | if (!upstream) return refuse(401, "This run's model token has expired, or its model connection was removed."); | |
| 239 | 275 | ||
| 240 | − | const path = url.pathname.slice("/anthropic".length) + url.search; | |
| 276 | + | const route = url.pathname.slice("/anthropic".length); | |
| 277 | + | const path = route + url.search; | |
| 278 | + | if (!runMayCall(route, request.method)) return refuse(404, `A run's model token reaches only /anthropic/v1/messages and /anthropic/v1/models, not ${route}.`); | |
| 279 | + | const hasBody = request.method !== "GET" && request.method !== "HEAD"; | |
| 280 | + | const text = hasBody ? await request.text() : null; | |
| 281 | + | const parsed = text === null ? null : parsedBody(text); | |
| 282 | + | ||
| 283 | + | // A model's answer costs the run: it must be under its cap to start | |
| 284 | + | // one, and the answer's cost is added to its count once it has gone by. | |
| 285 | + | let held: Held | null = null; | |
| 286 | + | if (isAnswer(route) && hasBody) { | |
| 287 | + | const spend = runSpend(env, upstream); | |
| 288 | + | const cap = capOf(upstream); | |
| 289 | + | let admitted; | |
| 290 | + | try { | |
| 291 | + | admitted = await spend.admit(cap); | |
| 292 | + | } catch (error) { | |
| 293 | + | console.error("models: a run's spend could not be checked", upstream.session, String(error)); | |
| 294 | + | return refuse(503, "g1t could not check this run's spending just now. Try again."); | |
| 295 | + | } | |
| 296 | + | if (!admitted.ok) return admitted.reason === "cap" ? capReached(cap, admitted.spent) : tooBusy(); | |
| 297 | + | held = { spend, ticket: admitted.ticket, requested: parsed as Partial<AnthropicRequest> | null, bodyLength: text?.length ?? 0 }; | |
| 298 | + | } | |
| 241 | 299 | // Both routes answer in Anthropic's shape, so one reading counts either. | |
| 242 | − | const answer = (response: Response) => (isAnswer(url.pathname.slice("/anthropic".length)) ? counted(response, upstream, env, ctx) : response); | |
| 243 | − | if (upstream.api === "openai") return answer(await viaChat(upstream, path, request)); | |
| 300 | + | const answer = (response: Response) => (held ? counted(response, upstream, env, ctx, held) : response); | |
| 301 | + | try { | |
| 302 | + | if (upstream.api === "openai") return answer(await viaChat(upstream, path, parsed as AnthropicRequest | null)); | |
| 244 | 303 | ||
| 245 | − | const { url: target, headers } = upstreamRequest(upstream, env, path, request.headers); | |
| 246 | − | // A route that names a model gets it for every request of the run, | |
| 247 | − | // including the harness's small background ones. | |
| 248 | − | let body: BodyInit | null = request.method === "GET" || request.method === "HEAD" ? null : request.body; | |
| 249 | − | if (upstream.model && body && path.startsWith("/v1/messages")) { | |
| 250 | − | const parsed = (await request.json()) as Record<string, unknown>; | |
| 251 | − | body = JSON.stringify({ ...parsed, model: upstream.model }); | |
| 252 | − | headers.delete("content-length"); | |
| 304 | + | const { url: target, headers } = upstreamRequest(upstream, env, path, request.headers); | |
| 305 | + | // A route that names a model gets it for every request of the run, | |
| 306 | + | // including the harness's small background ones. | |
| 307 | + | let body: string | null = text; | |
| 308 | + | if (upstream.model && parsed && path.startsWith("/v1/messages")) { | |
| 309 | + | body = JSON.stringify({ ...parsed, model: upstream.model }); | |
| 310 | + | headers.delete("content-length"); | |
| 311 | + | } | |
| 312 | + | return answer(await fetch(target, { method: request.method, headers, body })); | |
| 313 | + | } catch (error) { | |
| 314 | + | // No answer: it cost nothing, and gives its place back. | |
| 315 | + | if (held) ctx.waitUntil(held.spend.settle(held.ticket, 0).catch(() => undefined)); | |
| 316 | + | throw error; | |
| 253 | 317 | } | |
| 254 | − | return answer(await fetch(target, { method: request.method, headers, body })); | |
| 255 | 318 | }, | |
| 256 | 319 | } satisfies ExportedHandler<Env>; | |
| 3 | 3 | ||
| 4 | 4 | import type { ModelUpstream } from "@g1t/contracts"; | |
| 5 | 5 | ||
| 6 | − | import { isAnswer, tokenReport } from "./report.ts"; | |
| 6 | + | import { isAnswer, runMayCall, tokenReport } from "./report.ts"; | |
| 7 | 7 | import { NO_TOKENS, measure } from "./usage.ts"; | |
| 8 | 8 | ||
| 9 | 9 | const upstream: ModelUpstream = { | |
| ⋯ | |||
| 62 | 62 | await whole.response.text(); | |
| 63 | 63 | assert.equal(await whole.model, "claude-y"); | |
| 64 | 64 | }); | |
| 65 | + | ||
| 66 | + | test("a run's token reaches answers, token counts and the model list, and nothing else", () => { | |
| 67 | + | assert.ok(runMayCall("/v1/messages", "POST")); | |
| 68 | + | assert.ok(runMayCall("/v1/messages?beta=true", "POST")); | |
| 69 | + | assert.ok(runMayCall("/v1/messages/count_tokens", "POST")); | |
| 70 | + | assert.ok(runMayCall("/v1/models", "GET")); | |
| 71 | + | assert.ok(runMayCall("/v1/models/claude-sonnet-5-5", "GET")); | |
| 72 | + | assert.equal(runMayCall("/v1/messages/batches", "POST"), false); | |
| 73 | + | assert.equal(runMayCall("/v1//messages", "POST"), false); | |
| 74 | + | assert.equal(runMayCall("/v1/complete", "POST"), false); | |
| 75 | + | assert.equal(runMayCall("/v1/models", "POST"), false); | |
| 76 | + | assert.equal(runMayCall("/v1/files", "GET"), false); | |
| 77 | + | }); | |
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
Binary or large file; its contents are not shown.
This change is too large to show in full.