Merge remote-tracking branch 'origin/main' into workspace-chat
28 files+2255−1190/28 viewed
| 7 | 7 | # typescript type checks and tests of the apps and TS services, the | |
| 8 | 8 | # deploy and ops scripts, and the deploy manifest | |
| 9 | 9 | # build the site, sudo and the docs build as they deploy | |
| 10 | + | # | |
| 11 | + | # On a push to main only `rust` runs, to keep main's caches current: a pull | |
| 12 | + | # request's run restores from main's cache, never from another pull | |
| 13 | + | # request's, so without it each pull request would start from nothing. | |
| 10 | 14 | name: CI | |
| 11 | 15 | ||
| 12 | 16 | on: | |
| 13 | 17 | pull_request: | |
| 14 | 18 | branches: [main] | |
| 19 | + | push: | |
| 20 | + | branches: [main] | |
| 15 | 21 | workflow_dispatch: | |
| 16 | 22 | ||
| 17 | 23 | # Its token only reads: it checks the code out and nothing more. | |
| ⋯ | |||
| 47 | 53 | !target/debug/incremental | |
| 48 | 54 | key: cargo-test-${{ runner.os }}-${{ hashFiles('Cargo.lock', 'services/runner/base.json') }} | |
| 49 | 55 | restore-keys: cargo-test-${{ runner.os }}- | |
| 56 | + | # The workspace's library crates, which Cargo compiles again on every | |
| 57 | + | # checkout, come back from the repository's Actions cache when their | |
| 58 | + | # inputs did not change (scripts/sccache.sh). Test harnesses and | |
| 59 | + | # Workers' own crates are linked, and still compiled. | |
| 60 | + | - name: sccache | |
| 61 | + | run: bash scripts/sccache.sh install | |
| 50 | 62 | - name: Tests | |
| 51 | 63 | run: cargo test --workspace --locked --quiet | |
| 64 | + | - name: sccache's hits and misses | |
| 65 | + | if: ${{ always() }} | |
| 66 | + | run: bash scripts/sccache.sh stats | |
| 52 | 67 | ||
| 53 | 68 | typescript: | |
| 54 | 69 | name: TypeScript | |
| 70 | + | if: ${{ github.event_name != 'push' }} | |
| 55 | 71 | runs-on: ubuntu-latest | |
| 56 | 72 | timeout-minutes: 30 | |
| 57 | 73 | steps: | |
| ⋯ | |||
| 71 | 87 | ||
| 72 | 88 | build: | |
| 73 | 89 | name: Build | |
| 90 | + | if: ${{ github.event_name != 'push' }} | |
| 74 | 91 | runs-on: ubuntu-latest | |
| 75 | 92 | timeout-minutes: 30 | |
| 76 | 93 | steps: | |
| 180 | 180 | key: cargo-crates-${{ runner.os }}-${{ hashFiles('Cargo.lock') }} | |
| 181 | 181 | restore-keys: cargo-crates-${{ runner.os }}- | |
| 182 | 182 | # The compiled dependencies of this job's units, for wasm32 and the | |
| 183 | − | # build scripts and proc macros they run. The workspace's own crates | |
| 184 | − | # are compiled again whatever is cached (a checkout's sources are | |
| 185 | − | # newer), so an entry is saved only when the dependencies change: a | |
| 186 | − | # new Cargo.lock, or a new base image (base.json names its Rust). | |
| 187 | − | # Otherwise the nearest earlier entry, of any group, is a start. | |
| 183 | + | # build scripts and proc macros they run. Cargo calls rustc again for | |
| 184 | + | # the workspace's own crates whatever is cached (a checkout's sources | |
| 185 | + | # are newer); sccache, below, answers those calls. So an entry is | |
| 186 | + | # saved only when the dependencies change: a new Cargo.lock, or a new | |
| 187 | + | # base image (base.json names its Rust). Otherwise the nearest earlier | |
| 188 | + | # entry, of any group, is a start. | |
| 188 | 189 | - name: Cache the Cargo target | |
| 189 | 190 | if: ${{ matrix.rust }} | |
| 190 | 191 | uses: actions/cache@v4 | |
| ⋯ | |||
| 215 | 216 | !target/**/incremental | |
| 216 | 217 | key: runner-musl-${{ runner.os }}-${{ hashFiles('Cargo.lock', 'services/runner/base.json') }} | |
| 217 | 218 | restore-keys: runner-musl-${{ runner.os }}- | |
| 219 | + | # Every rustc call that makes a library (the workspace's crates, | |
| 220 | + | # which Cargo compiles again on every checkout, and any dependency | |
| 221 | + | # not restored above) is looked up by its inputs in the repository's | |
| 222 | + | # Actions cache: unchanged crates come back from it instead of being | |
| 223 | + | # compiled. A Worker's own crate (a cdylib) and the runner's binary | |
| 224 | + | # are still compiled. worker-build runs Cargo, so its wasm32 builds | |
| 225 | + | # go through it too; wasm-bindgen and wasm-opt are not rustc. | |
| 226 | + | # Pinned by version and sha256 in scripts/sccache.sh; without the | |
| 227 | + | # cache, the job builds as before. | |
| 228 | + | - name: sccache | |
| 229 | + | if: ${{ matrix.rust || matrix.image }} | |
| 230 | + | run: bash scripts/sccache.sh install | |
| 218 | 231 | - name: Install | |
| 219 | 232 | run: node scripts/deploy.mjs install --only "${{ matrix.units }}" | |
| 220 | 233 | - name: Deploy ${{ matrix.units }} | |
| ⋯ | |||
| 224 | 237 | # are not drafted as incidents. Optional: without it, nothing is sent. | |
| 225 | 238 | STATUS_DEPLOY_TOKEN: ${{ secrets.STATUS_DEPLOY_TOKEN }} | |
| 226 | 239 | run: node scripts/deploy.mjs deploy --only "${{ matrix.units }}" --force --no-migrations --concurrency 2 | |
| 240 | + | # Hits and misses, on the log and in the run's summary. | |
| 241 | + | - name: sccache's hits and misses | |
| 242 | + | if: ${{ always() && (matrix.rust || matrix.image) }} | |
| 243 | + | run: bash scripts/sccache.sh stats | |
| 227 | 244 | ||
| 228 | 245 | edge: | |
| 229 | 246 | name: edge (${{ matrix.group }}) | |
| 1105 | 1105 | "g1t-actions", | |
| 1106 | 1106 | "g1t-scan", | |
| 1107 | 1107 | "hex", | |
| 1108 | + | "libc", | |
| 1108 | 1109 | "serde", | |
| 1109 | 1110 | "serde_json", | |
| 1110 | 1111 | "serde_yaml", |
| 10 | 10 | //! `FinalizeCacheEntryUpload`, `GetCacheEntryDownloadURL`) and the | |
| 11 | 11 | //! artifacts' (`CreateArtifact`, `FinalizeArtifact`, `ListArtifacts`, | |
| 12 | 12 | //! `GetSignedArtifactURL`, `DeleteArtifact`). JSON, the toolkit's field | |
| 13 | − | //! names. | |
| 13 | + | //! names; the cache's methods also in protobuf (`application/protobuf`), | |
| 14 | + | //! which other clients of the protocol send (sccache, through OpenDAL). | |
| 14 | 15 | //! - The cache's older protocol, at `{ACTIONS_CACHE_URL}_apis/artifactcache/…`, | |
| 15 | 16 | //! which the toolkit's client uses whenever the server it runs against is | |
| 16 | − | //! not github.com: on g1t, that is the one it uses. | |
| 17 | + | //! not github.com: on g1t, that is the one it uses, and sccache's too. | |
| 18 | + | //! Its entries are sent in 32 MB chunks, or in one chunk of any size. | |
| 17 | 19 | //! - Blobs, at `/actions/toolkit/blobs/{token}`: the signed links those | |
| 18 | − | //! hand out. Downloads are a plain GET. Uploads speak the part of Azure | |
| 19 | − | //! Blob Storage's protocol the toolkit's client uses (Put Blob, Put | |
| 20 | − | //! Block, Put Block List), mapped onto an R2 multipart upload: a block's | |
| 21 | − | //! id ends in its index, which is its part's number. | |
| 20 | + | //! hand out. Downloads are a GET, of the whole blob or of one byte range | |
| 21 | + | //! (`Range`), as the toolkit's client fetches large entries in segments. | |
| 22 | + | //! Uploads speak the part of Azure Blob Storage's protocol the toolkit's | |
| 23 | + | //! client uses (Put Blob, Put Block, Put Block List), mapped onto an R2 | |
| 24 | + | //! multipart upload: a block's id ends in its index, which is its part's | |
| 25 | + | //! number. An upload link carries a query, as an Azure SAS link does, | |
| 26 | + | //! which clients that sign their requests with it need. | |
| 22 | 27 | //! | |
| 23 | 28 | //! Every call carries the job's runtime token; the actions service checks | |
| 24 | 29 | //! it and keeps the entries (cache.rs, artifacts.rs, runtime.rs there). | |
| ⋯ | |||
| 85 | 90 | ||
| 86 | 91 | // ── Twirp ─────────────────────────────────────────────────────────────────── | |
| 87 | 92 | ||
| 93 | + | /// Twirp's binary encoding. | |
| 94 | + | const PROTOBUF: &str = "application/protobuf"; | |
| 95 | + | ||
| 96 | + | /// Whether a request's `Content-Type` is Twirp's protobuf encoding. | |
| 97 | + | fn is_protobuf(content_type: &str) -> bool { | |
| 98 | + | let kind = content_type.split(';').next().unwrap_or_default().trim().to_ascii_lowercase(); | |
| 99 | + | kind == PROTOBUF || kind == "application/x-protobuf" | |
| 100 | + | } | |
| 101 | + | ||
| 102 | + | /// The cache service's messages in protobuf, read into and written from the | |
| 103 | + | /// JSON the handlers use (`results/api/v1/cache.proto`, field numbers as | |
| 104 | + | /// there). Only what the cache's three methods carry: strings, a repeated | |
| 105 | + | /// string, an int64 and a bool. `metadata` (field 1 of each request) is | |
| 106 | + | /// skipped: the runtime token says whose cache it is. | |
| 107 | + | mod proto { | |
| 108 | + | use serde_json::{Map, Value}; | |
| 109 | + | ||
| 110 | + | #[derive(Clone, Copy)] | |
| 111 | + | enum Kind { | |
| 112 | + | Text, | |
| 113 | + | Texts, | |
| 114 | + | Int, | |
| 115 | + | Bool, | |
| 116 | + | } | |
| 117 | + | ||
| 118 | + | /// A message's fields: number, JSON name, kind. | |
| 119 | + | type Fields = &'static [(u64, &'static str, Kind)]; | |
| 120 | + | ||
| 121 | + | fn request_fields(method: &str) -> Option<Fields> { | |
| 122 | + | Some(match method { | |
| 123 | + | "CreateCacheEntry" => &[(2, "key", Kind::Text), (3, "version", Kind::Text)], | |
| 124 | + | "FinalizeCacheEntryUpload" => &[(2, "key", Kind::Text), (3, "size_bytes", Kind::Int), (4, "version", Kind::Text)], | |
| 125 | + | "GetCacheEntryDownloadURL" => &[(2, "key", Kind::Text), (3, "restore_keys", Kind::Texts), (4, "version", Kind::Text)], | |
| 126 | + | _ => return None, | |
| 127 | + | }) | |
| 128 | + | } | |
| 129 | + | ||
| 130 | + | fn response_fields(method: &str) -> Fields { | |
| 131 | + | match method { | |
| 132 | + | "CreateCacheEntry" => &[(1, "ok", Kind::Bool), (2, "signed_upload_url", Kind::Text), (3, "message", Kind::Text)], | |
| 133 | + | "FinalizeCacheEntryUpload" => &[(1, "ok", Kind::Bool), (2, "entry_id", Kind::Int), (3, "message", Kind::Text)], | |
| 134 | + | "GetCacheEntryDownloadURL" => &[(1, "ok", Kind::Bool), (2, "signed_download_url", Kind::Text), (3, "matched_key", Kind::Text)], | |
| 135 | + | _ => &[], | |
| 136 | + | } | |
| 137 | + | } | |
| 138 | + | ||
| 139 | + | fn varint(bytes: &[u8], at: &mut usize) -> Option<u64> { | |
| 140 | + | let mut value = 0u64; | |
| 141 | + | for shift in (0..64).step_by(7) { | |
| 142 | + | let byte = *bytes.get(*at)?; | |
| 143 | + | *at += 1; | |
| 144 | + | value |= u64::from(byte & 0x7f) << shift; | |
| 145 | + | if byte & 0x80 == 0 { | |
| 146 | + | return Some(value); | |
| 147 | + | } | |
| 148 | + | } | |
| 149 | + | None | |
| 150 | + | } | |
| 151 | + | ||
| 152 | + | fn put_varint(out: &mut Vec<u8>, mut value: u64) { | |
| 153 | + | while value >= 0x80 { | |
| 154 | + | out.push((value as u8 & 0x7f) | 0x80); | |
| 155 | + | value >>= 7; | |
| 156 | + | } | |
| 157 | + | out.push(value as u8); | |
| 158 | + | } | |
| 159 | + | ||
| 160 | + | /// A request of `method` as JSON, or None when it is not one. | |
| 161 | + | pub fn request(method: &str, bytes: &[u8]) -> Option<Value> { | |
| 162 | + | let fields = request_fields(method)?; | |
| 163 | + | let mut out = Map::new(); | |
| 164 | + | let mut at = 0; | |
| 165 | + | while at < bytes.len() { | |
| 166 | + | let tag = varint(bytes, &mut at)?; | |
| 167 | + | let (number, wire) = (tag >> 3, tag & 7); | |
| 168 | + | let known = fields.iter().find(|(n, _, _)| *n == number); | |
| 169 | + | match wire { | |
| 170 | + | 0 => { | |
| 171 | + | let value = varint(bytes, &mut at)?; | |
| 172 | + | if let Some((_, name, Kind::Int)) = known { | |
| 173 | + | // An int64 is sent as its two's complement. | |
| 174 | + | out.insert((*name).to_owned(), Value::String((value as i64).to_string())); | |
| 175 | + | } | |
| 176 | + | } | |
| 177 | + | 2 => { | |
| 178 | + | let length = usize::try_from(varint(bytes, &mut at)?).ok()?; | |
| 179 | + | let end = at.checked_add(length).filter(|end| *end <= bytes.len())?; | |
| 180 | + | let raw = &bytes[at..end]; | |
| 181 | + | at = end; | |
| 182 | + | match known { | |
| 183 | + | Some((_, name, Kind::Text)) => { | |
| 184 | + | out.insert((*name).to_owned(), Value::String(String::from_utf8(raw.to_vec()).ok()?)); | |
| 185 | + | } | |
| 186 | + | Some((_, name, Kind::Texts)) => { | |
| 187 | + | let text = Value::String(String::from_utf8(raw.to_vec()).ok()?); | |
| 188 | + | match out.entry((*name).to_owned()).or_insert_with(|| Value::Array(Vec::new())) { | |
| 189 | + | Value::Array(list) => list.push(text), | |
| 190 | + | _ => return None, | |
| 191 | + | } | |
| 192 | + | } | |
| 193 | + | _ => {} | |
| 194 | + | } | |
| 195 | + | } | |
| 196 | + | 1 => at = at.checked_add(8).filter(|end| *end <= bytes.len())?, | |
| 197 | + | 5 => at = at.checked_add(4).filter(|end| *end <= bytes.len())?, | |
| 198 | + | _ => return None, | |
| 199 | + | } | |
| 200 | + | } | |
| 201 | + | Some(Value::Object(out)) | |
| 202 | + | } | |
| 203 | + | ||
| 204 | + | /// A response of `method` from its JSON. Defaults are left out, as | |
| 205 | + | /// proto3 does. | |
| 206 | + | pub fn response(method: &str, value: &Value) -> Vec<u8> { | |
| 207 | + | let mut out = Vec::new(); | |
| 208 | + | for (number, name, kind) in response_fields(method) { | |
| 209 | + | let field = &value[*name]; | |
| 210 | + | match kind { | |
| 211 | + | Kind::Bool if field.as_bool() == Some(true) => { | |
| 212 | + | put_varint(&mut out, number << 3); | |
| 213 | + | put_varint(&mut out, 1); | |
| 214 | + | } | |
| 215 | + | Kind::Int => { | |
| 216 | + | let n = field.as_i64().or_else(|| field.as_str().and_then(|s| s.parse().ok())).unwrap_or(0); | |
| 217 | + | if n != 0 { | |
| 218 | + | put_varint(&mut out, number << 3); | |
| 219 | + | put_varint(&mut out, n as u64); | |
| 220 | + | } | |
| 221 | + | } | |
| 222 | + | Kind::Text => { | |
| 223 | + | let text = field.as_str().unwrap_or_default(); | |
| 224 | + | if !text.is_empty() { | |
| 225 | + | put_varint(&mut out, (number << 3) | 2); | |
| 226 | + | put_varint(&mut out, text.len() as u64); | |
| 227 | + | out.extend_from_slice(text.as_bytes()); | |
| 228 | + | } | |
| 229 | + | } | |
| 230 | + | _ => {} | |
| 231 | + | } | |
| 232 | + | } | |
| 233 | + | out | |
| 234 | + | } | |
| 235 | + | } | |
| 236 | + | ||
| 88 | 237 | /// A Twirp error: its code and message, at the status Twirp gives it. | |
| 89 | 238 | fn twirp_error(code: &str, message: &str) -> Result<Response> { | |
| 90 | 239 | let status = match code { | |
| ⋯ | |||
| 92 | 241 | "permission_denied" => 403, | |
| 93 | 242 | "not_found" => 404, | |
| 94 | 243 | "already_exists" => 409, | |
| 95 | − | "invalid_argument" => 400, | |
| 244 | + | "invalid_argument" | "malformed" => 400, | |
| 245 | + | "bad_route" => 404, | |
| 96 | 246 | "failed_precondition" => 412, | |
| 97 | 247 | "resource_exhausted" => 429, | |
| 98 | 248 | _ => 500, | |
| ⋯ | |||
| 155 | 305 | }) | |
| 156 | 306 | } | |
| 157 | 307 | ||
| 308 | + | /// The Azure Storage version g1t's blob links answer as. | |
| 309 | + | const AZURE_VERSION: &str = "2024-11-04"; | |
| 310 | + | ||
| 311 | + | /// An upload link: the blob's, with a query as an Azure SAS link has one. | |
| 312 | + | /// A client that treats it as a container, a blob and a SAS token (OpenDAL, | |
| 313 | + | /// which sccache uses) refuses a link without one; the token in the path | |
| 314 | + | /// is what g1t checks. | |
| 315 | + | pub fn upload_url(api: &str, blob: &str) -> String { | |
| 316 | + | format!("{}?sv={AZURE_VERSION}", blob_url(api, blob)) | |
| 317 | + | } | |
| 318 | + | ||
| 158 | 319 | /// Starts an R2 upload for an entry the service reserved, and the signed | |
| 159 | 320 | /// link the toolkit sends it to. | |
| 160 | 321 | async fn start_upload(bucket: &Bucket, services: &Services, job: &str, token: &str, kind: &str, id: &str, object: &str) -> Result<Outcome<String>> { | |
| ⋯ | |||
| 167 | 328 | ) | |
| 168 | 329 | .await?; | |
| 169 | 330 | Ok(match signed { | |
| 170 | − | Outcome::Ok(blob) => Outcome::Ok(blob_url(&services.addresses.api, &blob)), | |
| 331 | + | Outcome::Ok(blob) => Outcome::Ok(upload_url(&services.addresses.api, &blob)), | |
| 171 | 332 | Outcome::Fail(refused) => Outcome::Fail(refused), | |
| 172 | 333 | }) | |
| 173 | 334 | } | |
| ⋯ | |||
| 178 | 339 | let Some(job) = runtime_job(&token) else { | |
| 179 | 340 | return twirp_error("unauthenticated", "Send the job's ACTIONS_RUNTIME_TOKEN as a bearer token."); | |
| 180 | 341 | }; | |
| 181 | − | let body: Value = request.json().await.unwrap_or(Value::Null); | |
| 342 | + | // Twirp clients send JSON or protobuf, and are answered in kind. | |
| 343 | + | let binary = is_protobuf(&request.headers().get("content-type")?.unwrap_or_default()); | |
| 344 | + | let body: Value = if binary { | |
| 345 | + | match proto::request(method, &request.bytes().await.unwrap_or_default()) { | |
| 346 | + | Some(body) => body, | |
| 347 | + | None => return twirp_error("malformed", "That is not a protobuf message this method takes."), | |
| 348 | + | } | |
| 349 | + | } else { | |
| 350 | + | request.json().await.unwrap_or(Value::Null) | |
| 351 | + | }; | |
| 352 | + | let answer = |value: Value| -> Result<Response> { | |
| 353 | + | if binary { | |
| 354 | + | let mut response = Response::from_bytes(proto::response(method, &value))?; | |
| 355 | + | response.headers_mut().set("content-type", PROTOBUF)?; | |
| 356 | + | Ok(response) | |
| 357 | + | } else { | |
| 358 | + | Response::from_json(&value) | |
| 359 | + | } | |
| 360 | + | }; | |
| 182 | 361 | let bucket = env.bucket("ACTIONS_CACHE")?; | |
| 183 | 362 | let actions = &services.actions; | |
| 184 | 363 | let (run, own_job) = backend_ids(&token); | |
| ⋯ | |||
| 201 | 380 | let found: Outcome<Option<CacheHit>> = g1t_kit::call(actions, "cache_lookup", &args).await?; | |
| 202 | 381 | match found { | |
| 203 | 382 | Outcome::Ok(Some(CacheHit { key, blob: Some(blob), .. })) => { | |
| 204 | − | Response::from_json(&json!({ "ok": true, "signed_download_url": blob_url(&services.addresses.api, &blob), "matched_key": key })) | |
| 383 | + | answer(json!({ "ok": true, "signed_download_url": blob_url(&services.addresses.api, &blob), "matched_key": key })) | |
| 205 | 384 | } | |
| 206 | − | Outcome::Ok(_) => Response::from_json(&json!({ "ok": false, "signed_download_url": "", "matched_key": "" })), | |
| 385 | + | Outcome::Ok(_) => answer(json!({ "ok": false, "signed_download_url": "", "matched_key": "" })), | |
| 207 | 386 | Outcome::Fail(refused) => twirp_failure(&refused), | |
| 208 | 387 | } | |
| 209 | 388 | } | |
| ⋯ | |||
| 212 | 391 | let reserved: Outcome<CacheReservation> = g1t_kit::call(actions, "cache_reserve", &args).await?; | |
| 213 | 392 | let reserved = match reserved { | |
| 214 | 393 | Outcome::Ok(reserved) => reserved, | |
| 215 | − | // The client warns with this and goes on, as for a key | |
| 216 | − | // another job is saving. | |
| 217 | − | Outcome::Fail(refused) => return Response::from_json(&json!({ "ok": false, "signed_upload_url": "", "message": refused.message })), | |
| 394 | + | // A key already saved, or being saved by another job, to a | |
| 395 | + | // protobuf client (OpenDAL's) is Twirp's `already_exists` | |
| 396 | + | // (409), which it takes as "someone else has it": sccache | |
| 397 | + | // then still writes. An `ok: false` would read as a broken | |
| 398 | + | // cache, and sccache would only read from it. | |
| 399 | + | Outcome::Fail(refused) if binary && refused.code == FailureCode::Conflict => return twirp_failure(&refused), | |
| 400 | + | // The toolkit's client logs this ("another job may be | |
| 401 | + | // creating this cache") and goes on. | |
| 402 | + | Outcome::Fail(refused) => return answer(json!({ "ok": false, "signed_upload_url": "", "message": refused.message })), | |
| 218 | 403 | }; | |
| 219 | 404 | match start_upload(&bucket, services, &job, &token, "cache", &reserved.id, &reserved.object).await? { | |
| 220 | − | Outcome::Ok(url) => Response::from_json(&json!({ "ok": true, "signed_upload_url": url })), | |
| 221 | − | Outcome::Fail(refused) => Response::from_json(&json!({ "ok": false, "signed_upload_url": "", "message": refused.message })), | |
| 405 | + | Outcome::Ok(url) => answer(json!({ "ok": true, "signed_upload_url": url })), | |
| 406 | + | Outcome::Fail(refused) => answer(json!({ "ok": false, "signed_upload_url": "", "message": refused.message })), | |
| 222 | 407 | } | |
| 223 | 408 | } | |
| 224 | 409 | (CACHE_SERVICE, "FinalizeCacheEntryUpload") => { | |
| ⋯ | |||
| 226 | 411 | let pending: Outcome<CacheReservation> = g1t_kit::call(actions, "cache_upload", &args).await?; | |
| 227 | 412 | let pending = match pending { | |
| 228 | 413 | Outcome::Ok(pending) => pending, | |
| 229 | − | Outcome::Fail(refused) => return Response::from_json(&json!({ "ok": false, "entry_id": "0", "message": refused.message })), | |
| 414 | + | Outcome::Fail(refused) => return answer(json!({ "ok": false, "entry_id": "0", "message": refused.message })), | |
| 230 | 415 | }; | |
| 231 | 416 | let Some(object) = bucket.head(&pending.object).await? else { | |
| 232 | − | return Response::from_json(&json!({ "ok": false, "entry_id": "0", "message": "Nothing was uploaded for that entry." })); | |
| 417 | + | return answer(json!({ "ok": false, "entry_id": "0", "message": "Nothing was uploaded for that entry." })); | |
| 233 | 418 | }; | |
| 234 | 419 | match commit_cache(&bucket, services, &job, &token, &pending.id, object.size()).await? { | |
| 235 | − | Outcome::Ok(()) => Response::from_json(&json!({ "ok": true, "entry_id": pending.number.to_string() })), | |
| 236 | − | Outcome::Fail(refused) => Response::from_json(&json!({ "ok": false, "entry_id": "0", "message": refused.message })), | |
| 420 | + | Outcome::Ok(()) => answer(json!({ "ok": true, "entry_id": pending.number.to_string() })), | |
| 421 | + | Outcome::Fail(refused) => answer(json!({ "ok": false, "entry_id": "0", "message": refused.message })), | |
| 237 | 422 | } | |
| 238 | 423 | } | |
| 239 | 424 | (ARTIFACT_SERVICE, "CreateArtifact") => { | |
| ⋯ | |||
| 321 | 506 | } | |
| 322 | 507 | ||
| 323 | 508 | /// The part a chunk of the older protocol is, from its `Content-Range`: | |
| 324 | − | /// chunks are `CACHE_PART_BYTES` apart, as the toolkit sends them. | |
| 509 | + | /// chunks are `CACHE_PART_BYTES` apart, as the toolkit sends them. A first | |
| 510 | + | /// chunk may be larger, up to `MAX_BLOCK_BYTES`: a client that sends an | |
| 511 | + | /// entry in one request (sccache does) sends a single chunk from 0. | |
| 325 | 512 | pub fn chunk_part(range: &str) -> Option<(u16, u64)> { | |
| 326 | 513 | let range = range.trim().strip_prefix("bytes ")?; | |
| 327 | 514 | let (span, _) = range.split_once('/')?; | |
| 328 | 515 | let (start, end) = span.split_once('-')?; | |
| 329 | 516 | let (start, end): (u64, u64) = (start.trim().parse().ok()?, end.trim().parse().ok()?); | |
| 330 | − | if end < start || start % CACHE_PART_BYTES != 0 || end - start + 1 > CACHE_PART_BYTES { | |
| 517 | + | if end < start { | |
| 331 | 518 | return None; | |
| 332 | 519 | } | |
| 333 | − | Some(((start / CACHE_PART_BYTES + 1) as u16, end - start + 1)) | |
| 520 | + | let length = end - start + 1; | |
| 521 | + | if start == 0 && length <= MAX_BLOCK_BYTES { | |
| 522 | + | return Some((1, length)); | |
| 523 | + | } | |
| 524 | + | if start % CACHE_PART_BYTES != 0 || length > CACHE_PART_BYTES { | |
| 525 | + | return None; | |
| 526 | + | } | |
| 527 | + | Some(((start / CACHE_PART_BYTES + 1) as u16, length)) | |
| 334 | 528 | } | |
| 335 | 529 | ||
| 530 | + | /// The bytes a download's `Range` header asks for, out of `size`: first and | |
| 531 | + | /// last, inclusive. None to send the whole blob (no header, or one this | |
| 532 | + | /// does not read, such as several ranges); `Some(None)` when the range is | |
| 533 | + | /// past the end (416). | |
| 534 | + | pub fn byte_range(header: &str, size: u64) -> Option<Option<(u64, u64)>> { | |
| 535 | + | let spec = header.trim().strip_prefix("bytes=")?.trim(); | |
| 536 | + | if spec.contains(',') { | |
| 537 | + | return None; | |
| 538 | + | } | |
| 539 | + | let (first, last) = spec.split_once('-')?; | |
| 540 | + | let (first, last) = (first.trim(), last.trim()); | |
| 541 | + | let range = if first.is_empty() { | |
| 542 | + | // The last `n` bytes. | |
| 543 | + | let n: u64 = last.parse().ok()?; | |
| 544 | + | if n == 0 || size == 0 { | |
| 545 | + | return Some(None); | |
| 546 | + | } | |
| 547 | + | (size.saturating_sub(n), size - 1) | |
| 548 | + | } else { | |
| 549 | + | let first: u64 = first.parse().ok()?; | |
| 550 | + | let last: u64 = if last.is_empty() { u64::MAX } else { last.parse().ok()? }; | |
| 551 | + | if last < first { | |
| 552 | + | return None; | |
| 553 | + | } | |
| 554 | + | if first >= size { | |
| 555 | + | return Some(None); | |
| 556 | + | } | |
| 557 | + | (first, last.min(size - 1)) | |
| 558 | + | }; | |
| 559 | + | Some(Some(range)) | |
| 560 | + | } | |
| 561 | + | ||
| 336 | 562 | /// `{ACTIONS_CACHE_URL}_apis/artifactcache/…`. `rest` is the path after it. | |
| 337 | 563 | pub async fn cache_v1(mut request: Request, env: &Env, services: &Services, method: &str, rest: &str) -> Result<Response> { | |
| 338 | 564 | let token = bearer(&request); | |
| ⋯ | |||
| 527 | 753 | headers.set("content-length", &size.to_string())?; | |
| 528 | 754 | headers.set("content-type", grant.content_type.as_deref().unwrap_or("application/octet-stream"))?; | |
| 529 | 755 | headers.set("x-ms-blob-type", "BlockBlob")?; | |
| 756 | + | headers.set("accept-ranges", "bytes")?; | |
| 530 | 757 | if let Some(name) = &grant.filename { | |
| 531 | 758 | headers.set("content-disposition", &format!("attachment; filename=\"{}\"", name.replace('"', "")))?; | |
| 532 | 759 | } | |
| ⋯ | |||
| 538 | 765 | headers(&mut response, object.size())?; | |
| 539 | 766 | return Ok(response); | |
| 540 | 767 | } | |
| 768 | + | // One byte range (`Range`, or Azure's `x-ms-range`): the | |
| 769 | + | // toolkit's client fetches a large entry in segments, side by | |
| 770 | + | // side, and writes each where its range says. | |
| 771 | + | let asked = match request.headers().get("x-ms-range")? { | |
| 772 | + | Some(range) => Some(range), | |
| 773 | + | None => request.headers().get("range")?, | |
| 774 | + | }; | |
| 775 | + | if let Some(asked) = asked.filter(|r| !r.trim().is_empty()) { | |
| 776 | + | let Some(object) = bucket.head(&grant.object).await? else { return azure_error(404, "BlobNotFound", "It is gone.") }; | |
| 777 | + | let size = object.size(); | |
| 778 | + | match byte_range(&asked, size) { | |
| 779 | + | Some(Some((first, last))) => { | |
| 780 | + | let length = last - first + 1; | |
| 781 | + | let Some(object) = bucket.get(&grant.object).range(worker::Range::OffsetWithLength { offset: first, length }).execute().await? else { | |
| 782 | + | return azure_error(404, "BlobNotFound", "It is gone."); | |
| 783 | + | }; | |
| 784 | + | let Some(body) = object.body() else { return azure_error(404, "BlobNotFound", "It is gone.") }; | |
| 785 | + | let mut response = Response::from_body(body.response_body()?)?.with_status(206); | |
| 786 | + | headers(&mut response, length)?; | |
| 787 | + | response.headers_mut().set("content-range", &format!("bytes {first}-{last}/{size}"))?; | |
| 788 | + | return Ok(response); | |
| 789 | + | } | |
| 790 | + | Some(None) => { | |
| 791 | + | let mut response = azure_error(416, "InvalidRange", "The range is past the end of the blob.")?; | |
| 792 | + | response.headers_mut().set("content-range", &format!("bytes */{size}"))?; | |
| 793 | + | return Ok(response); | |
| 794 | + | } | |
| 795 | + | // Not a range this reads: the whole blob. | |
| 796 | + | None => {} | |
| 797 | + | } | |
| 798 | + | } | |
| 541 | 799 | let Some(object) = bucket.get(&grant.object).execute().await? else { return azure_error(404, "BlobNotFound", "It is gone.") }; | |
| 542 | 800 | let size = object.size(); | |
| 543 | 801 | let Some(body) = object.body() else { return azure_error(404, "BlobNotFound", "It is gone.") }; | |
| ⋯ | |||
| 674 | 932 | let mb32 = CACHE_PART_BYTES; | |
| 675 | 933 | assert_eq!(chunk_part(&format!("bytes 0-{}/*", mb32 - 1)), Some((1, mb32))); | |
| 676 | 934 | assert_eq!(chunk_part(&format!("bytes {}-{}/*", mb32 * 2, mb32 * 2 + 99)), Some((3, 100))); | |
| 677 | − | // Not on a chunk's boundary, too long, or not a range. | |
| 935 | + | // A whole entry in one chunk, as sccache (OpenDAL) sends it: its | |
| 936 | + | // check file is 13 bytes, a compiled crate can be well over 32 MB. | |
| 937 | + | assert_eq!(chunk_part("bytes 0-12/*"), Some((1, 13))); | |
| 938 | + | assert_eq!(chunk_part(&format!("bytes 0-{}/*", mb32)), Some((1, mb32 + 1))); | |
| 939 | + | assert_eq!(chunk_part(&format!("bytes 0-{}/*", MAX_BLOCK_BYTES - 1)), Some((1, MAX_BLOCK_BYTES))); | |
| 940 | + | // Not on a chunk's boundary, too long, backwards, or not a range. | |
| 678 | 941 | assert_eq!(chunk_part("bytes 5-10/*"), None); | |
| 679 | − | assert_eq!(chunk_part(&format!("bytes 0-{}/*", mb32)), None); | |
| 942 | + | assert_eq!(chunk_part(&format!("bytes {mb32}-{}/*", mb32 * 2)), None); | |
| 943 | + | assert_eq!(chunk_part(&format!("bytes 0-{}/*", MAX_BLOCK_BYTES)), None); | |
| 944 | + | assert_eq!(chunk_part("bytes 10-5/*"), None); | |
| 680 | 945 | assert_eq!(chunk_part("0-10"), None); | |
| 681 | 946 | } | |
| 682 | 947 | ||
| 948 | + | #[test] | |
| 949 | + | fn downloads_read_one_byte_range() { | |
| 950 | + | // OpenDAL's stat: the first byte. | |
| 951 | + | assert_eq!(byte_range("bytes=0-0", 100), Some(Some((0, 0)))); | |
| 952 | + | // The toolkit's segments, the last one cut at the end. | |
| 953 | + | assert_eq!(byte_range("bytes=0-49", 100), Some(Some((0, 49)))); | |
| 954 | + | assert_eq!(byte_range("bytes=50-999", 100), Some(Some((50, 99)))); | |
| 955 | + | assert_eq!(byte_range("bytes=90-", 100), Some(Some((90, 99)))); | |
| 956 | + | assert_eq!(byte_range("bytes=-10", 100), Some(Some((90, 99)))); | |
| 957 | + | assert_eq!(byte_range("bytes=-1000", 100), Some(Some((0, 99)))); | |
| 958 | + | // Past the end: 416. | |
| 959 | + | assert_eq!(byte_range("bytes=100-200", 100), Some(None)); | |
| 960 | + | assert_eq!(byte_range("bytes=0-0", 0), Some(None)); | |
| 961 | + | assert_eq!(byte_range("bytes=-0", 100), Some(None)); | |
| 962 | + | // Not read: the whole blob. | |
| 963 | + | assert_eq!(byte_range("bytes=0-1,5-6", 100), None); | |
| 964 | + | assert_eq!(byte_range("bytes=9-3", 100), None); | |
| 965 | + | assert_eq!(byte_range("items=0-1", 100), None); | |
| 966 | + | assert_eq!(byte_range("bytes=a-b", 100), None); | |
| 967 | + | } | |
| 968 | + | ||
| 969 | + | #[test] | |
| 970 | + | fn upload_links_carry_a_query_as_sas_links_do() { | |
| 971 | + | let url = upload_url("https://api.g1t.sh", "tok.sig"); | |
| 972 | + | assert_eq!(url, "https://api.g1t.sh/actions/toolkit/blobs/tok.sig?sv=2024-11-04"); | |
| 973 | + | // How OpenDAL reads a signed upload link: a container, a blob in | |
| 974 | + | // it, and a SAS query, all of which must be there. | |
| 975 | + | let rest = url.strip_prefix("https://api.g1t.sh/").unwrap(); | |
| 976 | + | let (path, query) = rest.split_once('?').unwrap(); | |
| 977 | + | let (container, blob) = path.split_once('/').unwrap(); | |
| 978 | + | assert_eq!((container, blob, query), ("actions", "toolkit/blobs/tok.sig", "sv=2024-11-04")); | |
| 979 | + | } | |
| 980 | + | ||
| 981 | + | /// A protobuf length-delimited field, as prost writes it. | |
| 982 | + | fn pb_text(number: u8, text: &str) -> Vec<u8> { | |
| 983 | + | let mut out = vec![(number << 3) | 2, text.len() as u8]; | |
| 984 | + | out.extend_from_slice(text.as_bytes()); | |
| 985 | + | out | |
| 986 | + | } | |
| 987 | + | ||
| 988 | + | /// The requests sccache 0.18 sends (OpenDAL 0.58's `ghac` service, with | |
| 989 | + | /// prost): fields in number order, defaults left out, no metadata. | |
| 990 | + | #[test] | |
| 991 | + | fn twirp_reads_sccaches_protobuf_requests() { | |
| 992 | + | assert!(is_protobuf("application/protobuf")); | |
| 993 | + | assert!(is_protobuf("Application/Protobuf; charset=utf-8")); | |
| 994 | + | assert!(!is_protobuf("application/json")); | |
| 995 | + | assert!(!is_protobuf("")); | |
| 996 | + | ||
| 997 | + | let key = "sccache/f/c/b/fcb0a1d2e3"; | |
| 998 | + | let version = "sccache-v0.18.0"; | |
| 999 | + | let create = [pb_text(2, key), pb_text(3, version)].concat(); | |
| 1000 | + | let read = proto::request("CreateCacheEntry", &create).unwrap(); | |
| 1001 | + | assert_eq!((text(&read, "key"), text(&read, "version")), (key.to_owned(), version.to_owned())); | |
| 1002 | + | ||
| 1003 | + | // size_bytes is field 3, a varint: 300 is 0xac 0x02. | |
| 1004 | + | let finalize = [pb_text(2, key), vec![0x18, 0xac, 0x02], pb_text(4, version)].concat(); | |
| 1005 | + | let read = proto::request("FinalizeCacheEntryUpload", &finalize).unwrap(); | |
| 1006 | + | assert_eq!(number(&read, "size_bytes"), Some(300)); | |
| 1007 | + | assert_eq!(text(&read, "version"), version); | |
| 1008 | + | ||
| 1009 | + | let lookup = [pb_text(2, key), pb_text(4, version)].concat(); | |
| 1010 | + | let read = proto::request("GetCacheEntryDownloadURL", &lookup).unwrap(); | |
| 1011 | + | assert_eq!(text(&read, "key"), key); | |
| 1012 | + | assert!(field(&read, "restore_keys").is_null()); | |
| 1013 | + | // The toolkit's own lookup, with metadata (skipped) and restore keys. | |
| 1014 | + | let metadata = vec![0x0a, 0x02, 0x08, 0x07]; | |
| 1015 | + | let with_restore = [metadata, pb_text(2, "k"), pb_text(3, "k-"), pb_text(3, "x-"), pb_text(4, "v")].concat(); | |
| 1016 | + | let read = proto::request("GetCacheEntryDownloadURL", &with_restore).unwrap(); | |
| 1017 | + | assert_eq!(field(&read, "restore_keys"), &json!(["k-", "x-"])); | |
| 1018 | + | assert_eq!(text(&read, "version"), "v"); | |
| 1019 | + | ||
| 1020 | + | // The same three, as prost 0.14 encodes them with OpenDAL's | |
| 1021 | + | // generated types (the `ghac` crate, 0.3.0), byte for byte. | |
| 1022 | + | let recorded = |hex: &str| -> Vec<u8> { (0..hex.len()).step_by(2).map(|i| u8::from_str_radix(&hex[i..i + 2], 16).unwrap()).collect() }; | |
| 1023 | + | let prefix = "1218736363616368652f662f632f622f66636230613164326533"; | |
| 1024 | + | let suffix = "0f736363616368652d76302e31382e30"; | |
| 1025 | + | assert_eq!(recorded(&format!("{prefix}1a{suffix}")), create); | |
| 1026 | + | assert_eq!(recorded(&format!("{prefix}18ac0222{suffix}")), finalize); | |
| 1027 | + | assert_eq!(recorded(&format!("{prefix}22{suffix}")), lookup); | |
| 1028 | + | ||
| 1029 | + | // Cut short, or not a cache method. | |
| 1030 | + | assert!(proto::request("CreateCacheEntry", &create[..create.len() - 1]).is_none()); | |
| 1031 | + | assert!(proto::request("CreateCacheEntry", &[0x12, 0xff]).is_none()); | |
| 1032 | + | assert!(proto::request("CreateArtifact", &create).is_none()); | |
| 1033 | + | assert_eq!(proto::request("CreateCacheEntry", &[]), Some(json!({}))); | |
| 1034 | + | } | |
| 1035 | + | ||
| 1036 | + | #[test] | |
| 1037 | + | fn twirp_answers_in_protobuf_as_prost_reads_it() { | |
| 1038 | + | let url = "https://api.g1t.sh/actions/toolkit/blobs/t?sv=2024-11-04"; | |
| 1039 | + | let created = proto::response("CreateCacheEntry", &json!({ "ok": true, "signed_upload_url": url })); | |
| 1040 | + | assert_eq!(created, [vec![0x08, 0x01], pb_text(2, url)].concat()); | |
| 1041 | + | // Refused: ok false is the default, so only the message is sent. | |
| 1042 | + | let refused = proto::response("CreateCacheEntry", &json!({ "ok": false, "signed_upload_url": "", "message": "no" })); | |
| 1043 | + | assert_eq!(refused, pb_text(3, "no")); | |
| 1044 | + | // entry_id is an int64, given as a string in JSON. | |
| 1045 | + | let finalized = proto::response("FinalizeCacheEntryUpload", &json!({ "ok": true, "entry_id": "300" })); | |
| 1046 | + | assert_eq!(finalized, vec![0x08, 0x01, 0x10, 0xac, 0x02]); | |
| 1047 | + | let found = proto::response("GetCacheEntryDownloadURL", &json!({ "ok": true, "signed_download_url": "u", "matched_key": "k" })); | |
| 1048 | + | assert_eq!(found, [vec![0x08, 0x01], pb_text(2, "u"), pb_text(3, "k")].concat()); | |
| 1049 | + | // A miss is an empty message: ok false. | |
| 1050 | + | assert!(proto::response("GetCacheEntryDownloadURL", &json!({ "ok": false, "signed_download_url": "", "matched_key": "" })).is_empty()); | |
| 1051 | + | } | |
| 1052 | + | ||
| 683 | 1053 | /// The toolkit's requests, as `@actions/cache` 4 and `@actions/artifact` | |
| 684 | 1054 | /// 2 send them (protobuf-ts, proto field names, no defaults). | |
| 685 | 1055 | #[test] | |
| 48 | 48 | | `actions/upload-artifact`, `actions/download-artifact`, `actions/upload-artifact/merge` | The same inputs and outputs as version 4: `retention-days`, `overwrite`, `compression-level`, `include-hidden-files`, `!` exclusions, download by `pattern` with `merge-multiple`, and from another run with `run-id` and `github-token`. Up to 5 GiB each; see [artifacts](#artifacts). | | |
| 49 | 49 | | `actions/cache`, `actions/cache/restore`, `actions/cache/save` | Kept per repository and branch, found by `key` or the newest under a `restore-keys` prefix. `path` takes globs and `!` exclusions. Up to 2 GiB each; see [the cache](#the-cache). | | |
| 50 | 50 | | Actions that cache through the toolkit, such as `actions/setup-node` with `cache: npm` or `Swatinem/rust-cache` | The same: they save to and restore from the repository's cache. See [actions built on the toolkit](#actions-built-on-the-toolkit). | | |
| 51 | + | | sccache with `SCCACHE_GHA_ENABLED` | The same: its entries go to the repository's cache. See [caching Rust builds](#caching-rust-builds). | | |
| 51 | 52 | | `permissions: id-token: write` | The job can ask for an OIDC token, and trade it for a cloud provider's credentials. See [OIDC tokens](#oidc-tokens). | | |
| 52 | 53 | | `docker build`, `push`, `run`, `login`, `compose`, Buildx | The same, with a Docker Engine of the job's own. See [Docker](#docker). | | |
| 53 | 54 | | `services:` | The same: each service starts before the steps, health checks are waited for, and it is reached at `localhost` on its port and by its name. | | |
| ⋯ | |||
| 604 | 605 | `actions/download-artifact` and `actions/upload-artifact/merge` work: | |
| 605 | 606 | g1t runs those itself. | |
| 606 | 607 | ||
| 608 | + | ## Caching Rust builds | |
| 609 | + | ||
| 610 | + | A checkout gives every file a new modification time, so Cargo compiles | |
| 611 | + | your workspace's own crates again even when `target/` was restored from | |
| 612 | + | the cache. [sccache](https://github.com/mozilla/sccache) caches each | |
| 613 | + | compiler call by what it compiles (the source, the flags, the toolchain | |
| 614 | + | and the dependencies), so a crate that did not change comes back from the | |
| 615 | + | cache instead. Its GitHub Actions backend works on g1t as it is: it keeps | |
| 616 | + | its entries in the repository's cache, under [the cache's](#the-cache) | |
| 617 | + | limits and branch rules. | |
| 618 | + | ||
| 619 | + | 1. Install sccache, pinned to a release and checked against its checksum. | |
| 620 | + | 2. Set `RUSTC_WRAPPER: sccache` and `SCCACHE_GHA_ENABLED: "true"`. The | |
| 621 | + | job's `ACTIONS_RUNTIME_TOKEN` and `ACTIONS_CACHE_URL` are already in | |
| 622 | + | every step's environment; no step needs to export them. | |
| 623 | + | 3. Set `CARGO_INCREMENTAL: "0"`: sccache does not cache incremental | |
| 624 | + | builds, and a fresh checkout gains nothing from them. | |
| 625 | + | ||
| 626 | + | ```yaml | |
| 627 | + | name: CI | |
| 628 | + | on: | |
| 629 | + | pull_request: | |
| 630 | + | push: | |
| 631 | + | branches: [main] | |
| 632 | + | ||
| 633 | + | jobs: | |
| 634 | + | test: | |
| 635 | + | runs-on: ubuntu-latest | |
| 636 | + | env: | |
| 637 | + | RUSTC_WRAPPER: sccache | |
| 638 | + | SCCACHE_GHA_ENABLED: "true" | |
| 639 | + | CARGO_INCREMENTAL: "0" | |
| 640 | + | steps: | |
| 641 | + | - uses: actions/checkout@v5 | |
| 642 | + | - name: Install sccache | |
| 643 | + | env: | |
| 644 | + | VERSION: v0.18.0 | |
| 645 | + | SHA256: 45f1447fbe231e3037bde351ef70677dd212216c8d62ae7ca409fecc4d6acc89 | |
| 646 | + | run: | | |
| 647 | + | name="sccache-$VERSION-x86_64-unknown-linux-musl" | |
| 648 | + | curl -fsSL -o "$RUNNER_TEMP/sccache.tar.gz" \ | |
| 649 | + | "https://github.com/mozilla/sccache/releases/download/$VERSION/$name.tar.gz" | |
| 650 | + | echo "$SHA256 $RUNNER_TEMP/sccache.tar.gz" | sha256sum -c - | |
| 651 | + | tar -xzf "$RUNNER_TEMP/sccache.tar.gz" -C "$RUNNER_TEMP" | |
| 652 | + | install -m 0755 "$RUNNER_TEMP/$name/sccache" "$HOME/.cargo/bin/sccache" | |
| 653 | + | - run: cargo test --workspace --locked | |
| 654 | + | - name: sccache's hits and misses | |
| 655 | + | if: always() | |
| 656 | + | run: | | |
| 657 | + | echo '```' >> "$GITHUB_STEP_SUMMARY" | |
| 658 | + | sccache --show-stats | tee -a "$GITHUB_STEP_SUMMARY" | |
| 659 | + | echo '```' >> "$GITHUB_STEP_SUMMARY" | |
| 660 | + | ``` | |
| 661 | + | ||
| 662 | + | What to expect: | |
| 663 | + | ||
| 664 | + | | | | | |
| 665 | + | | --- | --- | | |
| 666 | + | | What is cached | Every library crate (`rlib`): your workspace's crates and their dependencies. | | |
| 667 | + | | What is compiled each time | What rustc links: binaries, `cdylib` crates, proc macros, build scripts and test harnesses. | | |
| 668 | + | | Its entries | One per compiled crate, often thousands for a workspace, each a few KB to a few MB. They count toward the repository's 10 GiB like any other entry, and the ones not restored for 7 days are deleted. | | |
| 669 | + | | Branches | A pull request reads what the default branch saved. Run the workflow on pushes to the default branch too, as above, or each pull request's first run starts with an empty cache. | | |
| 670 | + | | Starting over | Set `SCCACHE_GHA_VERSION` to any new value. A new sccache release starts over by itself. | | |
| 671 | + | | If the cache cannot be reached | sccache refuses to start. Set `SCCACHE_IGNORE_SERVER_IO_ERROR: "1"` to have rustc run without it instead. | | |
| 672 | + | ||
| 673 | + | sccache adds to `actions/cache` rather than replacing it: keep caching | |
| 674 | + | `~/.cargo/registry/cache` and `target/` keyed by `Cargo.lock`, so Cargo | |
| 675 | + | does not call rustc at all for dependencies that did not change, and | |
| 676 | + | sccache answers the calls it still makes for your own crates. | |
| 677 | + | `Swatinem/rust-cache` works the same way. | |
| 678 | + | ||
| 607 | 679 | ## OIDC tokens | |
| 608 | 680 | ||
| 609 | 681 | A job can prove which repository, branch and environment it runs for with | |
| 83 | 83 | return `${Math.floor(minutes / 60)}h ${minutes % 60}m`; | |
| 84 | 84 | } | |
| 85 | 85 | ||
| 86 | + | /** | |
| 87 | + | * How long something ran, as text. While it is still running the server and | |
| 88 | + | * the browser count to a different now, so the text may differ on hydration | |
| 89 | + | * (as TimeAgo's does). | |
| 90 | + | */ | |
| 91 | + | export function Duration({ start, end, className }: { start: string | null; end: string | null; className?: string }) { | |
| 92 | + | return ( | |
| 93 | + | <span className={className} suppressHydrationWarning> | |
| 94 | + | {duration(start, end)} | |
| 95 | + | </span> | |
| 96 | + | ); | |
| 97 | + | } | |
| 98 | + | ||
| 86 | 99 | /** `main` from `refs/heads/main`, `v1.2` from a tag, `#12` for a pull request. */ | |
| 87 | 100 | export function shortRef(ref: string): string { | |
| 88 | 101 | const pull = /^refs\/pull\/(\d+)\//.exec(ref); |
| 231 | 231 | {name} | |
| 232 | 232 | {detail && <span className="text-muted"> — {detail}</span>} | |
| 233 | 233 | </span> | |
| 234 | − | {time && <span className="shrink-0 animate-fade-in font-mono text-xs text-faint">{time}</span>} | |
| 234 | + | {time && ( | |
| 235 | + | // A running job's time counts to now, which differs between server and browser. | |
| 236 | + | <span className="shrink-0 animate-fade-in font-mono text-xs text-faint" suppressHydrationWarning> | |
| 237 | + | {time} | |
| 238 | + | </span> | |
| 239 | + | )} | |
| 235 | 240 | {timing && <SkeletonLine className="w-10 shrink-0 text-xs" />} | |
| 236 | 241 | {to && ( | |
| 237 | 242 | <Link to={to} className="shrink-0 text-xs text-muted hover:text-fg hover:underline"> |
| 31 | 31 | ||
| 32 | 32 | import type { Route } from "./+types/actions-run"; | |
| 33 | 33 | import { page } from "../../lib/meta"; | |
| 34 | − | import { LogText, Notes, StatusIcon, duration, shortRef, standingWord, useJobLog } from "../../components/actions"; | |
| 34 | + | import { Duration, LogText, Notes, StatusIcon, shortRef, standingWord, useJobLog } from "../../components/actions"; | |
| 35 | 35 | import { Markdown } from "../../components/markdown"; | |
| 36 | 36 | import { Button, ErrorText, SubmitButton, TimeAgo, usePending } from "../../components/ui"; | |
| 37 | 37 | import { CheckboxOption } from "../../components/ui/checkbox"; | |
| ⋯ | |||
| 150 | 150 | {matches} {matches === 1 ? "line" : "lines"} | |
| 151 | 151 | </span> | |
| 152 | 152 | )} | |
| 153 | − | <span className="ml-auto shrink-0 font-mono text-xs text-faint">{duration(step.startedAt, step.finishedAt)}</span> | |
| 153 | + | <Duration className="ml-auto shrink-0 font-mono text-xs text-faint" start={step.startedAt} end={step.finishedAt} /> | |
| 154 | 154 | </summary> | |
| 155 | 155 | <div className="border-t border-line bg-bg/60"> | |
| 156 | 156 | {text === undefined ? ( | |
| ⋯ | |||
| 365 | 365 | <h3 className="text-base font-semibold">{job.name}</h3> | |
| 366 | 366 | <span className="text-sm text-muted"> | |
| 367 | 367 | {job.cancelling ? "Cancelling: running its cleanup steps" : standingWord({ ...job, of: "job" })} | |
| 368 | − | {job.startedAt && ` · ${duration(job.startedAt, job.finishedAt)}`} | |
| 368 | + | {job.startedAt && ( | |
| 369 | + | <> | |
| 370 | + | {" · "} | |
| 371 | + | <Duration start={job.startedAt} end={job.finishedAt} /> | |
| 372 | + | </> | |
| 373 | + | )} | |
| 369 | 374 | </span> | |
| 370 | 375 | <span className="ml-auto flex items-center gap-2"> | |
| 371 | 376 | {rerun} | |
| ⋯ | |||
| 776 | 781 | {run.event} | |
| 777 | 782 | {run.actor && ` by ${run.actor}`} · <TimeAgo at={run.createdAt} /> | |
| 778 | 783 | </span> | |
| 779 | − | {run.startedAt && <span className="font-mono text-xs">{duration(run.startedAt, run.finishedAt)}</span>} | |
| 784 | + | {run.startedAt && <Duration className="font-mono text-xs" start={run.startedAt} end={run.finishedAt} />} | |
| 780 | 785 | {run.attempt > 1 && attempts.length <= 1 && <span>Attempt {run.attempt}</span>} | |
| 781 | 786 | {cancelling && <span className="text-warn">Its jobs are running their cleanup steps</span>} | |
| 782 | 787 | {detail.approval?.state === "approved" && detail.approval.approvedBy && <span>Approved by {detail.approval.approvedBy}</span>} | |
| ⋯ | |||
| 868 | 873 | > | |
| 869 | 874 | <StatusIcon status={job.status} conclusion={job.conclusion} of="job" environment={job.environment} size={14} /> | |
| 870 | 875 | <span className="min-w-0 truncate">{job.name}</span> | |
| 871 | − | <span className="ml-auto shrink-0 font-mono text-xs text-faint">{duration(job.startedAt, job.finishedAt)}</span> | |
| 876 | + | <Duration className="ml-auto shrink-0 font-mono text-xs text-faint" start={job.startedAt} end={job.finishedAt} /> | |
| 872 | 877 | </Link> | |
| 873 | 878 | ))} | |
| 874 | 879 | </nav> | |
| 6 | 6 | ||
| 7 | 7 | import type { Route } from "./+types/actions"; | |
| 8 | 8 | import { page } from "../../lib/meta"; | |
| 9 | − | import { Notes, StatusIcon, duration, shortRef } from "../../components/actions"; | |
| 9 | + | import { Duration, Notes, StatusIcon, shortRef } from "../../components/actions"; | |
| 10 | 10 | import { AddCiPrompt } from "../../components/add-ci"; | |
| 11 | 11 | import { Button, ComputeNote, CopyLine, EmptyState, ErrorText, SubmitButton, TimeAgo, usePending } from "../../components/ui"; | |
| 12 | 12 | import { CheckboxOption } from "../../components/ui/checkbox"; | |
| ⋯ | |||
| 120 | 120 | </span> | |
| 121 | 121 | <span className="w-28 shrink-0 text-right text-xs text-faint"> | |
| 122 | 122 | <TimeAgo at={run.createdAt} /> | |
| 123 | − | {run.startedAt && <span className="block font-mono">{duration(run.startedAt, run.finishedAt)}</span>} | |
| 123 | + | {run.startedAt && <Duration className="block font-mono" start={run.startedAt} end={run.finishedAt} />} | |
| 124 | 124 | </span> | |
| 125 | 125 | </Link> | |
| 126 | 126 | </li> | |
| 236 | 236 | assert!(source.contains("target/x86_64-unknown-linux-musl/release")); | |
| 237 | 237 | } | |
| 238 | 238 | ||
| 239 | + | /// Whether a step runs, for a job going `status`, with `matrix`. | |
| 240 | + | fn step_runs(step: &workflow::Step, matrix: Value, status: Status) -> bool { | |
| 241 | + | let mut contexts = Map::new(); | |
| 242 | + | contexts.insert("matrix".into(), matrix); | |
| 243 | + | let scope = Scope { contexts: &contexts, status, hash_files: None }; | |
| 244 | + | expr::condition(step.condition.as_deref().unwrap_or_default(), &scope).unwrap() | |
| 245 | + | } | |
| 246 | + | ||
| 247 | + | #[test] | |
| 248 | + | fn rust_builds_go_through_sccache_before_cargo_and_report_after() { | |
| 249 | + | let step = |job: &workflow::Job, run: &str| -> (usize, workflow::Step) { | |
| 250 | + | let at = job.steps.iter().position(|s| s.run.as_deref() == Some(run)).unwrap_or_else(|| panic!("{}: no step runs {run}", job.id)); | |
| 251 | + | (at, job.steps[at].clone()) | |
| 252 | + | }; | |
| 253 | + | let deploy = read("deploy.yml"); | |
| 254 | + | for stage in ["core", "edge", "front"] { | |
| 255 | + | let job = deploy.jobs.iter().find(|j| j.id == stage).unwrap(); | |
| 256 | + | let (install_at, install) = step(job, "bash scripts/sccache.sh install"); | |
| 257 | + | let (stats_at, stats) = step(job, "bash scripts/sccache.sh stats"); | |
| 258 | + | // Before anything runs Cargo (worker-build's install, the build), | |
| 259 | + | // and the statistics last. | |
| 260 | + | let install_step = job.steps.iter().position(|s| s.name.as_deref() == Some("Install")).unwrap(); | |
| 261 | + | assert!(install_at < install_step, "{stage}"); | |
| 262 | + | assert_eq!(stats_at, job.steps.len() - 1, "{stage}"); | |
| 263 | + | for (rust, image, uses) in [(true, false, true), (false, true, true), (false, false, false)] { | |
| 264 | + | let matrix = json!({ "group": "g", "units": "u", "rust": rust, "image": image }); | |
| 265 | + | assert_eq!(step_runs(&install, matrix.clone(), Status::Success), uses, "{stage} rust={rust} image={image}"); | |
| 266 | + | assert_eq!(step_runs(&stats, matrix.clone(), Status::Success), uses, "{stage}"); | |
| 267 | + | // A failed build still says what was cached. | |
| 268 | + | assert_eq!(step_runs(&stats, matrix, Status::Failure), uses, "{stage}"); | |
| 269 | + | } | |
| 270 | + | } | |
| 271 | + | let ci = read("ci.yml"); | |
| 272 | + | let rust = ci.jobs.iter().find(|j| j.id == "rust").unwrap(); | |
| 273 | + | let (install_at, _) = step(rust, "bash scripts/sccache.sh install"); | |
| 274 | + | let (tests_at, _) = step(rust, "cargo test --workspace --locked --quiet"); | |
| 275 | + | let (stats_at, stats) = step(rust, "bash scripts/sccache.sh stats"); | |
| 276 | + | assert!(install_at < tests_at && tests_at < stats_at); | |
| 277 | + | assert!(step_runs(&stats, json!({}), Status::Failure)); | |
| 278 | + | // main's runs keep the caches pull requests restore from: only Rust. | |
| 279 | + | assert!(ci.trigger("push").unwrap().branches.allows("main")); | |
| 280 | + | assert!(ci.trigger("pull_request").is_some()); | |
| 281 | + | for (id, on_push) in [("rust", true), ("typescript", false), ("build", false)] { | |
| 282 | + | let job = ci.jobs.iter().find(|j| j.id == id).unwrap(); | |
| 283 | + | for (event, expected) in [("push", on_push), ("pull_request", true)] { | |
| 284 | + | let mut contexts = Map::new(); | |
| 285 | + | contexts.insert("github".into(), json!({ "event_name": event, "ref": "refs/heads/main" })); | |
| 286 | + | let scope = Scope { contexts: &contexts, status: Status::Success, hash_files: None }; | |
| 287 | + | assert_eq!(expr::condition(job.condition.as_deref().unwrap_or_default(), &scope).unwrap(), expected, "{id} on {event}"); | |
| 288 | + | } | |
| 289 | + | } | |
| 290 | + | // The download is pinned by version and checksum. | |
| 291 | + | let script = std::fs::read_to_string(workflows_dir().join("../../scripts/sccache.sh")).unwrap(); | |
| 292 | + | assert!(script.contains("VERSION=0.18.0")); | |
| 293 | + | assert!(script.lines().any(|line| line.strip_prefix("SHA256=").is_some_and(|sum| sum.len() == 64))); | |
| 294 | + | assert!(script.contains("sha256sum -c")); | |
| 295 | + | assert!(script.contains("RUSTC_WRAPPER=")); | |
| 296 | + | assert!(script.contains("GITHUB_STEP_SUMMARY")); | |
| 297 | + | } | |
| 298 | + | ||
| 239 | 299 | #[test] | |
| 240 | 300 | fn the_runner_base_rebuilds_weekly_on_a_machine_with_docker() { | |
| 241 | 301 | let base = read("runner-base.yml"); |
| 25 | 25 | # native-certs: a guarded sandbox re-signs HTTPS with a certificate the | |
| 26 | 26 | # runner adds to the system store (guard.rs), which webpki-roots never sees. | |
| 27 | 27 | ureq = { version = "2", features = ["json", "native-certs"] } | |
| 28 | + | ||
| 29 | + | [target.'cfg(unix)'.dependencies] | |
| 30 | + | # A step's signals back to their defaults before it runs (actions/process.rs). | |
| 31 | + | libc = "0.2" |
| 130 | 130 | ||
| 131 | 131 | /// Sends `signal` to the process's group (it leads its own), so what the | |
| 132 | 132 | /// step started hears it too. Windows has no signals: it is left to `kill`. | |
| 133 | + | /// Sent with kill(2) itself: a machine without the `kill` program (procps | |
| 134 | + | /// is not in every image) would otherwise give the step no SIGINT at all, | |
| 135 | + | /// and leave what it started running after the step was killed. | |
| 136 | + | #[cfg(unix)] | |
| 133 | 137 | fn signal(child: &std::process::Child, signal: &str) { | |
| 134 | − | if cfg!(unix) { | |
| 135 | − | let _ = Command::new("kill") | |
| 136 | − | .args([format!("-{signal}"), "--".into(), format!("-{}", child.id())]) | |
| 137 | − | .stdout(Stdio::null()) | |
| 138 | − | .stderr(Stdio::null()) | |
| 139 | − | .status(); | |
| 138 | + | let number = match signal { | |
| 139 | + | "INT" => libc::SIGINT, | |
| 140 | + | "TERM" => libc::SIGTERM, | |
| 141 | + | _ => libc::SIGKILL, | |
| 142 | + | }; | |
| 143 | + | let Ok(group) = libc::pid_t::try_from(child.id()) else { | |
| 144 | + | return; | |
| 145 | + | }; | |
| 146 | + | // SAFETY: kill(2) with a negative pid signals that process group; it | |
| 147 | + | // reads no memory of ours. | |
| 148 | + | unsafe { | |
| 149 | + | libc::kill(-group, number); | |
| 140 | 150 | } | |
| 141 | 151 | } | |
| 142 | 152 | ||
| 153 | + | #[cfg(not(unix))] | |
| 154 | + | fn signal(_child: &std::process::Child, _signal: &str) {} | |
| 155 | + | ||
| 143 | 156 | /// Runs the command, sending its output (stdout and stderr together, a | |
| 144 | 157 | /// line at a time) through `commands` to the log, until it ends or | |
| 145 | 158 | /// `timeout` passes. | |
| ⋯ | |||
| 158 | 171 | command.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); | |
| 159 | 172 | // A group of its own, so a cancellation reaches what the step started. | |
| 160 | 173 | #[cfg(unix)] | |
| 161 | − | std::os::unix::process::CommandExt::process_group(&mut command, 0); | |
| 174 | + | { | |
| 175 | + | use std::os::unix::process::CommandExt; | |
| 176 | + | command.process_group(0); | |
| 177 | + | // A shell cannot trap a signal it was started with ignored, and a | |
| 178 | + | // runner started in the background (as a service, or under `&`) | |
| 179 | + | // passes SIGINT on ignored. Back to the defaults, so a cancelled | |
| 180 | + | // step hears SIGINT and its own trap runs. | |
| 181 | + | // SAFETY: only signal(), which is async-signal-safe, between fork and exec. | |
| 182 | + | unsafe { | |
| 183 | + | command.pre_exec(|| { | |
| 184 | + | for signal in [libc::SIGINT, libc::SIGTERM, libc::SIGQUIT] { | |
| 185 | + | libc::signal(signal, libc::SIG_DFL); | |
| 186 | + | } | |
| 187 | + | Ok(()) | |
| 188 | + | }); | |
| 189 | + | } | |
| 190 | + | } | |
| 162 | 191 | let mut child = command.spawn()?; | |
| 163 | 192 | let (sender, lines) = mpsc::channel::<String>(); | |
| 164 | 193 | let mut readers = Vec::new(); | |
| 387 | 387 | std::fs::create_dir_all(&origin).unwrap(); | |
| 388 | 388 | git_in(&origin, &["init", "--quiet", "--initial-branch=main"], None).unwrap(); | |
| 389 | 389 | commit(&origin, "a.txt", "one"); | |
| 390 | − | git_in(&origin, &["tag", "-a", "v1", "-m", "v1"], None).unwrap(); | |
| 390 | + | git_in(&origin, &["-c", "user.name=t", "-c", "user.email=t@example.com", "-c", "tag.gpgsign=false", "tag", "-a", "v1", "-m", "v1"], None).unwrap(); | |
| 391 | 391 | let mirror = root.join("mirror.git"); | |
| 392 | 392 | ||
| 393 | 393 | // Full. |
| 20 | 20 | ||
| 21 | 21 | | Source | What | Where it lands | | |
| 22 | 22 | | --- | --- | --- | | |
| 23 | − | | Billable usage, `GET /accounts/{account}/billable-usage?from=&to=` | One row per service per day in FOCUS columns: `ServiceFamilyName`, `ServiceName`, `ChargePeriodStart`, `ConsumedQuantity` (else `PricingQuantity`), `ContractedCost` / `BilledCost` / `EffectiveCost`. Every page is read (`result_info`: `cursor`, else `total_pages`), over whole billing cycles (see [The billing cycle](#the-billing-cycle)). Every product g1t uses appears here once it is used: Workers, Workers for Platforms, D1, KV, R2, Queues, Containers, Durable Objects, Artifacts, Browser Rendering, Workers AI, Vectorize, Cloudflare for SaaS, Email. While a cycle is open its rows carry no cost; `ListCost` is never taken, since it is before the included amounts. | `cost_lines`, source `billable_usage`: `quantity` (consumed), `billed_usd` (Cloudflare's own cost), `billable_quantity` and `cost_usd` (over the cycle), `basis`; the read itself in `cost_reads` | | |
| 23 | + | | Billable usage, `GET /accounts/{account}/billable-usage?from=&to=` | One row per service per day in FOCUS columns: `ServiceFamilyName`, `ServiceName`, `ChargePeriodStart`, `ConsumedQuantity` (else `PricingQuantity`), `ContractedCost` / `BilledCost` / `EffectiveCost`. Every page is read (`result_info`: `cursor`, else `total_pages`), over whole billing cycles (see [The billing cycle](#the-billing-cycle)). Every product g1t uses appears here once it is used: Workers, Workers for Platforms, D1, KV, R2, Queues, Containers, Durable Objects, Artifacts, Browser Rendering, Workers AI, Vectorize, Cloudflare for SaaS, Email. While a cycle is open its rows carry no cost; `ListCost` is never taken as a cost, since it is before the included amounts (the keeper reads it for a unit's rate, below). | `cost_lines`, source `billable_usage`: `quantity` (consumed), `billed_usd` (Cloudflare's own cost), `billable_quantity` and `cost_usd` (over the cycle), `basis`; the read itself in `cost_reads` | | |
| 24 | 24 | | GraphQL `artifactsEventsAdaptiveGroups` | Artifacts' own count by `date`, `eventType` and `repositoryName`. Operations are `create`, `fork`, `push`, `pull`, `delete`; errors (`rateLimited`, `serverError`, …) are kept but not counted. | `cost_lines`, source `artifacts_events`; per workspace (from the store key `<workspace>--<repo>`; a pull request's working copy, `pulls--<id>`, is its repository's workspace's, from repos' `pull_owners`) in `own_counts` as `cloudflare_git` | | |
| 25 | 25 | | GraphQL `aiGatewayRequestsAdaptiveGroups`, filtered to `AI_GATEWAY_ID` | What AI Gateway priced g1t's own provider traffic at, by `date`, `provider` and `model`: `count`, `sum.cost` (dollars), `sum.tokensIn`/`tokensOut`/`cacheReadTokens`/`cacheWriteTokens`; asked twice in one query, filtered `wholesale: 0` and `wholesale: 1`. `wholesale` is never a dimension: grouped by it, Cloudflare answers no rows and no error (until 2026-10-08 that left the gateway's side empty while it had logged $11.11). Read over its own window: the last 31 days until it has answered with a line, then the last few. Field names checked against Cloudflare's schema (introspection of `AccountAiGatewayRequestsAdaptiveGroups{Sum,Dimensions,Filter_InputObject}`). An adaptive (sampled) dataset: an estimate, close at g1t's volumes. Only g1t's hosted models go through this gateway: a workspace's own provider is called at its own address, never here. | `cost_lines`, source `ai_gateway`, product `ai_gateway_requests`: per day and model a line `<provider>_<model>` (requests, at the gateway's cost), and at no cost `…__tokens`, `…__cache_read_tokens`, `…__cache_write_tokens`; Cloudflare-billed (unified billing) requests are prefixed `wholesale__`. Mapped to `models` (migration 0036). A re-read day replaces all its gateway lines. | | |
| 26 | 26 | | The ledger | Every charge: its cost at the price book's cost, what it was charged at price, what paid for it. | read, never written | | |
| ⋯ | |||
| 299 | 299 | | Kind | When | What to do | | |
| 300 | 300 | | --- | --- | --- | | |
| 301 | 301 | | Count | g1t's count and Cloudflare's differ by more than the mapping's `drift_percent` (10%) | Find out what Cloudflare counts: compare its events with `own_counts` `artifacts_*` and `cost_operations`. If it counts more (binding reads, `ls-refs`), either change repos' `operation_mapping` so customers are charged for what Cloudflare counts, or leave it and let the per-unit cost rise (below). | | |
| 302 | − | | Cost | Cloudflare charged more than `drift_percent` away from the price book's cost of the same usage, with at least `min_daily_cost` | A price is stale: check the proposals. | | |
| 302 | + | | Cost | The price book's cost of a product's usage over the 7 days (the ledger's `cost_micros`) is more than `drift_percent` away from what the same usage comes to at Cloudflare's list prices before the included amounts, with at least `min_daily_cost` either side. The list cost is each billable-usage line mapped to the product, its whole quantity at `cycle::LIST_PRICES` (not rounded to whole millions; `margin::list_costs`). Never what Cloudflare billed: that is net of the cycle's included amounts while the price book costs every unit, so a product whose usage mostly fits in them (sandboxes inside 25 GiB-hours of Containers memory) would read as a stale price when nothing changed; what was billed is still the cost in the margins, and a leak. A product with usage on a meter that has no list price (Artifacts, Cloudflare for SaaS, R2 storage until priced) is not checked, since its usage cannot be priced like for like; add the price to `cycle::LIST_PRICES`. `cost_drift` is replaced every run and an alert whose drift is gone is resolved on the same run, so an alert raised by the old comparison (what was billed against the price book) closes on the next | A price is stale: check the proposals, and `cycle::LIST_PRICES` against Cloudflare's pricing page. | | |
| 303 | 303 | | Cost, on `models` | What AI Gateway priced g1t's own provider traffic at over the 7 days, against the ledger's model cost for the same days (billed to g1t: comped, free and trial use included, a workspace's own provider not) plus the model cost testing resets kept for those days (`reset_costs`), more than the `ai_gateway_requests` mapping's `drift_percent` (10%) apart, with at least `min_daily_cost`. A ledger with none of the gateway's cost is drift too, and so is a gateway that priced nothing against a ledger with at least `min_daily_cost` of model cost (no percentage): that is not agreement. Its detail says why as far as the run could tell: requests logged with no price (add the models' prices), a token Cloudflare refuses for the gateway (give it AI Gateway Read, or fix `AI_GATEWAY_ID`), a gateway the token sees with no requests (calls went around it), or a read that failed | The gateway higher: model calls g1t paid for and charged no one: runs not settled yet (they catch up within the hour), runs with no session, a run started without a billing ticket, or something else on g1t's gateway. The ledger higher: runs that reached a provider without the gateway. The detail adds why the gateway's own figure may be off: prompt-cache read and write tokens (the gateway prices them at its rates for cache tokens, which can lag the provider's; check against the provider's invoice), requests Cloudflare billed itself (unified billing: on Cloudflare's bill, not a provider's), and models with no price. A testing reset in the window is named in the detail: one that kept its cost says how much of the ledger's side it is; one from before resets kept their cost (the audit log has it, `reset_costs` does not) says the gateway's figure includes usage the ledger no longer has, so that part is not a leak, and the day it leaves the 7 days; while such a reset is in the window the `models` leak is not raised. Days are UTC by when a request ran (gateway) and when a charge was entered (ledger), so a run across midnight shifts a little between days; the 7-day sum absorbs it. | | |
| 304 | 304 | | Unpriced | Over the 7 days, a model in AI Gateway's analytics with tokens and $0 cost, or runs settled with `runs.gateway_note` (the gateway could not price all of a run) | The gateway has no price for a model g1t runs: add it in the gateway (custom cost) or route away from it. Until then those runs are charged no less than the sandbox reported (Claude Code's own price table), never $0 silently. | | |
| 305 | 305 | | Leak | Cost of at least `min_daily_cost` and nothing charged for it (never for `platform`), or a meter in `unmapped` | Map the meter (below), or decide it is overhead (`platform`). | | |
| ⋯ | |||
| 315 | 315 | - Proposals come from the keeper (sandbox seconds, app requests and CPU) | |
| 316 | 316 | and the reconciler (mappings with `scale_to_own`: today git operations). | |
| 317 | 317 | For git operations: Cloudflare's rate per its own operation (the median | |
| 318 | − | over charged days of cost ÷ quantity) × (Cloudflare's operations ÷ g1t's) | |
| 318 | + | over costed days of cost ÷ quantity) × (Cloudflare's operations ÷ g1t's) | |
| 319 | 319 | × 1,000. If Cloudflare counts three for each one g1t counts, the per-1,000 | |
| 320 | 320 | price triples. At least 1,000 of g1t's operations are needed. | |
| 321 | + | - A rate is always a price per unit before the included amounts, like the | |
| 322 | + | price book's: never what was billed over all of the quantity, which is | |
| 323 | + | net of them and reads as a cheaper unit. The reconciler takes a line's | |
| 324 | + | cost at `cycle::LIST_PRICES` where its meter has one (`margin::rate_line`), | |
| 325 | + | else Cloudflare's own cost (the median leaves out the day an included | |
| 326 | + | amount ran out). The keeper takes the bill's `ListCost` ÷ quantity | |
| 327 | + | (`keeper::billed_rate`), and the published rates where there is none. | |
| 321 | 328 | - Decision (`pricing::decide`): under 2% is noise; more than 4× either way | |
| 322 | 329 | is suspect and waits for staff; within `auto_apply_percent` (25%) it is | |
| 323 | 330 | applied on its own when `auto_apply` is on; anything else waits. | |
| 158 | 158 | 1. Plans: for each unit, the live commit; `git diff` from it to `HEAD`; | |
| 159 | 159 | whether the changed files touch the unit (its folder, the crates and | |
| 160 | 160 | packages it is built from, its inputs, lockfile changes that reach it). | |
| 161 | + | A Rust crate's `tests/`, `benches/` and `examples/` are not what it is | |
| 162 | + | built from, and neither is a source file compiled only for tests: one | |
| 163 | + | whose every `mod` declaration is under `#[cfg(test)]` (with or without | |
| 164 | + | `#[path]`), or inside a module that is. A change to a crate's tests | |
| 165 | + | alone deploys nothing (`scripts/deploy/stack.mjs`, `testOnlySource`). | |
| 161 | 166 | 2. Refuses if uncommitted changes touch what would deploy (`--allow-dirty`). | |
| 162 | 167 | 3. Applies every pending migration (`wrangler d1 migrations apply --remote`), | |
| 163 | 168 | in parallel. Any failure stops the deploy before code. | |
| ⋯ | |||
| 386 | 391 | Not yet seen on Cloudflare itself: watch the first runs' logs for the | |
| 387 | 392 | `Docker: started` line, and `dockerd.log` if it does not come. | |
| 388 | 393 | ||
| 394 | + | #### Sandbox errors | |
| 395 | + | ||
| 396 | + | Each sandbox is a Durable Object of `g1t-runner`, and Cloudflare counts | |
| 397 | + | each of its invocations: the `run` call that starts it, `halt`, | |
| 398 | + | `noteBlocked` and `flagAbuse`, and its alarms. The containers library keeps | |
| 399 | + | an alarm going for as long as the container runs (each one waits up to | |
| 400 | + | three minutes), and runs `onStop` from the alarm after the container | |
| 401 | + | exits, so most of a sandbox's invocations are alarms. | |
| 402 | + | ||
| 403 | + | To see what the errors are, by namespace and status, and what was thrown: | |
| 404 | + | ||
| 405 | + | ```sh | |
| 406 | + | CLOUDFLARE_API_TOKEN=<token> node scripts/ops/runner-errors.mjs # the last 7 days | |
| 407 | + | CLOUDFLARE_API_TOKEN=<token> node scripts/ops/runner-errors.mjs --days 30 --json | |
| 408 | + | ``` | |
| 409 | + | ||
| 410 | + | The token needs Account Analytics: Read and Workers Observability: Read | |
| 411 | + | (Workers Scripts: Read adds namespace names). The report puts each message | |
| 412 | + | in a bucket and says whether it is expected: | |
| 413 | + | ||
| 414 | + | | Bucket | Expected | What it is | | |
| 415 | + | | --- | --- | --- | | |
| 416 | + | | `deploy_reset` | Yes | A runner deploy resets every sandbox's object, failing the alarm or call in flight. The container keeps running and the next alarm picks it up. | | |
| 417 | + | | `no_capacity` | Yes | No container instance was free (`max_instances`). The work fails to start and says so. | | |
| 418 | + | | `container_exited`, `caller_gone` | Yes | A container that stopped, or a caller that went away first. | | |
| 419 | + | | `stop_not_reported` | No | `onStop` could not tell a service that the sandbox stopped after two tries. The five-minute sweep catches the work up. | | |
| 420 | + | | `alarm_failed`, `not_started`, `storage`, `limits`, `other` | No | Read the message. | | |
| 421 | + | ||
| 422 | + | The runner logs these at error level, each with a fixed prefix you can | |
| 423 | + | search for in Workers Logs: | |
| 424 | + | ||
| 425 | + | - `sandbox not started` | |
| 426 | + | - `sandbox stop not reported` | |
| 427 | + | - `sandbox alarm failed` | |
| 428 | + | - `sandbox container error` | |
| 429 | + | - `sandbox not destroyed` | |
| 430 | + | ||
| 431 | + | `onStop` never throws. A throw would fail the alarm, which Cloudflare | |
| 432 | + | retries and counts as an error each time, running the whole stop again. | |
| 433 | + | A sandbox's run inside its time cap is not stopped for inactivity. The | |
| 434 | + | library's `sleepAfter` (100 minutes) would otherwise stop a run whose | |
| 435 | + | guardrails allow longer, up to 240 minutes, because the runner never | |
| 436 | + | fetches the container, so to the library every sandbox looks idle. | |
| 437 | + | ||
| 389 | 438 | ## Build speed | |
| 390 | 439 | ||
| 391 | 440 | Measured on the development machine (Windows, 32 cores, warm Cargo cache), | |
| ⋯ | |||
| 433 | 482 | | The runner binary, cold (builder container) | | 99 s | | |
| 434 | 483 | | A Rust CI job's build (events, search, repos; 4 vCPUs) | | 51 s cold, 13 s with the Cargo target restored (107 MB zstd entry) | | |
| 435 | 484 | ||
| 485 | + | - **sccache** (`scripts/sccache.sh`, pinned by version and sha256): a | |
| 486 | + | checkout gives every source a new mtime, so Cargo calls rustc again for | |
| 487 | + | every workspace crate a unit uses, however much of `target/` was | |
| 488 | + | restored. With `RUSTC_WRAPPER=sccache` and its GitHub Actions backend, | |
| 489 | + | each call that makes a library is looked up by its inputs in the | |
| 490 | + | repository's Actions cache (the same cache as `actions/cache`, scoped to | |
| 491 | + | `main`), and an unchanged crate comes back from it. Measured on the | |
| 492 | + | development machine on 2026-10-08, `repos` and `events` for wasm32, | |
| 493 | + | release, `-j 4`, the dependencies already built, every workspace source | |
| 494 | + | touched as a checkout does (sccache's local disk cache; the Actions | |
| 495 | + | cache adds a download per hit): | |
| 496 | + | ||
| 497 | + | | | Time | | |
| 498 | + | | --- | --- | | |
| 499 | + | | Without sccache (before) | 44.3 s, 44.6 s | | |
| 500 | + | | sccache, empty cache (its first run, filling it) | 48.8 s | | |
| 501 | + | | sccache, filled (6 hits: `contracts`, `kit`, `rules`, `scan`, `secrets`, `blobstore`) | 24.2 s | | |
| 502 | + | ||
| 503 | + | CI's build the same way (`cargo test --workspace --no-run`, debug, | |
| 504 | + | `-j 4`, `CARGO_INCREMENTAL=0`): 70.2 s and 74.6 s without sccache, 66.5 s | |
| 505 | + | filling it, 40.6 s filled (7 library hits; 23 test harnesses and | |
| 506 | + | cdylibs compiled). | |
| 507 | + | ||
| 508 | + | What is left is each Worker's own crate: a `cdylib` is linked, and | |
| 509 | + | sccache does not cache what rustc links (binaries, cdylibs, proc | |
| 510 | + | macros, build scripts, test harnesses). That compile is real work | |
| 511 | + | anyway: a unit deploys because it, or a crate it uses, changed. The | |
| 512 | + | same holds for CI's `cargo test`: the workspace's libraries come back | |
| 513 | + | from the cache, the test harnesses are compiled. Every Rust job of | |
| 514 | + | Deploy and CI's Rust job end with sccache's hits and misses in the | |
| 515 | + | run's summary. If the cache cannot be reached, or the download fails | |
| 516 | + | its checksum, the job warns and builds as before. On g1t, sccache | |
| 517 | + | speaks the toolkit cache's older protocol (`GITHUB_SERVER_URL` is not | |
| 518 | + | github.com): a lookup, a reservation, one `PATCH` of the whole entry | |
| 519 | + | (`Content-Range: bytes 0-N/*`) and a commit per miss; a lookup and a | |
| 520 | + | blob `GET` per hit. Its storage check saves `sccache/.sccache_check` | |
| 521 | + | once, and takes the `409` on later runs as "already there". | |
| 522 | + | - **Not restoring mtimes:** setting each file's mtime from git (so Cargo | |
| 523 | + | would trust the restored `target/`) was considered and left out. A | |
| 524 | + | restored `target/` may come from another commit (the cache's | |
| 525 | + | `restore-keys` take the nearest earlier entry, of any group), and a | |
| 526 | + | file changed by an older commit than that build would look unchanged: | |
| 527 | + | a stale crate deployed. sccache looks at what is compiled, not when. | |
| 436 | 528 | - **Only what changed** is the largest saving: a change to one service | |
| 437 | − | deploys one service. | |
| 529 | + | deploys one service, and a change only to a crate's tests deploys | |
| 530 | + | nothing. | |
| 438 | 531 | ||
| 439 | 532 | ## The workflow | |
| 440 | 533 | ||
| ⋯ | |||
| 463 | 556 | be rebuilt alone (`image: true` in the matrix, also on `g1t-4core`, where | |
| 464 | 557 | it builds and pushes the image with the job's own Docker Engine). `fail-fast: false`, so one failed job does not cut | |
| 465 | 558 | another off mid-upload; the next stage then does not start. | |
| 466 | − | - **Tests:** there is no CI workflow on g1t yet; `main` is kept passing by | |
| 467 | − | the merge queue's checks. `check` runs the deploy tool's own tests. When a | |
| 468 | − | CI workflow is added, make `migrate` and the stages wait for it (`workflow_run`, or a job in | |
| 469 | − | this file). | |
| 559 | + | - **Tests:** `.g1t/workflows/ci.yml` tests every pull request into `main` | |
| 560 | + | (and, on pushes to `main`, runs only its Rust job, to keep the caches | |
| 561 | + | pull requests restore from current). `check` runs the deploy tool's own | |
| 562 | + | tests. Deploy does not wait for CI. | |
| 470 | 563 | - **Machines:** Rust jobs and the runner's image run on `g1t-4core` (4 vCPUs, | |
| 471 | 564 | 12 GiB, 20 GB), the others on the standard machine | |
| 472 | 565 | (`runs-on: ${{ (matrix.rust || matrix.image) && 'g1t-4core' || 'ubuntu-latest' }}`). | |
| ⋯ | |||
| 477 | 570 | downloaded tools, `~/.cargo/registry/cache`, and the Cargo target's | |
| 478 | 571 | release dependencies (`target/release` and | |
| 479 | 572 | `target/wasm32-unknown-unknown/release`, without `incremental` or | |
| 480 | − | `.wasm`), keyed by the build group, `Cargo.lock` and `base.json`. The | |
| 481 | − | workspace's own crates are compiled again on every run (a checkout's | |
| 482 | − | sources are newer than any cache); the crates.io dependencies are not. | |
| 483 | − | npm's cache is not kept: every job runs `npm ci` of only what its units | |
| 484 | − | need (`deploy.mjs install`: Wrangler alone for Rust jobs). | |
| 573 | + | `.wasm`), keyed by the build group, `Cargo.lock` and `base.json`. Cargo | |
| 574 | + | calls rustc again for the workspace's own crates on every run (a | |
| 575 | + | checkout's sources are newer than any cache); **sccache** answers those | |
| 576 | + | calls from the repository's Actions cache when a crate's inputs did not | |
| 577 | + | change (see [Build speed](#build-speed)). npm's cache is not kept: every | |
| 578 | + | job runs `npm ci` of only what its units need (`deploy.mjs install`: | |
| 579 | + | Wrangler alone for Rust jobs). | |
| 485 | 580 | - **Conditions:** each stage runs with `!failure() && !cancelled()`, which | |
| 486 | 581 | on g1t (as on GitHub) is true when no job before it failed, however far | |
| 487 | 582 | back: a `migrate` job skipped for having nothing to apply does not stop | |
| ⋯ | |||
| 511 | 606 | miss. The image job adds `x86_64-unknown-linux-musl` (about 30 MB from | |
| 512 | 607 | `static.rust-lang.org`) and keeps its Cargo target in the cache. worker-build | |
| 513 | 608 | fetches wasm-bindgen and wasm-opt from GitHub releases and esbuild from | |
| 514 | − | npm. All of those hosts are on the list every workflow job may reach. | |
| 609 | + | npm. Each Rust job downloads sccache (about 10 MB) from its GitHub release | |
| 610 | + | too (`scripts/sccache.sh`, which checks its sha256); it is not in the base | |
| 611 | + | image, and putting it there means a pinned download in the base's | |
| 612 | + | Dockerfile and a base rebuild (`build-base`). All of those hosts are on | |
| 613 | + | the list every workflow job may reach. | |
| 515 | 614 | ||
| 516 | 615 | ### Network | |
| 517 | 616 | ||
| 39 | 39 | versionFrom, | |
| 40 | 40 | } from "./cloudflare.mjs"; | |
| 41 | 41 | import { changedNames, parseCargoLock, parseNpmLock, reaches } from "./lockfiles.mjs"; | |
| 42 | − | import { decide, planJson, pool } from "./plan.mjs"; | |
| 42 | + | import { decide, git, planJson, pool } from "./plan.mjs"; | |
| 43 | 43 | import { | |
| 44 | 44 | ROOT, | |
| 45 | 45 | buildGroups, | |
| ⋯ | |||
| 53 | 53 | problems, | |
| 54 | 54 | resolveStack, | |
| 55 | 55 | resolvedStack, | |
| 56 | + | testOnlySource, | |
| 56 | 57 | touches, | |
| 57 | 58 | touchesBase, | |
| 58 | 59 | touchesImage, | |
| ⋯ | |||
| 328 | 329 | assert.ok(!ids(stack.units, ["packages/contracts/src/index.ts"]).includes("events")); | |
| 329 | 330 | }); | |
| 330 | 331 | ||
| 332 | + | test("a crate's tests, benches and examples deploy nothing", () => { | |
| 333 | + | // 0cdd221 changed only this file, and the plan sent three Rust Workers. | |
| 334 | + | assert.deepEqual(ids(stack.units, ["crates/actions/tests/repository_workflows.rs"]), []); | |
| 335 | + | assert.ok(!touchesImage(unit("runner"), ["crates/actions/tests/repository_workflows.rs"])); | |
| 336 | + | assert.deepEqual(ids(stack.units, ["crates/kit/benches/wire.rs", "crates/scan/examples/scan.rs", "services/repos/tests/git.rs"]), []); | |
| 337 | + | // Their sources still do, and a folder merely named like one. | |
| 338 | + | assert.deepEqual(ids(stack.units, ["crates/actions/src/expr.rs"]), ["actions", "security", "runner"]); | |
| 339 | + | assert.ok(touchesImage(unit("runner"), ["crates/actions/src/expr.rs"])); | |
| 340 | + | assert.deepEqual(ids(stack.units, ["services/repos/src/tests/helpers.rs"]), ["repos"]); | |
| 341 | + | assert.deepEqual(ids(stack.units, ["crates/kit/testsuite/a.rs"]), ids(stack.units, ["crates/kit/src/lib.rs"])); | |
| 342 | + | // A unit that is not a crate keeps its whole folder. | |
| 343 | + | assert.deepEqual(unit("web").crateDirs, []); | |
| 344 | + | assert.deepEqual(unit("events").crateDirs, ["crates/contracts", "crates/kit", "services/events"]); | |
| 345 | + | assert.ok(unit("runner").crateDirs.includes("crates/runner")); | |
| 346 | + | }); | |
| 347 | + | ||
| 348 | + | /** Rust sources by path, as `testOnlySource` reads them. */ | |
| 349 | + | const sources = (files) => ({ | |
| 350 | + | read: (path) => files[path] ?? "", | |
| 351 | + | list: (dir) => Object.keys(files).filter((path) => path.slice(0, path.lastIndexOf("/")) === dir), | |
| 352 | + | }); | |
| 353 | + | ||
| 354 | + | test("a Rust file compiled only for tests is found from what declares it", () => { | |
| 355 | + | const crate = ["crates/x"]; | |
| 356 | + | const tree = sources({ | |
| 357 | + | "crates/x/src/lib.rs": "pub mod api;\nmod store;\n#[cfg(test)]\nmod tests;\n#[cfg(test)] pub(crate) mod fixtures;\n#[cfg(any(test, feature = \"x\"))]\nmod both;\n", | |
| 358 | + | "crates/x/src/api.rs": "mod wire;\n#[cfg(test)]\n#[path = \"api_tests.rs\"]\nmod tests;\n", | |
| 359 | + | "crates/x/src/api/wire.rs": "", | |
| 360 | + | "crates/x/src/api_tests.rs": "use super::*;", | |
| 361 | + | "crates/x/src/store/mod.rs": "#[cfg(test)]\nmod memory;\n", | |
| 362 | + | "crates/x/src/store/memory.rs": "mod deep;", | |
| 363 | + | "crates/x/src/store/memory/deep.rs": "", | |
| 364 | + | "crates/x/src/tests.rs": "", | |
| 365 | + | "crates/x/src/fixtures/mod.rs": "", | |
| 366 | + | "crates/x/src/both.rs": "", | |
| 367 | + | "crates/x/src/bin/tool.rs": "", | |
| 368 | + | }); | |
| 369 | + | const only = (file) => testOnlySource(file, crate, tree); | |
| 370 | + | for (const file of ["crates/x/src/tests.rs", "crates/x/src/fixtures/mod.rs", "crates/x/src/api_tests.rs", "crates/x/src/store/memory.rs", "crates/x/src/store/memory/deep.rs"]) { | |
| 371 | + | assert.ok(only(file), file); | |
| 372 | + | } | |
| 373 | + | // Built: roots, ordinary modules, a cfg that is not only test, a file | |
| 374 | + | // nothing declares (new, or deleted), and files outside src or the crate. | |
| 375 | + | for (const file of [ | |
| 376 | + | "crates/x/src/lib.rs", | |
| 377 | + | "crates/x/src/api.rs", | |
| 378 | + | "crates/x/src/api/wire.rs", | |
| 379 | + | "crates/x/src/store/mod.rs", | |
| 380 | + | "crates/x/src/both.rs", | |
| 381 | + | "crates/x/src/bin/tool.rs", | |
| 382 | + | "crates/x/src/new.rs", | |
| 383 | + | "crates/x/build.rs", | |
| 384 | + | "crates/y/src/tests.rs", | |
| 385 | + | "crates/x/src/notes.md", | |
| 386 | + | ]) { | |
| 387 | + | assert.ok(!only(file), file); | |
| 388 | + | } | |
| 389 | + | }); | |
| 390 | + | ||
| 391 | + | test("this repository's test-only sources are read from git", () => { | |
| 392 | + | const head = git.head(); | |
| 393 | + | const atHead = { read: (path) => git.show(head, path), list: (dir) => git.list(head, dir) }; | |
| 394 | + | const only = (u, file) => testOnlySource(file, unit(u).crateDirs, atHead); | |
| 395 | + | assert.ok(only("api", "apps/api/src/responses.rs")); | |
| 396 | + | assert.ok(only("billing", "services/billing/src/catalogue_tests.rs")); | |
| 397 | + | assert.ok(only("billing", "services/billing/src/gateway_tests.rs")); | |
| 398 | + | assert.ok(!only("billing", "services/billing/src/catalogue.rs")); | |
| 399 | + | assert.ok(!only("api", "apps/api/src/toolkit.rs")); | |
| 400 | + | }); | |
| 401 | + | ||
| 402 | + | test("decide: a change only to tests deploys nothing", () => { | |
| 403 | + | const units = ["actions", "billing", "runner"].map(unit); | |
| 404 | + | const live = { actions: { sha: OLD }, billing: { sha: OLD }, runner: { sha: OLD } }; | |
| 405 | + | const files = { | |
| 406 | + | [`${HEAD}:services/billing/src/gateway.rs`]: "#[cfg(test)]\n#[path = \"gateway_tests.rs\"]\nmod tests;\n", | |
| 407 | + | [`${HEAD}:services/billing/src/lib.rs`]: "mod gateway;\n", | |
| 408 | + | }; | |
| 409 | + | const gitApi = { | |
| 410 | + | ...fakeGit(["crates/actions/tests/repository_workflows.rs", "services/billing/src/gateway_tests.rs"]), | |
| 411 | + | show: (sha, path) => files[`${sha}:${path}`] ?? "", | |
| 412 | + | list: (sha, dir) => Object.keys(files).map((k) => k.slice(sha.length + 1)).filter((p) => p.startsWith(`${dir}/`)).concat(dir === "services/billing/src" ? ["services/billing/src/gateway_tests.rs"] : []), | |
| 413 | + | }; | |
| 414 | + | const decisions = decide(units, { live, head: HEAD, gitApi }); | |
| 415 | + | assert.deepEqual(decisions.map((d) => [d.unit.id, d.deploy, d.image]), [["actions", false, false], ["billing", false, false], ["runner", false, false]]); | |
| 416 | + | // With a source that ships beside them, only that is listed. | |
| 417 | + | const mixed = decide([unit("billing")], { live, head: HEAD, gitApi: { ...gitApi, changed: () => ["services/billing/src/gateway_tests.rs", "services/billing/src/gateway.rs"] } }); | |
| 418 | + | assert.deepEqual([mixed[0].deploy, mixed[0].files], [true, ["services/billing/src/gateway.rs"]]); | |
| 419 | + | }); | |
| 420 | + | ||
| 331 | 421 | test("own folders, declared inputs and root files", () => { | |
| 332 | 422 | assert.deepEqual(ids(stack.units, ["services/pages/src/index.ts"]), ["pages"]); | |
| 333 | 423 | assert.deepEqual(ids(stack.units, ["apps/web/app/lib/roadmap.ts"]), ["og", "web"]); | |
| 5 | 5 | import { execFileSync } from "node:child_process"; | |
| 6 | 6 | ||
| 7 | 7 | import { changedNames, lockRoots, parseCargoLock, parseNpmLock, reaches } from "./lockfiles.mjs"; | |
| 8 | − | import { ROOT, byStage, buildGroups, codeStages, touches, touchesImage } from "./stack.mjs"; | |
| 8 | + | import { ROOT, byStage, buildGroups, codeStages, testOnlySource, touches, touchesImage } from "./stack.mjs"; | |
| 9 | 9 | ||
| 10 | 10 | /** Git, read-only. */ | |
| 11 | 11 | export const git = { | |
| ⋯ | |||
| 21 | 21 | return false; | |
| 22 | 22 | } | |
| 23 | 23 | }, | |
| 24 | − | /** Files changed between two commits. */ | |
| 25 | 24 | /** A file's text at a commit, or "" if it is not there. */ | |
| 26 | 25 | show: (sha, path) => { | |
| 27 | 26 | try { | |
| ⋯ | |||
| 30 | 29 | return ""; | |
| 31 | 30 | } | |
| 32 | 31 | }, | |
| 32 | + | /** The files directly in a folder at a commit (repository-relative). */ | |
| 33 | + | list: (sha, dir) => { | |
| 34 | + | try { | |
| 35 | + | return run(["ls-tree", "--name-only", sha, "--", `${dir}/`]).split("\n").filter(Boolean); | |
| 36 | + | } catch { | |
| 37 | + | return []; | |
| 38 | + | } | |
| 39 | + | }, | |
| 40 | + | /** Files changed between two commits. */ | |
| 33 | 41 | changed: (from, to) => run(["diff", "--name-only", "--no-renames", from, to]).split("\n").filter(Boolean), | |
| 34 | 42 | /** Whether `older` is in `newer`'s history (and not the same commit). */ | |
| 35 | 43 | isAncestor: (older, newer) => { | |
| ⋯ | |||
| 82 | 90 | } | |
| 83 | 91 | return locks.get(key); | |
| 84 | 92 | }; | |
| 93 | + | // A Rust source compiled only for tests (`#[cfg(test)] mod tests;`) is | |
| 94 | + | // not in what deploys. Read at the commit being deployed, each file once. | |
| 95 | + | const texts = new Map(); | |
| 96 | + | const listings = new Map(); | |
| 97 | + | const atHead = { | |
| 98 | + | read: (path) => { | |
| 99 | + | if (!texts.has(path)) texts.set(path, gitApi.show(head, path)); | |
| 100 | + | return texts.get(path); | |
| 101 | + | }, | |
| 102 | + | list: (dir) => { | |
| 103 | + | if (!listings.has(dir)) listings.set(dir, gitApi.list?.(head, dir) ?? []); | |
| 104 | + | return listings.get(dir); | |
| 105 | + | }, | |
| 106 | + | }; | |
| 107 | + | const testOnly = new Map(); | |
| 108 | + | const onlyForTests = (unit, file) => { | |
| 109 | + | if (!file.endsWith(".rs") || !unit.crateDirs?.some((dir) => file.startsWith(`${dir}/src/`))) return false; | |
| 110 | + | if (!testOnly.has(file)) testOnly.set(file, testOnlySource(file, unit.crateDirs, atHead)); | |
| 111 | + | return testOnly.get(file); | |
| 112 | + | }; | |
| 85 | 113 | const relevant = (unit, sha, files) => | |
| 86 | 114 | files.filter((file) => { | |
| 115 | + | if (onlyForTests(unit, file)) return false; | |
| 87 | 116 | if (file !== "Cargo.lock" && file !== "package-lock.json") return true; | |
| 88 | 117 | const { after, names } = lockChange(sha, file); | |
| 89 | 118 | const roots = lockRoots(unit)[file === "Cargo.lock" ? "cargo" : "npm"]; | |
| 173 | 173 | unit.crate = crate; | |
| 174 | 174 | unit.pkg = pkg; | |
| 175 | 175 | unit.dependsOn = [...dirs].sort(); | |
| 176 | + | // The Rust crates it is built from, whose tests, benches and examples | |
| 177 | + | // are not. | |
| 178 | + | unit.crateDirs = crate ? [unit.path, ...closure(cargo, crate).map((name) => cargo.get(name).dir)].sort() : []; | |
| 176 | 179 | unit.inputs = [...new Set([...(unit.inputs ?? []), ...globalInputs(unit.kind)])].sort(); | |
| 177 | 180 | if (unit.image) { | |
| 178 | 181 | const name = unit.image.crate; | |
| ⋯ | |||
| 189 | 192 | }; | |
| 190 | 193 | for (const dir of unit.image.dirs) if (dir !== unit.path) unit.dependsOn.push(dir); | |
| 191 | 194 | unit.dependsOn = [...new Set(unit.dependsOn)].sort(); | |
| 195 | + | unit.crateDirs = [...new Set([...unit.crateDirs, ...unit.image.dirs])].sort(); | |
| 192 | 196 | unit.inputs = [...new Set([...unit.inputs, "Cargo.toml", "Cargo.lock", "scripts/build-runner.mjs"])].sort(); | |
| 193 | 197 | } | |
| 194 | 198 | } | |
| ⋯ | |||
| 202 | 206 | ||
| 203 | 207 | const under = (file, dir) => file === dir || file.startsWith(`${dir}/`); | |
| 204 | 208 | ||
| 209 | + | /** A Rust crate's folders that its library and binaries are never built from. */ | |
| 210 | + | export const CRATE_TEST_DIRS = ["tests", "benches", "examples"]; | |
| 211 | + | ||
| 212 | + | /** | |
| 213 | + | * Whether `file` is in the tests, benches or examples of one of | |
| 214 | + | * `crateDirs`: Cargo builds those only for `cargo test`, `cargo bench` and | |
| 215 | + | * `--example`, never into what deploys. | |
| 216 | + | */ | |
| 217 | + | export function inCrateTests(file, crateDirs = []) { | |
| 218 | + | return crateDirs.some((dir) => CRATE_TEST_DIRS.some((sub) => under(file, `${dir}/${sub}`))); | |
| 219 | + | } | |
| 220 | + | ||
| 221 | + | const dirOf = (path) => path.slice(0, Math.max(0, path.lastIndexOf("/"))); | |
| 222 | + | const baseOf = (path) => path.slice(path.lastIndexOf("/") + 1); | |
| 223 | + | ||
| 224 | + | /** | |
| 225 | + | * The `mod name;` declarations in Rust source `text` that `match` (by name, | |
| 226 | + | * or by a `#[path]`), each with whether it is under `#[cfg(test)]`. | |
| 227 | + | * Only outline modules (`mod x;`); the attributes are those directly above. | |
| 228 | + | */ | |
| 229 | + | function declarations(text, match) { | |
| 230 | + | const found = []; | |
| 231 | + | const pattern = /((?:#\[[^\]]*\]\s*)*)(?:pub(?:\([^)]*\))?\s+)?mod\s+(r#)?(\w+)\s*;/g; | |
| 232 | + | for (const [, attrs, , name] of text.matchAll(pattern)) { | |
| 233 | + | const path = /#\[\s*path\s*=\s*"([^"]+)"\s*\]/.exec(attrs)?.[1] ?? null; | |
| 234 | + | if (!match(name, path)) continue; | |
| 235 | + | found.push(/#\[\s*cfg\s*\(\s*test\s*\)\s*\]/.test(attrs)); | |
| 236 | + | } | |
| 237 | + | return found; | |
| 238 | + | } | |
| 239 | + | ||
| 240 | + | /** | |
| 241 | + | * Whether Rust source `file`, in the `src/` of one of `crateDirs`, is | |
| 242 | + | * compiled only for tests: every module declaration of it found is under | |
| 243 | + | * `#[cfg(test)]`, or is in a file that is itself only for tests. A file | |
| 244 | + | * nothing is found to declare (a new crate root, a deleted file, a module | |
| 245 | + | * declared some way this does not read) counts as built, so a change to it | |
| 246 | + | * deploys. | |
| 247 | + | * | |
| 248 | + | * read(path): a file's text at the commit being deployed, "" if absent | |
| 249 | + | * list(dir): the files directly in a folder at that commit | |
| 250 | + | */ | |
| 251 | + | export function testOnlySource(file, crateDirs, { read, list = () => [] }, depth = 0) { | |
| 252 | + | if (!file.endsWith(".rs") || depth > 16) return false; | |
| 253 | + | const crateDir = crateDirs.find((dir) => file.startsWith(`${dir}/src/`)); | |
| 254 | + | if (!crateDir) return false; | |
| 255 | + | const src = `${crateDir}/src`; | |
| 256 | + | const dir = dirOf(file); | |
| 257 | + | const base = baseOf(file).slice(0, -".rs".length); | |
| 258 | + | // Crate roots and binaries are not declared by anything. | |
| 259 | + | if (dir === src && ["lib", "main"].includes(base)) return false; | |
| 260 | + | if (dir === `${src}/bin` || (base === "main" && dirOf(dir) === `${src}/bin`)) return false; | |
| 261 | + | // Where `mod name;` for this file would be: `a/name.rs` and `a/name/mod.rs` | |
| 262 | + | // are declared by `a.rs` or `a/mod.rs` (by `lib.rs` or `main.rs` in src). | |
| 263 | + | const name = base === "mod" ? baseOf(dir) : base; | |
| 264 | + | const parentDir = base === "mod" ? dirOf(dir) : dir; | |
| 265 | + | const parents = | |
| 266 | + | parentDir === src ? [`${src}/lib.rs`, `${src}/main.rs`] : [`${parentDir}/mod.rs`, `${dirOf(parentDir)}/${baseOf(parentDir)}.rs`]; | |
| 267 | + | // One declaration that is built is enough to stop: most changed files | |
| 268 | + | // are ordinary modules, settled by reading their parent. | |
| 269 | + | let declared = false; | |
| 270 | + | const testOnly = (declarer, isTest) => { | |
| 271 | + | declared = true; | |
| 272 | + | return isTest || testOnlySource(declarer, crateDirs, { read, list }, depth + 1); | |
| 273 | + | }; | |
| 274 | + | for (const parent of parents) { | |
| 275 | + | for (const isTest of declarations(read(parent), (found, path) => found === name && path === null)) { | |
| 276 | + | if (!testOnly(parent, isTest)) return false; | |
| 277 | + | } | |
| 278 | + | } | |
| 279 | + | // `#[path = "x.rs"] mod y;` in a file beside it names it directly. | |
| 280 | + | const fileName = baseOf(file); | |
| 281 | + | for (const sibling of list(dir).filter((path) => path.endsWith(".rs") && path !== file)) { | |
| 282 | + | for (const isTest of declarations(read(sibling), (_, path) => path === fileName || path === `./${fileName}`)) { | |
| 283 | + | if (!testOnly(sibling, isTest)) return false; | |
| 284 | + | } | |
| 285 | + | } | |
| 286 | + | return declared; | |
| 287 | + | } | |
| 288 | + | ||
| 205 | 289 | /** | |
| 206 | 290 | * Why a set of changed files (repository-relative, `/`-separated) touches | |
| 207 | 291 | * a unit: the first file that does, and through what. Null if none does. | |
| 292 | + | * Its crates' tests, benches and examples do not. | |
| 208 | 293 | */ | |
| 209 | 294 | export function touches(unit, files) { | |
| 295 | + | files = files.filter((file) => !inCrateTests(file, unit.crateDirs)); | |
| 210 | 296 | for (const file of files) { | |
| 211 | 297 | if (under(file, unit.path)) return { file, via: "its own folder" }; | |
| 212 | 298 | } | |
| ⋯ | |||
| 221 | 307 | /** Whether changed files touch a unit's Containers image. */ | |
| 222 | 308 | export function touchesImage(unit, files) { | |
| 223 | 309 | if (!unit.image) return false; | |
| 224 | − | return files.some((file) => unit.image.files.includes(file) || unit.image.dirs.some((dir) => under(file, dir))); | |
| 310 | + | return files.some( | |
| 311 | + | (file) => unit.image.files.includes(file) || (unit.image.dirs.some((dir) => under(file, dir)) && !inCrateTests(file, unit.image.dirs)), | |
| 312 | + | ); | |
| 225 | 313 | } | |
| 226 | 314 | ||
| 227 | 315 | /** Whether changed files touch the folder a unit's base image is built from. */ | |
| 1 | + | #!/usr/bin/env node | |
| 2 | + | // What are the runner's Durable Object errors? The sandboxes that run agent | |
| 3 | + | // attempts, workflow jobs and merge-queue builds are Durable Objects of the | |
| 4 | + | // g1t-runner Worker (services/runner: AttemptSandbox, Sandbox2Core, | |
| 5 | + | // Sandbox4Core). This asks Cloudflare, over the last N days: | |
| 6 | + | // | |
| 7 | + | // 1. GraphQL Analytics (durableObjectsInvocationsAdaptiveGroups): requests | |
| 8 | + | // and errors by namespace and status, and by day and status. | |
| 9 | + | // 2. Workers Observability (the telemetry query API): the runner's failed | |
| 10 | + | // invocations by class, event type (alarm, rpc, fetch) and outcome; the | |
| 11 | + | // exceptions they threw, by message; the runner's own error-level logs | |
| 12 | + | // (`sandbox stop not reported`, `sandbox alarm failed`, `sandbox not | |
| 13 | + | // started`, `sandbox container error`), by message; and a few recent | |
| 14 | + | // failed invocations in full. | |
| 15 | + | // | |
| 16 | + | // Each message is put in a bucket (`classify`): a deploy resetting the | |
| 17 | + | // object, no container free, a container that exited, and so on, so what is | |
| 18 | + | // expected and what is a bug can be told apart at a glance. | |
| 19 | + | // | |
| 20 | + | // Read-only: GraphQL queries, one Durable Objects listing for names, and | |
| 21 | + | // telemetry queries. Never prints the token. | |
| 22 | + | // | |
| 23 | + | // CLOUDFLARE_API_TOKEN=<token> node scripts/ops/runner-errors.mjs [--days 7] [--json] | |
| 24 | + | // | |
| 25 | + | // The token needs Account Analytics: Read (GraphQL) and Workers Observability: | |
| 26 | + | // Read (logs); Workers Scripts: Read adds namespace names. A part the token | |
| 27 | + | // cannot read is a note in the report, never the end of it. | |
| 28 | + | ||
| 29 | + | import { ACCOUNT_ID, cloudflareAuth } from "../deploy/cloudflare.mjs"; | |
| 30 | + | ||
| 31 | + | const API = "https://api.cloudflare.com/client/v4"; | |
| 32 | + | ||
| 33 | + | const HELP = `node scripts/ops/runner-errors.mjs [--days N] [--script NAME] [--samples N] [--json] [--keys] [--account <id>] | |
| 34 | + | ||
| 35 | + | The runner's Durable Object errors: requests and errors by namespace and | |
| 36 | + | status (GraphQL Analytics), and what failed and why (Workers Observability). | |
| 37 | + | ||
| 38 | + | --days N how far back, in days (default 7, at most 31) | |
| 39 | + | --script NAME the Worker (default g1t-runner) | |
| 40 | + | --samples N recent failed invocations shown in full (default 10) | |
| 41 | + | --json one JSON object, snake_case keys | |
| 42 | + | --keys also list the telemetry keys the runner's events have | |
| 43 | + | --account <id> the Cloudflare account (default CLOUDFLARE_ACCOUNT_ID, else g1t's) | |
| 44 | + | --help this text | |
| 45 | + | ||
| 46 | + | Needs CLOUDFLARE_API_TOKEN (Account Analytics: Read, Workers Observability: | |
| 47 | + | Read), or CLOUDFLARE_API_KEY with CLOUDFLARE_EMAIL. Exits 1 when nothing | |
| 48 | + | could be read.`; | |
| 49 | + | ||
| 50 | + | /** The command line. */ | |
| 51 | + | export function parseArgs(argv, env = process.env) { | |
| 52 | + | const option = (name) => { | |
| 53 | + | const at = argv.indexOf(name); | |
| 54 | + | return at >= 0 && argv[at + 1] && !argv[at + 1].startsWith("--") ? argv[at + 1] : null; | |
| 55 | + | }; | |
| 56 | + | const days = Number(option("--days") ?? NaN); | |
| 57 | + | const samples = Number(option("--samples") ?? NaN); | |
| 58 | + | return { | |
| 59 | + | help: argv.includes("--help") || argv.includes("-h"), | |
| 60 | + | json: argv.includes("--json"), | |
| 61 | + | keys: argv.includes("--keys"), | |
| 62 | + | days: Number.isFinite(days) && days > 0 ? Math.min(31, days) : 7, | |
| 63 | + | samples: Number.isFinite(samples) && samples >= 0 ? Math.min(100, Math.floor(samples)) : 10, | |
| 64 | + | script: option("--script") || "g1t-runner", | |
| 65 | + | account: option("--account") || env.CLOUDFLARE_ACCOUNT_ID || ACCOUNT_ID, | |
| 66 | + | }; | |
| 67 | + | } | |
| 68 | + | ||
| 69 | + | /** The window: the last `days` days up to `now`. */ | |
| 70 | + | export function windowOf(now, days) { | |
| 71 | + | return { start: new Date(now.getTime() - days * 86_400_000).toISOString(), end: now.toISOString() }; | |
| 72 | + | } | |
| 73 | + | ||
| 74 | + | /** | |
| 75 | + | * The GraphQL queries, each with variants tried in order when Cloudflare | |
| 76 | + | * refuses a field. Rows come back under `rows`. | |
| 77 | + | */ | |
| 78 | + | export const DO_QUERIES = [ | |
| 79 | + | { | |
| 80 | + | key: "by_namespace", | |
| 81 | + | label: "Durable Object requests by namespace and status", | |
| 82 | + | variants: [ | |
| 83 | + | { sum: ["requests", "errors", "wallTime"], dims: ["namespaceId", "status"] }, | |
| 84 | + | { sum: ["requests", "errors"], dims: ["namespaceId", "status"] }, | |
| 85 | + | { sum: ["requests"], dims: ["namespaceId", "status"] }, | |
| 86 | + | { sum: ["requests", "errors"], dims: ["namespaceId"] }, | |
| 87 | + | ], | |
| 88 | + | }, | |
| 89 | + | { | |
| 90 | + | key: "by_day", | |
| 91 | + | label: "Durable Object requests by day and status", | |
| 92 | + | variants: [ | |
| 93 | + | { sum: ["requests", "errors"], dims: ["date", "status"] }, | |
| 94 | + | { sum: ["requests"], dims: ["date", "status"] }, | |
| 95 | + | ], | |
| 96 | + | }, | |
| 97 | + | ]; | |
| 98 | + | ||
| 99 | + | /** One DO invocations query for one script, for one variant. */ | |
| 100 | + | export function doQuery(variant) { | |
| 101 | + | return `query RunnerErrors($accountTag: String!, $start: Time!, $end: Time!, $script: String!) { | |
| 102 | + | viewer { | |
| 103 | + | accounts(filter: { accountTag: $accountTag }) { | |
| 104 | + | rows: durableObjectsInvocationsAdaptiveGroups(limit: 10000, filter: { datetime_geq: $start, datetime_leq: $end, scriptName: $script }) { | |
| 105 | + | sum { ${variant.sum.join(" ")} } | |
| 106 | + | dimensions { ${variant.dims.join(" ")} } | |
| 107 | + | } | |
| 108 | + | } | |
| 109 | + | } | |
| 110 | + | }`; | |
| 111 | + | } | |
| 112 | + | ||
| 113 | + | const snake = (name) => name.replace(/[A-Z]/g, (letter) => `_${letter.toLowerCase()}`); | |
| 114 | + | ||
| 115 | + | /** | |
| 116 | + | * A GraphQL answer as rows: one per dimension combination, the first | |
| 117 | + | * dimension named through `names` (namespace ids to names), with each sum | |
| 118 | + | * and the error rate, largest first by requests. Throws with Cloudflare's | |
| 119 | + | * message when the answer has errors. | |
| 120 | + | */ | |
| 121 | + | export function parseDoGroups(body, variant, names = {}) { | |
| 122 | + | if (body?.errors?.length) throw new Error(body.errors.map((error) => error.message).join("; ").slice(0, 400)); | |
| 123 | + | const groups = body?.data?.viewer?.accounts?.[0]?.rows; | |
| 124 | + | if (!Array.isArray(groups)) throw new Error("no rows in the answer"); | |
| 125 | + | const rows = groups.map((group) => { | |
| 126 | + | const row = {}; | |
| 127 | + | variant.dims.forEach((dim, at) => { | |
| 128 | + | const value = group.dimensions?.[dim]; | |
| 129 | + | const text = value == null || value === "" ? "(none)" : String(value); | |
| 130 | + | row[snake(dim)] = at === 0 && names[text] ? names[text] : text; | |
| 131 | + | }); | |
| 132 | + | for (const field of variant.sum) row[snake(field)] = Number(group.sum?.[field] ?? 0); | |
| 133 | + | return row; | |
| 134 | + | }); | |
| 135 | + | const sortKey = variant.dims[0] === "date" ? null : "requests"; | |
| 136 | + | rows.sort((a, b) => (sortKey ? b.requests - a.requests : 0) || String(a[snake(variant.dims[0])]).localeCompare(String(b[snake(variant.dims[0])]))); | |
| 137 | + | const totals = Object.fromEntries(variant.sum.map((field) => [snake(field), rows.reduce((sum, row) => sum + row[snake(field)], 0)])); | |
| 138 | + | return { dims: variant.dims.map(snake), metrics: variant.sum.map(snake), rows, totals }; | |
| 139 | + | } | |
| 140 | + | ||
| 141 | + | /** | |
| 142 | + | * Requests and errors per status, summed over namespaces: which statuses | |
| 143 | + | * the errors are (`scriptThrewException`, `internalError`, | |
| 144 | + | * `clientDisconnected`, `exceededResources`, ...). | |
| 145 | + | */ | |
| 146 | + | export function byStatus(parsed) { | |
| 147 | + | if (!parsed.dims.includes("status")) return []; | |
| 148 | + | const out = new Map(); | |
| 149 | + | for (const row of parsed.rows) { | |
| 150 | + | const entry = out.get(row.status) ?? { status: row.status, requests: 0, errors: 0 }; | |
| 151 | + | entry.requests += row.requests ?? 0; | |
| 152 | + | entry.errors += row.errors ?? 0; | |
| 153 | + | out.set(row.status, entry); | |
| 154 | + | } | |
| 155 | + | return [...out.values()].sort((a, b) => b.requests - a.requests); | |
| 156 | + | } | |
| 157 | + | ||
| 158 | + | /** The telemetry keys the questions below group by and filter on. */ | |
| 159 | + | export const KEYS = { | |
| 160 | + | outcome: "$workers.outcome", | |
| 161 | + | eventType: "$workers.eventType", | |
| 162 | + | entrypoint: "$workers.entrypoint", | |
| 163 | + | error: "$metadata.error", | |
| 164 | + | message: "$metadata.message", | |
| 165 | + | level: "$metadata.level", | |
| 166 | + | }; | |
| 167 | + | ||
| 168 | + | /** Which key names the Worker: tried in order, the next when one finds nothing. */ | |
| 169 | + | export const SERVICE_KEYS = ["$workers.scriptName", "$metadata.service"]; | |
| 170 | + | ||
| 171 | + | const filter = (key, operation, value) => (value === undefined ? { key, operation, type: "string" } : { key, operation, type: "string", value }); | |
| 172 | + | ||
| 173 | + | /** | |
| 174 | + | * The telemetry questions: each a body for `POST .../workers/observability/ | |
| 175 | + | * telemetry/query`, for the Worker named under `serviceKey`. | |
| 176 | + | */ | |
| 177 | + | export function telemetryQuestions({ script, from, to, samples, serviceKey = SERVICE_KEYS[0] }) { | |
| 178 | + | const service = filter(serviceKey, "eq", script); | |
| 179 | + | const failed = filter(KEYS.outcome, "neq", "ok"); | |
| 180 | + | const base = { timeframe: { from, to }, limit: 100 }; | |
| 181 | + | const count = [{ operator: "count", alias: "events" }]; | |
| 182 | + | const group = (...keys) => keys.map((value) => ({ type: "string", value })); | |
| 183 | + | return [ | |
| 184 | + | { | |
| 185 | + | key: "failed_invocations", | |
| 186 | + | label: "Failed invocations by class, event and outcome", | |
| 187 | + | body: { | |
| 188 | + | ...base, | |
| 189 | + | queryId: "g1t-runner-failed-invocations", | |
| 190 | + | view: "calculations", | |
| 191 | + | parameters: { datasets: ["cloudflare-workers"], filters: [service, failed], calculations: count, groupBys: group(KEYS.entrypoint, KEYS.eventType, KEYS.outcome) }, | |
| 192 | + | }, | |
| 193 | + | }, | |
| 194 | + | { | |
| 195 | + | key: "exceptions", | |
| 196 | + | label: "Exceptions by message", | |
| 197 | + | body: { | |
| 198 | + | ...base, | |
| 199 | + | queryId: "g1t-runner-exceptions", | |
| 200 | + | view: "calculations", | |
| 201 | + | parameters: { datasets: ["cloudflare-workers"], filters: [service, filter(KEYS.error, "exists")], calculations: count, groupBys: group(KEYS.error) }, | |
| 202 | + | }, | |
| 203 | + | }, | |
| 204 | + | { | |
| 205 | + | key: "error_logs", | |
| 206 | + | label: "The runner's error-level logs by message", | |
| 207 | + | body: { | |
| 208 | + | ...base, | |
| 209 | + | queryId: "g1t-runner-error-logs", | |
| 210 | + | view: "calculations", | |
| 211 | + | parameters: { datasets: ["cloudflare-workers"], filters: [service, filter(KEYS.level, "eq", "error")], calculations: count, groupBys: group(KEYS.message) }, | |
| 212 | + | }, | |
| 213 | + | }, | |
| 214 | + | { | |
| 215 | + | key: "samples", | |
| 216 | + | label: "Recent failed invocations", | |
| 217 | + | body: { | |
| 218 | + | ...base, | |
| 219 | + | limit: Math.max(1, samples), | |
| 220 | + | queryId: "g1t-runner-failed-samples", | |
| 221 | + | view: "events", | |
| 222 | + | parameters: { datasets: ["cloudflare-workers"], filters: [service, failed] }, | |
| 223 | + | }, | |
| 224 | + | }, | |
| 225 | + | ]; | |
| 226 | + | } | |
| 227 | + | ||
| 228 | + | /** | |
| 229 | + | * A telemetry calculations answer as rows: each group's key values and its | |
| 230 | + | * count, largest first. Throws with Cloudflare's message when it failed. | |
| 231 | + | */ | |
| 232 | + | export function parseCalculations(body) { | |
| 233 | + | if (body?.success === false || body?.errors?.length) throw new Error(messagesOf(body)); | |
| 234 | + | const calculation = body?.result?.calculations?.[0]; | |
| 235 | + | if (!calculation) throw new Error("no calculations in the answer"); | |
| 236 | + | const rows = (calculation.aggregates ?? []).map((aggregate) => { | |
| 237 | + | const row = {}; | |
| 238 | + | for (const group of aggregate.groups ?? []) row[group.key] = group.value == null || group.value === "" ? "(none)" : String(group.value); | |
| 239 | + | row.count = Number(aggregate.value ?? aggregate.count ?? 0); | |
| 240 | + | return row; | |
| 241 | + | }); | |
| 242 | + | return rows.sort((a, b) => b.count - a.count); | |
| 243 | + | } | |
| 244 | + | ||
| 245 | + | /** A telemetry events answer as plain records: when, which class, event, outcome, and what went wrong. */ | |
| 246 | + | export function parseEvents(body) { | |
| 247 | + | if (body?.success === false || body?.errors?.length) throw new Error(messagesOf(body)); | |
| 248 | + | const events = body?.result?.events?.events ?? body?.result?.events ?? []; | |
| 249 | + | if (!Array.isArray(events)) throw new Error("no events in the answer"); | |
| 250 | + | return events.map((event) => { | |
| 251 | + | const workers = event.$workers ?? {}; | |
| 252 | + | const metadata = event.$metadata ?? {}; | |
| 253 | + | const at = event.timestamp ?? metadata.startTime ?? workers.timestamp; | |
| 254 | + | return { | |
| 255 | + | at: typeof at === "number" ? new Date(at).toISOString() : (at ?? null), | |
| 256 | + | entrypoint: workers.entrypoint ?? null, | |
| 257 | + | event_type: workers.eventType ?? null, | |
| 258 | + | outcome: workers.outcome ?? null, | |
| 259 | + | object: typeof workers.durableObjectId === "string" ? workers.durableObjectId.slice(0, 12) : null, | |
| 260 | + | error: metadata.error ?? null, | |
| 261 | + | message: metadata.message ?? (typeof event.source === "string" ? event.source : (event.source?.message ?? null)), | |
| 262 | + | }; | |
| 263 | + | }); | |
| 264 | + | } | |
| 265 | + | ||
| 266 | + | function messagesOf(body) { | |
| 267 | + | const errors = body?.errors ?? []; | |
| 268 | + | const text = errors.map((error) => error.message ?? JSON.stringify(error)).join("; "); | |
| 269 | + | return (text || `request failed${body?.status ? ` with ${body.status}` : ""}`).slice(0, 400); | |
| 270 | + | } | |
| 271 | + | ||
| 272 | + | /** | |
| 273 | + | * What a failure most likely is, from its message or outcome: a bucket and | |
| 274 | + | * whether it is expected. Order matters: the first match wins. | |
| 275 | + | */ | |
| 276 | + | export const BUCKETS = [ | |
| 277 | + | { bucket: "deploy_reset", expected: true, why: "a runner deploy reset the object mid-invocation", test: /code (was|has been) updated|reset because its code|new version of the (script|worker)|durable object reset/i }, | |
| 278 | + | { bucket: "stop_not_reported", expected: false, why: "onStop could not tell a service the sandbox stopped (now logged, not thrown)", test: /sandbox stop not reported/i }, | |
| 279 | + | { bucket: "alarm_failed", expected: false, why: "the sandbox's alarm threw (retried by Cloudflare)", test: /sandbox alarm failed/i }, | |
| 280 | + | { bucket: "no_capacity", expected: true, why: "no container instance free (max_instances, or provisioning)", test: /no container instance|max(imum)? concurrent instance|too many containers per second/i }, | |
| 281 | + | { bucket: "not_started", expected: false, why: "a sandbox could not start (guardrails unreadable, or the container would not start)", test: /sandbox not started|could not read this project's guardrails|did not start after|failed to start container/i }, | |
| 282 | + | { bucket: "container_exited", expected: true, why: "the container exited or was stopped (a finished run, a time cap, a stop)", test: /container exited|runtime signalled|exited before we could determine|crashed while checking for ports|exit code/i }, | |
| 283 | + | { bucket: "connection_lost", expected: false, why: "the connection to the container was lost", test: /network connection lost|disconnected/i }, | |
| 284 | + | { bucket: "storage", expected: false, why: "Durable Object storage failed or was overloaded", test: /storage|sqlite|overloaded/i }, | |
| 285 | + | { bucket: "limits", expected: false, why: "a CPU, memory or subrequest limit", test: /exceeded|too many subrequests|memory limit/i }, | |
| 286 | + | { bucket: "container_error", expected: false, why: "the containers library reported an error", test: /sandbox container error|container error/i }, | |
| 287 | + | ]; | |
| 288 | + | ||
| 289 | + | /** The bucket of one failure's message (or, failing that, its outcome). */ | |
| 290 | + | export function classify(message, outcome = null) { | |
| 291 | + | const text = String(message ?? ""); | |
| 292 | + | for (const bucket of BUCKETS) if (text && bucket.test.test(text)) return { bucket: bucket.bucket, expected: bucket.expected, why: bucket.why }; | |
| 293 | + | if (/canceled|cancelled|clientdisconnected|responsestreamdisconnected/i.test(String(outcome ?? ""))) { | |
| 294 | + | return { bucket: "caller_gone", expected: true, why: "the caller went away before the object answered" }; | |
| 295 | + | } | |
| 296 | + | return { bucket: "other", expected: false, why: "not recognised: read the message" }; | |
| 297 | + | } | |
| 298 | + | ||
| 299 | + | /** Rows of messages with counts, summed into buckets, largest first. */ | |
| 300 | + | export function bucketsOf(rows, messageKey) { | |
| 301 | + | const out = new Map(); | |
| 302 | + | for (const row of rows) { | |
| 303 | + | const found = classify(row[messageKey], row[KEYS.outcome]); | |
| 304 | + | const entry = out.get(found.bucket) ?? { ...found, count: 0 }; | |
| 305 | + | entry.count += row.count; | |
| 306 | + | out.set(found.bucket, entry); | |
| 307 | + | } | |
| 308 | + | return [...out.values()].sort((a, b) => b.count - a.count); | |
| 309 | + | } | |
| 310 | + | ||
| 311 | + | const number = (value) => (Number.isInteger(value) ? value.toLocaleString("en-US") : value.toLocaleString("en-US", { maximumFractionDigits: 2 })); | |
| 312 | + | const percent = (part, whole) => (whole > 0 ? `${((100 * part) / whole).toFixed(1)}%` : "-"); | |
| 313 | + | ||
| 314 | + | function table(rows, columns) { | |
| 315 | + | if (!rows.length) return [" (nothing in this window)"]; | |
| 316 | + | const cells = [columns.map((c) => c.title), ...rows.map((row) => columns.map((c) => c.value(row)))]; | |
| 317 | + | const widths = columns.map((_, at) => Math.max(...cells.map((row) => String(row[at]).length))); | |
| 318 | + | return cells.map((row) => " " + row.map((cell, at) => (columns[at].right ? String(cell).padStart(widths[at]) : String(cell).padEnd(widths[at]))).join(" ")); | |
| 319 | + | } | |
| 320 | + | ||
| 321 | + | /** The report as plain text. */ | |
| 322 | + | export function format(report, top = 25) { | |
| 323 | + | const lines = [`Durable Object errors of ${report.script} on account ${report.account}, ${report.start} to ${report.end}`]; | |
| 324 | + | for (const query of report.analytics) { | |
| 325 | + | lines.push("", query.label); | |
| 326 | + | if (query.error) { | |
| 327 | + | lines.push(` could not read: ${query.error}`); | |
| 328 | + | continue; | |
| 329 | + | } | |
| 330 | + | const columns = [ | |
| 331 | + | ...query.dims.map((dim) => ({ title: dim, value: (row) => row[dim] })), | |
| 332 | + | ...query.metrics.map((metric) => ({ title: metric, right: true, value: (row) => number(row[metric]) })), | |
| 333 | + | ]; | |
| 334 | + | if (query.metrics.includes("errors")) columns.push({ title: "error_rate", right: true, value: (row) => percent(row.errors, row.requests) }); | |
| 335 | + | lines.push(...table(query.rows.slice(0, top), columns)); | |
| 336 | + | lines.push(` total: ${Object.entries(query.totals).map(([metric, value]) => `${metric} ${number(value)}`).join(", ")}`); | |
| 337 | + | if (query.by_status?.length) { | |
| 338 | + | lines.push(" by status:"); | |
| 339 | + | lines.push(...table(query.by_status, [ | |
| 340 | + | { title: "status", value: (row) => row.status }, | |
| 341 | + | { title: "requests", right: true, value: (row) => number(row.requests) }, | |
| 342 | + | { title: "errors", right: true, value: (row) => number(row.errors) }, | |
| 343 | + | ]).map((line) => ` ${line}`)); | |
| 344 | + | } | |
| 345 | + | } | |
| 346 | + | for (const question of report.telemetry) { | |
| 347 | + | lines.push("", question.label); | |
| 348 | + | if (question.error) { | |
| 349 | + | lines.push(` could not read: ${question.error}`); | |
| 350 | + | continue; | |
| 351 | + | } | |
| 352 | + | if (question.key === "samples") { | |
| 353 | + | if (!question.rows.length) lines.push(" (none)"); | |
| 354 | + | for (const row of question.rows) { | |
| 355 | + | lines.push(` ${row.at ?? "?"} ${row.entrypoint ?? "?"} ${row.event_type ?? "?"} ${row.outcome ?? "?"}${row.object ? ` ${row.object}` : ""}`); | |
| 356 | + | if (row.error || row.message) lines.push(` ${String(row.error ?? row.message).slice(0, 300)}`); | |
| 357 | + | } | |
| 358 | + | continue; | |
| 359 | + | } | |
| 360 | + | const keys = Object.keys(question.rows[0] ?? {}).filter((key) => key !== "count"); | |
| 361 | + | lines.push(...table(question.rows.slice(0, top), [ | |
| 362 | + | ...keys.map((key) => ({ title: key, value: (row) => String(row[key] ?? "").slice(0, 120) })), | |
| 363 | + | { title: "count", right: true, value: (row) => number(row.count) }, | |
| 364 | + | ])); | |
| 365 | + | if (question.buckets?.length) { | |
| 366 | + | lines.push(" most likely:"); | |
| 367 | + | for (const bucket of question.buckets) lines.push(` ${String(number(bucket.count)).padStart(6)} ${bucket.bucket}${bucket.expected ? " (expected)" : ""}: ${bucket.why}`); | |
| 368 | + | } | |
| 369 | + | } | |
| 370 | + | if (report.keys) lines.push("", "Telemetry keys:", ...report.keys.map((key) => ` ${key}`)); | |
| 371 | + | if (report.notes.length) lines.push("", "Notes:", ...report.notes.map((note) => ` ${note}`)); | |
| 372 | + | return lines.join("\n"); | |
| 373 | + | } | |
| 374 | + | ||
| 375 | + | async function post(auth, path, body) { | |
| 376 | + | const response = await fetch(`${API}${path}`, { | |
| 377 | + | method: "POST", | |
| 378 | + | headers: { ...auth, "content-type": "application/json", "user-agent": "g1t-ops" }, | |
| 379 | + | body: JSON.stringify(body), | |
| 380 | + | }); | |
| 381 | + | const answer = await response.json().catch(() => ({ success: false, errors: [{ message: `HTTP ${response.status}, not JSON` }] })); | |
| 382 | + | if (!response.ok && !answer.errors?.length) answer.errors = [{ message: `HTTP ${response.status}` }]; | |
| 383 | + | return answer; | |
| 384 | + | } | |
| 385 | + | ||
| 386 | + | async function durableNames(auth, account) { | |
| 387 | + | const out = {}; | |
| 388 | + | try { | |
| 389 | + | for (let page = 1; page <= 10; page++) { | |
| 390 | + | const response = await fetch(`${API}/accounts/${account}/workers/durable_objects/namespaces?per_page=100&page=${page}`, { headers: { ...auth, "user-agent": "g1t-ops" } }); | |
| 391 | + | const body = await response.json(); | |
| 392 | + | if (!response.ok || !Array.isArray(body.result)) break; | |
| 393 | + | for (const item of body.result) if (item.id) out[item.id] = item.name ?? item.id; | |
| 394 | + | if (body.result.length < 100) break; | |
| 395 | + | } | |
| 396 | + | } catch { | |
| 397 | + | // Ids stand in for names. | |
| 398 | + | } | |
| 399 | + | return out; | |
| 400 | + | } | |
| 401 | + | ||
| 402 | + | async function analytics(auth, options, window, names) { | |
| 403 | + | return Promise.all( | |
| 404 | + | DO_QUERIES.map(async (query) => { | |
| 405 | + | let error = null; | |
| 406 | + | for (const variant of query.variants) { | |
| 407 | + | try { | |
| 408 | + | const body = await post(auth, "/graphql", { query: doQuery(variant), variables: { accountTag: options.account, start: window.start, end: window.end, script: options.script } }); | |
| 409 | + | const parsed = parseDoGroups(body, variant, names); | |
| 410 | + | return { key: query.key, label: query.label, ...parsed, by_status: query.key === "by_namespace" ? byStatus(parsed) : undefined }; | |
| 411 | + | } catch (thrown) { | |
| 412 | + | error ??= String(thrown.message ?? thrown); | |
| 413 | + | } | |
| 414 | + | } | |
| 415 | + | return { key: query.key, label: query.label, error }; | |
| 416 | + | }), | |
| 417 | + | ); | |
| 418 | + | } | |
| 419 | + | ||
| 420 | + | async function telemetry(auth, options, window) { | |
| 421 | + | const path = `/accounts/${options.account}/workers/observability/telemetry/query`; | |
| 422 | + | const from = Date.parse(window.start); | |
| 423 | + | const to = Date.parse(window.end); | |
| 424 | + | let results = null; | |
| 425 | + | for (const serviceKey of SERVICE_KEYS) { | |
| 426 | + | results = await Promise.all( | |
| 427 | + | telemetryQuestions({ script: options.script, from, to, samples: options.samples, serviceKey }).map(async (question) => { | |
| 428 | + | try { | |
| 429 | + | const body = await post(auth, path, question.body); | |
| 430 | + | if (question.key === "samples") return { key: question.key, label: question.label, rows: parseEvents(body) }; | |
| 431 | + | const rows = parseCalculations(body); | |
| 432 | + | const messageKey = question.key === "exceptions" ? KEYS.error : question.key === "error_logs" ? KEYS.message : null; | |
| 433 | + | return { key: question.key, label: question.label, rows, buckets: messageKey ? bucketsOf(rows, messageKey) : undefined }; | |
| 434 | + | } catch (error) { | |
| 435 | + | return { key: question.key, label: question.label, error: String(error.message ?? error) }; | |
| 436 | + | } | |
| 437 | + | }), | |
| 438 | + | ); | |
| 439 | + | // Another key names the Worker when this one found nothing at all. | |
| 440 | + | if (results.some((result) => result.rows?.length)) return { serviceKey, results }; | |
| 441 | + | } | |
| 442 | + | return { serviceKey: SERVICE_KEYS.at(-1), results }; | |
| 443 | + | } | |
| 444 | + | ||
| 445 | + | async function telemetryKeys(auth, options, window) { | |
| 446 | + | const body = await post(auth, `/accounts/${options.account}/workers/observability/telemetry/keys`, { | |
| 447 | + | timeframe: { from: Date.parse(window.start), to: Date.parse(window.end) }, | |
| 448 | + | datasets: ["cloudflare-workers"], | |
| 449 | + | filters: [filter(SERVICE_KEYS[0], "eq", options.script)], | |
| 450 | + | limit: 500, | |
| 451 | + | }); | |
| 452 | + | if (body?.success === false || body?.errors?.length) throw new Error(messagesOf(body)); | |
| 453 | + | return (body.result ?? []).map((key) => (typeof key === "string" ? key : `${key.key} (${key.type})`)).sort(); | |
| 454 | + | } | |
| 455 | + | ||
| 456 | + | async function main() { | |
| 457 | + | const options = parseArgs(process.argv.slice(2)); | |
| 458 | + | if (options.help) { | |
| 459 | + | console.log(HELP); | |
| 460 | + | return 0; | |
| 461 | + | } | |
| 462 | + | const auth = cloudflareAuth(); | |
| 463 | + | if (!auth) { | |
| 464 | + | console.error(`Set CLOUDFLARE_API_TOKEN to a token with Account Analytics: Read and Workers Observability: Read on account ${options.account}.`); | |
| 465 | + | return 2; | |
| 466 | + | } | |
| 467 | + | const window = windowOf(new Date(), options.days); | |
| 468 | + | const names = await durableNames(auth, options.account); | |
| 469 | + | const [graph, logs, keys] = await Promise.all([ | |
| 470 | + | analytics(auth, options, window, names), | |
| 471 | + | telemetry(auth, options, window), | |
| 472 | + | options.keys ? telemetryKeys(auth, options, window).catch((error) => [`could not list keys: ${error.message}`]) : Promise.resolve(null), | |
| 473 | + | ]); | |
| 474 | + | const notes = []; | |
| 475 | + | if (!Object.keys(names).length) notes.push("Namespace ids are not named: the token cannot read the Durable Objects listing (Workers Scripts: Read)."); | |
| 476 | + | notes.push(`Telemetry is filtered on ${logs.serviceKey}.`); | |
| 477 | + | const report = { account: options.account, script: options.script, start: window.start, end: window.end, analytics: graph, telemetry: logs.results, keys, notes }; | |
| 478 | + | console.log(options.json ? JSON.stringify(report, null, 2) : format(report)); | |
| 479 | + | const read = [...graph, ...logs.results].some((part) => !part.error); | |
| 480 | + | return read ? 0 : 1; | |
| 481 | + | } | |
| 482 | + | ||
| 483 | + | if (process.argv[1]?.replaceAll("\\", "/").endsWith("scripts/ops/runner-errors.mjs")) { | |
| 484 | + | main().then( | |
| 485 | + | (code) => process.exit(code), | |
| 486 | + | (error) => { | |
| 487 | + | console.error(`runner-errors: ${error.message}`); | |
| 488 | + | process.exit(1); | |
| 489 | + | }, | |
| 490 | + | ); | |
| 491 | + | } |
| 1 | + | // The runner's Durable Object error report, from answers shaped as | |
| 2 | + | // Cloudflare gives them, without the network. | |
| 3 | + | ||
| 4 | + | import assert from "node:assert/strict"; | |
| 5 | + | import { test } from "node:test"; | |
| 6 | + | ||
| 7 | + | import { | |
| 8 | + | DO_QUERIES, | |
| 9 | + | KEYS, | |
| 10 | + | SERVICE_KEYS, | |
| 11 | + | bucketsOf, | |
| 12 | + | byStatus, | |
| 13 | + | classify, | |
| 14 | + | doQuery, | |
| 15 | + | format, | |
| 16 | + | parseArgs, | |
| 17 | + | parseCalculations, | |
| 18 | + | parseDoGroups, | |
| 19 | + | parseEvents, | |
| 20 | + | telemetryQuestions, | |
| 21 | + | windowOf, | |
| 22 | + | } from "./runner-errors.mjs"; | |
| 23 | + | ||
| 24 | + | const answer = (rows) => ({ data: { viewer: { accounts: [{ rows }] } } }); | |
| 25 | + | const byNamespace = DO_QUERIES.find((query) => query.key === "by_namespace").variants[1]; | |
| 26 | + | ||
| 27 | + | test("the command line has defaults and bounds", () => { | |
| 28 | + | assert.deepEqual(parseArgs([], {}), { help: false, json: false, keys: false, days: 7, samples: 10, script: "g1t-runner", account: "1e6f2cffa3f445920836e8ebe446bb58" }); | |
| 29 | + | const options = parseArgs(["--days", "90", "--samples", "3", "--script", "other", "--json", "--keys"], { CLOUDFLARE_ACCOUNT_ID: "acct" }); | |
| 30 | + | assert.equal(options.days, 31); | |
| 31 | + | assert.equal(options.samples, 3); | |
| 32 | + | assert.equal(options.script, "other"); | |
| 33 | + | assert.equal(options.account, "acct"); | |
| 34 | + | assert.ok(options.json && options.keys); | |
| 35 | + | assert.equal(parseArgs(["--days", "soon"], {}).days, 7); | |
| 36 | + | }); | |
| 37 | + | ||
| 38 | + | test("the window is the last N days", () => { | |
| 39 | + | assert.deepEqual(windowOf(new Date("2026-10-08T12:00:00Z"), 7), { start: "2026-10-01T12:00:00.000Z", end: "2026-10-08T12:00:00.000Z" }); | |
| 40 | + | }); | |
| 41 | + | ||
| 42 | + | test("the query is for one script, with the variant's sums and dimensions", () => { | |
| 43 | + | const query = doQuery(byNamespace); | |
| 44 | + | assert.match(query, /durableObjectsInvocationsAdaptiveGroups\(limit: 10000, filter: \{ datetime_geq: \$start, datetime_leq: \$end, scriptName: \$script \}\)/); | |
| 45 | + | assert.match(query, /sum \{ requests errors \}/); | |
| 46 | + | assert.match(query, /dimensions \{ namespaceId status \}/); | |
| 47 | + | assert.match(query, /\$script: String!/); | |
| 48 | + | }); | |
| 49 | + | ||
| 50 | + | test("rows are named, summed per status and sorted by requests", () => { | |
| 51 | + | const body = answer([ | |
| 52 | + | { sum: { requests: 900, errors: 300 }, dimensions: { namespaceId: "ns1", status: "scriptThrewException" } }, | |
| 53 | + | { sum: { requests: 1000, errors: 0 }, dimensions: { namespaceId: "ns1", status: "success" } }, | |
| 54 | + | { sum: { requests: 40, errors: 20 }, dimensions: { namespaceId: "ns2", status: "internalError" } }, | |
| 55 | + | { sum: { requests: 10, errors: 0 }, dimensions: { namespaceId: "ns2", status: "success" } }, | |
| 56 | + | ]); | |
| 57 | + | const parsed = parseDoGroups(body, byNamespace, { ns1: "g1t-runner_AttemptSandbox" }); | |
| 58 | + | assert.deepEqual(parsed.dims, ["namespace_id", "status"]); | |
| 59 | + | assert.deepEqual(parsed.rows[0], { namespace_id: "g1t-runner_AttemptSandbox", status: "success", requests: 1000, errors: 0 }); | |
| 60 | + | assert.equal(parsed.rows.at(-1).namespace_id, "ns2"); | |
| 61 | + | assert.deepEqual(parsed.totals, { requests: 1950, errors: 320 }); | |
| 62 | + | assert.deepEqual(byStatus(parsed), [ | |
| 63 | + | { status: "success", requests: 1010, errors: 0 }, | |
| 64 | + | { status: "scriptThrewException", requests: 900, errors: 300 }, | |
| 65 | + | { status: "internalError", requests: 40, errors: 20 }, | |
| 66 | + | ]); | |
| 67 | + | }); | |
| 68 | + | ||
| 69 | + | test("Cloudflare's refusal becomes the error, so the next variant is tried", () => { | |
| 70 | + | assert.throws(() => parseDoGroups({ errors: [{ message: "unknown field wallTime" }] }, byNamespace), /unknown field wallTime/); | |
| 71 | + | assert.throws(() => parseDoGroups({ data: {} }, byNamespace), /no rows/); | |
| 72 | + | }); | |
| 73 | + | ||
| 74 | + | test("the telemetry questions filter on the script and on failures", () => { | |
| 75 | + | const questions = telemetryQuestions({ script: "g1t-runner", from: 1, to: 2, samples: 5 }); | |
| 76 | + | assert.deepEqual(questions.map((q) => q.key), ["failed_invocations", "exceptions", "error_logs", "samples"]); | |
| 77 | + | const failed = questions[0].body; | |
| 78 | + | assert.equal(failed.view, "calculations"); | |
| 79 | + | assert.deepEqual(failed.timeframe, { from: 1, to: 2 }); | |
| 80 | + | assert.deepEqual(failed.parameters.filters, [ | |
| 81 | + | { key: SERVICE_KEYS[0], operation: "eq", type: "string", value: "g1t-runner" }, | |
| 82 | + | { key: KEYS.outcome, operation: "neq", type: "string", value: "ok" }, | |
| 83 | + | ]); | |
| 84 | + | assert.deepEqual(failed.parameters.groupBys.map((g) => g.value), [KEYS.entrypoint, KEYS.eventType, KEYS.outcome]); | |
| 85 | + | assert.equal(questions[3].body.view, "events"); | |
| 86 | + | assert.equal(questions[3].body.limit, 5); | |
| 87 | + | const other = telemetryQuestions({ script: "g1t-runner", from: 1, to: 2, samples: 5, serviceKey: "$metadata.service" }); | |
| 88 | + | assert.equal(other[1].body.parameters.filters[0].key, "$metadata.service"); | |
| 89 | + | }); | |
| 90 | + | ||
| 91 | + | test("a calculations answer becomes rows with counts, largest first", () => { | |
| 92 | + | const body = { | |
| 93 | + | success: true, | |
| 94 | + | result: { | |
| 95 | + | calculations: [ | |
| 96 | + | { | |
| 97 | + | aggregates: [ | |
| 98 | + | { groups: [{ key: KEYS.entrypoint, value: "AttemptSandbox" }, { key: KEYS.eventType, value: "rpc" }], value: 12 }, | |
| 99 | + | { groups: [{ key: KEYS.entrypoint, value: "AttemptSandbox" }, { key: KEYS.eventType, value: "alarm" }], value: 540 }, | |
| 100 | + | { groups: [{ key: KEYS.entrypoint, value: "" }, { key: KEYS.eventType, value: "alarm" }], count: 3 }, | |
| 101 | + | ], | |
| 102 | + | }, | |
| 103 | + | ], | |
| 104 | + | }, | |
| 105 | + | }; | |
| 106 | + | assert.deepEqual(parseCalculations(body), [ | |
| 107 | + | { [KEYS.entrypoint]: "AttemptSandbox", [KEYS.eventType]: "alarm", count: 540 }, | |
| 108 | + | { [KEYS.entrypoint]: "AttemptSandbox", [KEYS.eventType]: "rpc", count: 12 }, | |
| 109 | + | { [KEYS.entrypoint]: "(none)", [KEYS.eventType]: "alarm", count: 3 }, | |
| 110 | + | ]); | |
| 111 | + | assert.throws(() => parseCalculations({ success: false, errors: [{ message: "Unauthorized" }] }), /Unauthorized/); | |
| 112 | + | assert.throws(() => parseCalculations({ success: true, result: {} }), /no calculations/); | |
| 113 | + | }); | |
| 114 | + | ||
| 115 | + | test("an events answer becomes plain records, the object id cut short", () => { | |
| 116 | + | const body = { | |
| 117 | + | success: true, | |
| 118 | + | result: { | |
| 119 | + | events: { | |
| 120 | + | events: [ | |
| 121 | + | { | |
| 122 | + | timestamp: Date.UTC(2026, 9, 7, 3, 4, 5), | |
| 123 | + | $workers: { entrypoint: "AttemptSandbox", eventType: "alarm", outcome: "exception", durableObjectId: "0123456789abcdef0123" }, | |
| 124 | + | $metadata: { error: "Durable Object reset because its code was updated." }, | |
| 125 | + | }, | |
| 126 | + | { $workers: { eventType: "rpc", outcome: "canceled" }, source: { message: "sandbox not started" } }, | |
| 127 | + | ], | |
| 128 | + | }, | |
| 129 | + | }, | |
| 130 | + | }; | |
| 131 | + | assert.deepEqual(parseEvents(body), [ | |
| 132 | + | { at: "2026-10-07T03:04:05.000Z", entrypoint: "AttemptSandbox", event_type: "alarm", outcome: "exception", object: "0123456789ab", error: "Durable Object reset because its code was updated.", message: null }, | |
| 133 | + | { at: null, entrypoint: null, event_type: "rpc", outcome: "canceled", object: null, error: null, message: "sandbox not started" }, | |
| 134 | + | ]); | |
| 135 | + | }); | |
| 136 | + | ||
| 137 | + | test("messages fall into buckets, expected or not", () => { | |
| 138 | + | assert.equal(classify("Durable Object reset because its code was updated.").bucket, "deploy_reset"); | |
| 139 | + | assert.equal(classify("Durable Object reset because its code was updated.").expected, true); | |
| 140 | + | assert.equal(classify("sandbox stop not reported checks report_checks failed with status 500").bucket, "stop_not_reported"); | |
| 141 | + | assert.equal(classify("sandbox alarm failed abc retry 2 storage overloaded").bucket, "alarm_failed"); | |
| 142 | + | assert.equal(classify("there is no container instance that can be provided to this durable object").bucket, "no_capacity"); | |
| 143 | + | assert.equal(classify("sandbox not started agent g1t could not read this project's guardrails: down").bucket, "not_started"); | |
| 144 | + | assert.equal(classify("container exited with unexpected exit code: 137").bucket, "container_exited"); | |
| 145 | + | assert.equal(classify("Network connection lost.").bucket, "connection_lost"); | |
| 146 | + | assert.equal(classify("", "canceled").bucket, "caller_gone"); | |
| 147 | + | assert.equal(classify("something new").bucket, "other"); | |
| 148 | + | assert.equal(classify("something new").expected, false); | |
| 149 | + | }); | |
| 150 | + | ||
| 151 | + | test("buckets sum the counts of their messages", () => { | |
| 152 | + | const rows = [ | |
| 153 | + | { [KEYS.error]: "Durable Object reset because its code was updated.", count: 200 }, | |
| 154 | + | { [KEYS.error]: "Network connection lost.", count: 7 }, | |
| 155 | + | { [KEYS.error]: "Durable Object reset because its code was updated (2).", count: 50 }, | |
| 156 | + | ]; | |
| 157 | + | assert.deepEqual( | |
| 158 | + | bucketsOf(rows, KEYS.error).map((b) => [b.bucket, b.count]), | |
| 159 | + | [ | |
| 160 | + | ["deploy_reset", 250], | |
| 161 | + | ["connection_lost", 7], | |
| 162 | + | ], | |
| 163 | + | ); | |
| 164 | + | }); | |
| 165 | + | ||
| 166 | + | test("the text report shows rates, statuses, buckets and what could not be read", () => { | |
| 167 | + | const parsed = parseDoGroups( | |
| 168 | + | answer([ | |
| 169 | + | { sum: { requests: 100, errors: 31 }, dimensions: { namespaceId: "ns1", status: "scriptThrewException" } }, | |
| 170 | + | { sum: { requests: 200, errors: 0 }, dimensions: { namespaceId: "ns1", status: "success" } }, | |
| 171 | + | ]), | |
| 172 | + | byNamespace, | |
| 173 | + | { ns1: "g1t-runner_AttemptSandbox" }, | |
| 174 | + | ); | |
| 175 | + | const rows = [{ [KEYS.error]: "Durable Object reset because its code was updated.", count: 31 }]; | |
| 176 | + | const text = format({ | |
| 177 | + | account: "acct", | |
| 178 | + | script: "g1t-runner", | |
| 179 | + | start: "2026-10-01T00:00:00.000Z", | |
| 180 | + | end: "2026-10-08T00:00:00.000Z", | |
| 181 | + | analytics: [ | |
| 182 | + | { key: "by_namespace", label: "By namespace", ...parsed, by_status: byStatus(parsed) }, | |
| 183 | + | { key: "by_day", label: "By day", error: "unknown field date" }, | |
| 184 | + | ], | |
| 185 | + | telemetry: [ | |
| 186 | + | { key: "exceptions", label: "Exceptions", rows, buckets: bucketsOf(rows, KEYS.error) }, | |
| 187 | + | { key: "samples", label: "Samples", rows: [{ at: "2026-10-07T00:00:00.000Z", entrypoint: "AttemptSandbox", event_type: "alarm", outcome: "exception", object: "abc", error: "boom", message: null }] }, | |
| 188 | + | ], | |
| 189 | + | keys: null, | |
| 190 | + | notes: ["Telemetry is filtered on $workers.scriptName."], | |
| 191 | + | }); | |
| 192 | + | assert.match(text, /g1t-runner_AttemptSandbox\s+scriptThrewException\s+100\s+31\s+31\.0%/); | |
| 193 | + | assert.match(text, /by status:/); | |
| 194 | + | assert.match(text, /could not read: unknown field date/); | |
| 195 | + | assert.match(text, /31\s+deploy_reset \(expected\)/); | |
| 196 | + | assert.match(text, /AttemptSandbox alarm exception abc/); | |
| 197 | + | assert.match(text, /\n {4}boom/); | |
| 198 | + | assert.doesNotMatch(text, /Bearer|authorization/i); | |
| 199 | + | }); |
| 1 | + | #!/usr/bin/env bash | |
| 2 | + | # sccache for a workflow's Rust builds. Every rustc call that makes a | |
| 3 | + | # library is looked up by what it compiles (sources, flags, toolchain, | |
| 4 | + | # dependencies) in the repository's Actions cache, the one `actions/cache` | |
| 5 | + | # uses, through sccache's GitHub Actions backend. A checkout gives every | |
| 6 | + | # source file a new mtime, so Cargo calls rustc again for the workspace's | |
| 7 | + | # own crates however much of `target/` was restored; with sccache, the | |
| 8 | + | # crates whose inputs did not change come back from the cache instead of | |
| 9 | + | # being compiled. Binaries, cdylibs (a Worker's own crate), proc macros and | |
| 10 | + | # test harnesses are linked, and sccache compiles those as before. | |
| 11 | + | # | |
| 12 | + | # bash scripts/sccache.sh install # a step before the first cargo | |
| 13 | + | # bash scripts/sccache.sh stats # the last step, with if: always() | |
| 14 | + | # | |
| 15 | + | # install downloads the pinned release and checks its sha256, starts the | |
| 16 | + | # server (which reads and writes a check entry in the cache), and only then | |
| 17 | + | # sets RUSTC_WRAPPER for the steps after it. If anything there fails, the | |
| 18 | + | # job builds as it did without sccache, and the step says why. | |
| 19 | + | # stats prints the server's hits and misses, and adds them to the job's | |
| 20 | + | # summary. | |
| 21 | + | # | |
| 22 | + | # See docs/DEPLOYING.md ("Build speed") and the Actions guide's | |
| 23 | + | # "Caching Rust builds". | |
| 24 | + | set -euo pipefail | |
| 25 | + | ||
| 26 | + | VERSION=0.18.0 | |
| 27 | + | TARGET=x86_64-unknown-linux-musl | |
| 28 | + | # sccache-v0.18.0-x86_64-unknown-linux-musl.tar.gz, from the release's | |
| 29 | + | # .sha256 file. | |
| 30 | + | SHA256=45f1447fbe231e3037bde351ef70677dd212216c8d62ae7ca409fecc4d6acc89 | |
| 31 | + | ||
| 32 | + | BIN="$HOME/.cargo/bin" | |
| 33 | + | SCCACHE="$BIN/sccache" | |
| 34 | + | ||
| 35 | + | # What sccache and Cargo read, here and in the steps after this one. | |
| 36 | + | export SCCACHE_GHA_ENABLED=true | |
| 37 | + | # One server for the whole job: its statistics are the job's. | |
| 38 | + | export SCCACHE_IDLE_TIMEOUT=0 | |
| 39 | + | # If the server goes away mid-build, rustc runs without it. | |
| 40 | + | export SCCACHE_IGNORE_SERVER_IO_ERROR=1 | |
| 41 | + | # sccache cannot cache incremental builds; a fresh checkout gains nothing | |
| 42 | + | # from them anyway. | |
| 43 | + | export CARGO_INCREMENTAL=0 | |
| 44 | + | ||
| 45 | + | warn() { echo "::warning title=sccache::$1, so this job builds without it"; } | |
| 46 | + | ||
| 47 | + | fetch() { | |
| 48 | + | local name="sccache-v${VERSION}-${TARGET}" | |
| 49 | + | local dir="${RUNNER_TEMP:-/tmp}/sccache-${VERSION}" | |
| 50 | + | if [ -x "$SCCACHE" ] && [ "$("$SCCACHE" --version 2>/dev/null | awk '{print $2}')" = "$VERSION" ]; then | |
| 51 | + | return 0 | |
| 52 | + | fi | |
| 53 | + | mkdir -p "$dir" "$BIN" | |
| 54 | + | curl -fsSL --retry 3 -o "$dir/$name.tar.gz" "https://github.com/mozilla/sccache/releases/download/v${VERSION}/${name}.tar.gz" || return 1 | |
| 55 | + | echo "${SHA256} $dir/$name.tar.gz" | sha256sum -c --quiet - || return 1 | |
| 56 | + | tar -xzf "$dir/$name.tar.gz" -C "$dir" || return 1 | |
| 57 | + | install -m 0755 "$dir/$name/sccache" "$SCCACHE" | |
| 58 | + | } | |
| 59 | + | ||
| 60 | + | install_sccache() { | |
| 61 | + | if [ -z "${ACTIONS_RUNTIME_TOKEN:-}" ] || [ -z "${ACTIONS_CACHE_URL:-}${ACTIONS_RESULTS_URL:-}" ]; then | |
| 62 | + | warn "The job has no Actions cache to use (ACTIONS_RUNTIME_TOKEN, ACTIONS_CACHE_URL)" | |
| 63 | + | return 0 | |
| 64 | + | fi | |
| 65 | + | if ! fetch; then | |
| 66 | + | warn "sccache ${VERSION} could not be downloaded or did not match its checksum" | |
| 67 | + | return 0 | |
| 68 | + | fi | |
| 69 | + | if ! "$SCCACHE" --start-server; then | |
| 70 | + | warn "sccache's server did not start (it could not read the cache)" | |
| 71 | + | return 0 | |
| 72 | + | fi | |
| 73 | + | { | |
| 74 | + | echo "RUSTC_WRAPPER=$SCCACHE" | |
| 75 | + | echo "SCCACHE_GHA_ENABLED=$SCCACHE_GHA_ENABLED" | |
| 76 | + | echo "SCCACHE_IDLE_TIMEOUT=$SCCACHE_IDLE_TIMEOUT" | |
| 77 | + | echo "SCCACHE_IGNORE_SERVER_IO_ERROR=$SCCACHE_IGNORE_SERVER_IO_ERROR" | |
| 78 | + | echo "CARGO_INCREMENTAL=$CARGO_INCREMENTAL" | |
| 79 | + | } >> "$GITHUB_ENV" | |
| 80 | + | echo "sccache ${VERSION}: rustc goes through it from here on, cached in this repository's Actions cache." | |
| 81 | + | } | |
| 82 | + | ||
| 83 | + | stats() { | |
| 84 | + | if [ -z "${RUSTC_WRAPPER:-}" ] || [ ! -x "$SCCACHE" ]; then | |
| 85 | + | echo "sccache was not used in this job." | |
| 86 | + | return 0 | |
| 87 | + | fi | |
| 88 | + | local out | |
| 89 | + | out="$("$SCCACHE" --show-stats 2>&1)" || true | |
| 90 | + | echo "$out" | |
| 91 | + | if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then | |
| 92 | + | { | |
| 93 | + | echo "### sccache" | |
| 94 | + | echo | |
| 95 | + | echo '```' | |
| 96 | + | echo "$out" | |
| 97 | + | echo '```' | |
| 98 | + | } >> "$GITHUB_STEP_SUMMARY" | |
| 99 | + | fi | |
| 100 | + | } | |
| 101 | + | ||
| 102 | + | case "${1:-}" in | |
| 103 | + | install) install_sccache ;; | |
| 104 | + | stats) stats ;; | |
| 105 | + | *) | |
| 106 | + | echo "usage: bash scripts/sccache.sh install|stats" >&2 | |
| 107 | + | exit 2 | |
| 108 | + | ;; | |
| 109 | + | esac |
| 86 | 86 | out | |
| 87 | 87 | } | |
| 88 | 88 | ||
| 89 | + | /// Whether a repository holding `total` bytes must evict to stay within | |
| 90 | + | /// `quota`: what `to_evict` would take something from. | |
| 91 | + | pub(crate) fn over_quota(total: u64, quota: u64) -> bool { | |
| 92 | + | total > quota | |
| 93 | + | } | |
| 94 | + | ||
| 89 | 95 | /// A month's GB-months from the bytes held each day so far: each day's | |
| 90 | 96 | /// bytes over 30 days. | |
| 91 | 97 | pub(crate) fn gb_months(days: &[u64]) -> f64 { | |
| ⋯ | |||
| 328 | 334 | if ready.is_none() { | |
| 329 | 335 | return Ok(fail(FailureCode::NotFound, "No upload of that entry is in progress.")); | |
| 330 | 336 | } | |
| 337 | + | // What the repository holds, summed: only past the quota are its | |
| 338 | + | // entries listed to choose what to evict. A tool that saves an | |
| 339 | + | // entry per compiled file (sccache) commits thousands of small | |
| 340 | + | // entries a build, and listing them all on each was quadratic. | |
| 331 | 341 | #[derive(Deserialize)] | |
| 342 | + | struct Total { | |
| 343 | + | bytes: Option<f64>, | |
| 344 | + | } | |
| 345 | + | let total = self | |
| 346 | + | .db | |
| 347 | + | .prepare("SELECT SUM(size) AS bytes FROM cache_entries WHERE repo_id = ? AND status = 'ready'") | |
| 348 | + | .bind(&[job.repo_id.as_str().into()])? | |
| 349 | + | .first::<Total>(None) | |
| 350 | + | .await? | |
| 351 | + | .and_then(|t| t.bytes) | |
| 352 | + | .unwrap_or(0.0); | |
| 353 | + | if !over_quota(total as u64, CACHE_REPO_QUOTA_BYTES) { | |
| 354 | + | return Ok(Outcome::Ok(CacheCommitted { evicted: Vec::new() })); | |
| 355 | + | } | |
| 356 | + | #[derive(Deserialize)] | |
| 332 | 357 | struct Held { | |
| 333 | 358 | id: String, | |
| 334 | 359 | object: String, | |
| ⋯ | |||
| 510 | 535 | assert_eq!(to_evict(&held, "new", 7), ["old", "c"]); | |
| 511 | 536 | // The entry just saved stays, even when it alone is past the quota. | |
| 512 | 537 | assert_eq!(to_evict(&entries(&[("big", 20), ("a", 1)]), "big", 10), ["a"]); | |
| 538 | + | // Entries are listed only when the sum is over: at or under the | |
| 539 | + | // quota, nothing would be evicted. | |
| 540 | + | for (held, quota) in [(entries(&[("a", 4), ("b", 6)]), 10), (entries(&[("a", 1)]), 12)] { | |
| 541 | + | let total = held.iter().map(|(_, size)| size).sum(); | |
| 542 | + | assert!(!over_quota(total, quota)); | |
| 543 | + | assert!(to_evict(&held, "a", quota).is_empty()); | |
| 544 | + | } | |
| 545 | + | assert!(over_quota(11, 10)); | |
| 513 | 546 | } | |
| 514 | 547 | ||
| 515 | 548 | #[test] | |
| 203 | 203 | LIST_PRICES.iter().filter(|p| p.product == product && meter.starts_with(p.meter)).max_by_key(|p| p.meter.len()) | |
| 204 | 204 | } | |
| 205 | 205 | ||
| 206 | + | /// What `quantity` of a meter comes to at its list price before any | |
| 207 | + | /// included amount, and not in whole blocks: the price book costs every | |
| 208 | + | /// unit, so this, not what Cloudflare billed past the included amounts, is | |
| 209 | + | /// what it is checked against. None for a meter with no list price. | |
| 210 | + | pub(crate) fn list_cost(product: &str, meter: &str, quantity: f64) -> Option<f64> { | |
| 211 | + | list_price(product, meter).map(|p| quantity.max(0.0) * p.usd / p.per) | |
| 212 | + | } | |
| 213 | + | ||
| 206 | 214 | /// Puts a cost on every billable-usage line, cycle by cycle (`anchor`): | |
| 207 | 215 | /// Cloudflare's own where it put one on any of the meter's lines in the | |
| 208 | 216 | /// cycle, else the list price past the included amount, landing on the | |
| ⋯ | |||
| 400 | 408 | } | |
| 401 | 409 | ||
| 402 | 410 | #[test] | |
| 411 | + | fn a_list_cost_is_every_unit_at_the_list_price() { | |
| 412 | + | // 126.87k GiB-seconds is $0.32 at list, though only the 36.87k past | |
| 413 | + | // the included 90k were billed ($0.09). | |
| 414 | + | let memory = list_cost("containers", "container_memory_per_gib_second", 126_870.0).unwrap(); | |
| 415 | + | assert!((memory - 0.317_175).abs() < 1e-9); | |
| 416 | + | // Per-million meters are not rounded up to a whole million. | |
| 417 | + | assert!((list_cost("workers", "workers_cpu_ms", 11_160_000.0).unwrap() - 0.2232).abs() < 1e-9); | |
| 418 | + | assert!(list_cost("email", "email_service_emails_sent", 7.0).is_none()); | |
| 419 | + | } | |
| 420 | + | ||
| 421 | + | #[test] | |
| 403 | 422 | fn subscriptions_accrue_by_the_cycles_days() { | |
| 404 | 423 | // A whole cycle is the month's price, whatever its length. | |
| 405 | 424 | assert_eq!(accrued(30_000_000, "2026-09-28", "2026-10-27", 28), 30_000_000); | |
| 326 | 326 | seconds.max(0) as f64 * base_per_second + cpu_seconds.max(0.0) * per_vcpu_second | |
| 327 | 327 | } | |
| 328 | 328 | ||
| 329 | − | /// A unit's marginal rate from the bill: the median, over the days that | |
| 330 | − | /// were charged, of cost over quantity. None while nothing was charged. | |
| 329 | + | /// A unit's rate from the bill: the median, over the days with a list | |
| 330 | + | /// cost, of list cost over quantity. Never what was billed: that is net of | |
| 331 | + | /// the included amounts (nothing while a cycle is inside them, part of a | |
| 332 | + | /// day's usage on the day it passes one), so over all of the quantity it | |
| 333 | + | /// reads as a lower price when nothing changed. None without a list cost; | |
| 334 | + | /// the published rates stand in. | |
| 331 | 335 | pub(crate) fn billed_rate(rows: &[&UsageRow]) -> Option<f64> { | |
| 332 | 336 | let mut rates: Vec<f64> = rows | |
| 333 | 337 | .iter() | |
| 334 | − | .filter(|r| r.cost > 0.0 && r.quantity > 0.0) | |
| 335 | − | .map(|r| r.cost / r.quantity) | |
| 338 | + | .filter(|r| r.list_cost > 0.0 && r.quantity > 0.0) | |
| 339 | + | .map(|r| r.list_cost / r.quantity) | |
| 336 | 340 | .collect(); | |
| 337 | 341 | if rates.is_empty() { | |
| 338 | 342 | return None; | |
| ⋯ | |||
| 350 | 354 | unit: String, | |
| 351 | 355 | quantity: f64, | |
| 352 | 356 | cost: f64, | |
| 357 | + | /// The quantity at list price, before the included amounts. | |
| 358 | + | list_cost: f64, | |
| 353 | 359 | } | |
| 354 | 360 | ||
| 355 | 361 | impl UsageRow { | |
| ⋯ | |||
| 376 | 382 | cost: Some(number(&["ContractedCost", "BilledCost", "contracted_cost"])) | |
| 377 | 383 | .filter(|cost| *cost > 0.0) | |
| 378 | 384 | .unwrap_or_else(|| number(&["ListCost", "list_cost"])), | |
| 385 | + | list_cost: number(&["ListCost", "list_cost"]), | |
| 379 | 386 | }) | |
| 380 | 387 | } | |
| 381 | 388 | } | |
| ⋯ | |||
| 879 | 886 | .collect() | |
| 880 | 887 | }; | |
| 881 | 888 | ||
| 882 | − | // Containers: each resource at what the bill shows it costs, or | |
| 883 | − | // the published rate while the included amount still covers it, | |
| 889 | + | // Containers: each resource at the list cost the bill shows for it | |
| 890 | + | // (before the included amounts), or the published rate without one, | |
| 884 | 891 | // over how much CPU g1t's sandboxes really use per second. | |
| 885 | 892 | let memory = billed_rate(&named(&["container memory"])); | |
| 886 | 893 | let disk = billed_rate(&named(&["container disk"])); | |
| ⋯ | |||
| 894 | 901 | durable_object.unwrap_or(LIST_DO_GB_SECOND), | |
| 895 | 902 | ); | |
| 896 | 903 | // The parts, for runs that report their own CPU. | |
| 897 | − | let parts_reason = "Cloudflare's Containers and Durable Objects rates, as billed or published"; | |
| 904 | + | let parts_reason = "Cloudflare's Containers and Durable Objects rates, as listed on the bill or published"; | |
| 898 | 905 | self.measure("sandbox_base_second", sandbox_base_micros(rates.0, rates.1, rates.3), parts_reason).await?; | |
| 899 | 906 | self.measure("sandbox_cpu_second", rates.2 * MICROS_PER_DOLLAR as f64, parts_reason).await?; | |
| 900 | 907 | if let Some(per_second) = sandbox_second_micros(usage, rates.0, rates.1, rates.2, rates.3) { | |
| ⋯ | |||
| 911 | 918 | if billed.is_empty() { | |
| 912 | 919 | "rates are Cloudflare's published ones".to_owned() | |
| 913 | 920 | } else { | |
| 914 | − | format!("{} at what Cloudflare billed", billed.join(", ")) | |
| 921 | + | format!("{} at the list cost on Cloudflare's bill", billed.join(", ")) | |
| 915 | 922 | }, | |
| 916 | 923 | ); | |
| 917 | 924 | for meter in ["sandbox_second", "build_second"] { | |
| ⋯ | |||
| 928 | 935 | ]; | |
| 929 | 936 | for (meter, words, unit) in app_meters { | |
| 930 | 937 | if let Some(rate) = billed_rate(&named(words)) { | |
| 931 | − | let reason = format!("Cloudflare billed Workers {unit} at ${:.2} per million", rate * 1e6); | |
| 938 | + | let reason = format!("Cloudflare's bill lists Workers {unit} at ${:.2} per million", rate * 1e6); | |
| 932 | 939 | self.measure(meter, rate * 1e6 * MICROS_PER_DOLLAR as f64, &reason).await?; | |
| 933 | 940 | } | |
| 934 | 941 | } | |
| ⋯ | |||
| 1071 | 1078 | ||
| 1072 | 1079 | #[test] | |
| 1073 | 1080 | fn a_billed_rate_is_the_median_of_the_charged_days() { | |
| 1074 | − | let row = |quantity: f64, cost: f64| UsageRow { | |
| 1081 | + | let row = |quantity: f64, list_cost: f64| UsageRow { | |
| 1075 | 1082 | period_start: String::new(), | |
| 1076 | 1083 | period_end: String::new(), | |
| 1077 | 1084 | service: "Containers / Container Memory".into(), | |
| 1078 | 1085 | unit: "Count".into(), | |
| 1079 | 1086 | quantity, | |
| 1080 | − | cost, | |
| 1087 | + | cost: list_cost, | |
| 1088 | + | list_cost, | |
| 1081 | 1089 | }; | |
| 1082 | 1090 | let rows = [row(100.0, 0.0), row(100.0, 0.0002), row(100.0, 0.00025), row(100.0, 0.00025)]; | |
| 1083 | 1091 | assert_eq!(billed_rate(&rows.iter().collect::<Vec<_>>()), Some(0.000_002_5)); | |
| ⋯ | |||
| 1085 | 1093 | } | |
| 1086 | 1094 | ||
| 1087 | 1095 | #[test] | |
| 1096 | + | fn a_rate_is_the_list_price_not_what_was_billed_past_the_included_amount() { | |
| 1097 | + | // 126,870 GiB-seconds, of which the 36,870 past the included 90,000 | |
| 1098 | + | // were billed: $0.09 billed, $0.32 at list. The rate is the list's. | |
| 1099 | + | let row = UsageRow { | |
| 1100 | + | period_start: String::new(), | |
| 1101 | + | period_end: String::new(), | |
| 1102 | + | service: "Containers / Container Memory".into(), | |
| 1103 | + | unit: "GiB-seconds".into(), | |
| 1104 | + | quantity: 126_870.0, | |
| 1105 | + | cost: 0.092_175, | |
| 1106 | + | list_cost: 0.317_175, | |
| 1107 | + | }; | |
| 1108 | + | assert!((billed_rate(&[&row]).unwrap() - 0.000_002_5).abs() < 1e-15); | |
| 1109 | + | // Billed with no list cost says nothing about the price. | |
| 1110 | + | assert_eq!(billed_rate(&[&UsageRow { list_cost: 0.0, ..row }]), None); | |
| 1111 | + | } | |
| 1112 | + | ||
| 1113 | + | #[test] | |
| 1088 | 1114 | fn usage_rows_are_read_by_their_focus_names() { | |
| 1089 | 1115 | let row = UsageRow::from_value(&json!({ | |
| 1090 | 1116 | "ServiceFamilyName": "Containers", | |
| ⋯ | |||
| 1092 | 1118 | "PricingUnit": "GiB-seconds", | |
| 1093 | 1119 | "PricingQuantity": "1200.5", | |
| 1094 | 1120 | "ContractedCost": 0.003, | |
| 1121 | + | "ListCost": 0.0030, | |
| 1095 | 1122 | "ChargePeriodStart": "2026-10-01", | |
| 1096 | 1123 | })) | |
| 1097 | 1124 | .unwrap(); | |
| 1098 | 1125 | assert_eq!(row.service, "Containers / Memory"); | |
| 1099 | 1126 | assert_eq!(row.quantity, 1200.5); | |
| 1100 | 1127 | assert_eq!(row.cost, 0.003); | |
| 1128 | + | assert_eq!(row.list_cost, 0.003); | |
| 1101 | 1129 | assert!(UsageRow::from_value(&json!({ "nothing": 1 })).is_none()); | |
| 1102 | 1130 | } | |
| 1103 | 1131 | } | |
| 495 | 495 | pub(crate) enum DriftKind { | |
| 496 | 496 | /// g1t counted a different number of units than Cloudflare did. | |
| 497 | 497 | Count, | |
| 498 | − | /// What Cloudflare charged differs from what the price book says the | |
| 499 | − | /// same usage cost. | |
| 498 | + | /// The price book's cost of a bucket's usage differs from what the | |
| 499 | + | /// same usage comes to at Cloudflare's list prices, before the included | |
| 500 | + | /// amounts (a price may be stale); on `models`, the ledger's model cost | |
| 501 | + | /// differs from what AI Gateway priced the same traffic at. | |
| 500 | 502 | Cost, | |
| 501 | 503 | /// Cloudflare charged for something nothing charges customers for. | |
| 502 | 504 | Leak, | |
| ⋯ | |||
| 526 | 528 | } | |
| 527 | 529 | ||
| 528 | 530 | /// Drift over a window for one bucket: counts more than `threshold` | |
| 529 | − | /// percent apart, a bill that far from the price book's cost of the same | |
| 530 | − | /// usage, and cost with nothing charged for it. Under `min_cost_micros` | |
| 531 | − | /// in all, cost says nothing. | |
| 532 | − | pub(crate) fn drifts(bucket: &str, days: &[ProductDay], threshold: f64, counted: bool, min_cost_micros: i64) -> Vec<Drift> { | |
| 531 | + | /// percent apart, the price book's cost of the usage that far from what the | |
| 532 | + | /// same usage comes to at Cloudflare's list prices (`list_micros`, from | |
| 533 | + | /// `list_costs`), and cost with nothing charged for it. Under | |
| 534 | + | /// `min_cost_micros` in all, cost says nothing. | |
| 535 | + | /// | |
| 536 | + | /// The price book is set against the list cost, never against what | |
| 537 | + | /// Cloudflare billed: the bill is net of the included amounts and the price | |
| 538 | + | /// book's cost is of every unit, so a bucket whose usage mostly fits in | |
| 539 | + | /// them would read as a stale price when nothing changed. Without a list | |
| 540 | + | /// cost (a meter in the bucket with no list price) the price book is not | |
| 541 | + | /// checked. A leak is still what Cloudflare billed: money spent. | |
| 542 | + | pub(crate) fn drifts(bucket: &str, days: &[ProductDay], threshold: f64, counted: bool, min_cost_micros: i64, list_micros: Option<f64>) -> Vec<Drift> { | |
| 533 | 543 | let overhead = OVERHEAD.contains(&bucket); | |
| 534 | 544 | let sum = |f: &dyn Fn(&ProductDay) -> f64| days.iter().map(f).sum::<f64>(); | |
| 535 | 545 | let cf_cost = sum(&|d| d.cf_cost_micros as f64); | |
| ⋯ | |||
| 548 | 558 | out.push(Drift { bucket: bucket.into(), kind: DriftKind::Count, ours: own_quantity, cloudflare: cf_quantity, delta_percent: delta }); | |
| 549 | 559 | } | |
| 550 | 560 | } | |
| 551 | − | let enough = cf_cost.max(own_cost) >= min_cost_micros as f64; | |
| 552 | 561 | // Models: what AI Gateway priced g1t's own provider traffic at (its | |
| 553 | 562 | // lines, as "Cloudflare's" side) against the ledger's model cost. Only | |
| 554 | 563 | // once the gateway has been read; then the ledger having none of it is | |
| 555 | − | // drift too (traffic no run was charged for). | |
| 556 | − | let models = NOT_CLOUDFLARE.contains(&bucket) && cf_cost > 0.0; | |
| 557 | − | if enough && !overhead && cf_cost > 0.0 && (own_cost > 0.0 || models) { | |
| 558 | − | let delta = delta_percent(own_cost, cf_cost); | |
| 564 | + | // drift too (traffic no run was charged for). Every other bucket: its | |
| 565 | + | // usage at Cloudflare's list prices, before the included amounts. | |
| 566 | + | let not_cloudflare = NOT_CLOUDFLARE.contains(&bucket); | |
| 567 | + | let theirs = if not_cloudflare { cf_cost } else { list_micros.unwrap_or(0.0) }; | |
| 568 | + | let enough = theirs.max(own_cost) >= min_cost_micros as f64; | |
| 569 | + | let models = not_cloudflare && cf_cost > 0.0; | |
| 570 | + | if enough && !overhead && theirs > 0.0 && (own_cost > 0.0 || models) { | |
| 571 | + | let delta = delta_percent(own_cost, theirs); | |
| 559 | 572 | if delta.is_some_and(|d| d.abs() > threshold) { | |
| 560 | − | out.push(Drift { bucket: bucket.into(), kind: DriftKind::Cost, ours: own_cost, cloudflare: cf_cost, delta_percent: delta }); | |
| 573 | + | out.push(Drift { bucket: bucket.into(), kind: DriftKind::Cost, ours: own_cost, cloudflare: theirs, delta_percent: delta }); | |
| 561 | 574 | } | |
| 562 | 575 | } | |
| 563 | 576 | // The ledger has model cost and the gateway priced none of it: a token | |
| ⋯ | |||
| 572 | 585 | out | |
| 573 | 586 | } | |
| 574 | 587 | ||
| 588 | + | /// What each bucket's billable usage over a window comes to at | |
| 589 | + | /// Cloudflare's list prices, before the included amounts (`cycle::list_cost` | |
| 590 | + | /// line by line, in micros): what the price book's cost of the same usage | |
| 591 | + | /// is checked against. None for a bucket with usage on a meter that has no | |
| 592 | + | /// list price: its usage cannot be priced like for like, so it is not | |
| 593 | + | /// checked. Other sources (Artifacts events, AI Gateway) are left out. | |
| 594 | + | pub(crate) fn list_costs(rules: &[Rule], lines: &[LineRow]) -> BTreeMap<String, Option<f64>> { | |
| 595 | + | let mut out: BTreeMap<String, Option<f64>> = BTreeMap::new(); | |
| 596 | + | for line in lines.iter().filter(|l| l.source == SOURCE_BILLABLE) { | |
| 597 | + | let bucket = costs::classify(rules, &line.product, &line.meter).map_or(UNMAPPED, |r| r.bucket.as_str()); | |
| 598 | + | let entry = out.entry(bucket.to_owned()).or_insert(Some(0.0)); | |
| 599 | + | if line.quantity <= 0.0 { | |
| 600 | + | continue; | |
| 601 | + | } | |
| 602 | + | *entry = match (*entry, crate::cycle::list_cost(&line.product, &line.meter, line.quantity)) { | |
| 603 | + | (Some(sum), Some(cost)) => Some(sum + cost * 1_000_000.0), | |
| 604 | + | _ => None, | |
| 605 | + | }; | |
| 606 | + | } | |
| 607 | + | out | |
| 608 | + | } | |
| 609 | + | ||
| 575 | 610 | /// What can make AI Gateway's cost differ from what the providers bill, | |
| 576 | 611 | /// said for staff: cache tokens (priced by the gateway at its own rates for | |
| 577 | 612 | /// them, which may lag the provider's), requests Cloudflare billed itself, | |
| ⋯ | |||
| 893 | 928 | out | |
| 894 | 929 | } | |
| 895 | 930 | ||
| 896 | − | /// Cloudflare's marginal rate for one of its units: the median over the | |
| 897 | − | /// charged days of cost over quantity, in dollars. None while the included | |
| 898 | − | /// amounts still cover it. Each item is a day's (quantity, cost). | |
| 931 | + | /// Cloudflare's rate for one of its units: the median over the costed | |
| 932 | + | /// days of cost over quantity, in dollars. None while nothing is costed. | |
| 933 | + | /// Each item is a day's (quantity, cost), from `rate_line`. | |
| 899 | 934 | pub(crate) fn billed_rate(days: &[(f64, f64)]) -> Option<f64> { | |
| 900 | 935 | let mut rates: Vec<f64> = days.iter().filter(|(q, c)| *q > 0.0 && *c > 0.0).map(|(q, c)| c / q).collect(); | |
| 901 | 936 | if rates.is_empty() { | |
| ⋯ | |||
| 905 | 940 | Some(rates[rates.len() / 2]) | |
| 906 | 941 | } | |
| 907 | 942 | ||
| 943 | + | /// A line's (quantity, cost) for `billed_rate`: at its list price for all | |
| 944 | + | /// of the quantity where the meter has one, never what the cycle billed, | |
| 945 | + | /// which is net of the included amounts (nothing until the cycle passes | |
| 946 | + | /// them, part of a day's usage on the day it does, then whole millions) and | |
| 947 | + | /// so says nothing about the price of each unit. A meter with no list | |
| 948 | + | /// price has only what Cloudflare billed; the median over the charged days | |
| 949 | + | /// leaves out the day its included amount ran out. | |
| 950 | + | pub(crate) fn rate_line(product: &str, meter: &str, quantity: f64, cost_usd: f64) -> (f64, f64) { | |
| 951 | + | (quantity, crate::cycle::list_cost(product, meter, quantity).unwrap_or(cost_usd)) | |
| 952 | + | } | |
| 953 | + | ||
| 908 | 954 | /// What one of g1t's units costs, from Cloudflare's rate per its own unit | |
| 909 | 955 | /// and how many of Cloudflare's units each of g1t's took: if Cloudflare | |
| 910 | 956 | /// counts three operations for every git operation g1t counts, a git | |
| ⋯ | |||
| 1568 | 1614 | } | |
| 1569 | 1615 | let caveats = self.gateway_caveats(&since, until).await?; | |
| 1570 | 1616 | let resets = self.resets_since(&since, until).await?; | |
| 1617 | + | // The same days' usage at list prices, before the included amounts: | |
| 1618 | + | // what the price book is checked against (never the bill, which is | |
| 1619 | + | // net of them). | |
| 1620 | + | let lines = self | |
| 1621 | + | .db | |
| 1622 | + | .prepare("SELECT day, source, product, meter, quantity, cost_usd FROM cost_lines WHERE source = ?1 AND day >= ?2 AND day <= ?3") | |
| 1623 | + | .bind(&[SOURCE_BILLABLE.into(), since.as_str().into(), until.into()])? | |
| 1624 | + | .all() | |
| 1625 | + | .await? | |
| 1626 | + | .results::<LineRow>()?; | |
| 1627 | + | let list = list_costs(&rules, &lines); | |
| 1571 | 1628 | let mut found = Vec::new(); | |
| 1572 | 1629 | if let Some(drift) = unpriced_drift(&caveats) { | |
| 1573 | 1630 | found.push(drift); | |
| ⋯ | |||
| 1577 | 1634 | let threshold = bucket_rules.iter().map(|r| r.drift_percent).fold(f64::INFINITY, f64::min); | |
| 1578 | 1635 | let threshold = if threshold.is_finite() { threshold } else { 10.0 }; | |
| 1579 | 1636 | let counted = bucket_rules.iter().any(|r| r.own_meter.is_some()); | |
| 1580 | − | for drift in drifts(bucket, days, threshold, counted, settings.min_daily_cost_micros) { | |
| 1637 | + | let list_micros = list.get(bucket).copied().flatten(); | |
| 1638 | + | for drift in drifts(bucket, days, threshold, counted, settings.min_daily_cost_micros, list_micros) { | |
| 1581 | 1639 | if wiped_not_leaked(&drift, &resets) { | |
| 1582 | 1640 | continue; | |
| 1583 | 1641 | } | |
| ⋯ | |||
| 1591 | 1649 | drift.delta_percent.unwrap_or(0.0) | |
| 1592 | 1650 | ), | |
| 1593 | 1651 | DriftKind::Cost => format!( | |
| 1594 | − | "{title}: Cloudflare charged {} over the last {DRIFT_DAYS} days; the price book's cost of the same usage is {} ({:+.1}%). A price may be stale: see the proposals.", | |
| 1652 | + | "{title}: the last {DRIFT_DAYS} days' usage comes to {} at Cloudflare's list prices, before the included amounts; the price book's cost of the same usage is {} ({:+.1}%). A price may be stale: see the proposals.", | |
| 1595 | 1653 | dollars(drift.cloudflare as i64), | |
| 1596 | 1654 | dollars(drift.ours as i64), | |
| 1597 | 1655 | drift.delta_percent.unwrap_or(0.0) | |
| ⋯ | |||
| 1676 | 1734 | let mine: Vec<(f64, f64)> = lines | |
| 1677 | 1735 | .iter() | |
| 1678 | 1736 | .filter(|l| costs::classify(&rules, &l.product, &l.meter).is_some_and(|r| r.product == s.product && r.meter == s.meter)) | |
| 1679 | − | .map(|l| (l.quantity, l.cost_usd)) | |
| 1737 | + | .map(|l| rate_line(&l.product, &l.meter, l.quantity, l.cost_usd)) | |
| 1680 | 1738 | .collect(); | |
| 1681 | 1739 | let Some(rate) = billed_rate(&mine) else { continue }; | |
| 1682 | 1740 | let cf_units: f64 = mine.iter().map(|(q, _)| q).sum(); | |
| ⋯ | |||
| 2584 | 2642 | fn counts_more_than_the_threshold_apart_are_drift() { | |
| 2585 | 2643 | // Cloudflare counted 30,000 operations where g1t counted 10,000: | |
| 2586 | 2644 | // binding reads, perhaps. -66.7%. | |
| 2587 | − | let drift = drifts("git", &[day("git", 3_000_000, 1_500_000, 1_800_000, 30_000.0, 10_000.0)], 10.0, true, 100_000); | |
| 2645 | + | let drift = drifts("git", &[day("git", 3_000_000, 1_500_000, 1_800_000, 30_000.0, 10_000.0)], 10.0, true, 100_000, Some(3_000_000.0)); | |
| 2588 | 2646 | assert_eq!(drift.iter().map(|d| d.kind).collect::<Vec<_>>(), vec![DriftKind::Count, DriftKind::Cost]); | |
| 2589 | 2647 | assert!((drift[0].delta_percent.unwrap() + 66.666).abs() < 0.01); | |
| 2590 | 2648 | // 9% apart: within 10%. | |
| 2591 | − | assert!(drifts("git", &[day("git", 1_000_000, 1_000_000, 1_200_000, 10_000.0, 10_900.0)], 10.0, true, 100_000).is_empty()); | |
| 2649 | + | assert!(drifts("git", &[day("git", 1_000_000, 1_000_000, 1_200_000, 10_000.0, 10_900.0)], 10.0, true, 100_000, Some(1_000_000.0)).is_empty()); | |
| 2592 | 2650 | // Uncounted products have no count drift. | |
| 2593 | − | assert!(drifts("sandboxes", &[day("sandboxes", 1_000_000, 1_050_000, 1_200_000, 5.0, 0.0)], 10.0, false, 100_000).is_empty()); | |
| 2651 | + | assert!(drifts("sandboxes", &[day("sandboxes", 1_000_000, 1_050_000, 1_200_000, 5.0, 0.0)], 10.0, false, 100_000, Some(1_000_000.0)).is_empty()); | |
| 2594 | 2652 | } | |
| 2595 | 2653 | ||
| 2596 | 2654 | #[test] | |
| 2597 | 2655 | fn cost_with_no_revenue_is_a_leak_but_not_for_running_g1t() { | |
| 2598 | − | let leak = drifts("actions_cache", &[day("actions_cache", 400_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000); | |
| 2656 | + | let leak = drifts("actions_cache", &[day("actions_cache", 400_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None); | |
| 2599 | 2657 | assert_eq!(leak.len(), 1); | |
| 2600 | 2658 | assert_eq!(leak[0].kind, DriftKind::Leak); | |
| 2601 | − | assert!(drifts("platform", &[day("platform", 5_000_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000).is_empty()); | |
| 2659 | + | assert!(drifts("platform", &[day("platform", 5_000_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty()); | |
| 2602 | 2660 | // Pennies say nothing. | |
| 2603 | − | assert!(drifts("actions_cache", &[day("actions_cache", 50_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000).is_empty()); | |
| 2604 | − | assert!(drifts(UNMAPPED, &[day(UNMAPPED, 250_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000)[0].kind == DriftKind::Leak); | |
| 2661 | + | assert!(drifts("actions_cache", &[day("actions_cache", 50_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty()); | |
| 2662 | + | assert!(drifts(UNMAPPED, &[day(UNMAPPED, 250_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None)[0].kind == DriftKind::Leak); | |
| 2663 | + | } | |
| 2664 | + | ||
| 2665 | + | /// A week of sandboxes mostly inside the cycle's included amounts: | |
| 2666 | + | /// Cloudflare billed $0.0922 (the memory past 25 GiB-hours), and the | |
| 2667 | + | /// same usage is $1.70 at list prices. | |
| 2668 | + | fn sandbox_week() -> (Vec<Rule>, Vec<LineRow>) { | |
| 2669 | + | let mut rules = rules(); | |
| 2670 | + | rules.push(rule("durable_objects", "durable_objects_compute_duration", "sandboxes", None)); | |
| 2671 | + | let lines = vec![ | |
| 2672 | + | line("2026-10-07", SOURCE_BILLABLE, "containers", "container_memory_per_gib_second", 126_870.0, 0.092_175), | |
| 2673 | + | line("2026-10-07", SOURCE_BILLABLE, "containers", "container_vcpu", 12_000.0, 0.0), | |
| 2674 | + | line("2026-10-07", SOURCE_BILLABLE, "containers", "container_disk_per_gb_second", 253_740.0, 0.0), | |
| 2675 | + | line("2026-10-07", SOURCE_BILLABLE, "durable_objects", "durable_objects_compute_duration", 90_000.0, 0.0), | |
| 2676 | + | // Not billable usage: no part of the list cost. | |
| 2677 | + | line("2026-10-07", SOURCE_ARTIFACTS, "artifacts", "events_push", 40.0, 0.0), | |
| 2678 | + | ]; | |
| 2679 | + | (rules, lines) | |
| 2680 | + | } | |
| 2681 | + | ||
| 2682 | + | #[test] | |
| 2683 | + | fn usage_inside_the_included_amounts_is_not_a_stale_price() { | |
| 2684 | + | let (rules, lines) = sandbox_week(); | |
| 2685 | + | let list = list_costs(&rules, &lines); | |
| 2686 | + | let sandboxes = list["sandboxes"].unwrap(); | |
| 2687 | + | assert!((sandboxes - 1_699_936.8).abs() < 1.0, "{sandboxes}"); | |
| 2688 | + | // The price book's cost of the same usage is $1.69: within 1% of | |
| 2689 | + | // the list, though Cloudflare billed $0.0922 after the included | |
| 2690 | + | // amounts. Before, that read as +1733.6%. | |
| 2691 | + | let week = day("sandboxes", 92_175, 1_690_000, 2_028_000, 0.0, 0.0); | |
| 2692 | + | assert!(drifts("sandboxes", std::slice::from_ref(&week), 10.0, false, 100_000, Some(sandboxes)).is_empty()); | |
| 2693 | + | let net = drifts("sandboxes", &[week], 10.0, false, 100_000, Some(92_175.0)); | |
| 2694 | + | assert_eq!(net[0].kind, DriftKind::Cost, "billed against the price book was the false alarm"); | |
| 2695 | + | } | |
| 2696 | + | ||
| 2697 | + | #[test] | |
| 2698 | + | fn a_stale_price_is_still_drift() { | |
| 2699 | + | let (rules, lines) = sandbox_week(); | |
| 2700 | + | let sandboxes = list_costs(&rules, &lines)["sandboxes"].unwrap(); | |
| 2701 | + | // The price book still costs the same usage at $1.20: a list price | |
| 2702 | + | // rose and the book did not follow. | |
| 2703 | + | let found = drifts("sandboxes", &[day("sandboxes", 92_175, 1_200_000, 1_440_000, 0.0, 0.0)], 10.0, false, 100_000, Some(sandboxes)); | |
| 2704 | + | assert_eq!(found.len(), 1); | |
| 2705 | + | assert_eq!((found[0].kind, found[0].ours, found[0].cloudflare.round()), (DriftKind::Cost, 1_200_000.0, 1_699_937.0)); | |
| 2706 | + | assert!((found[0].delta_percent.unwrap() + 29.4).abs() < 0.1); | |
| 2707 | + | // Nothing billed at all (the whole week inside the included | |
| 2708 | + | // amounts) is checked all the same. | |
| 2709 | + | assert_eq!(drifts("sandboxes", &[day("sandboxes", 0, 1_200_000, 1_440_000, 0.0, 0.0)], 10.0, false, 100_000, Some(sandboxes)).len(), 1); | |
| 2710 | + | } | |
| 2711 | + | ||
| 2712 | + | #[test] | |
| 2713 | + | fn a_bucket_with_a_meter_with_no_list_price_is_not_checked() { | |
| 2714 | + | let (rules, mut lines) = sandbox_week(); | |
| 2715 | + | lines.push(line("2026-10-07", SOURCE_BILLABLE, "artifacts", "storage", 3.0, 0.4)); | |
| 2716 | + | lines.push(line("2026-10-07", SOURCE_BILLABLE, "artifacts", "operations", 0.0, 0.0)); | |
| 2717 | + | let list = list_costs(&rules, &lines); | |
| 2718 | + | assert_eq!(list["git"], None); | |
| 2719 | + | assert!(list["sandboxes"].is_some()); | |
| 2720 | + | // No list cost: no cost drift, but a leak is still what was billed. | |
| 2721 | + | assert!(drifts("git", &[day("git", 3_000_000, 1_000_000, 1_200_000, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty()); | |
| 2722 | + | assert_eq!(drifts("git", &[day("git", 3_000_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None)[0].kind, DriftKind::Leak); | |
| 2723 | + | } | |
| 2724 | + | ||
| 2725 | + | #[test] | |
| 2726 | + | fn a_unit_rate_is_the_list_price_where_there_is_one() { | |
| 2727 | + | // Billed $0.20 for 11.16M CPU ms (9.16M past the included 30M, in | |
| 2728 | + | // whole millions): $0.02 a million at list, whatever was billed. | |
| 2729 | + | let (q, c) = rate_line("workers", "workers_cpu_ms", 11_160_000.0, 0.20); | |
| 2730 | + | assert!((billed_rate(&[(q, c)]).unwrap() * 1e6 - 0.02).abs() < 1e-12); | |
| 2731 | + | // No list price: what Cloudflare billed. | |
| 2732 | + | assert_eq!(rate_line("artifacts", "operations", 1_000.0, 0.5), (1_000.0, 0.5)); | |
| 2605 | 2733 | } | |
| 2606 | 2734 | ||
| 2607 | 2735 | #[test] | |
| ⋯ | |||
| 2636 | 2764 | let on = |day: &str, cf: f64, own: f64| ProductDay { day: day.into(), bucket: "git".into(), cf_quantity: cf, own_quantity: own, ..ProductDay::default() }; | |
| 2637 | 2765 | // Five days of Cloudflare's count before g1t's meter, then two that match. | |
| 2638 | 2766 | let days = vec![on("2026-10-01", 500.0, 0.0), on("2026-10-05", 300.0, 0.0), on("2026-10-06", 210.0, 231.0), on("2026-10-07", 450.0, 458.0)]; | |
| 2639 | − | assert!(drifts("git", &days, 10.0, true, 0).iter().all(|d| d.kind != DriftKind::Count)); | |
| 2767 | + | assert!(drifts("git", &days, 10.0, true, 0, None).iter().all(|d| d.kind != DriftKind::Count)); | |
| 2640 | 2768 | // A real gap on the days both counted still shows. | |
| 2641 | 2769 | let days = vec![on("2026-10-01", 500.0, 0.0), on("2026-10-06", 400.0, 231.0), on("2026-10-07", 600.0, 300.0)]; | |
| 2642 | − | let found = drifts("git", &days, 10.0, true, 0); | |
| 2770 | + | let found = drifts("git", &days, 10.0, true, 0, None); | |
| 2643 | 2771 | let count = found.iter().find(|d| d.kind == DriftKind::Count).unwrap(); | |
| 2644 | 2772 | assert_eq!((count.ours, count.cloudflare), (531.0, 1000.0)); | |
| 2645 | 2773 | // A meter that never counted is compared over every day. | |
| 2646 | 2774 | let days = vec![on("2026-10-06", 400.0, 0.0)]; | |
| 2647 | − | assert!(drifts("git", &days, 10.0, true, 0).iter().any(|d| d.kind == DriftKind::Count)); | |
| 2775 | + | assert!(drifts("git", &days, 10.0, true, 0, None).iter().any(|d| d.kind == DriftKind::Count)); | |
| 2648 | 2776 | } | |
| 2649 | 2777 | ||
| 2650 | 2778 | #[test] | |
| ⋯ | |||
| 2752 | 2880 | #[test] | |
| 2753 | 2881 | fn the_gateways_total_against_the_ledgers_model_cost_is_drift() { | |
| 2754 | 2882 | // The gateway priced $5 of g1t's own traffic; the ledger has $3. | |
| 2755 | − | let short = drifts("models", &[day("models", 5_000_000, 3_000_000, 3_600_000, 0.0, 0.0)], 10.0, false, 100_000); | |
| 2883 | + | let short = drifts("models", &[day("models", 5_000_000, 3_000_000, 3_600_000, 0.0, 0.0)], 10.0, false, 100_000, None); | |
| 2756 | 2884 | assert_eq!(short.iter().map(|d| d.kind).collect::<Vec<_>>(), vec![DriftKind::Cost]); | |
| 2757 | 2885 | assert!((short[0].delta_percent.unwrap() + 40.0).abs() < 1e-9); | |
| 2758 | 2886 | // Gateway traffic with nothing on the ledger at all: cost drift and a leak. | |
| 2759 | − | let none = drifts("models", &[day("models", 2_000_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000); | |
| 2887 | + | let none = drifts("models", &[day("models", 2_000_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None); | |
| 2760 | 2888 | assert_eq!(none.iter().map(|d| d.kind).collect::<Vec<_>>(), vec![DriftKind::Cost, DriftKind::Leak]); | |
| 2761 | 2889 | // Within the threshold: nothing. | |
| 2762 | − | assert!(drifts("models", &[day("models", 1_050_000, 1_000_000, 1_200_000, 0.0, 0.0)], 10.0, false, 100_000).is_empty()); | |
| 2890 | + | assert!(drifts("models", &[day("models", 1_050_000, 1_000_000, 1_200_000, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty()); | |
| 2763 | 2891 | // The gateway priced nothing against a ledger that has model cost: | |
| 2764 | 2892 | // not agreement (a token that cannot see AI Gateway reads as no | |
| 2765 | 2893 | // rows), so it is said. Under the minimum, or no model cost: nothing. | |
| 2766 | − | let silent = drifts("models", &[day("models", 0, 1_000_000, 1_200_000, 0.0, 0.0)], 10.0, false, 100_000); | |
| 2894 | + | let silent = drifts("models", &[day("models", 0, 1_000_000, 1_200_000, 0.0, 0.0)], 10.0, false, 100_000, None); | |
| 2767 | 2895 | assert_eq!(silent, vec![Drift { bucket: "models".into(), kind: DriftKind::Cost, ours: 1_000_000.0, cloudflare: 0.0, delta_percent: None }]); | |
| 2768 | 2896 | // Why it is empty, as far as the run could tell. | |
| 2769 | 2897 | let why = |caveats: &costs::GatewayCaveats, read: costs::GatewayRead| models_detail(&silent[0], caveats, &[], &read); | |
| ⋯ | |||
| 2780 | 2908 | let unpriced = costs::GatewayCaveats { requests: 42.0, unpriced: vec!["anthropic_claude_new_1".into()], ..Default::default() }; | |
| 2781 | 2909 | let said = why(&unpriced, costs::GatewayRead::Rows); | |
| 2782 | 2910 | assert!(said.contains("logged 42 requests") && said.contains("no price for the models used (anthropic_claude_new_1)"), "{said}"); | |
| 2783 | − | assert!(drifts("models", &[day("models", 0, 50_000, 60_000, 0.0, 0.0)], 10.0, false, 100_000).is_empty()); | |
| 2784 | − | assert!(drifts("models", &[day("models", 0, 0, 0, 0.0, 0.0)], 10.0, false, 100_000).is_empty()); | |
| 2911 | + | assert!(drifts("models", &[day("models", 0, 50_000, 60_000, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty()); | |
| 2912 | + | assert!(drifts("models", &[day("models", 0, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty()); | |
| 2785 | 2913 | // The detail says which way and why it may be off. | |
| 2786 | 2914 | let caveats = costs::GatewayCaveats { cache_read_tokens: 3_000_000.0, unpriced: vec!["anthropic_claude_new_1".into()], ..Default::default() }; | |
| 2787 | 2915 | let detail = models_detail(&short[0], &caveats, &[], &costs::GatewayRead::default()); | |
| ⋯ | |||
| 2851 | 2979 | let models = |days: &[ProductDay]| days.iter().find(|d| d.bucket == "models").cloned().unwrap(); | |
| 2852 | 2980 | // Without it: AI Gateway's $11.11 against the ledger's $2.49. | |
| 2853 | 2981 | let (days, _) = gateway_and_ledger(false); | |
| 2854 | − | let drift = drifts("models", &[models(&days)], 10.0, false, 100_000); | |
| 2982 | + | let drift = drifts("models", &[models(&days)], 10.0, false, 100_000, None); | |
| 2855 | 2983 | assert_eq!(drift.iter().map(|d| d.kind).collect::<Vec<_>>(), vec![DriftKind::Cost]); | |
| 2856 | 2984 | assert_eq!((drift[0].ours, drift[0].cloudflare), (2_490_000.0, 11_110_000.0)); | |
| 2857 | 2985 | // With it: the ledger's model cost and the reset's add up to the gateway's. | |
| 2858 | 2986 | let (days, _) = gateway_and_ledger(true); | |
| 2859 | 2987 | let m = models(&days); | |
| 2860 | 2988 | assert_eq!(m.own_cost_micros, 11_110_000); | |
| 2861 | − | assert!(drifts("models", &[m], 10.0, false, 100_000).is_empty()); | |
| 2989 | + | assert!(drifts("models", &[m], 10.0, false, 100_000, None).is_empty()); | |
| 2862 | 2990 | // The reset's own row makes no bucket of its own. | |
| 2863 | 2991 | assert!(!days.iter().any(|d| d.bucket.is_empty())); | |
| 2864 | 2992 | } | |
| 95 | 95 | import { buildMentionPrompt, describeThread, handleMention, jobTokenRefusal, planMention } from "./mentions"; | |
| 96 | 96 | import { instructionsFor, repoInstructions, withBlock } from "./repo-instructions"; | |
| 97 | 97 | import { cancelTask, enqueueTask, handedOverStep, selfHostedRoute, taskEnv, taskRepo } from "./self-hosted"; | |
| 98 | + | import { answered, describeError, tellStopped, withinTimeCap } from "./lifecycle"; | |
| 98 | 99 | import { | |
| 99 | 100 | ABUSE_EXIT_CODE, | |
| 100 | 101 | ABUSE_HOST, | |
| ⋯ | |||
| 412 | 413 | * it and cleans up if it dies without reporting. | |
| 413 | 414 | */ | |
| 414 | 415 | export class AttemptSandbox extends Container<RunnerEnv> { | |
| 415 | − | // Past the longest time cap (implement, 90 minutes) and its alarm, so a | |
| 416 | − | // long run is never put to sleep before its own cap ends it. A finished | |
| 416 | + | // Past the longest default time cap (implement, 90 minutes) and its | |
| 417 | + | // alarm. A run whose guardrails allow longer (up to 240 minutes) is kept | |
| 418 | + | // past it by `onActivityExpired`, so only its own cap ends it. A finished | |
| 417 | 419 | // run's process exits and stops the sandbox well before this. | |
| 418 | 420 | sleepAfter = "100m"; | |
| 419 | 421 | // A guarded sandbox's HTTPS goes through `egress` too (guard.ts). | |
| ⋯ | |||
| 439 | 441 | ? withPlanLimits(await buildGuardFor(this.env.WORK, build.repo, build.kind, build.minutes, build.repoId, build.job, build.hosts), limits) | |
| 440 | 442 | : null; | |
| 441 | 443 | } catch (error) { | |
| 444 | + | // Thrown to the caller, which says why the work did not start; logged | |
| 445 | + | // here too, so a sandbox that never started is traceable on its own. | |
| 446 | + | console.error("sandbox not started", run.kind, describeError(error)); | |
| 442 | 447 | await this.settle(0); | |
| 443 | 448 | throw error; | |
| 444 | 449 | } | |
| ⋯ | |||
| 501 | 506 | await this.schedule(cap * 60 + ALARM_GRACE_SECONDS, "timeUp"); | |
| 502 | 507 | } | |
| 503 | 508 | } catch (error) { | |
| 509 | + | console.error("sandbox not started", run.kind, describeError(error)); | |
| 504 | 510 | await revokeCredentials(this.env.IDENTITY, this.ctx.storage, this.env.INTEGRATIONS); | |
| 505 | 511 | if (tracked) await this.closeRun("failed", `The sandbox could not start: ${String(error)}`); | |
| 506 | 512 | await this.settle(0); | |
| ⋯ | |||
| 632 | 638 | await this.remoteEnded(1, reason); | |
| 633 | 639 | return; | |
| 634 | 640 | } | |
| 635 | − | await this.destroy(); | |
| 641 | + | // A container already gone has nothing to stop; one that will not stop | |
| 642 | + | // is logged, and its time cap still ends it. | |
| 643 | + | await this.destroy().catch((error: unknown) => console.error("sandbox not destroyed", reason, describeError(error))); | |
| 644 | + | } | |
| 645 | + | ||
| 646 | + | /** | |
| 647 | + | * The library's `sleepAfter` has passed. The runner never fetches its | |
| 648 | + | * container, so to the library every sandbox looks idle: a run inside its | |
| 649 | + | * time cap keeps going, and the cap's own alarm (`timeUp`) ends it. Only | |
| 650 | + | * a sandbox with no cap is stopped for inactivity. | |
| 651 | + | */ | |
| 652 | + | override async onActivityExpired(): Promise<void> { | |
| 653 | + | const started = await this.ctx.storage.get<number>("started"); | |
| 654 | + | const cap = await this.ctx.storage.get<number>("timeCap"); | |
| 655 | + | if (withinTimeCap(started, cap, Date.now(), ALARM_GRACE_SECONDS)) return; | |
| 656 | + | await super.onActivityExpired(); | |
| 657 | + | } | |
| 658 | + | ||
| 659 | + | /** | |
| 660 | + | * A container that crashed or could not be reached, as the library tells | |
| 661 | + | * it. Logged at error level with the sandbox, never thrown: the library | |
| 662 | + | * ignores what this throws, and the stop that follows is handled by | |
| 663 | + | * `onStop` or by `run`, which says why the work did not start. | |
| 664 | + | */ | |
| 665 | + | override onError(error: unknown): void { | |
| 666 | + | console.error("sandbox container error", this.ctx.id.toString(), describeError(error)); | |
| 636 | 667 | } | |
| 637 | 668 | ||
| 638 | 669 | /** | |
| 670 | + | * The library's alarm: scheduled callbacks, the container's keep-alive, | |
| 671 | + | * and `onStop` once it has stopped. A failure is logged with the retry it | |
| 672 | + | * was, then thrown so Cloudflare tries the alarm again. | |
| 673 | + | */ | |
| 674 | + | override async alarm(alarmProps?: AlarmInvocationInfo): Promise<void> { | |
| 675 | + | try { | |
| 676 | + | await super.alarm(alarmProps); | |
| 677 | + | } catch (error) { | |
| 678 | + | console.error("sandbox alarm failed", this.ctx.id.toString(), `retry ${alarmProps?.retryCount ?? 0}`, describeError(error)); | |
| 679 | + | throw error; | |
| 680 | + | } | |
| 681 | + | } | |
| 682 | + | ||
| 683 | + | /** | |
| 639 | 684 | * A self-hosted runner's task ended (the actions service says so, or g1t | |
| 640 | 685 | * stopped it): everything a container's stop does, once. | |
| 641 | 686 | */ | |
| ⋯ | |||
| 713 | 758 | if (ended === "stopped" && run && STOP_ENDS.has(run.kind)) return; | |
| 714 | 759 | console.log("sandbox stopped", run?.kind, "exit", exitCode, reason); | |
| 715 | 760 | if (!run) return; | |
| 761 | + | // Never thrown: this runs in the sandbox's alarm, which a throw would | |
| 762 | + | // fail, retry and count as an error, running all of the above again. | |
| 763 | + | // What could not be told is logged, and the sweep catches it up. | |
| 764 | + | await tellStopped(run.kind, () => this.reportStopped(run, why, exitCode)); | |
| 765 | + | } | |
| 766 | + | ||
| 767 | + | /** | |
| 768 | + | * Tells whoever is waiting on the sandbox's work that it stopped without | |
| 769 | + | * finishing it. Each is refused harmlessly when the sandbox reported its | |
| 770 | + | * end before it stopped. Throws when the service could not be reached. | |
| 771 | + | */ | |
| 772 | + | private async reportStopped(run: Run, why: string | null, exitCode: number): Promise<void> { | |
| 716 | 773 | if (run.kind === "actions") { | |
| 717 | 774 | // Refused harmlessly if the job reported its end before it stopped. | |
| 718 | − | await this.env.ACTIONS.fetch("https://actions/rpc/job_report", { | |
| 775 | + | const response = await this.env.ACTIONS.fetch("https://actions/rpc/job_report", { | |
| 719 | 776 | method: "POST", | |
| 720 | 777 | headers: { "content-type": "application/json" }, | |
| 721 | 778 | body: JSON.stringify({ | |
| ⋯ | |||
| 724 | 781 | report: { kind: "done", conclusion: "failure", reason: why ?? "The runner stopped before the job finished." }, | |
| 725 | 782 | }), | |
| 726 | 783 | }); | |
| 784 | + | await answered("job_report", response); | |
| 727 | 785 | return; | |
| 728 | 786 | } | |
| 729 | 787 | // Nothing was pushed, so no pull request opens; why is in its log. | |
| ⋯ | |||
| 731 | 789 | if (run.kind === "backup") { | |
| 732 | 790 | // Refused harmlessly if the sandbox reported before it stopped; the | |
| 733 | 791 | // job is otherwise tried again later tonight. | |
| 734 | − | await reposClient(this.env.REPOS) | |
| 735 | − | .failBackup(run.jobId, run.token, why ?? `The sandbox exited with ${exitCode}.`) | |
| 736 | − | .catch((error: unknown) => console.log("backup failure not reported", run.jobId, String(error))); | |
| 792 | + | await reposClient(this.env.REPOS).failBackup(run.jobId, run.token, why ?? `The sandbox exited with ${exitCode}.`); | |
| 737 | 793 | return; | |
| 738 | 794 | } | |
| 739 | 795 | if (run.kind === "deploy") { | |
| 740 | 796 | // Refused harmlessly if the build reported its end before it stopped. | |
| 741 | − | await this.env.DEPLOYMENTS.fetch(`https://deployments/jobs/${run.deployId}/fail`, { | |
| 797 | + | const response = await this.env.DEPLOYMENTS.fetch(`https://deployments/jobs/${run.deployId}/fail`, { | |
| 742 | 798 | method: "POST", | |
| 743 | 799 | headers: { "content-type": "application/json" }, | |
| 744 | 800 | body: JSON.stringify({ token: run.token, message: why ?? "The build stopped before it finished." }), | |
| 745 | 801 | }); | |
| 802 | + | await answered("deploy fail", response); | |
| 746 | 803 | return; | |
| 747 | 804 | } | |
| 748 | 805 | const work = workClient(this.env.WORK); | |
| 1 | + | import assert from "node:assert/strict"; | |
| 2 | + | import { test } from "node:test"; | |
| 3 | + | ||
| 4 | + | import { answered, describeError, tellStopped, withinTimeCap } from "./lifecycle.ts"; | |
| 5 | + | ||
| 6 | + | const quiet = { waitMs: 0 }; | |
| 7 | + | ||
| 8 | + | test("a stop that is told the first time is told once and logs nothing", async () => { | |
| 9 | + | const logged: unknown[][] = []; | |
| 10 | + | let calls = 0; | |
| 11 | + | const told = await tellStopped("checks", async () => void calls++, { ...quiet, log: (...data) => logged.push(data) }); | |
| 12 | + | assert.equal(told, true); | |
| 13 | + | assert.equal(calls, 1); | |
| 14 | + | assert.deepEqual(logged, []); | |
| 15 | + | }); | |
| 16 | + | ||
| 17 | + | test("a stop that fails once is tried again, and gets through", async () => { | |
| 18 | + | const logged: unknown[][] = []; | |
| 19 | + | let calls = 0; | |
| 20 | + | const told = await tellStopped( | |
| 21 | + | "queue", | |
| 22 | + | async () => { | |
| 23 | + | calls++; | |
| 24 | + | if (calls === 1) throw new Error("report_queue failed with status 500"); | |
| 25 | + | }, | |
| 26 | + | { ...quiet, log: (...data) => logged.push(data) }, | |
| 27 | + | ); | |
| 28 | + | assert.equal(told, true); | |
| 29 | + | assert.equal(calls, 2); | |
| 30 | + | assert.deepEqual(logged, []); | |
| 31 | + | }); | |
| 32 | + | ||
| 33 | + | test("a stop that cannot be told never throws, and is logged with what and why", async () => { | |
| 34 | + | const logged: unknown[][] = []; | |
| 35 | + | let calls = 0; | |
| 36 | + | const told = await tellStopped( | |
| 37 | + | "actions", | |
| 38 | + | async () => { | |
| 39 | + | calls++; | |
| 40 | + | throw new Error("job_report answered 503"); | |
| 41 | + | }, | |
| 42 | + | { ...quiet, log: (...data) => logged.push(data) }, | |
| 43 | + | ); | |
| 44 | + | assert.equal(told, false); | |
| 45 | + | assert.equal(calls, 2); | |
| 46 | + | assert.deepEqual(logged, [["sandbox stop not reported", "actions", "job_report answered 503"]]); | |
| 47 | + | }); | |
| 48 | + | ||
| 49 | + | test("whatever is thrown, even nothing, is logged as one line", async () => { | |
| 50 | + | const logged: unknown[][] = []; | |
| 51 | + | await tellStopped("plan", () => Promise.reject(undefined), { ...quiet, attempts: 1, log: (...data) => logged.push(data) }); | |
| 52 | + | assert.deepEqual(logged, [["sandbox stop not reported", "plan", "no reason given"]]); | |
| 53 | + | assert.equal(describeError("gone"), "gone"); | |
| 54 | + | assert.equal(describeError({ code: 7 }), '{"code":7}'); | |
| 55 | + | assert.equal(describeError(new TypeError("bad")), "bad"); | |
| 56 | + | }); | |
| 57 | + | ||
| 58 | + | test("a service that cannot answer is a failure; one that refuses is not", async () => { | |
| 59 | + | await answered("job_report", new Response("{}", { status: 200 })); | |
| 60 | + | // The deployments service's answer for a build that already finished. | |
| 61 | + | await answered("deploy fail", new Response("{}", { status: 404 })); | |
| 62 | + | await answered("job_report", new Response("{}", { status: 409 })); | |
| 63 | + | await assert.rejects(answered("job_report", new Response("worker threw", { status: 500 })), /job_report answered 500: worker threw/); | |
| 64 | + | await assert.rejects(answered("deploy fail", new Response("", { status: 503 })), /^Error: deploy fail answered 503$/); | |
| 65 | + | await assert.rejects(answered("job_report", new Response("slow down", { status: 429 })), /429/); | |
| 66 | + | }); | |
| 67 | + | ||
| 68 | + | test("a run inside its time cap is not stopped for inactivity", () => { | |
| 69 | + | const started = Date.UTC(2026, 9, 8, 12, 0, 0); | |
| 70 | + | const minute = 60_000; | |
| 71 | + | const grace = 180; | |
| 72 | + | // A 240-minute cap outlives the 100-minute sleepAfter. | |
| 73 | + | assert.equal(withinTimeCap(started, 240, started + 100 * minute, grace), true); | |
| 74 | + | assert.equal(withinTimeCap(started, 240, started + 242 * minute, grace), true); | |
| 75 | + | // Past the cap and its alarm's grace: the library may stop it. | |
| 76 | + | assert.equal(withinTimeCap(started, 240, started + 243 * minute, grace), false); | |
| 77 | + | assert.equal(withinTimeCap(started, 90, started + 100 * minute, grace), false); | |
| 78 | + | }); | |
| 79 | + | ||
| 80 | + | test("a sandbox with no cap or no start is stopped for inactivity as before", () => { | |
| 81 | + | const now = Date.now(); | |
| 82 | + | assert.equal(withinTimeCap(undefined, 90, now, 180), false); | |
| 83 | + | assert.equal(withinTimeCap(now - 1000, undefined, now, 180), false); | |
| 84 | + | assert.equal(withinTimeCap(now - 1000, null, now, 180), false); | |
| 85 | + | assert.equal(withinTimeCap(now - 1000, 0, now, 180), false); | |
| 86 | + | }); |
| 1 | + | /** | |
| 2 | + | * How a sandbox's Durable Object handles the end of its container without | |
| 3 | + | * failing the invocation it happens in. The containers library calls | |
| 4 | + | * `onStop` from the object's alarm: a stop hook that throws fails that | |
| 5 | + | * alarm, which Cloudflare retries and counts as an error each time, and | |
| 6 | + | * which runs the whole hook again. So the hook's calls to other services | |
| 7 | + | * go through `tellStopped`, which tries twice and logs at error level, and | |
| 8 | + | * never throws. Pure, so it is tested on its own. | |
| 9 | + | */ | |
| 10 | + | ||
| 11 | + | /** How long `tellStopped` waits before its second try. */ | |
| 12 | + | export const RETRY_AFTER_MS = 500; | |
| 13 | + | ||
| 14 | + | /** | |
| 15 | + | * Tells another service that a sandbox stopped: `step` runs, and once more | |
| 16 | + | * if it throws. Never throws: a step that fails twice is logged with | |
| 17 | + | * `log` as `sandbox stop not reported`, with what and why, so it shows in | |
| 18 | + | * the runner's logs at error level and the five-minute sweep picks the | |
| 19 | + | * work up instead. Returns whether it got through. | |
| 20 | + | */ | |
| 21 | + | export async function tellStopped( | |
| 22 | + | what: string, | |
| 23 | + | step: () => Promise<unknown>, | |
| 24 | + | options: { log?: (...data: unknown[]) => void; attempts?: number; waitMs?: number } = {}, | |
| 25 | + | ): Promise<boolean> { | |
| 26 | + | const log = options.log ?? console.error; | |
| 27 | + | const attempts = Math.max(1, options.attempts ?? 2); | |
| 28 | + | let last: unknown = null; | |
| 29 | + | for (let attempt = 1; attempt <= attempts; attempt++) { | |
| 30 | + | try { | |
| 31 | + | await step(); | |
| 32 | + | return true; | |
| 33 | + | } catch (error) { | |
| 34 | + | last = error; | |
| 35 | + | if (attempt < attempts) await new Promise((resolve) => setTimeout(resolve, options.waitMs ?? RETRY_AFTER_MS)); | |
| 36 | + | } | |
| 37 | + | } | |
| 38 | + | log("sandbox stop not reported", what, describeError(last)); | |
| 39 | + | return false; | |
| 40 | + | } | |
| 41 | + | ||
| 42 | + | /** | |
| 43 | + | * A service binding's answer as a step's outcome: a 5xx (or 429) throws, so | |
| 44 | + | * `tellStopped` tries again and logs it. A refusal is not a failure: the | |
| 45 | + | * work already reported its end, and services say so with `ok: false` or a | |
| 46 | + | * 4xx (the deployments service answers 404 for a build that has finished). | |
| 47 | + | */ | |
| 48 | + | export async function answered(what: string, response: Response): Promise<void> { | |
| 49 | + | if (response.status < 500 && response.status !== 429) return; | |
| 50 | + | const body = await response.text().catch(() => ""); | |
| 51 | + | throw new Error(`${what} answered ${response.status}${body ? `: ${body.slice(0, 200)}` : ""}`); | |
| 52 | + | } | |
| 53 | + | ||
| 54 | + | /** | |
| 55 | + | * Whether a sandbox whose `sleepAfter` has passed is still inside its run's | |
| 56 | + | * time cap, and so keeps running: the cap, not inactivity, ends a run (the | |
| 57 | + | * runner never fetches the container, so to the library every sandbox looks | |
| 58 | + | * idle). Without a cap or a start time, `sleepAfter` applies. | |
| 59 | + | */ | |
| 60 | + | export function withinTimeCap( | |
| 61 | + | started: number | null | undefined, | |
| 62 | + | capMinutes: number | null | undefined, | |
| 63 | + | now: number, | |
| 64 | + | graceSeconds: number, | |
| 65 | + | ): boolean { | |
| 66 | + | if (typeof started !== "number" || typeof capMinutes !== "number" || capMinutes <= 0) return false; | |
| 67 | + | return now < started + (capMinutes * 60 + graceSeconds) * 1000; | |
| 68 | + | } | |
| 69 | + | ||
| 70 | + | /** An error as one line, for a log: its message, or what was thrown. */ | |
| 71 | + | export function describeError(error: unknown): string { | |
| 72 | + | if (error instanceof Error) return error.message || error.name; | |
| 73 | + | if (error === undefined) return "no reason given"; | |
| 74 | + | if (typeof error === "string") return error; | |
| 75 | + | try { | |
| 76 | + | return JSON.stringify(error) ?? String(error); | |
| 77 | + | } catch { | |
| 78 | + | return String(error); | |
| 79 | + | } | |
| 80 | + | } |