Skip to content

Commit

Merge remote-tracking branch 'origin/main' into workspace-chat

syntaqxcommitted Parentsf7fbb1328427c6Browse files
28 files+2255−1190/28 viewed
+17−0
77 # typescript type checks and tests of the apps and TS services, the
88 # deploy and ops scripts, and the deploy manifest
99 # build the site, sudo and the docs build as they deploy
10+#
11+# On a push to main only `rust` runs, to keep main's caches current: a pull
12+# request's run restores from main's cache, never from another pull
13+# request's, so without it each pull request would start from nothing.
1014 name: CI
1115
1216 on:
1317 pull_request:
1418 branches: [main]
19+ push:
20+ branches: [main]
1521 workflow_dispatch:
1622
1723 # Its token only reads: it checks the code out and nothing more.
4753 !target/debug/incremental
4854 key: cargo-test-${{ runner.os }}-${{ hashFiles('Cargo.lock', 'services/runner/base.json') }}
4955 restore-keys: cargo-test-${{ runner.os }}-
56+ # The workspace's library crates, which Cargo compiles again on every
57+ # checkout, come back from the repository's Actions cache when their
58+ # inputs did not change (scripts/sccache.sh). Test harnesses and
59+ # Workers' own crates are linked, and still compiled.
60+ - name: sccache
61+ run: bash scripts/sccache.sh install
5062 - name: Tests
5163 run: cargo test --workspace --locked --quiet
64+ - name: sccache's hits and misses
65+ if: ${{ always() }}
66+ run: bash scripts/sccache.sh stats
5267
5368 typescript:
5469 name: TypeScript
70+ if: ${{ github.event_name != 'push' }}
5571 runs-on: ubuntu-latest
5672 timeout-minutes: 30
5773 steps:
7187
7288 build:
7389 name: Build
90+ if: ${{ github.event_name != 'push' }}
7491 runs-on: ubuntu-latest
7592 timeout-minutes: 30
7693 steps:
+22−5
180180 key: cargo-crates-${{ runner.os }}-${{ hashFiles('Cargo.lock') }}
181181 restore-keys: cargo-crates-${{ runner.os }}-
182182 # The compiled dependencies of this job's units, for wasm32 and the
183− # build scripts and proc macros they run. The workspace's own crates
184− # are compiled again whatever is cached (a checkout's sources are
185− # newer), so an entry is saved only when the dependencies change: a
186− # new Cargo.lock, or a new base image (base.json names its Rust).
187− # Otherwise the nearest earlier entry, of any group, is a start.
183+ # build scripts and proc macros they run. Cargo calls rustc again for
184+ # the workspace's own crates whatever is cached (a checkout's sources
185+ # are newer); sccache, below, answers those calls. So an entry is
186+ # saved only when the dependencies change: a new Cargo.lock, or a new
187+ # base image (base.json names its Rust). Otherwise the nearest earlier
188+ # entry, of any group, is a start.
188189 - name: Cache the Cargo target
189190 if: ${{ matrix.rust }}
190191 uses: actions/cache@v4
215216 !target/**/incremental
216217 key: runner-musl-${{ runner.os }}-${{ hashFiles('Cargo.lock', 'services/runner/base.json') }}
217218 restore-keys: runner-musl-${{ runner.os }}-
219+ # Every rustc call that makes a library (the workspace's crates,
220+ # which Cargo compiles again on every checkout, and any dependency
221+ # not restored above) is looked up by its inputs in the repository's
222+ # Actions cache: unchanged crates come back from it instead of being
223+ # compiled. A Worker's own crate (a cdylib) and the runner's binary
224+ # are still compiled. worker-build runs Cargo, so its wasm32 builds
225+ # go through it too; wasm-bindgen and wasm-opt are not rustc.
226+ # Pinned by version and sha256 in scripts/sccache.sh; without the
227+ # cache, the job builds as before.
228+ - name: sccache
229+ if: ${{ matrix.rust || matrix.image }}
230+ run: bash scripts/sccache.sh install
218231 - name: Install
219232 run: node scripts/deploy.mjs install --only "${{ matrix.units }}"
220233 - name: Deploy ${{ matrix.units }}
224237 # are not drafted as incidents. Optional: without it, nothing is sent.
225238 STATUS_DEPLOY_TOKEN: ${{ secrets.STATUS_DEPLOY_TOKEN }}
226239 run: node scripts/deploy.mjs deploy --only "${{ matrix.units }}" --force --no-migrations --concurrency 2
240+ # Hits and misses, on the log and in the run's summary.
241+ - name: sccache's hits and misses
242+ if: ${{ always() && (matrix.rust || matrix.image) }}
243+ run: bash scripts/sccache.sh stats
227244
228245 edge:
229246 name: edge (${{ matrix.group }})
+1−0
11051105 "g1t-actions",
11061106 "g1t-scan",
11071107 "hex",
1108+ "libc",
11081109 "serde",
11091110 "serde_json",
11101111 "serde_yaml",
+395−25
1010 //! `FinalizeCacheEntryUpload`, `GetCacheEntryDownloadURL`) and the
1111 //! artifacts' (`CreateArtifact`, `FinalizeArtifact`, `ListArtifacts`,
1212 //! `GetSignedArtifactURL`, `DeleteArtifact`). JSON, the toolkit's field
13−//! names.
13+//! names; the cache's methods also in protobuf (`application/protobuf`),
14+//! which other clients of the protocol send (sccache, through OpenDAL).
1415 //! - The cache's older protocol, at `{ACTIONS_CACHE_URL}_apis/artifactcache/…`,
1516 //! which the toolkit's client uses whenever the server it runs against is
16−//! not github.com: on g1t, that is the one it uses.
17+//! not github.com: on g1t, that is the one it uses, and sccache's too.
18+//! Its entries are sent in 32 MB chunks, or in one chunk of any size.
1719 //! - Blobs, at `/actions/toolkit/blobs/{token}`: the signed links those
18−//! hand out. Downloads are a plain GET. Uploads speak the part of Azure
19−//! Blob Storage's protocol the toolkit's client uses (Put Blob, Put
20−//! Block, Put Block List), mapped onto an R2 multipart upload: a block's
21−//! id ends in its index, which is its part's number.
20+//! hand out. Downloads are a GET, of the whole blob or of one byte range
21+//! (`Range`), as the toolkit's client fetches large entries in segments.
22+//! Uploads speak the part of Azure Blob Storage's protocol the toolkit's
23+//! client uses (Put Blob, Put Block, Put Block List), mapped onto an R2
24+//! multipart upload: a block's id ends in its index, which is its part's
25+//! number. An upload link carries a query, as an Azure SAS link does,
26+//! which clients that sign their requests with it need.
2227 //!
2328 //! Every call carries the job's runtime token; the actions service checks
2429 //! it and keeps the entries (cache.rs, artifacts.rs, runtime.rs there).
8590
8691 // ── Twirp ───────────────────────────────────────────────────────────────────
8792
93+/// Twirp's binary encoding.
94+const PROTOBUF: &str = "application/protobuf";
95+
96+/// Whether a request's `Content-Type` is Twirp's protobuf encoding.
97+fn is_protobuf(content_type: &str) -> bool {
98+ let kind = content_type.split(';').next().unwrap_or_default().trim().to_ascii_lowercase();
99+ kind == PROTOBUF || kind == "application/x-protobuf"
100+}
101+
102+/// The cache service's messages in protobuf, read into and written from the
103+/// JSON the handlers use (`results/api/v1/cache.proto`, field numbers as
104+/// there). Only what the cache's three methods carry: strings, a repeated
105+/// string, an int64 and a bool. `metadata` (field 1 of each request) is
106+/// skipped: the runtime token says whose cache it is.
107+mod proto {
108+ use serde_json::{Map, Value};
109+
110+ #[derive(Clone, Copy)]
111+ enum Kind {
112+ Text,
113+ Texts,
114+ Int,
115+ Bool,
116+ }
117+
118+ /// A message's fields: number, JSON name, kind.
119+ type Fields = &'static [(u64, &'static str, Kind)];
120+
121+ fn request_fields(method: &str) -> Option<Fields> {
122+ Some(match method {
123+ "CreateCacheEntry" => &[(2, "key", Kind::Text), (3, "version", Kind::Text)],
124+ "FinalizeCacheEntryUpload" => &[(2, "key", Kind::Text), (3, "size_bytes", Kind::Int), (4, "version", Kind::Text)],
125+ "GetCacheEntryDownloadURL" => &[(2, "key", Kind::Text), (3, "restore_keys", Kind::Texts), (4, "version", Kind::Text)],
126+ _ => return None,
127+ })
128+ }
129+
130+ fn response_fields(method: &str) -> Fields {
131+ match method {
132+ "CreateCacheEntry" => &[(1, "ok", Kind::Bool), (2, "signed_upload_url", Kind::Text), (3, "message", Kind::Text)],
133+ "FinalizeCacheEntryUpload" => &[(1, "ok", Kind::Bool), (2, "entry_id", Kind::Int), (3, "message", Kind::Text)],
134+ "GetCacheEntryDownloadURL" => &[(1, "ok", Kind::Bool), (2, "signed_download_url", Kind::Text), (3, "matched_key", Kind::Text)],
135+ _ => &[],
136+ }
137+ }
138+
139+ fn varint(bytes: &[u8], at: &mut usize) -> Option<u64> {
140+ let mut value = 0u64;
141+ for shift in (0..64).step_by(7) {
142+ let byte = *bytes.get(*at)?;
143+ *at += 1;
144+ value |= u64::from(byte & 0x7f) << shift;
145+ if byte & 0x80 == 0 {
146+ return Some(value);
147+ }
148+ }
149+ None
150+ }
151+
152+ fn put_varint(out: &mut Vec<u8>, mut value: u64) {
153+ while value >= 0x80 {
154+ out.push((value as u8 & 0x7f) | 0x80);
155+ value >>= 7;
156+ }
157+ out.push(value as u8);
158+ }
159+
160+ /// A request of `method` as JSON, or None when it is not one.
161+ pub fn request(method: &str, bytes: &[u8]) -> Option<Value> {
162+ let fields = request_fields(method)?;
163+ let mut out = Map::new();
164+ let mut at = 0;
165+ while at < bytes.len() {
166+ let tag = varint(bytes, &mut at)?;
167+ let (number, wire) = (tag >> 3, tag & 7);
168+ let known = fields.iter().find(|(n, _, _)| *n == number);
169+ match wire {
170+ 0 => {
171+ let value = varint(bytes, &mut at)?;
172+ if let Some((_, name, Kind::Int)) = known {
173+ // An int64 is sent as its two's complement.
174+ out.insert((*name).to_owned(), Value::String((value as i64).to_string()));
175+ }
176+ }
177+ 2 => {
178+ let length = usize::try_from(varint(bytes, &mut at)?).ok()?;
179+ let end = at.checked_add(length).filter(|end| *end <= bytes.len())?;
180+ let raw = &bytes[at..end];
181+ at = end;
182+ match known {
183+ Some((_, name, Kind::Text)) => {
184+ out.insert((*name).to_owned(), Value::String(String::from_utf8(raw.to_vec()).ok()?));
185+ }
186+ Some((_, name, Kind::Texts)) => {
187+ let text = Value::String(String::from_utf8(raw.to_vec()).ok()?);
188+ match out.entry((*name).to_owned()).or_insert_with(|| Value::Array(Vec::new())) {
189+ Value::Array(list) => list.push(text),
190+ _ => return None,
191+ }
192+ }
193+ _ => {}
194+ }
195+ }
196+ 1 => at = at.checked_add(8).filter(|end| *end <= bytes.len())?,
197+ 5 => at = at.checked_add(4).filter(|end| *end <= bytes.len())?,
198+ _ => return None,
199+ }
200+ }
201+ Some(Value::Object(out))
202+ }
203+
204+ /// A response of `method` from its JSON. Defaults are left out, as
205+ /// proto3 does.
206+ pub fn response(method: &str, value: &Value) -> Vec<u8> {
207+ let mut out = Vec::new();
208+ for (number, name, kind) in response_fields(method) {
209+ let field = &value[*name];
210+ match kind {
211+ Kind::Bool if field.as_bool() == Some(true) => {
212+ put_varint(&mut out, number << 3);
213+ put_varint(&mut out, 1);
214+ }
215+ Kind::Int => {
216+ let n = field.as_i64().or_else(|| field.as_str().and_then(|s| s.parse().ok())).unwrap_or(0);
217+ if n != 0 {
218+ put_varint(&mut out, number << 3);
219+ put_varint(&mut out, n as u64);
220+ }
221+ }
222+ Kind::Text => {
223+ let text = field.as_str().unwrap_or_default();
224+ if !text.is_empty() {
225+ put_varint(&mut out, (number << 3) | 2);
226+ put_varint(&mut out, text.len() as u64);
227+ out.extend_from_slice(text.as_bytes());
228+ }
229+ }
230+ _ => {}
231+ }
232+ }
233+ out
234+ }
235+}
236+
88237 /// A Twirp error: its code and message, at the status Twirp gives it.
89238 fn twirp_error(code: &str, message: &str) -> Result<Response> {
90239 let status = match code {
92241 "permission_denied" => 403,
93242 "not_found" => 404,
94243 "already_exists" => 409,
95− "invalid_argument" => 400,
244+ "invalid_argument" | "malformed" => 400,
245+ "bad_route" => 404,
96246 "failed_precondition" => 412,
97247 "resource_exhausted" => 429,
98248 _ => 500,
155305 })
156306 }
157307
308+/// The Azure Storage version g1t's blob links answer as.
309+const AZURE_VERSION: &str = "2024-11-04";
310+
311+/// An upload link: the blob's, with a query as an Azure SAS link has one.
312+/// A client that treats it as a container, a blob and a SAS token (OpenDAL,
313+/// which sccache uses) refuses a link without one; the token in the path
314+/// is what g1t checks.
315+pub fn upload_url(api: &str, blob: &str) -> String {
316+ format!("{}?sv={AZURE_VERSION}", blob_url(api, blob))
317+}
318+
158319 /// Starts an R2 upload for an entry the service reserved, and the signed
159320 /// link the toolkit sends it to.
160321 async fn start_upload(bucket: &Bucket, services: &Services, job: &str, token: &str, kind: &str, id: &str, object: &str) -> Result<Outcome<String>> {
167328 )
168329 .await?;
169330 Ok(match signed {
170− Outcome::Ok(blob) => Outcome::Ok(blob_url(&services.addresses.api, &blob)),
331+ Outcome::Ok(blob) => Outcome::Ok(upload_url(&services.addresses.api, &blob)),
171332 Outcome::Fail(refused) => Outcome::Fail(refused),
172333 })
173334 }
178339 let Some(job) = runtime_job(&token) else {
179340 return twirp_error("unauthenticated", "Send the job's ACTIONS_RUNTIME_TOKEN as a bearer token.");
180341 };
181− let body: Value = request.json().await.unwrap_or(Value::Null);
342+ // Twirp clients send JSON or protobuf, and are answered in kind.
343+ let binary = is_protobuf(&request.headers().get("content-type")?.unwrap_or_default());
344+ let body: Value = if binary {
345+ match proto::request(method, &request.bytes().await.unwrap_or_default()) {
346+ Some(body) => body,
347+ None => return twirp_error("malformed", "That is not a protobuf message this method takes."),
348+ }
349+ } else {
350+ request.json().await.unwrap_or(Value::Null)
351+ };
352+ let answer = |value: Value| -> Result<Response> {
353+ if binary {
354+ let mut response = Response::from_bytes(proto::response(method, &value))?;
355+ response.headers_mut().set("content-type", PROTOBUF)?;
356+ Ok(response)
357+ } else {
358+ Response::from_json(&value)
359+ }
360+ };
182361 let bucket = env.bucket("ACTIONS_CACHE")?;
183362 let actions = &services.actions;
184363 let (run, own_job) = backend_ids(&token);
201380 let found: Outcome<Option<CacheHit>> = g1t_kit::call(actions, "cache_lookup", &args).await?;
202381 match found {
203382 Outcome::Ok(Some(CacheHit { key, blob: Some(blob), .. })) => {
204− Response::from_json(&json!({ "ok": true, "signed_download_url": blob_url(&services.addresses.api, &blob), "matched_key": key }))
383+ answer(json!({ "ok": true, "signed_download_url": blob_url(&services.addresses.api, &blob), "matched_key": key }))
205384 }
206− Outcome::Ok(_) => Response::from_json(&json!({ "ok": false, "signed_download_url": "", "matched_key": "" })),
385+ Outcome::Ok(_) => answer(json!({ "ok": false, "signed_download_url": "", "matched_key": "" })),
207386 Outcome::Fail(refused) => twirp_failure(&refused),
208387 }
209388 }
212391 let reserved: Outcome<CacheReservation> = g1t_kit::call(actions, "cache_reserve", &args).await?;
213392 let reserved = match reserved {
214393 Outcome::Ok(reserved) => reserved,
215− // The client warns with this and goes on, as for a key
216− // another job is saving.
217− Outcome::Fail(refused) => return Response::from_json(&json!({ "ok": false, "signed_upload_url": "", "message": refused.message })),
394+ // A key already saved, or being saved by another job, to a
395+ // protobuf client (OpenDAL's) is Twirp's `already_exists`
396+ // (409), which it takes as "someone else has it": sccache
397+ // then still writes. An `ok: false` would read as a broken
398+ // cache, and sccache would only read from it.
399+ Outcome::Fail(refused) if binary && refused.code == FailureCode::Conflict => return twirp_failure(&refused),
400+ // The toolkit's client logs this ("another job may be
401+ // creating this cache") and goes on.
402+ Outcome::Fail(refused) => return answer(json!({ "ok": false, "signed_upload_url": "", "message": refused.message })),
218403 };
219404 match start_upload(&bucket, services, &job, &token, "cache", &reserved.id, &reserved.object).await? {
220− Outcome::Ok(url) => Response::from_json(&json!({ "ok": true, "signed_upload_url": url })),
221− Outcome::Fail(refused) => Response::from_json(&json!({ "ok": false, "signed_upload_url": "", "message": refused.message })),
405+ Outcome::Ok(url) => answer(json!({ "ok": true, "signed_upload_url": url })),
406+ Outcome::Fail(refused) => answer(json!({ "ok": false, "signed_upload_url": "", "message": refused.message })),
222407 }
223408 }
224409 (CACHE_SERVICE, "FinalizeCacheEntryUpload") => {
226411 let pending: Outcome<CacheReservation> = g1t_kit::call(actions, "cache_upload", &args).await?;
227412 let pending = match pending {
228413 Outcome::Ok(pending) => pending,
229− Outcome::Fail(refused) => return Response::from_json(&json!({ "ok": false, "entry_id": "0", "message": refused.message })),
414+ Outcome::Fail(refused) => return answer(json!({ "ok": false, "entry_id": "0", "message": refused.message })),
230415 };
231416 let Some(object) = bucket.head(&pending.object).await? else {
232− return Response::from_json(&json!({ "ok": false, "entry_id": "0", "message": "Nothing was uploaded for that entry." }));
417+ return answer(json!({ "ok": false, "entry_id": "0", "message": "Nothing was uploaded for that entry." }));
233418 };
234419 match commit_cache(&bucket, services, &job, &token, &pending.id, object.size()).await? {
235− Outcome::Ok(()) => Response::from_json(&json!({ "ok": true, "entry_id": pending.number.to_string() })),
236− Outcome::Fail(refused) => Response::from_json(&json!({ "ok": false, "entry_id": "0", "message": refused.message })),
420+ Outcome::Ok(()) => answer(json!({ "ok": true, "entry_id": pending.number.to_string() })),
421+ Outcome::Fail(refused) => answer(json!({ "ok": false, "entry_id": "0", "message": refused.message })),
237422 }
238423 }
239424 (ARTIFACT_SERVICE, "CreateArtifact") => {
321506 }
322507
323508 /// The part a chunk of the older protocol is, from its `Content-Range`:
324−/// chunks are `CACHE_PART_BYTES` apart, as the toolkit sends them.
509+/// chunks are `CACHE_PART_BYTES` apart, as the toolkit sends them. A first
510+/// chunk may be larger, up to `MAX_BLOCK_BYTES`: a client that sends an
511+/// entry in one request (sccache does) sends a single chunk from 0.
325512 pub fn chunk_part(range: &str) -> Option<(u16, u64)> {
326513 let range = range.trim().strip_prefix("bytes ")?;
327514 let (span, _) = range.split_once('/')?;
328515 let (start, end) = span.split_once('-')?;
329516 let (start, end): (u64, u64) = (start.trim().parse().ok()?, end.trim().parse().ok()?);
330− if end < start || start % CACHE_PART_BYTES != 0 || end - start + 1 > CACHE_PART_BYTES {
517+ if end < start {
331518 return None;
332519 }
333− Some(((start / CACHE_PART_BYTES + 1) as u16, end - start + 1))
520+ let length = end - start + 1;
521+ if start == 0 && length <= MAX_BLOCK_BYTES {
522+ return Some((1, length));
523+ }
524+ if start % CACHE_PART_BYTES != 0 || length > CACHE_PART_BYTES {
525+ return None;
526+ }
527+ Some(((start / CACHE_PART_BYTES + 1) as u16, length))
334528 }
335529
530+/// The bytes a download's `Range` header asks for, out of `size`: first and
531+/// last, inclusive. None to send the whole blob (no header, or one this
532+/// does not read, such as several ranges); `Some(None)` when the range is
533+/// past the end (416).
534+pub fn byte_range(header: &str, size: u64) -> Option<Option<(u64, u64)>> {
535+ let spec = header.trim().strip_prefix("bytes=")?.trim();
536+ if spec.contains(',') {
537+ return None;
538+ }
539+ let (first, last) = spec.split_once('-')?;
540+ let (first, last) = (first.trim(), last.trim());
541+ let range = if first.is_empty() {
542+ // The last `n` bytes.
543+ let n: u64 = last.parse().ok()?;
544+ if n == 0 || size == 0 {
545+ return Some(None);
546+ }
547+ (size.saturating_sub(n), size - 1)
548+ } else {
549+ let first: u64 = first.parse().ok()?;
550+ let last: u64 = if last.is_empty() { u64::MAX } else { last.parse().ok()? };
551+ if last < first {
552+ return None;
553+ }
554+ if first >= size {
555+ return Some(None);
556+ }
557+ (first, last.min(size - 1))
558+ };
559+ Some(Some(range))
560+}
561+
336562 /// `{ACTIONS_CACHE_URL}_apis/artifactcache/…`. `rest` is the path after it.
337563 pub async fn cache_v1(mut request: Request, env: &Env, services: &Services, method: &str, rest: &str) -> Result<Response> {
338564 let token = bearer(&request);
527753 headers.set("content-length", &size.to_string())?;
528754 headers.set("content-type", grant.content_type.as_deref().unwrap_or("application/octet-stream"))?;
529755 headers.set("x-ms-blob-type", "BlockBlob")?;
756+ headers.set("accept-ranges", "bytes")?;
530757 if let Some(name) = &grant.filename {
531758 headers.set("content-disposition", &format!("attachment; filename=\"{}\"", name.replace('"', "")))?;
532759 }
538765 headers(&mut response, object.size())?;
539766 return Ok(response);
540767 }
768+ // One byte range (`Range`, or Azure's `x-ms-range`): the
769+ // toolkit's client fetches a large entry in segments, side by
770+ // side, and writes each where its range says.
771+ let asked = match request.headers().get("x-ms-range")? {
772+ Some(range) => Some(range),
773+ None => request.headers().get("range")?,
774+ };
775+ if let Some(asked) = asked.filter(|r| !r.trim().is_empty()) {
776+ let Some(object) = bucket.head(&grant.object).await? else { return azure_error(404, "BlobNotFound", "It is gone.") };
777+ let size = object.size();
778+ match byte_range(&asked, size) {
779+ Some(Some((first, last))) => {
780+ let length = last - first + 1;
781+ let Some(object) = bucket.get(&grant.object).range(worker::Range::OffsetWithLength { offset: first, length }).execute().await? else {
782+ return azure_error(404, "BlobNotFound", "It is gone.");
783+ };
784+ let Some(body) = object.body() else { return azure_error(404, "BlobNotFound", "It is gone.") };
785+ let mut response = Response::from_body(body.response_body()?)?.with_status(206);
786+ headers(&mut response, length)?;
787+ response.headers_mut().set("content-range", &format!("bytes {first}-{last}/{size}"))?;
788+ return Ok(response);
789+ }
790+ Some(None) => {
791+ let mut response = azure_error(416, "InvalidRange", "The range is past the end of the blob.")?;
792+ response.headers_mut().set("content-range", &format!("bytes */{size}"))?;
793+ return Ok(response);
794+ }
795+ // Not a range this reads: the whole blob.
796+ None => {}
797+ }
798+ }
541799 let Some(object) = bucket.get(&grant.object).execute().await? else { return azure_error(404, "BlobNotFound", "It is gone.") };
542800 let size = object.size();
543801 let Some(body) = object.body() else { return azure_error(404, "BlobNotFound", "It is gone.") };
674932 let mb32 = CACHE_PART_BYTES;
675933 assert_eq!(chunk_part(&format!("bytes 0-{}/*", mb32 - 1)), Some((1, mb32)));
676934 assert_eq!(chunk_part(&format!("bytes {}-{}/*", mb32 * 2, mb32 * 2 + 99)), Some((3, 100)));
677− // Not on a chunk's boundary, too long, or not a range.
935+ // A whole entry in one chunk, as sccache (OpenDAL) sends it: its
936+ // check file is 13 bytes, a compiled crate can be well over 32 MB.
937+ assert_eq!(chunk_part("bytes 0-12/*"), Some((1, 13)));
938+ assert_eq!(chunk_part(&format!("bytes 0-{}/*", mb32)), Some((1, mb32 + 1)));
939+ assert_eq!(chunk_part(&format!("bytes 0-{}/*", MAX_BLOCK_BYTES - 1)), Some((1, MAX_BLOCK_BYTES)));
940+ // Not on a chunk's boundary, too long, backwards, or not a range.
678941 assert_eq!(chunk_part("bytes 5-10/*"), None);
679− assert_eq!(chunk_part(&format!("bytes 0-{}/*", mb32)), None);
942+ assert_eq!(chunk_part(&format!("bytes {mb32}-{}/*", mb32 * 2)), None);
943+ assert_eq!(chunk_part(&format!("bytes 0-{}/*", MAX_BLOCK_BYTES)), None);
944+ assert_eq!(chunk_part("bytes 10-5/*"), None);
680945 assert_eq!(chunk_part("0-10"), None);
681946 }
682947
948+ #[test]
949+ fn downloads_read_one_byte_range() {
950+ // OpenDAL's stat: the first byte.
951+ assert_eq!(byte_range("bytes=0-0", 100), Some(Some((0, 0))));
952+ // The toolkit's segments, the last one cut at the end.
953+ assert_eq!(byte_range("bytes=0-49", 100), Some(Some((0, 49))));
954+ assert_eq!(byte_range("bytes=50-999", 100), Some(Some((50, 99))));
955+ assert_eq!(byte_range("bytes=90-", 100), Some(Some((90, 99))));
956+ assert_eq!(byte_range("bytes=-10", 100), Some(Some((90, 99))));
957+ assert_eq!(byte_range("bytes=-1000", 100), Some(Some((0, 99))));
958+ // Past the end: 416.
959+ assert_eq!(byte_range("bytes=100-200", 100), Some(None));
960+ assert_eq!(byte_range("bytes=0-0", 0), Some(None));
961+ assert_eq!(byte_range("bytes=-0", 100), Some(None));
962+ // Not read: the whole blob.
963+ assert_eq!(byte_range("bytes=0-1,5-6", 100), None);
964+ assert_eq!(byte_range("bytes=9-3", 100), None);
965+ assert_eq!(byte_range("items=0-1", 100), None);
966+ assert_eq!(byte_range("bytes=a-b", 100), None);
967+ }
968+
969+ #[test]
970+ fn upload_links_carry_a_query_as_sas_links_do() {
971+ let url = upload_url("https://api.g1t.sh", "tok.sig");
972+ assert_eq!(url, "https://api.g1t.sh/actions/toolkit/blobs/tok.sig?sv=2024-11-04");
973+ // How OpenDAL reads a signed upload link: a container, a blob in
974+ // it, and a SAS query, all of which must be there.
975+ let rest = url.strip_prefix("https://api.g1t.sh/").unwrap();
976+ let (path, query) = rest.split_once('?').unwrap();
977+ let (container, blob) = path.split_once('/').unwrap();
978+ assert_eq!((container, blob, query), ("actions", "toolkit/blobs/tok.sig", "sv=2024-11-04"));
979+ }
980+
981+ /// A protobuf length-delimited field, as prost writes it.
982+ fn pb_text(number: u8, text: &str) -> Vec<u8> {
983+ let mut out = vec![(number << 3) | 2, text.len() as u8];
984+ out.extend_from_slice(text.as_bytes());
985+ out
986+ }
987+
988+ /// The requests sccache 0.18 sends (OpenDAL 0.58's `ghac` service, with
989+ /// prost): fields in number order, defaults left out, no metadata.
990+ #[test]
991+ fn twirp_reads_sccaches_protobuf_requests() {
992+ assert!(is_protobuf("application/protobuf"));
993+ assert!(is_protobuf("Application/Protobuf; charset=utf-8"));
994+ assert!(!is_protobuf("application/json"));
995+ assert!(!is_protobuf(""));
996+
997+ let key = "sccache/f/c/b/fcb0a1d2e3";
998+ let version = "sccache-v0.18.0";
999+ let create = [pb_text(2, key), pb_text(3, version)].concat();
1000+ let read = proto::request("CreateCacheEntry", &create).unwrap();
1001+ assert_eq!((text(&read, "key"), text(&read, "version")), (key.to_owned(), version.to_owned()));
1002+
1003+ // size_bytes is field 3, a varint: 300 is 0xac 0x02.
1004+ let finalize = [pb_text(2, key), vec![0x18, 0xac, 0x02], pb_text(4, version)].concat();
1005+ let read = proto::request("FinalizeCacheEntryUpload", &finalize).unwrap();
1006+ assert_eq!(number(&read, "size_bytes"), Some(300));
1007+ assert_eq!(text(&read, "version"), version);
1008+
1009+ let lookup = [pb_text(2, key), pb_text(4, version)].concat();
1010+ let read = proto::request("GetCacheEntryDownloadURL", &lookup).unwrap();
1011+ assert_eq!(text(&read, "key"), key);
1012+ assert!(field(&read, "restore_keys").is_null());
1013+ // The toolkit's own lookup, with metadata (skipped) and restore keys.
1014+ let metadata = vec![0x0a, 0x02, 0x08, 0x07];
1015+ let with_restore = [metadata, pb_text(2, "k"), pb_text(3, "k-"), pb_text(3, "x-"), pb_text(4, "v")].concat();
1016+ let read = proto::request("GetCacheEntryDownloadURL", &with_restore).unwrap();
1017+ assert_eq!(field(&read, "restore_keys"), &json!(["k-", "x-"]));
1018+ assert_eq!(text(&read, "version"), "v");
1019+
1020+ // The same three, as prost 0.14 encodes them with OpenDAL's
1021+ // generated types (the `ghac` crate, 0.3.0), byte for byte.
1022+ let recorded = |hex: &str| -> Vec<u8> { (0..hex.len()).step_by(2).map(|i| u8::from_str_radix(&hex[i..i + 2], 16).unwrap()).collect() };
1023+ let prefix = "1218736363616368652f662f632f622f66636230613164326533";
1024+ let suffix = "0f736363616368652d76302e31382e30";
1025+ assert_eq!(recorded(&format!("{prefix}1a{suffix}")), create);
1026+ assert_eq!(recorded(&format!("{prefix}18ac0222{suffix}")), finalize);
1027+ assert_eq!(recorded(&format!("{prefix}22{suffix}")), lookup);
1028+
1029+ // Cut short, or not a cache method.
1030+ assert!(proto::request("CreateCacheEntry", &create[..create.len() - 1]).is_none());
1031+ assert!(proto::request("CreateCacheEntry", &[0x12, 0xff]).is_none());
1032+ assert!(proto::request("CreateArtifact", &create).is_none());
1033+ assert_eq!(proto::request("CreateCacheEntry", &[]), Some(json!({})));
1034+ }
1035+
1036+ #[test]
1037+ fn twirp_answers_in_protobuf_as_prost_reads_it() {
1038+ let url = "https://api.g1t.sh/actions/toolkit/blobs/t?sv=2024-11-04";
1039+ let created = proto::response("CreateCacheEntry", &json!({ "ok": true, "signed_upload_url": url }));
1040+ assert_eq!(created, [vec![0x08, 0x01], pb_text(2, url)].concat());
1041+ // Refused: ok false is the default, so only the message is sent.
1042+ let refused = proto::response("CreateCacheEntry", &json!({ "ok": false, "signed_upload_url": "", "message": "no" }));
1043+ assert_eq!(refused, pb_text(3, "no"));
1044+ // entry_id is an int64, given as a string in JSON.
1045+ let finalized = proto::response("FinalizeCacheEntryUpload", &json!({ "ok": true, "entry_id": "300" }));
1046+ assert_eq!(finalized, vec![0x08, 0x01, 0x10, 0xac, 0x02]);
1047+ let found = proto::response("GetCacheEntryDownloadURL", &json!({ "ok": true, "signed_download_url": "u", "matched_key": "k" }));
1048+ assert_eq!(found, [vec![0x08, 0x01], pb_text(2, "u"), pb_text(3, "k")].concat());
1049+ // A miss is an empty message: ok false.
1050+ assert!(proto::response("GetCacheEntryDownloadURL", &json!({ "ok": false, "signed_download_url": "", "matched_key": "" })).is_empty());
1051+ }
1052+
6831053 /// The toolkit's requests, as `@actions/cache` 4 and `@actions/artifact`
6841054 /// 2 send them (protobuf-ts, proto field names, no defaults).
6851055 #[test]
+72−0
4848 | `actions/upload-artifact`, `actions/download-artifact`, `actions/upload-artifact/merge` | The same inputs and outputs as version 4: `retention-days`, `overwrite`, `compression-level`, `include-hidden-files`, `!` exclusions, download by `pattern` with `merge-multiple`, and from another run with `run-id` and `github-token`. Up to 5 GiB each; see [artifacts](#artifacts). |
4949 | `actions/cache`, `actions/cache/restore`, `actions/cache/save` | Kept per repository and branch, found by `key` or the newest under a `restore-keys` prefix. `path` takes globs and `!` exclusions. Up to 2 GiB each; see [the cache](#the-cache). |
5050 | Actions that cache through the toolkit, such as `actions/setup-node` with `cache: npm` or `Swatinem/rust-cache` | The same: they save to and restore from the repository's cache. See [actions built on the toolkit](#actions-built-on-the-toolkit). |
51+| sccache with `SCCACHE_GHA_ENABLED` | The same: its entries go to the repository's cache. See [caching Rust builds](#caching-rust-builds). |
5152 | `permissions: id-token: write` | The job can ask for an OIDC token, and trade it for a cloud provider's credentials. See [OIDC tokens](#oidc-tokens). |
5253 | `docker build`, `push`, `run`, `login`, `compose`, Buildx | The same, with a Docker Engine of the job's own. See [Docker](#docker). |
5354 | `services:` | The same: each service starts before the steps, health checks are waited for, and it is reached at `localhost` on its port and by its name. |
604605 `actions/download-artifact` and `actions/upload-artifact/merge` work:
605606 g1t runs those itself.
606607
608+## Caching Rust builds
609+
610+A checkout gives every file a new modification time, so Cargo compiles
611+your workspace's own crates again even when `target/` was restored from
612+the cache. [sccache](https://github.com/mozilla/sccache) caches each
613+compiler call by what it compiles (the source, the flags, the toolchain
614+and the dependencies), so a crate that did not change comes back from the
615+cache instead. Its GitHub Actions backend works on g1t as it is: it keeps
616+its entries in the repository's cache, under [the cache's](#the-cache)
617+limits and branch rules.
618+
619+1. Install sccache, pinned to a release and checked against its checksum.
620+2. Set `RUSTC_WRAPPER: sccache` and `SCCACHE_GHA_ENABLED: "true"`. The
621+ job's `ACTIONS_RUNTIME_TOKEN` and `ACTIONS_CACHE_URL` are already in
622+ every step's environment; no step needs to export them.
623+3. Set `CARGO_INCREMENTAL: "0"`: sccache does not cache incremental
624+ builds, and a fresh checkout gains nothing from them.
625+
626+```yaml
627+name: CI
628+on:
629+ pull_request:
630+ push:
631+ branches: [main]
632+
633+jobs:
634+ test:
635+ runs-on: ubuntu-latest
636+ env:
637+ RUSTC_WRAPPER: sccache
638+ SCCACHE_GHA_ENABLED: "true"
639+ CARGO_INCREMENTAL: "0"
640+ steps:
641+ - uses: actions/checkout@v5
642+ - name: Install sccache
643+ env:
644+ VERSION: v0.18.0
645+ SHA256: 45f1447fbe231e3037bde351ef70677dd212216c8d62ae7ca409fecc4d6acc89
646+ run: |
647+ name="sccache-$VERSION-x86_64-unknown-linux-musl"
648+ curl -fsSL -o "$RUNNER_TEMP/sccache.tar.gz" \
649+ "https://github.com/mozilla/sccache/releases/download/$VERSION/$name.tar.gz"
650+ echo "$SHA256 $RUNNER_TEMP/sccache.tar.gz" | sha256sum -c -
651+ tar -xzf "$RUNNER_TEMP/sccache.tar.gz" -C "$RUNNER_TEMP"
652+ install -m 0755 "$RUNNER_TEMP/$name/sccache" "$HOME/.cargo/bin/sccache"
653+ - run: cargo test --workspace --locked
654+ - name: sccache's hits and misses
655+ if: always()
656+ run: |
657+ echo '```' >> "$GITHUB_STEP_SUMMARY"
658+ sccache --show-stats | tee -a "$GITHUB_STEP_SUMMARY"
659+ echo '```' >> "$GITHUB_STEP_SUMMARY"
660+```
661+
662+What to expect:
663+
664+| | |
665+| --- | --- |
666+| What is cached | Every library crate (`rlib`): your workspace's crates and their dependencies. |
667+| What is compiled each time | What rustc links: binaries, `cdylib` crates, proc macros, build scripts and test harnesses. |
668+| Its entries | One per compiled crate, often thousands for a workspace, each a few KB to a few MB. They count toward the repository's 10 GiB like any other entry, and the ones not restored for 7 days are deleted. |
669+| Branches | A pull request reads what the default branch saved. Run the workflow on pushes to the default branch too, as above, or each pull request's first run starts with an empty cache. |
670+| Starting over | Set `SCCACHE_GHA_VERSION` to any new value. A new sccache release starts over by itself. |
671+| If the cache cannot be reached | sccache refuses to start. Set `SCCACHE_IGNORE_SERVER_IO_ERROR: "1"` to have rustc run without it instead. |
672+
673+sccache adds to `actions/cache` rather than replacing it: keep caching
674+`~/.cargo/registry/cache` and `target/` keyed by `Cargo.lock`, so Cargo
675+does not call rustc at all for dependencies that did not change, and
676+sccache answers the calls it still makes for your own crates.
677+`Swatinem/rust-cache` works the same way.
678+
607679 ## OIDC tokens
608680
609681 A job can prove which repository, branch and environment it runs for with
+13−0
8383 return `${Math.floor(minutes / 60)}h ${minutes % 60}m`;
8484 }
8585
86+/**
87+ * How long something ran, as text. While it is still running the server and
88+ * the browser count to a different now, so the text may differ on hydration
89+ * (as TimeAgo's does).
90+ */
91+export function Duration({ start, end, className }: { start: string | null; end: string | null; className?: string }) {
92+ return (
93+ <span className={className} suppressHydrationWarning>
94+ {duration(start, end)}
95+ </span>
96+ );
97+}
98+
8699 /** `main` from `refs/heads/main`, `v1.2` from a tag, `#12` for a pull request. */
87100 export function shortRef(ref: string): string {
88101 const pull = /^refs\/pull\/(\d+)\//.exec(ref);
+6−1
231231 {name}
232232 {detail && <span className="text-muted"> — {detail}</span>}
233233 </span>
234− {time && <span className="shrink-0 animate-fade-in font-mono text-xs text-faint">{time}</span>}
234+ {time && (
235+ // A running job's time counts to now, which differs between server and browser.
236+ <span className="shrink-0 animate-fade-in font-mono text-xs text-faint" suppressHydrationWarning>
237+ {time}
238+ </span>
239+ )}
235240 {timing && <SkeletonLine className="w-10 shrink-0 text-xs" />}
236241 {to && (
237242 <Link to={to} className="shrink-0 text-xs text-muted hover:text-fg hover:underline">
+10−5
3131
3232 import type { Route } from "./+types/actions-run";
3333 import { page } from "../../lib/meta";
34−import { LogText, Notes, StatusIcon, duration, shortRef, standingWord, useJobLog } from "../../components/actions";
34+import { Duration, LogText, Notes, StatusIcon, shortRef, standingWord, useJobLog } from "../../components/actions";
3535 import { Markdown } from "../../components/markdown";
3636 import { Button, ErrorText, SubmitButton, TimeAgo, usePending } from "../../components/ui";
3737 import { CheckboxOption } from "../../components/ui/checkbox";
150150 {matches} {matches === 1 ? "line" : "lines"}
151151 </span>
152152 )}
153− <span className="ml-auto shrink-0 font-mono text-xs text-faint">{duration(step.startedAt, step.finishedAt)}</span>
153+ <Duration className="ml-auto shrink-0 font-mono text-xs text-faint" start={step.startedAt} end={step.finishedAt} />
154154 </summary>
155155 <div className="border-t border-line bg-bg/60">
156156 {text === undefined ? (
365365 <h3 className="text-base font-semibold">{job.name}</h3>
366366 <span className="text-sm text-muted">
367367 {job.cancelling ? "Cancelling: running its cleanup steps" : standingWord({ ...job, of: "job" })}
368− {job.startedAt && ` · ${duration(job.startedAt, job.finishedAt)}`}
368+ {job.startedAt && (
369+ <>
370+ {" · "}
371+ <Duration start={job.startedAt} end={job.finishedAt} />
372+ </>
373+ )}
369374 </span>
370375 <span className="ml-auto flex items-center gap-2">
371376 {rerun}
776781 {run.event}
777782 {run.actor && ` by ${run.actor}`} · <TimeAgo at={run.createdAt} />
778783 </span>
779− {run.startedAt && <span className="font-mono text-xs">{duration(run.startedAt, run.finishedAt)}</span>}
784+ {run.startedAt && <Duration className="font-mono text-xs" start={run.startedAt} end={run.finishedAt} />}
780785 {run.attempt > 1 && attempts.length <= 1 && <span>Attempt {run.attempt}</span>}
781786 {cancelling && <span className="text-warn">Its jobs are running their cleanup steps</span>}
782787 {detail.approval?.state === "approved" && detail.approval.approvedBy && <span>Approved by {detail.approval.approvedBy}</span>}
868873 >
869874 <StatusIcon status={job.status} conclusion={job.conclusion} of="job" environment={job.environment} size={14} />
870875 <span className="min-w-0 truncate">{job.name}</span>
871− <span className="ml-auto shrink-0 font-mono text-xs text-faint">{duration(job.startedAt, job.finishedAt)}</span>
876+ <Duration className="ml-auto shrink-0 font-mono text-xs text-faint" start={job.startedAt} end={job.finishedAt} />
872877 </Link>
873878 ))}
874879 </nav>
+2−2
66
77 import type { Route } from "./+types/actions";
88 import { page } from "../../lib/meta";
9−import { Notes, StatusIcon, duration, shortRef } from "../../components/actions";
9+import { Duration, Notes, StatusIcon, shortRef } from "../../components/actions";
1010 import { AddCiPrompt } from "../../components/add-ci";
1111 import { Button, ComputeNote, CopyLine, EmptyState, ErrorText, SubmitButton, TimeAgo, usePending } from "../../components/ui";
1212 import { CheckboxOption } from "../../components/ui/checkbox";
120120 </span>
121121 <span className="w-28 shrink-0 text-right text-xs text-faint">
122122 <TimeAgo at={run.createdAt} />
123− {run.startedAt && <span className="block font-mono">{duration(run.startedAt, run.finishedAt)}</span>}
123+ {run.startedAt && <Duration className="block font-mono" start={run.startedAt} end={run.finishedAt} />}
124124 </span>
125125 </Link>
126126 </li>
+60−0
236236 assert!(source.contains("target/x86_64-unknown-linux-musl/release"));
237237 }
238238
239+/// Whether a step runs, for a job going `status`, with `matrix`.
240+fn step_runs(step: &workflow::Step, matrix: Value, status: Status) -> bool {
241+ let mut contexts = Map::new();
242+ contexts.insert("matrix".into(), matrix);
243+ let scope = Scope { contexts: &contexts, status, hash_files: None };
244+ expr::condition(step.condition.as_deref().unwrap_or_default(), &scope).unwrap()
245+}
246+
247+#[test]
248+fn rust_builds_go_through_sccache_before_cargo_and_report_after() {
249+ let step = |job: &workflow::Job, run: &str| -> (usize, workflow::Step) {
250+ let at = job.steps.iter().position(|s| s.run.as_deref() == Some(run)).unwrap_or_else(|| panic!("{}: no step runs {run}", job.id));
251+ (at, job.steps[at].clone())
252+ };
253+ let deploy = read("deploy.yml");
254+ for stage in ["core", "edge", "front"] {
255+ let job = deploy.jobs.iter().find(|j| j.id == stage).unwrap();
256+ let (install_at, install) = step(job, "bash scripts/sccache.sh install");
257+ let (stats_at, stats) = step(job, "bash scripts/sccache.sh stats");
258+ // Before anything runs Cargo (worker-build's install, the build),
259+ // and the statistics last.
260+ let install_step = job.steps.iter().position(|s| s.name.as_deref() == Some("Install")).unwrap();
261+ assert!(install_at < install_step, "{stage}");
262+ assert_eq!(stats_at, job.steps.len() - 1, "{stage}");
263+ for (rust, image, uses) in [(true, false, true), (false, true, true), (false, false, false)] {
264+ let matrix = json!({ "group": "g", "units": "u", "rust": rust, "image": image });
265+ assert_eq!(step_runs(&install, matrix.clone(), Status::Success), uses, "{stage} rust={rust} image={image}");
266+ assert_eq!(step_runs(&stats, matrix.clone(), Status::Success), uses, "{stage}");
267+ // A failed build still says what was cached.
268+ assert_eq!(step_runs(&stats, matrix, Status::Failure), uses, "{stage}");
269+ }
270+ }
271+ let ci = read("ci.yml");
272+ let rust = ci.jobs.iter().find(|j| j.id == "rust").unwrap();
273+ let (install_at, _) = step(rust, "bash scripts/sccache.sh install");
274+ let (tests_at, _) = step(rust, "cargo test --workspace --locked --quiet");
275+ let (stats_at, stats) = step(rust, "bash scripts/sccache.sh stats");
276+ assert!(install_at < tests_at && tests_at < stats_at);
277+ assert!(step_runs(&stats, json!({}), Status::Failure));
278+ // main's runs keep the caches pull requests restore from: only Rust.
279+ assert!(ci.trigger("push").unwrap().branches.allows("main"));
280+ assert!(ci.trigger("pull_request").is_some());
281+ for (id, on_push) in [("rust", true), ("typescript", false), ("build", false)] {
282+ let job = ci.jobs.iter().find(|j| j.id == id).unwrap();
283+ for (event, expected) in [("push", on_push), ("pull_request", true)] {
284+ let mut contexts = Map::new();
285+ contexts.insert("github".into(), json!({ "event_name": event, "ref": "refs/heads/main" }));
286+ let scope = Scope { contexts: &contexts, status: Status::Success, hash_files: None };
287+ assert_eq!(expr::condition(job.condition.as_deref().unwrap_or_default(), &scope).unwrap(), expected, "{id} on {event}");
288+ }
289+ }
290+ // The download is pinned by version and checksum.
291+ let script = std::fs::read_to_string(workflows_dir().join("../../scripts/sccache.sh")).unwrap();
292+ assert!(script.contains("VERSION=0.18.0"));
293+ assert!(script.lines().any(|line| line.strip_prefix("SHA256=").is_some_and(|sum| sum.len() == 64)));
294+ assert!(script.contains("sha256sum -c"));
295+ assert!(script.contains("RUSTC_WRAPPER="));
296+ assert!(script.contains("GITHUB_STEP_SUMMARY"));
297+}
298+
239299 #[test]
240300 fn the_runner_base_rebuilds_weekly_on_a_machine_with_docker() {
241301 let base = read("runner-base.yml");
+4−0
2525 # native-certs: a guarded sandbox re-signs HTTPS with a certificate the
2626 # runner adds to the system store (guard.rs), which webpki-roots never sees.
2727 ureq = { version = "2", features = ["json", "native-certs"] }
28+
29+[target.'cfg(unix)'.dependencies]
30+# A step's signals back to their defaults before it runs (actions/process.rs).
31+libc = "0.2"
+36−7
130130
131131 /// Sends `signal` to the process's group (it leads its own), so what the
132132 /// step started hears it too. Windows has no signals: it is left to `kill`.
133+/// Sent with kill(2) itself: a machine without the `kill` program (procps
134+/// is not in every image) would otherwise give the step no SIGINT at all,
135+/// and leave what it started running after the step was killed.
136+#[cfg(unix)]
133137 fn signal(child: &std::process::Child, signal: &str) {
134− if cfg!(unix) {
135− let _ = Command::new("kill")
136− .args([format!("-{signal}"), "--".into(), format!("-{}", child.id())])
137− .stdout(Stdio::null())
138− .stderr(Stdio::null())
139− .status();
138+ let number = match signal {
139+ "INT" => libc::SIGINT,
140+ "TERM" => libc::SIGTERM,
141+ _ => libc::SIGKILL,
142+ };
143+ let Ok(group) = libc::pid_t::try_from(child.id()) else {
144+ return;
145+ };
146+ // SAFETY: kill(2) with a negative pid signals that process group; it
147+ // reads no memory of ours.
148+ unsafe {
149+ libc::kill(-group, number);
140150 }
141151 }
142152
153+#[cfg(not(unix))]
154+fn signal(_child: &std::process::Child, _signal: &str) {}
155+
143156 /// Runs the command, sending its output (stdout and stderr together, a
144157 /// line at a time) through `commands` to the log, until it ends or
145158 /// `timeout` passes.
158171 command.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped());
159172 // A group of its own, so a cancellation reaches what the step started.
160173 #[cfg(unix)]
161− std::os::unix::process::CommandExt::process_group(&mut command, 0);
174+ {
175+ use std::os::unix::process::CommandExt;
176+ command.process_group(0);
177+ // A shell cannot trap a signal it was started with ignored, and a
178+ // runner started in the background (as a service, or under `&`)
179+ // passes SIGINT on ignored. Back to the defaults, so a cancelled
180+ // step hears SIGINT and its own trap runs.
181+ // SAFETY: only signal(), which is async-signal-safe, between fork and exec.
182+ unsafe {
183+ command.pre_exec(|| {
184+ for signal in [libc::SIGINT, libc::SIGTERM, libc::SIGQUIT] {
185+ libc::signal(signal, libc::SIG_DFL);
186+ }
187+ Ok(())
188+ });
189+ }
190+ }
162191 let mut child = command.spawn()?;
163192 let (sender, lines) = mpsc::channel::<String>();
164193 let mut readers = Vec::new();
+1−1
387387 std::fs::create_dir_all(&origin).unwrap();
388388 git_in(&origin, &["init", "--quiet", "--initial-branch=main"], None).unwrap();
389389 commit(&origin, "a.txt", "one");
390− git_in(&origin, &["tag", "-a", "v1", "-m", "v1"], None).unwrap();
390+ git_in(&origin, &["-c", "user.name=t", "-c", "user.email=t@example.com", "-c", "tag.gpgsign=false", "tag", "-a", "v1", "-m", "v1"], None).unwrap();
391391 let mirror = root.join("mirror.git");
392392
393393 // Full.
+10−3
2020
2121 | Source | What | Where it lands |
2222 | --- | --- | --- |
23−| Billable usage, `GET /accounts/{account}/billable-usage?from=&to=` | One row per service per day in FOCUS columns: `ServiceFamilyName`, `ServiceName`, `ChargePeriodStart`, `ConsumedQuantity` (else `PricingQuantity`), `ContractedCost` / `BilledCost` / `EffectiveCost`. Every page is read (`result_info`: `cursor`, else `total_pages`), over whole billing cycles (see [The billing cycle](#the-billing-cycle)). Every product g1t uses appears here once it is used: Workers, Workers for Platforms, D1, KV, R2, Queues, Containers, Durable Objects, Artifacts, Browser Rendering, Workers AI, Vectorize, Cloudflare for SaaS, Email. While a cycle is open its rows carry no cost; `ListCost` is never taken, since it is before the included amounts. | `cost_lines`, source `billable_usage`: `quantity` (consumed), `billed_usd` (Cloudflare's own cost), `billable_quantity` and `cost_usd` (over the cycle), `basis`; the read itself in `cost_reads` |
23+| Billable usage, `GET /accounts/{account}/billable-usage?from=&to=` | One row per service per day in FOCUS columns: `ServiceFamilyName`, `ServiceName`, `ChargePeriodStart`, `ConsumedQuantity` (else `PricingQuantity`), `ContractedCost` / `BilledCost` / `EffectiveCost`. Every page is read (`result_info`: `cursor`, else `total_pages`), over whole billing cycles (see [The billing cycle](#the-billing-cycle)). Every product g1t uses appears here once it is used: Workers, Workers for Platforms, D1, KV, R2, Queues, Containers, Durable Objects, Artifacts, Browser Rendering, Workers AI, Vectorize, Cloudflare for SaaS, Email. While a cycle is open its rows carry no cost; `ListCost` is never taken as a cost, since it is before the included amounts (the keeper reads it for a unit's rate, below). | `cost_lines`, source `billable_usage`: `quantity` (consumed), `billed_usd` (Cloudflare's own cost), `billable_quantity` and `cost_usd` (over the cycle), `basis`; the read itself in `cost_reads` |
2424 | GraphQL `artifactsEventsAdaptiveGroups` | Artifacts' own count by `date`, `eventType` and `repositoryName`. Operations are `create`, `fork`, `push`, `pull`, `delete`; errors (`rateLimited`, `serverError`, …) are kept but not counted. | `cost_lines`, source `artifacts_events`; per workspace (from the store key `<workspace>--<repo>`; a pull request's working copy, `pulls--<id>`, is its repository's workspace's, from repos' `pull_owners`) in `own_counts` as `cloudflare_git` |
2525 | GraphQL `aiGatewayRequestsAdaptiveGroups`, filtered to `AI_GATEWAY_ID` | What AI Gateway priced g1t's own provider traffic at, by `date`, `provider` and `model`: `count`, `sum.cost` (dollars), `sum.tokensIn`/`tokensOut`/`cacheReadTokens`/`cacheWriteTokens`; asked twice in one query, filtered `wholesale: 0` and `wholesale: 1`. `wholesale` is never a dimension: grouped by it, Cloudflare answers no rows and no error (until 2026-10-08 that left the gateway's side empty while it had logged $11.11). Read over its own window: the last 31 days until it has answered with a line, then the last few. Field names checked against Cloudflare's schema (introspection of `AccountAiGatewayRequestsAdaptiveGroups{Sum,Dimensions,Filter_InputObject}`). An adaptive (sampled) dataset: an estimate, close at g1t's volumes. Only g1t's hosted models go through this gateway: a workspace's own provider is called at its own address, never here. | `cost_lines`, source `ai_gateway`, product `ai_gateway_requests`: per day and model a line `<provider>_<model>` (requests, at the gateway's cost), and at no cost `…__tokens`, `…__cache_read_tokens`, `…__cache_write_tokens`; Cloudflare-billed (unified billing) requests are prefixed `wholesale__`. Mapped to `models` (migration 0036). A re-read day replaces all its gateway lines. |
2626 | The ledger | Every charge: its cost at the price book's cost, what it was charged at price, what paid for it. | read, never written |
299299 | Kind | When | What to do |
300300 | --- | --- | --- |
301301 | Count | g1t's count and Cloudflare's differ by more than the mapping's `drift_percent` (10%) | Find out what Cloudflare counts: compare its events with `own_counts` `artifacts_*` and `cost_operations`. If it counts more (binding reads, `ls-refs`), either change repos' `operation_mapping` so customers are charged for what Cloudflare counts, or leave it and let the per-unit cost rise (below). |
302−| Cost | Cloudflare charged more than `drift_percent` away from the price book's cost of the same usage, with at least `min_daily_cost` | A price is stale: check the proposals. |
302+| Cost | The price book's cost of a product's usage over the 7 days (the ledger's `cost_micros`) is more than `drift_percent` away from what the same usage comes to at Cloudflare's list prices before the included amounts, with at least `min_daily_cost` either side. The list cost is each billable-usage line mapped to the product, its whole quantity at `cycle::LIST_PRICES` (not rounded to whole millions; `margin::list_costs`). Never what Cloudflare billed: that is net of the cycle's included amounts while the price book costs every unit, so a product whose usage mostly fits in them (sandboxes inside 25 GiB-hours of Containers memory) would read as a stale price when nothing changed; what was billed is still the cost in the margins, and a leak. A product with usage on a meter that has no list price (Artifacts, Cloudflare for SaaS, R2 storage until priced) is not checked, since its usage cannot be priced like for like; add the price to `cycle::LIST_PRICES`. `cost_drift` is replaced every run and an alert whose drift is gone is resolved on the same run, so an alert raised by the old comparison (what was billed against the price book) closes on the next | A price is stale: check the proposals, and `cycle::LIST_PRICES` against Cloudflare's pricing page. |
303303 | Cost, on `models` | What AI Gateway priced g1t's own provider traffic at over the 7 days, against the ledger's model cost for the same days (billed to g1t: comped, free and trial use included, a workspace's own provider not) plus the model cost testing resets kept for those days (`reset_costs`), more than the `ai_gateway_requests` mapping's `drift_percent` (10%) apart, with at least `min_daily_cost`. A ledger with none of the gateway's cost is drift too, and so is a gateway that priced nothing against a ledger with at least `min_daily_cost` of model cost (no percentage): that is not agreement. Its detail says why as far as the run could tell: requests logged with no price (add the models' prices), a token Cloudflare refuses for the gateway (give it AI Gateway Read, or fix `AI_GATEWAY_ID`), a gateway the token sees with no requests (calls went around it), or a read that failed | The gateway higher: model calls g1t paid for and charged no one: runs not settled yet (they catch up within the hour), runs with no session, a run started without a billing ticket, or something else on g1t's gateway. The ledger higher: runs that reached a provider without the gateway. The detail adds why the gateway's own figure may be off: prompt-cache read and write tokens (the gateway prices them at its rates for cache tokens, which can lag the provider's; check against the provider's invoice), requests Cloudflare billed itself (unified billing: on Cloudflare's bill, not a provider's), and models with no price. A testing reset in the window is named in the detail: one that kept its cost says how much of the ledger's side it is; one from before resets kept their cost (the audit log has it, `reset_costs` does not) says the gateway's figure includes usage the ledger no longer has, so that part is not a leak, and the day it leaves the 7 days; while such a reset is in the window the `models` leak is not raised. Days are UTC by when a request ran (gateway) and when a charge was entered (ledger), so a run across midnight shifts a little between days; the 7-day sum absorbs it. |
304304 | Unpriced | Over the 7 days, a model in AI Gateway's analytics with tokens and $0 cost, or runs settled with `runs.gateway_note` (the gateway could not price all of a run) | The gateway has no price for a model g1t runs: add it in the gateway (custom cost) or route away from it. Until then those runs are charged no less than the sandbox reported (Claude Code's own price table), never $0 silently. |
305305 | Leak | Cost of at least `min_daily_cost` and nothing charged for it (never for `platform`), or a meter in `unmapped` | Map the meter (below), or decide it is overhead (`platform`). |
315315 - Proposals come from the keeper (sandbox seconds, app requests and CPU)
316316 and the reconciler (mappings with `scale_to_own`: today git operations).
317317 For git operations: Cloudflare's rate per its own operation (the median
318− over charged days of cost ÷ quantity) × (Cloudflare's operations ÷ g1t's)
318+ over costed days of cost ÷ quantity) × (Cloudflare's operations ÷ g1t's)
319319 × 1,000. If Cloudflare counts three for each one g1t counts, the per-1,000
320320 price triples. At least 1,000 of g1t's operations are needed.
321+- A rate is always a price per unit before the included amounts, like the
322+ price book's: never what was billed over all of the quantity, which is
323+ net of them and reads as a cheaper unit. The reconciler takes a line's
324+ cost at `cycle::LIST_PRICES` where its meter has one (`margin::rate_line`),
325+ else Cloudflare's own cost (the median leaves out the day an included
326+ amount ran out). The keeper takes the bill's `ListCost` ÷ quantity
327+ (`keeper::billed_rate`), and the published rates where there is none.
321328 - Decision (`pricing::decide`): under 2% is noise; more than 4× either way
322329 is suspect and waits for staff; within `auto_apply_percent` (25%) it is
323330 applied on its own when `auto_apply` is on; anything else waits.
+110−11
158158 1. Plans: for each unit, the live commit; `git diff` from it to `HEAD`;
159159 whether the changed files touch the unit (its folder, the crates and
160160 packages it is built from, its inputs, lockfile changes that reach it).
161+ A Rust crate's `tests/`, `benches/` and `examples/` are not what it is
162+ built from, and neither is a source file compiled only for tests: one
163+ whose every `mod` declaration is under `#[cfg(test)]` (with or without
164+ `#[path]`), or inside a module that is. A change to a crate's tests
165+ alone deploys nothing (`scripts/deploy/stack.mjs`, `testOnlySource`).
161166 2. Refuses if uncommitted changes touch what would deploy (`--allow-dirty`).
162167 3. Applies every pending migration (`wrangler d1 migrations apply --remote`),
163168 in parallel. Any failure stops the deploy before code.
386391 Not yet seen on Cloudflare itself: watch the first runs' logs for the
387392 `Docker: started` line, and `dockerd.log` if it does not come.
388393
394+#### Sandbox errors
395+
396+Each sandbox is a Durable Object of `g1t-runner`, and Cloudflare counts
397+each of its invocations: the `run` call that starts it, `halt`,
398+`noteBlocked` and `flagAbuse`, and its alarms. The containers library keeps
399+an alarm going for as long as the container runs (each one waits up to
400+three minutes), and runs `onStop` from the alarm after the container
401+exits, so most of a sandbox's invocations are alarms.
402+
403+To see what the errors are, by namespace and status, and what was thrown:
404+
405+```sh
406+CLOUDFLARE_API_TOKEN=<token> node scripts/ops/runner-errors.mjs # the last 7 days
407+CLOUDFLARE_API_TOKEN=<token> node scripts/ops/runner-errors.mjs --days 30 --json
408+```
409+
410+The token needs Account Analytics: Read and Workers Observability: Read
411+(Workers Scripts: Read adds namespace names). The report puts each message
412+in a bucket and says whether it is expected:
413+
414+| Bucket | Expected | What it is |
415+| --- | --- | --- |
416+| `deploy_reset` | Yes | A runner deploy resets every sandbox's object, failing the alarm or call in flight. The container keeps running and the next alarm picks it up. |
417+| `no_capacity` | Yes | No container instance was free (`max_instances`). The work fails to start and says so. |
418+| `container_exited`, `caller_gone` | Yes | A container that stopped, or a caller that went away first. |
419+| `stop_not_reported` | No | `onStop` could not tell a service that the sandbox stopped after two tries. The five-minute sweep catches the work up. |
420+| `alarm_failed`, `not_started`, `storage`, `limits`, `other` | No | Read the message. |
421+
422+The runner logs these at error level, each with a fixed prefix you can
423+search for in Workers Logs:
424+
425+- `sandbox not started`
426+- `sandbox stop not reported`
427+- `sandbox alarm failed`
428+- `sandbox container error`
429+- `sandbox not destroyed`
430+
431+`onStop` never throws. A throw would fail the alarm, which Cloudflare
432+retries and counts as an error each time, running the whole stop again.
433+A sandbox's run inside its time cap is not stopped for inactivity. The
434+library's `sleepAfter` (100 minutes) would otherwise stop a run whose
435+guardrails allow longer, up to 240 minutes, because the runner never
436+fetches the container, so to the library every sandbox looks idle.
437+
389438 ## Build speed
390439
391440 Measured on the development machine (Windows, 32 cores, warm Cargo cache),
433482 | The runner binary, cold (builder container) | | 99 s |
434483 | A Rust CI job's build (events, search, repos; 4 vCPUs) | | 51 s cold, 13 s with the Cargo target restored (107 MB zstd entry) |
435484
485+- **sccache** (`scripts/sccache.sh`, pinned by version and sha256): a
486+ checkout gives every source a new mtime, so Cargo calls rustc again for
487+ every workspace crate a unit uses, however much of `target/` was
488+ restored. With `RUSTC_WRAPPER=sccache` and its GitHub Actions backend,
489+ each call that makes a library is looked up by its inputs in the
490+ repository's Actions cache (the same cache as `actions/cache`, scoped to
491+ `main`), and an unchanged crate comes back from it. Measured on the
492+ development machine on 2026-10-08, `repos` and `events` for wasm32,
493+ release, `-j 4`, the dependencies already built, every workspace source
494+ touched as a checkout does (sccache's local disk cache; the Actions
495+ cache adds a download per hit):
496+
497+ | | Time |
498+ | --- | --- |
499+ | Without sccache (before) | 44.3 s, 44.6 s |
500+ | sccache, empty cache (its first run, filling it) | 48.8 s |
501+ | sccache, filled (6 hits: `contracts`, `kit`, `rules`, `scan`, `secrets`, `blobstore`) | 24.2 s |
502+
503+ CI's build the same way (`cargo test --workspace --no-run`, debug,
504+ `-j 4`, `CARGO_INCREMENTAL=0`): 70.2 s and 74.6 s without sccache, 66.5 s
505+ filling it, 40.6 s filled (7 library hits; 23 test harnesses and
506+ cdylibs compiled).
507+
508+ What is left is each Worker's own crate: a `cdylib` is linked, and
509+ sccache does not cache what rustc links (binaries, cdylibs, proc
510+ macros, build scripts, test harnesses). That compile is real work
511+ anyway: a unit deploys because it, or a crate it uses, changed. The
512+ same holds for CI's `cargo test`: the workspace's libraries come back
513+ from the cache, the test harnesses are compiled. Every Rust job of
514+ Deploy and CI's Rust job end with sccache's hits and misses in the
515+ run's summary. If the cache cannot be reached, or the download fails
516+ its checksum, the job warns and builds as before. On g1t, sccache
517+ speaks the toolkit cache's older protocol (`GITHUB_SERVER_URL` is not
518+ github.com): a lookup, a reservation, one `PATCH` of the whole entry
519+ (`Content-Range: bytes 0-N/*`) and a commit per miss; a lookup and a
520+ blob `GET` per hit. Its storage check saves `sccache/.sccache_check`
521+ once, and takes the `409` on later runs as "already there".
522+- **Not restoring mtimes:** setting each file's mtime from git (so Cargo
523+ would trust the restored `target/`) was considered and left out. A
524+ restored `target/` may come from another commit (the cache's
525+ `restore-keys` take the nearest earlier entry, of any group), and a
526+ file changed by an older commit than that build would look unchanged:
527+ a stale crate deployed. sccache looks at what is compiled, not when.
436528 - **Only what changed** is the largest saving: a change to one service
437− deploys one service.
529+ deploys one service, and a change only to a crate's tests deploys
530+ nothing.
438531
439532 ## The workflow
440533
463556 be rebuilt alone (`image: true` in the matrix, also on `g1t-4core`, where
464557 it builds and pushes the image with the job's own Docker Engine). `fail-fast: false`, so one failed job does not cut
465558 another off mid-upload; the next stage then does not start.
466−- **Tests:** there is no CI workflow on g1t yet; `main` is kept passing by
467− the merge queue's checks. `check` runs the deploy tool's own tests. When a
468− CI workflow is added, make `migrate` and the stages wait for it (`workflow_run`, or a job in
469− this file).
559+- **Tests:** `.g1t/workflows/ci.yml` tests every pull request into `main`
560+ (and, on pushes to `main`, runs only its Rust job, to keep the caches
561+ pull requests restore from current). `check` runs the deploy tool's own
562+ tests. Deploy does not wait for CI.
470563 - **Machines:** Rust jobs and the runner's image run on `g1t-4core` (4 vCPUs,
471564 12 GiB, 20 GB), the others on the standard machine
472565 (`runs-on: ${{ (matrix.rust || matrix.image) && 'g1t-4core' || 'ubuntu-latest' }}`).
477570 downloaded tools, `~/.cargo/registry/cache`, and the Cargo target's
478571 release dependencies (`target/release` and
479572 `target/wasm32-unknown-unknown/release`, without `incremental` or
480− `.wasm`), keyed by the build group, `Cargo.lock` and `base.json`. The
481− workspace's own crates are compiled again on every run (a checkout's
482− sources are newer than any cache); the crates.io dependencies are not.
483− npm's cache is not kept: every job runs `npm ci` of only what its units
484− need (`deploy.mjs install`: Wrangler alone for Rust jobs).
573+ `.wasm`), keyed by the build group, `Cargo.lock` and `base.json`. Cargo
574+ calls rustc again for the workspace's own crates on every run (a
575+ checkout's sources are newer than any cache); **sccache** answers those
576+ calls from the repository's Actions cache when a crate's inputs did not
577+ change (see [Build speed](#build-speed)). npm's cache is not kept: every
578+ job runs `npm ci` of only what its units need (`deploy.mjs install`:
579+ Wrangler alone for Rust jobs).
485580 - **Conditions:** each stage runs with `!failure() && !cancelled()`, which
486581 on g1t (as on GitHub) is true when no job before it failed, however far
487582 back: a `migrate` job skipped for having nothing to apply does not stop
511606 miss. The image job adds `x86_64-unknown-linux-musl` (about 30 MB from
512607 `static.rust-lang.org`) and keeps its Cargo target in the cache. worker-build
513608 fetches wasm-bindgen and wasm-opt from GitHub releases and esbuild from
514−npm. All of those hosts are on the list every workflow job may reach.
609+npm. Each Rust job downloads sccache (about 10 MB) from its GitHub release
610+too (`scripts/sccache.sh`, which checks its sha256); it is not in the base
611+image, and putting it there means a pinned download in the base's
612+Dockerfile and a base rebuild (`build-base`). All of those hosts are on
613+the list every workflow job may reach.
515614
516615 ### Network
517616
+91−1
3939 versionFrom,
4040 } from "./cloudflare.mjs";
4141 import { changedNames, parseCargoLock, parseNpmLock, reaches } from "./lockfiles.mjs";
42−import { decide, planJson, pool } from "./plan.mjs";
42+import { decide, git, planJson, pool } from "./plan.mjs";
4343 import {
4444 ROOT,
4545 buildGroups,
5353 problems,
5454 resolveStack,
5555 resolvedStack,
56+ testOnlySource,
5657 touches,
5758 touchesBase,
5859 touchesImage,
328329 assert.ok(!ids(stack.units, ["packages/contracts/src/index.ts"]).includes("events"));
329330 });
330331
332+test("a crate's tests, benches and examples deploy nothing", () => {
333+ // 0cdd221 changed only this file, and the plan sent three Rust Workers.
334+ assert.deepEqual(ids(stack.units, ["crates/actions/tests/repository_workflows.rs"]), []);
335+ assert.ok(!touchesImage(unit("runner"), ["crates/actions/tests/repository_workflows.rs"]));
336+ assert.deepEqual(ids(stack.units, ["crates/kit/benches/wire.rs", "crates/scan/examples/scan.rs", "services/repos/tests/git.rs"]), []);
337+ // Their sources still do, and a folder merely named like one.
338+ assert.deepEqual(ids(stack.units, ["crates/actions/src/expr.rs"]), ["actions", "security", "runner"]);
339+ assert.ok(touchesImage(unit("runner"), ["crates/actions/src/expr.rs"]));
340+ assert.deepEqual(ids(stack.units, ["services/repos/src/tests/helpers.rs"]), ["repos"]);
341+ assert.deepEqual(ids(stack.units, ["crates/kit/testsuite/a.rs"]), ids(stack.units, ["crates/kit/src/lib.rs"]));
342+ // A unit that is not a crate keeps its whole folder.
343+ assert.deepEqual(unit("web").crateDirs, []);
344+ assert.deepEqual(unit("events").crateDirs, ["crates/contracts", "crates/kit", "services/events"]);
345+ assert.ok(unit("runner").crateDirs.includes("crates/runner"));
346+});
347+
348+/** Rust sources by path, as `testOnlySource` reads them. */
349+const sources = (files) => ({
350+ read: (path) => files[path] ?? "",
351+ list: (dir) => Object.keys(files).filter((path) => path.slice(0, path.lastIndexOf("/")) === dir),
352+});
353+
354+test("a Rust file compiled only for tests is found from what declares it", () => {
355+ const crate = ["crates/x"];
356+ const tree = sources({
357+ "crates/x/src/lib.rs": "pub mod api;\nmod store;\n#[cfg(test)]\nmod tests;\n#[cfg(test)] pub(crate) mod fixtures;\n#[cfg(any(test, feature = \"x\"))]\nmod both;\n",
358+ "crates/x/src/api.rs": "mod wire;\n#[cfg(test)]\n#[path = \"api_tests.rs\"]\nmod tests;\n",
359+ "crates/x/src/api/wire.rs": "",
360+ "crates/x/src/api_tests.rs": "use super::*;",
361+ "crates/x/src/store/mod.rs": "#[cfg(test)]\nmod memory;\n",
362+ "crates/x/src/store/memory.rs": "mod deep;",
363+ "crates/x/src/store/memory/deep.rs": "",
364+ "crates/x/src/tests.rs": "",
365+ "crates/x/src/fixtures/mod.rs": "",
366+ "crates/x/src/both.rs": "",
367+ "crates/x/src/bin/tool.rs": "",
368+ });
369+ const only = (file) => testOnlySource(file, crate, tree);
370+ for (const file of ["crates/x/src/tests.rs", "crates/x/src/fixtures/mod.rs", "crates/x/src/api_tests.rs", "crates/x/src/store/memory.rs", "crates/x/src/store/memory/deep.rs"]) {
371+ assert.ok(only(file), file);
372+ }
373+ // Built: roots, ordinary modules, a cfg that is not only test, a file
374+ // nothing declares (new, or deleted), and files outside src or the crate.
375+ for (const file of [
376+ "crates/x/src/lib.rs",
377+ "crates/x/src/api.rs",
378+ "crates/x/src/api/wire.rs",
379+ "crates/x/src/store/mod.rs",
380+ "crates/x/src/both.rs",
381+ "crates/x/src/bin/tool.rs",
382+ "crates/x/src/new.rs",
383+ "crates/x/build.rs",
384+ "crates/y/src/tests.rs",
385+ "crates/x/src/notes.md",
386+ ]) {
387+ assert.ok(!only(file), file);
388+ }
389+});
390+
391+test("this repository's test-only sources are read from git", () => {
392+ const head = git.head();
393+ const atHead = { read: (path) => git.show(head, path), list: (dir) => git.list(head, dir) };
394+ const only = (u, file) => testOnlySource(file, unit(u).crateDirs, atHead);
395+ assert.ok(only("api", "apps/api/src/responses.rs"));
396+ assert.ok(only("billing", "services/billing/src/catalogue_tests.rs"));
397+ assert.ok(only("billing", "services/billing/src/gateway_tests.rs"));
398+ assert.ok(!only("billing", "services/billing/src/catalogue.rs"));
399+ assert.ok(!only("api", "apps/api/src/toolkit.rs"));
400+});
401+
402+test("decide: a change only to tests deploys nothing", () => {
403+ const units = ["actions", "billing", "runner"].map(unit);
404+ const live = { actions: { sha: OLD }, billing: { sha: OLD }, runner: { sha: OLD } };
405+ const files = {
406+ [`${HEAD}:services/billing/src/gateway.rs`]: "#[cfg(test)]\n#[path = \"gateway_tests.rs\"]\nmod tests;\n",
407+ [`${HEAD}:services/billing/src/lib.rs`]: "mod gateway;\n",
408+ };
409+ const gitApi = {
410+ ...fakeGit(["crates/actions/tests/repository_workflows.rs", "services/billing/src/gateway_tests.rs"]),
411+ show: (sha, path) => files[`${sha}:${path}`] ?? "",
412+ list: (sha, dir) => Object.keys(files).map((k) => k.slice(sha.length + 1)).filter((p) => p.startsWith(`${dir}/`)).concat(dir === "services/billing/src" ? ["services/billing/src/gateway_tests.rs"] : []),
413+ };
414+ const decisions = decide(units, { live, head: HEAD, gitApi });
415+ assert.deepEqual(decisions.map((d) => [d.unit.id, d.deploy, d.image]), [["actions", false, false], ["billing", false, false], ["runner", false, false]]);
416+ // With a source that ships beside them, only that is listed.
417+ const mixed = decide([unit("billing")], { live, head: HEAD, gitApi: { ...gitApi, changed: () => ["services/billing/src/gateway_tests.rs", "services/billing/src/gateway.rs"] } });
418+ assert.deepEqual([mixed[0].deploy, mixed[0].files], [true, ["services/billing/src/gateway.rs"]]);
419+});
420+
331421 test("own folders, declared inputs and root files", () => {
332422 assert.deepEqual(ids(stack.units, ["services/pages/src/index.ts"]), ["pages"]);
333423 assert.deepEqual(ids(stack.units, ["apps/web/app/lib/roadmap.ts"]), ["og", "web"]);
+31−2
55 import { execFileSync } from "node:child_process";
66
77 import { changedNames, lockRoots, parseCargoLock, parseNpmLock, reaches } from "./lockfiles.mjs";
8−import { ROOT, byStage, buildGroups, codeStages, touches, touchesImage } from "./stack.mjs";
8+import { ROOT, byStage, buildGroups, codeStages, testOnlySource, touches, touchesImage } from "./stack.mjs";
99
1010 /** Git, read-only. */
1111 export const git = {
2121 return false;
2222 }
2323 },
24− /** Files changed between two commits. */
2524 /** A file's text at a commit, or "" if it is not there. */
2625 show: (sha, path) => {
2726 try {
3029 return "";
3130 }
3231 },
32+ /** The files directly in a folder at a commit (repository-relative). */
33+ list: (sha, dir) => {
34+ try {
35+ return run(["ls-tree", "--name-only", sha, "--", `${dir}/`]).split("\n").filter(Boolean);
36+ } catch {
37+ return [];
38+ }
39+ },
40+ /** Files changed between two commits. */
3341 changed: (from, to) => run(["diff", "--name-only", "--no-renames", from, to]).split("\n").filter(Boolean),
3442 /** Whether `older` is in `newer`'s history (and not the same commit). */
3543 isAncestor: (older, newer) => {
8290 }
8391 return locks.get(key);
8492 };
93+ // A Rust source compiled only for tests (`#[cfg(test)] mod tests;`) is
94+ // not in what deploys. Read at the commit being deployed, each file once.
95+ const texts = new Map();
96+ const listings = new Map();
97+ const atHead = {
98+ read: (path) => {
99+ if (!texts.has(path)) texts.set(path, gitApi.show(head, path));
100+ return texts.get(path);
101+ },
102+ list: (dir) => {
103+ if (!listings.has(dir)) listings.set(dir, gitApi.list?.(head, dir) ?? []);
104+ return listings.get(dir);
105+ },
106+ };
107+ const testOnly = new Map();
108+ const onlyForTests = (unit, file) => {
109+ if (!file.endsWith(".rs") || !unit.crateDirs?.some((dir) => file.startsWith(`${dir}/src/`))) return false;
110+ if (!testOnly.has(file)) testOnly.set(file, testOnlySource(file, unit.crateDirs, atHead));
111+ return testOnly.get(file);
112+ };
85113 const relevant = (unit, sha, files) =>
86114 files.filter((file) => {
115+ if (onlyForTests(unit, file)) return false;
87116 if (file !== "Cargo.lock" && file !== "package-lock.json") return true;
88117 const { after, names } = lockChange(sha, file);
89118 const roots = lockRoots(unit)[file === "Cargo.lock" ? "cargo" : "npm"];
+89−1
173173 unit.crate = crate;
174174 unit.pkg = pkg;
175175 unit.dependsOn = [...dirs].sort();
176+ // The Rust crates it is built from, whose tests, benches and examples
177+ // are not.
178+ unit.crateDirs = crate ? [unit.path, ...closure(cargo, crate).map((name) => cargo.get(name).dir)].sort() : [];
176179 unit.inputs = [...new Set([...(unit.inputs ?? []), ...globalInputs(unit.kind)])].sort();
177180 if (unit.image) {
178181 const name = unit.image.crate;
189192 };
190193 for (const dir of unit.image.dirs) if (dir !== unit.path) unit.dependsOn.push(dir);
191194 unit.dependsOn = [...new Set(unit.dependsOn)].sort();
195+ unit.crateDirs = [...new Set([...unit.crateDirs, ...unit.image.dirs])].sort();
192196 unit.inputs = [...new Set([...unit.inputs, "Cargo.toml", "Cargo.lock", "scripts/build-runner.mjs"])].sort();
193197 }
194198 }
202206
203207 const under = (file, dir) => file === dir || file.startsWith(`${dir}/`);
204208
209+/** A Rust crate's folders that its library and binaries are never built from. */
210+export const CRATE_TEST_DIRS = ["tests", "benches", "examples"];
211+
212+/**
213+ * Whether `file` is in the tests, benches or examples of one of
214+ * `crateDirs`: Cargo builds those only for `cargo test`, `cargo bench` and
215+ * `--example`, never into what deploys.
216+ */
217+export function inCrateTests(file, crateDirs = []) {
218+ return crateDirs.some((dir) => CRATE_TEST_DIRS.some((sub) => under(file, `${dir}/${sub}`)));
219+}
220+
221+const dirOf = (path) => path.slice(0, Math.max(0, path.lastIndexOf("/")));
222+const baseOf = (path) => path.slice(path.lastIndexOf("/") + 1);
223+
224+/**
225+ * The `mod name;` declarations in Rust source `text` that `match` (by name,
226+ * or by a `#[path]`), each with whether it is under `#[cfg(test)]`.
227+ * Only outline modules (`mod x;`); the attributes are those directly above.
228+ */
229+function declarations(text, match) {
230+ const found = [];
231+ const pattern = /((?:#\[[^\]]*\]\s*)*)(?:pub(?:\([^)]*\))?\s+)?mod\s+(r#)?(\w+)\s*;/g;
232+ for (const [, attrs, , name] of text.matchAll(pattern)) {
233+ const path = /#\[\s*path\s*=\s*"([^"]+)"\s*\]/.exec(attrs)?.[1] ?? null;
234+ if (!match(name, path)) continue;
235+ found.push(/#\[\s*cfg\s*\(\s*test\s*\)\s*\]/.test(attrs));
236+ }
237+ return found;
238+}
239+
240+/**
241+ * Whether Rust source `file`, in the `src/` of one of `crateDirs`, is
242+ * compiled only for tests: every module declaration of it found is under
243+ * `#[cfg(test)]`, or is in a file that is itself only for tests. A file
244+ * nothing is found to declare (a new crate root, a deleted file, a module
245+ * declared some way this does not read) counts as built, so a change to it
246+ * deploys.
247+ *
248+ * read(path): a file's text at the commit being deployed, "" if absent
249+ * list(dir): the files directly in a folder at that commit
250+ */
251+export function testOnlySource(file, crateDirs, { read, list = () => [] }, depth = 0) {
252+ if (!file.endsWith(".rs") || depth > 16) return false;
253+ const crateDir = crateDirs.find((dir) => file.startsWith(`${dir}/src/`));
254+ if (!crateDir) return false;
255+ const src = `${crateDir}/src`;
256+ const dir = dirOf(file);
257+ const base = baseOf(file).slice(0, -".rs".length);
258+ // Crate roots and binaries are not declared by anything.
259+ if (dir === src && ["lib", "main"].includes(base)) return false;
260+ if (dir === `${src}/bin` || (base === "main" && dirOf(dir) === `${src}/bin`)) return false;
261+ // Where `mod name;` for this file would be: `a/name.rs` and `a/name/mod.rs`
262+ // are declared by `a.rs` or `a/mod.rs` (by `lib.rs` or `main.rs` in src).
263+ const name = base === "mod" ? baseOf(dir) : base;
264+ const parentDir = base === "mod" ? dirOf(dir) : dir;
265+ const parents =
266+ parentDir === src ? [`${src}/lib.rs`, `${src}/main.rs`] : [`${parentDir}/mod.rs`, `${dirOf(parentDir)}/${baseOf(parentDir)}.rs`];
267+ // One declaration that is built is enough to stop: most changed files
268+ // are ordinary modules, settled by reading their parent.
269+ let declared = false;
270+ const testOnly = (declarer, isTest) => {
271+ declared = true;
272+ return isTest || testOnlySource(declarer, crateDirs, { read, list }, depth + 1);
273+ };
274+ for (const parent of parents) {
275+ for (const isTest of declarations(read(parent), (found, path) => found === name && path === null)) {
276+ if (!testOnly(parent, isTest)) return false;
277+ }
278+ }
279+ // `#[path = "x.rs"] mod y;` in a file beside it names it directly.
280+ const fileName = baseOf(file);
281+ for (const sibling of list(dir).filter((path) => path.endsWith(".rs") && path !== file)) {
282+ for (const isTest of declarations(read(sibling), (_, path) => path === fileName || path === `./${fileName}`)) {
283+ if (!testOnly(sibling, isTest)) return false;
284+ }
285+ }
286+ return declared;
287+}
288+
205289 /**
206290 * Why a set of changed files (repository-relative, `/`-separated) touches
207291 * a unit: the first file that does, and through what. Null if none does.
292+ * Its crates' tests, benches and examples do not.
208293 */
209294 export function touches(unit, files) {
295+ files = files.filter((file) => !inCrateTests(file, unit.crateDirs));
210296 for (const file of files) {
211297 if (under(file, unit.path)) return { file, via: "its own folder" };
212298 }
221307 /** Whether changed files touch a unit's Containers image. */
222308 export function touchesImage(unit, files) {
223309 if (!unit.image) return false;
224− return files.some((file) => unit.image.files.includes(file) || unit.image.dirs.some((dir) => under(file, dir)));
310+ return files.some(
311+ (file) => unit.image.files.includes(file) || (unit.image.dirs.some((dir) => under(file, dir)) && !inCrateTests(file, unit.image.dirs)),
312+ );
225313 }
226314
227315 /** Whether changed files touch the folder a unit's base image is built from. */
+491−0
1+#!/usr/bin/env node
2+// What are the runner's Durable Object errors? The sandboxes that run agent
3+// attempts, workflow jobs and merge-queue builds are Durable Objects of the
4+// g1t-runner Worker (services/runner: AttemptSandbox, Sandbox2Core,
5+// Sandbox4Core). This asks Cloudflare, over the last N days:
6+//
7+// 1. GraphQL Analytics (durableObjectsInvocationsAdaptiveGroups): requests
8+// and errors by namespace and status, and by day and status.
9+// 2. Workers Observability (the telemetry query API): the runner's failed
10+// invocations by class, event type (alarm, rpc, fetch) and outcome; the
11+// exceptions they threw, by message; the runner's own error-level logs
12+// (`sandbox stop not reported`, `sandbox alarm failed`, `sandbox not
13+// started`, `sandbox container error`), by message; and a few recent
14+// failed invocations in full.
15+//
16+// Each message is put in a bucket (`classify`): a deploy resetting the
17+// object, no container free, a container that exited, and so on, so what is
18+// expected and what is a bug can be told apart at a glance.
19+//
20+// Read-only: GraphQL queries, one Durable Objects listing for names, and
21+// telemetry queries. Never prints the token.
22+//
23+// CLOUDFLARE_API_TOKEN=<token> node scripts/ops/runner-errors.mjs [--days 7] [--json]
24+//
25+// The token needs Account Analytics: Read (GraphQL) and Workers Observability:
26+// Read (logs); Workers Scripts: Read adds namespace names. A part the token
27+// cannot read is a note in the report, never the end of it.
28+
29+import { ACCOUNT_ID, cloudflareAuth } from "../deploy/cloudflare.mjs";
30+
31+const API = "https://api.cloudflare.com/client/v4";
32+
33+const HELP = `node scripts/ops/runner-errors.mjs [--days N] [--script NAME] [--samples N] [--json] [--keys] [--account <id>]
34+
35+The runner's Durable Object errors: requests and errors by namespace and
36+status (GraphQL Analytics), and what failed and why (Workers Observability).
37+
38+ --days N how far back, in days (default 7, at most 31)
39+ --script NAME the Worker (default g1t-runner)
40+ --samples N recent failed invocations shown in full (default 10)
41+ --json one JSON object, snake_case keys
42+ --keys also list the telemetry keys the runner's events have
43+ --account <id> the Cloudflare account (default CLOUDFLARE_ACCOUNT_ID, else g1t's)
44+ --help this text
45+
46+Needs CLOUDFLARE_API_TOKEN (Account Analytics: Read, Workers Observability:
47+Read), or CLOUDFLARE_API_KEY with CLOUDFLARE_EMAIL. Exits 1 when nothing
48+could be read.`;
49+
50+/** The command line. */
51+export function parseArgs(argv, env = process.env) {
52+ const option = (name) => {
53+ const at = argv.indexOf(name);
54+ return at >= 0 && argv[at + 1] && !argv[at + 1].startsWith("--") ? argv[at + 1] : null;
55+ };
56+ const days = Number(option("--days") ?? NaN);
57+ const samples = Number(option("--samples") ?? NaN);
58+ return {
59+ help: argv.includes("--help") || argv.includes("-h"),
60+ json: argv.includes("--json"),
61+ keys: argv.includes("--keys"),
62+ days: Number.isFinite(days) && days > 0 ? Math.min(31, days) : 7,
63+ samples: Number.isFinite(samples) && samples >= 0 ? Math.min(100, Math.floor(samples)) : 10,
64+ script: option("--script") || "g1t-runner",
65+ account: option("--account") || env.CLOUDFLARE_ACCOUNT_ID || ACCOUNT_ID,
66+ };
67+}
68+
69+/** The window: the last `days` days up to `now`. */
70+export function windowOf(now, days) {
71+ return { start: new Date(now.getTime() - days * 86_400_000).toISOString(), end: now.toISOString() };
72+}
73+
74+/**
75+ * The GraphQL queries, each with variants tried in order when Cloudflare
76+ * refuses a field. Rows come back under `rows`.
77+ */
78+export const DO_QUERIES = [
79+ {
80+ key: "by_namespace",
81+ label: "Durable Object requests by namespace and status",
82+ variants: [
83+ { sum: ["requests", "errors", "wallTime"], dims: ["namespaceId", "status"] },
84+ { sum: ["requests", "errors"], dims: ["namespaceId", "status"] },
85+ { sum: ["requests"], dims: ["namespaceId", "status"] },
86+ { sum: ["requests", "errors"], dims: ["namespaceId"] },
87+ ],
88+ },
89+ {
90+ key: "by_day",
91+ label: "Durable Object requests by day and status",
92+ variants: [
93+ { sum: ["requests", "errors"], dims: ["date", "status"] },
94+ { sum: ["requests"], dims: ["date", "status"] },
95+ ],
96+ },
97+];
98+
99+/** One DO invocations query for one script, for one variant. */
100+export function doQuery(variant) {
101+ return `query RunnerErrors($accountTag: String!, $start: Time!, $end: Time!, $script: String!) {
102+ viewer {
103+ accounts(filter: { accountTag: $accountTag }) {
104+ rows: durableObjectsInvocationsAdaptiveGroups(limit: 10000, filter: { datetime_geq: $start, datetime_leq: $end, scriptName: $script }) {
105+ sum { ${variant.sum.join(" ")} }
106+ dimensions { ${variant.dims.join(" ")} }
107+ }
108+ }
109+ }
110+}`;
111+}
112+
113+const snake = (name) => name.replace(/[A-Z]/g, (letter) => `_${letter.toLowerCase()}`);
114+
115+/**
116+ * A GraphQL answer as rows: one per dimension combination, the first
117+ * dimension named through `names` (namespace ids to names), with each sum
118+ * and the error rate, largest first by requests. Throws with Cloudflare's
119+ * message when the answer has errors.
120+ */
121+export function parseDoGroups(body, variant, names = {}) {
122+ if (body?.errors?.length) throw new Error(body.errors.map((error) => error.message).join("; ").slice(0, 400));
123+ const groups = body?.data?.viewer?.accounts?.[0]?.rows;
124+ if (!Array.isArray(groups)) throw new Error("no rows in the answer");
125+ const rows = groups.map((group) => {
126+ const row = {};
127+ variant.dims.forEach((dim, at) => {
128+ const value = group.dimensions?.[dim];
129+ const text = value == null || value === "" ? "(none)" : String(value);
130+ row[snake(dim)] = at === 0 && names[text] ? names[text] : text;
131+ });
132+ for (const field of variant.sum) row[snake(field)] = Number(group.sum?.[field] ?? 0);
133+ return row;
134+ });
135+ const sortKey = variant.dims[0] === "date" ? null : "requests";
136+ rows.sort((a, b) => (sortKey ? b.requests - a.requests : 0) || String(a[snake(variant.dims[0])]).localeCompare(String(b[snake(variant.dims[0])])));
137+ const totals = Object.fromEntries(variant.sum.map((field) => [snake(field), rows.reduce((sum, row) => sum + row[snake(field)], 0)]));
138+ return { dims: variant.dims.map(snake), metrics: variant.sum.map(snake), rows, totals };
139+}
140+
141+/**
142+ * Requests and errors per status, summed over namespaces: which statuses
143+ * the errors are (`scriptThrewException`, `internalError`,
144+ * `clientDisconnected`, `exceededResources`, ...).
145+ */
146+export function byStatus(parsed) {
147+ if (!parsed.dims.includes("status")) return [];
148+ const out = new Map();
149+ for (const row of parsed.rows) {
150+ const entry = out.get(row.status) ?? { status: row.status, requests: 0, errors: 0 };
151+ entry.requests += row.requests ?? 0;
152+ entry.errors += row.errors ?? 0;
153+ out.set(row.status, entry);
154+ }
155+ return [...out.values()].sort((a, b) => b.requests - a.requests);
156+}
157+
158+/** The telemetry keys the questions below group by and filter on. */
159+export const KEYS = {
160+ outcome: "$workers.outcome",
161+ eventType: "$workers.eventType",
162+ entrypoint: "$workers.entrypoint",
163+ error: "$metadata.error",
164+ message: "$metadata.message",
165+ level: "$metadata.level",
166+};
167+
168+/** Which key names the Worker: tried in order, the next when one finds nothing. */
169+export const SERVICE_KEYS = ["$workers.scriptName", "$metadata.service"];
170+
171+const filter = (key, operation, value) => (value === undefined ? { key, operation, type: "string" } : { key, operation, type: "string", value });
172+
173+/**
174+ * The telemetry questions: each a body for `POST .../workers/observability/
175+ * telemetry/query`, for the Worker named under `serviceKey`.
176+ */
177+export function telemetryQuestions({ script, from, to, samples, serviceKey = SERVICE_KEYS[0] }) {
178+ const service = filter(serviceKey, "eq", script);
179+ const failed = filter(KEYS.outcome, "neq", "ok");
180+ const base = { timeframe: { from, to }, limit: 100 };
181+ const count = [{ operator: "count", alias: "events" }];
182+ const group = (...keys) => keys.map((value) => ({ type: "string", value }));
183+ return [
184+ {
185+ key: "failed_invocations",
186+ label: "Failed invocations by class, event and outcome",
187+ body: {
188+ ...base,
189+ queryId: "g1t-runner-failed-invocations",
190+ view: "calculations",
191+ parameters: { datasets: ["cloudflare-workers"], filters: [service, failed], calculations: count, groupBys: group(KEYS.entrypoint, KEYS.eventType, KEYS.outcome) },
192+ },
193+ },
194+ {
195+ key: "exceptions",
196+ label: "Exceptions by message",
197+ body: {
198+ ...base,
199+ queryId: "g1t-runner-exceptions",
200+ view: "calculations",
201+ parameters: { datasets: ["cloudflare-workers"], filters: [service, filter(KEYS.error, "exists")], calculations: count, groupBys: group(KEYS.error) },
202+ },
203+ },
204+ {
205+ key: "error_logs",
206+ label: "The runner's error-level logs by message",
207+ body: {
208+ ...base,
209+ queryId: "g1t-runner-error-logs",
210+ view: "calculations",
211+ parameters: { datasets: ["cloudflare-workers"], filters: [service, filter(KEYS.level, "eq", "error")], calculations: count, groupBys: group(KEYS.message) },
212+ },
213+ },
214+ {
215+ key: "samples",
216+ label: "Recent failed invocations",
217+ body: {
218+ ...base,
219+ limit: Math.max(1, samples),
220+ queryId: "g1t-runner-failed-samples",
221+ view: "events",
222+ parameters: { datasets: ["cloudflare-workers"], filters: [service, failed] },
223+ },
224+ },
225+ ];
226+}
227+
228+/**
229+ * A telemetry calculations answer as rows: each group's key values and its
230+ * count, largest first. Throws with Cloudflare's message when it failed.
231+ */
232+export function parseCalculations(body) {
233+ if (body?.success === false || body?.errors?.length) throw new Error(messagesOf(body));
234+ const calculation = body?.result?.calculations?.[0];
235+ if (!calculation) throw new Error("no calculations in the answer");
236+ const rows = (calculation.aggregates ?? []).map((aggregate) => {
237+ const row = {};
238+ for (const group of aggregate.groups ?? []) row[group.key] = group.value == null || group.value === "" ? "(none)" : String(group.value);
239+ row.count = Number(aggregate.value ?? aggregate.count ?? 0);
240+ return row;
241+ });
242+ return rows.sort((a, b) => b.count - a.count);
243+}
244+
245+/** A telemetry events answer as plain records: when, which class, event, outcome, and what went wrong. */
246+export function parseEvents(body) {
247+ if (body?.success === false || body?.errors?.length) throw new Error(messagesOf(body));
248+ const events = body?.result?.events?.events ?? body?.result?.events ?? [];
249+ if (!Array.isArray(events)) throw new Error("no events in the answer");
250+ return events.map((event) => {
251+ const workers = event.$workers ?? {};
252+ const metadata = event.$metadata ?? {};
253+ const at = event.timestamp ?? metadata.startTime ?? workers.timestamp;
254+ return {
255+ at: typeof at === "number" ? new Date(at).toISOString() : (at ?? null),
256+ entrypoint: workers.entrypoint ?? null,
257+ event_type: workers.eventType ?? null,
258+ outcome: workers.outcome ?? null,
259+ object: typeof workers.durableObjectId === "string" ? workers.durableObjectId.slice(0, 12) : null,
260+ error: metadata.error ?? null,
261+ message: metadata.message ?? (typeof event.source === "string" ? event.source : (event.source?.message ?? null)),
262+ };
263+ });
264+}
265+
266+function messagesOf(body) {
267+ const errors = body?.errors ?? [];
268+ const text = errors.map((error) => error.message ?? JSON.stringify(error)).join("; ");
269+ return (text || `request failed${body?.status ? ` with ${body.status}` : ""}`).slice(0, 400);
270+}
271+
272+/**
273+ * What a failure most likely is, from its message or outcome: a bucket and
274+ * whether it is expected. Order matters: the first match wins.
275+ */
276+export const BUCKETS = [
277+ { bucket: "deploy_reset", expected: true, why: "a runner deploy reset the object mid-invocation", test: /code (was|has been) updated|reset because its code|new version of the (script|worker)|durable object reset/i },
278+ { bucket: "stop_not_reported", expected: false, why: "onStop could not tell a service the sandbox stopped (now logged, not thrown)", test: /sandbox stop not reported/i },
279+ { bucket: "alarm_failed", expected: false, why: "the sandbox's alarm threw (retried by Cloudflare)", test: /sandbox alarm failed/i },
280+ { bucket: "no_capacity", expected: true, why: "no container instance free (max_instances, or provisioning)", test: /no container instance|max(imum)? concurrent instance|too many containers per second/i },
281+ { bucket: "not_started", expected: false, why: "a sandbox could not start (guardrails unreadable, or the container would not start)", test: /sandbox not started|could not read this project's guardrails|did not start after|failed to start container/i },
282+ { bucket: "container_exited", expected: true, why: "the container exited or was stopped (a finished run, a time cap, a stop)", test: /container exited|runtime signalled|exited before we could determine|crashed while checking for ports|exit code/i },
283+ { bucket: "connection_lost", expected: false, why: "the connection to the container was lost", test: /network connection lost|disconnected/i },
284+ { bucket: "storage", expected: false, why: "Durable Object storage failed or was overloaded", test: /storage|sqlite|overloaded/i },
285+ { bucket: "limits", expected: false, why: "a CPU, memory or subrequest limit", test: /exceeded|too many subrequests|memory limit/i },
286+ { bucket: "container_error", expected: false, why: "the containers library reported an error", test: /sandbox container error|container error/i },
287+];
288+
289+/** The bucket of one failure's message (or, failing that, its outcome). */
290+export function classify(message, outcome = null) {
291+ const text = String(message ?? "");
292+ for (const bucket of BUCKETS) if (text && bucket.test.test(text)) return { bucket: bucket.bucket, expected: bucket.expected, why: bucket.why };
293+ if (/canceled|cancelled|clientdisconnected|responsestreamdisconnected/i.test(String(outcome ?? ""))) {
294+ return { bucket: "caller_gone", expected: true, why: "the caller went away before the object answered" };
295+ }
296+ return { bucket: "other", expected: false, why: "not recognised: read the message" };
297+}
298+
299+/** Rows of messages with counts, summed into buckets, largest first. */
300+export function bucketsOf(rows, messageKey) {
301+ const out = new Map();
302+ for (const row of rows) {
303+ const found = classify(row[messageKey], row[KEYS.outcome]);
304+ const entry = out.get(found.bucket) ?? { ...found, count: 0 };
305+ entry.count += row.count;
306+ out.set(found.bucket, entry);
307+ }
308+ return [...out.values()].sort((a, b) => b.count - a.count);
309+}
310+
311+const number = (value) => (Number.isInteger(value) ? value.toLocaleString("en-US") : value.toLocaleString("en-US", { maximumFractionDigits: 2 }));
312+const percent = (part, whole) => (whole > 0 ? `${((100 * part) / whole).toFixed(1)}%` : "-");
313+
314+function table(rows, columns) {
315+ if (!rows.length) return [" (nothing in this window)"];
316+ const cells = [columns.map((c) => c.title), ...rows.map((row) => columns.map((c) => c.value(row)))];
317+ const widths = columns.map((_, at) => Math.max(...cells.map((row) => String(row[at]).length)));
318+ return cells.map((row) => " " + row.map((cell, at) => (columns[at].right ? String(cell).padStart(widths[at]) : String(cell).padEnd(widths[at]))).join(" "));
319+}
320+
321+/** The report as plain text. */
322+export function format(report, top = 25) {
323+ const lines = [`Durable Object errors of ${report.script} on account ${report.account}, ${report.start} to ${report.end}`];
324+ for (const query of report.analytics) {
325+ lines.push("", query.label);
326+ if (query.error) {
327+ lines.push(` could not read: ${query.error}`);
328+ continue;
329+ }
330+ const columns = [
331+ ...query.dims.map((dim) => ({ title: dim, value: (row) => row[dim] })),
332+ ...query.metrics.map((metric) => ({ title: metric, right: true, value: (row) => number(row[metric]) })),
333+ ];
334+ if (query.metrics.includes("errors")) columns.push({ title: "error_rate", right: true, value: (row) => percent(row.errors, row.requests) });
335+ lines.push(...table(query.rows.slice(0, top), columns));
336+ lines.push(` total: ${Object.entries(query.totals).map(([metric, value]) => `${metric} ${number(value)}`).join(", ")}`);
337+ if (query.by_status?.length) {
338+ lines.push(" by status:");
339+ lines.push(...table(query.by_status, [
340+ { title: "status", value: (row) => row.status },
341+ { title: "requests", right: true, value: (row) => number(row.requests) },
342+ { title: "errors", right: true, value: (row) => number(row.errors) },
343+ ]).map((line) => ` ${line}`));
344+ }
345+ }
346+ for (const question of report.telemetry) {
347+ lines.push("", question.label);
348+ if (question.error) {
349+ lines.push(` could not read: ${question.error}`);
350+ continue;
351+ }
352+ if (question.key === "samples") {
353+ if (!question.rows.length) lines.push(" (none)");
354+ for (const row of question.rows) {
355+ lines.push(` ${row.at ?? "?"} ${row.entrypoint ?? "?"} ${row.event_type ?? "?"} ${row.outcome ?? "?"}${row.object ? ` ${row.object}` : ""}`);
356+ if (row.error || row.message) lines.push(` ${String(row.error ?? row.message).slice(0, 300)}`);
357+ }
358+ continue;
359+ }
360+ const keys = Object.keys(question.rows[0] ?? {}).filter((key) => key !== "count");
361+ lines.push(...table(question.rows.slice(0, top), [
362+ ...keys.map((key) => ({ title: key, value: (row) => String(row[key] ?? "").slice(0, 120) })),
363+ { title: "count", right: true, value: (row) => number(row.count) },
364+ ]));
365+ if (question.buckets?.length) {
366+ lines.push(" most likely:");
367+ for (const bucket of question.buckets) lines.push(` ${String(number(bucket.count)).padStart(6)} ${bucket.bucket}${bucket.expected ? " (expected)" : ""}: ${bucket.why}`);
368+ }
369+ }
370+ if (report.keys) lines.push("", "Telemetry keys:", ...report.keys.map((key) => ` ${key}`));
371+ if (report.notes.length) lines.push("", "Notes:", ...report.notes.map((note) => ` ${note}`));
372+ return lines.join("\n");
373+}
374+
375+async function post(auth, path, body) {
376+ const response = await fetch(`${API}${path}`, {
377+ method: "POST",
378+ headers: { ...auth, "content-type": "application/json", "user-agent": "g1t-ops" },
379+ body: JSON.stringify(body),
380+ });
381+ const answer = await response.json().catch(() => ({ success: false, errors: [{ message: `HTTP ${response.status}, not JSON` }] }));
382+ if (!response.ok && !answer.errors?.length) answer.errors = [{ message: `HTTP ${response.status}` }];
383+ return answer;
384+}
385+
386+async function durableNames(auth, account) {
387+ const out = {};
388+ try {
389+ for (let page = 1; page <= 10; page++) {
390+ const response = await fetch(`${API}/accounts/${account}/workers/durable_objects/namespaces?per_page=100&page=${page}`, { headers: { ...auth, "user-agent": "g1t-ops" } });
391+ const body = await response.json();
392+ if (!response.ok || !Array.isArray(body.result)) break;
393+ for (const item of body.result) if (item.id) out[item.id] = item.name ?? item.id;
394+ if (body.result.length < 100) break;
395+ }
396+ } catch {
397+ // Ids stand in for names.
398+ }
399+ return out;
400+}
401+
402+async function analytics(auth, options, window, names) {
403+ return Promise.all(
404+ DO_QUERIES.map(async (query) => {
405+ let error = null;
406+ for (const variant of query.variants) {
407+ try {
408+ const body = await post(auth, "/graphql", { query: doQuery(variant), variables: { accountTag: options.account, start: window.start, end: window.end, script: options.script } });
409+ const parsed = parseDoGroups(body, variant, names);
410+ return { key: query.key, label: query.label, ...parsed, by_status: query.key === "by_namespace" ? byStatus(parsed) : undefined };
411+ } catch (thrown) {
412+ error ??= String(thrown.message ?? thrown);
413+ }
414+ }
415+ return { key: query.key, label: query.label, error };
416+ }),
417+ );
418+}
419+
420+async function telemetry(auth, options, window) {
421+ const path = `/accounts/${options.account}/workers/observability/telemetry/query`;
422+ const from = Date.parse(window.start);
423+ const to = Date.parse(window.end);
424+ let results = null;
425+ for (const serviceKey of SERVICE_KEYS) {
426+ results = await Promise.all(
427+ telemetryQuestions({ script: options.script, from, to, samples: options.samples, serviceKey }).map(async (question) => {
428+ try {
429+ const body = await post(auth, path, question.body);
430+ if (question.key === "samples") return { key: question.key, label: question.label, rows: parseEvents(body) };
431+ const rows = parseCalculations(body);
432+ const messageKey = question.key === "exceptions" ? KEYS.error : question.key === "error_logs" ? KEYS.message : null;
433+ return { key: question.key, label: question.label, rows, buckets: messageKey ? bucketsOf(rows, messageKey) : undefined };
434+ } catch (error) {
435+ return { key: question.key, label: question.label, error: String(error.message ?? error) };
436+ }
437+ }),
438+ );
439+ // Another key names the Worker when this one found nothing at all.
440+ if (results.some((result) => result.rows?.length)) return { serviceKey, results };
441+ }
442+ return { serviceKey: SERVICE_KEYS.at(-1), results };
443+}
444+
445+async function telemetryKeys(auth, options, window) {
446+ const body = await post(auth, `/accounts/${options.account}/workers/observability/telemetry/keys`, {
447+ timeframe: { from: Date.parse(window.start), to: Date.parse(window.end) },
448+ datasets: ["cloudflare-workers"],
449+ filters: [filter(SERVICE_KEYS[0], "eq", options.script)],
450+ limit: 500,
451+ });
452+ if (body?.success === false || body?.errors?.length) throw new Error(messagesOf(body));
453+ return (body.result ?? []).map((key) => (typeof key === "string" ? key : `${key.key} (${key.type})`)).sort();
454+}
455+
456+async function main() {
457+ const options = parseArgs(process.argv.slice(2));
458+ if (options.help) {
459+ console.log(HELP);
460+ return 0;
461+ }
462+ const auth = cloudflareAuth();
463+ if (!auth) {
464+ console.error(`Set CLOUDFLARE_API_TOKEN to a token with Account Analytics: Read and Workers Observability: Read on account ${options.account}.`);
465+ return 2;
466+ }
467+ const window = windowOf(new Date(), options.days);
468+ const names = await durableNames(auth, options.account);
469+ const [graph, logs, keys] = await Promise.all([
470+ analytics(auth, options, window, names),
471+ telemetry(auth, options, window),
472+ options.keys ? telemetryKeys(auth, options, window).catch((error) => [`could not list keys: ${error.message}`]) : Promise.resolve(null),
473+ ]);
474+ const notes = [];
475+ if (!Object.keys(names).length) notes.push("Namespace ids are not named: the token cannot read the Durable Objects listing (Workers Scripts: Read).");
476+ notes.push(`Telemetry is filtered on ${logs.serviceKey}.`);
477+ const report = { account: options.account, script: options.script, start: window.start, end: window.end, analytics: graph, telemetry: logs.results, keys, notes };
478+ console.log(options.json ? JSON.stringify(report, null, 2) : format(report));
479+ const read = [...graph, ...logs.results].some((part) => !part.error);
480+ return read ? 0 : 1;
481+}
482+
483+if (process.argv[1]?.replaceAll("\\", "/").endsWith("scripts/ops/runner-errors.mjs")) {
484+ main().then(
485+ (code) => process.exit(code),
486+ (error) => {
487+ console.error(`runner-errors: ${error.message}`);
488+ process.exit(1);
489+ },
490+ );
491+}
+199−0
1+// The runner's Durable Object error report, from answers shaped as
2+// Cloudflare gives them, without the network.
3+
4+import assert from "node:assert/strict";
5+import { test } from "node:test";
6+
7+import {
8+ DO_QUERIES,
9+ KEYS,
10+ SERVICE_KEYS,
11+ bucketsOf,
12+ byStatus,
13+ classify,
14+ doQuery,
15+ format,
16+ parseArgs,
17+ parseCalculations,
18+ parseDoGroups,
19+ parseEvents,
20+ telemetryQuestions,
21+ windowOf,
22+} from "./runner-errors.mjs";
23+
24+const answer = (rows) => ({ data: { viewer: { accounts: [{ rows }] } } });
25+const byNamespace = DO_QUERIES.find((query) => query.key === "by_namespace").variants[1];
26+
27+test("the command line has defaults and bounds", () => {
28+ assert.deepEqual(parseArgs([], {}), { help: false, json: false, keys: false, days: 7, samples: 10, script: "g1t-runner", account: "1e6f2cffa3f445920836e8ebe446bb58" });
29+ const options = parseArgs(["--days", "90", "--samples", "3", "--script", "other", "--json", "--keys"], { CLOUDFLARE_ACCOUNT_ID: "acct" });
30+ assert.equal(options.days, 31);
31+ assert.equal(options.samples, 3);
32+ assert.equal(options.script, "other");
33+ assert.equal(options.account, "acct");
34+ assert.ok(options.json && options.keys);
35+ assert.equal(parseArgs(["--days", "soon"], {}).days, 7);
36+});
37+
38+test("the window is the last N days", () => {
39+ assert.deepEqual(windowOf(new Date("2026-10-08T12:00:00Z"), 7), { start: "2026-10-01T12:00:00.000Z", end: "2026-10-08T12:00:00.000Z" });
40+});
41+
42+test("the query is for one script, with the variant's sums and dimensions", () => {
43+ const query = doQuery(byNamespace);
44+ assert.match(query, /durableObjectsInvocationsAdaptiveGroups\(limit: 10000, filter: \{ datetime_geq: \$start, datetime_leq: \$end, scriptName: \$script \}\)/);
45+ assert.match(query, /sum \{ requests errors \}/);
46+ assert.match(query, /dimensions \{ namespaceId status \}/);
47+ assert.match(query, /\$script: String!/);
48+});
49+
50+test("rows are named, summed per status and sorted by requests", () => {
51+ const body = answer([
52+ { sum: { requests: 900, errors: 300 }, dimensions: { namespaceId: "ns1", status: "scriptThrewException" } },
53+ { sum: { requests: 1000, errors: 0 }, dimensions: { namespaceId: "ns1", status: "success" } },
54+ { sum: { requests: 40, errors: 20 }, dimensions: { namespaceId: "ns2", status: "internalError" } },
55+ { sum: { requests: 10, errors: 0 }, dimensions: { namespaceId: "ns2", status: "success" } },
56+ ]);
57+ const parsed = parseDoGroups(body, byNamespace, { ns1: "g1t-runner_AttemptSandbox" });
58+ assert.deepEqual(parsed.dims, ["namespace_id", "status"]);
59+ assert.deepEqual(parsed.rows[0], { namespace_id: "g1t-runner_AttemptSandbox", status: "success", requests: 1000, errors: 0 });
60+ assert.equal(parsed.rows.at(-1).namespace_id, "ns2");
61+ assert.deepEqual(parsed.totals, { requests: 1950, errors: 320 });
62+ assert.deepEqual(byStatus(parsed), [
63+ { status: "success", requests: 1010, errors: 0 },
64+ { status: "scriptThrewException", requests: 900, errors: 300 },
65+ { status: "internalError", requests: 40, errors: 20 },
66+ ]);
67+});
68+
69+test("Cloudflare's refusal becomes the error, so the next variant is tried", () => {
70+ assert.throws(() => parseDoGroups({ errors: [{ message: "unknown field wallTime" }] }, byNamespace), /unknown field wallTime/);
71+ assert.throws(() => parseDoGroups({ data: {} }, byNamespace), /no rows/);
72+});
73+
74+test("the telemetry questions filter on the script and on failures", () => {
75+ const questions = telemetryQuestions({ script: "g1t-runner", from: 1, to: 2, samples: 5 });
76+ assert.deepEqual(questions.map((q) => q.key), ["failed_invocations", "exceptions", "error_logs", "samples"]);
77+ const failed = questions[0].body;
78+ assert.equal(failed.view, "calculations");
79+ assert.deepEqual(failed.timeframe, { from: 1, to: 2 });
80+ assert.deepEqual(failed.parameters.filters, [
81+ { key: SERVICE_KEYS[0], operation: "eq", type: "string", value: "g1t-runner" },
82+ { key: KEYS.outcome, operation: "neq", type: "string", value: "ok" },
83+ ]);
84+ assert.deepEqual(failed.parameters.groupBys.map((g) => g.value), [KEYS.entrypoint, KEYS.eventType, KEYS.outcome]);
85+ assert.equal(questions[3].body.view, "events");
86+ assert.equal(questions[3].body.limit, 5);
87+ const other = telemetryQuestions({ script: "g1t-runner", from: 1, to: 2, samples: 5, serviceKey: "$metadata.service" });
88+ assert.equal(other[1].body.parameters.filters[0].key, "$metadata.service");
89+});
90+
91+test("a calculations answer becomes rows with counts, largest first", () => {
92+ const body = {
93+ success: true,
94+ result: {
95+ calculations: [
96+ {
97+ aggregates: [
98+ { groups: [{ key: KEYS.entrypoint, value: "AttemptSandbox" }, { key: KEYS.eventType, value: "rpc" }], value: 12 },
99+ { groups: [{ key: KEYS.entrypoint, value: "AttemptSandbox" }, { key: KEYS.eventType, value: "alarm" }], value: 540 },
100+ { groups: [{ key: KEYS.entrypoint, value: "" }, { key: KEYS.eventType, value: "alarm" }], count: 3 },
101+ ],
102+ },
103+ ],
104+ },
105+ };
106+ assert.deepEqual(parseCalculations(body), [
107+ { [KEYS.entrypoint]: "AttemptSandbox", [KEYS.eventType]: "alarm", count: 540 },
108+ { [KEYS.entrypoint]: "AttemptSandbox", [KEYS.eventType]: "rpc", count: 12 },
109+ { [KEYS.entrypoint]: "(none)", [KEYS.eventType]: "alarm", count: 3 },
110+ ]);
111+ assert.throws(() => parseCalculations({ success: false, errors: [{ message: "Unauthorized" }] }), /Unauthorized/);
112+ assert.throws(() => parseCalculations({ success: true, result: {} }), /no calculations/);
113+});
114+
115+test("an events answer becomes plain records, the object id cut short", () => {
116+ const body = {
117+ success: true,
118+ result: {
119+ events: {
120+ events: [
121+ {
122+ timestamp: Date.UTC(2026, 9, 7, 3, 4, 5),
123+ $workers: { entrypoint: "AttemptSandbox", eventType: "alarm", outcome: "exception", durableObjectId: "0123456789abcdef0123" },
124+ $metadata: { error: "Durable Object reset because its code was updated." },
125+ },
126+ { $workers: { eventType: "rpc", outcome: "canceled" }, source: { message: "sandbox not started" } },
127+ ],
128+ },
129+ },
130+ };
131+ assert.deepEqual(parseEvents(body), [
132+ { at: "2026-10-07T03:04:05.000Z", entrypoint: "AttemptSandbox", event_type: "alarm", outcome: "exception", object: "0123456789ab", error: "Durable Object reset because its code was updated.", message: null },
133+ { at: null, entrypoint: null, event_type: "rpc", outcome: "canceled", object: null, error: null, message: "sandbox not started" },
134+ ]);
135+});
136+
137+test("messages fall into buckets, expected or not", () => {
138+ assert.equal(classify("Durable Object reset because its code was updated.").bucket, "deploy_reset");
139+ assert.equal(classify("Durable Object reset because its code was updated.").expected, true);
140+ assert.equal(classify("sandbox stop not reported checks report_checks failed with status 500").bucket, "stop_not_reported");
141+ assert.equal(classify("sandbox alarm failed abc retry 2 storage overloaded").bucket, "alarm_failed");
142+ assert.equal(classify("there is no container instance that can be provided to this durable object").bucket, "no_capacity");
143+ assert.equal(classify("sandbox not started agent g1t could not read this project's guardrails: down").bucket, "not_started");
144+ assert.equal(classify("container exited with unexpected exit code: 137").bucket, "container_exited");
145+ assert.equal(classify("Network connection lost.").bucket, "connection_lost");
146+ assert.equal(classify("", "canceled").bucket, "caller_gone");
147+ assert.equal(classify("something new").bucket, "other");
148+ assert.equal(classify("something new").expected, false);
149+});
150+
151+test("buckets sum the counts of their messages", () => {
152+ const rows = [
153+ { [KEYS.error]: "Durable Object reset because its code was updated.", count: 200 },
154+ { [KEYS.error]: "Network connection lost.", count: 7 },
155+ { [KEYS.error]: "Durable Object reset because its code was updated (2).", count: 50 },
156+ ];
157+ assert.deepEqual(
158+ bucketsOf(rows, KEYS.error).map((b) => [b.bucket, b.count]),
159+ [
160+ ["deploy_reset", 250],
161+ ["connection_lost", 7],
162+ ],
163+ );
164+});
165+
166+test("the text report shows rates, statuses, buckets and what could not be read", () => {
167+ const parsed = parseDoGroups(
168+ answer([
169+ { sum: { requests: 100, errors: 31 }, dimensions: { namespaceId: "ns1", status: "scriptThrewException" } },
170+ { sum: { requests: 200, errors: 0 }, dimensions: { namespaceId: "ns1", status: "success" } },
171+ ]),
172+ byNamespace,
173+ { ns1: "g1t-runner_AttemptSandbox" },
174+ );
175+ const rows = [{ [KEYS.error]: "Durable Object reset because its code was updated.", count: 31 }];
176+ const text = format({
177+ account: "acct",
178+ script: "g1t-runner",
179+ start: "2026-10-01T00:00:00.000Z",
180+ end: "2026-10-08T00:00:00.000Z",
181+ analytics: [
182+ { key: "by_namespace", label: "By namespace", ...parsed, by_status: byStatus(parsed) },
183+ { key: "by_day", label: "By day", error: "unknown field date" },
184+ ],
185+ telemetry: [
186+ { key: "exceptions", label: "Exceptions", rows, buckets: bucketsOf(rows, KEYS.error) },
187+ { key: "samples", label: "Samples", rows: [{ at: "2026-10-07T00:00:00.000Z", entrypoint: "AttemptSandbox", event_type: "alarm", outcome: "exception", object: "abc", error: "boom", message: null }] },
188+ ],
189+ keys: null,
190+ notes: ["Telemetry is filtered on $workers.scriptName."],
191+ });
192+ assert.match(text, /g1t-runner_AttemptSandbox\s+scriptThrewException\s+100\s+31\s+31\.0%/);
193+ assert.match(text, /by status:/);
194+ assert.match(text, /could not read: unknown field date/);
195+ assert.match(text, /31\s+deploy_reset \(expected\)/);
196+ assert.match(text, /AttemptSandbox alarm exception abc/);
197+ assert.match(text, /\n {4}boom/);
198+ assert.doesNotMatch(text, /Bearer|authorization/i);
199+});
+109−0
1+#!/usr/bin/env bash
2+# sccache for a workflow's Rust builds. Every rustc call that makes a
3+# library is looked up by what it compiles (sources, flags, toolchain,
4+# dependencies) in the repository's Actions cache, the one `actions/cache`
5+# uses, through sccache's GitHub Actions backend. A checkout gives every
6+# source file a new mtime, so Cargo calls rustc again for the workspace's
7+# own crates however much of `target/` was restored; with sccache, the
8+# crates whose inputs did not change come back from the cache instead of
9+# being compiled. Binaries, cdylibs (a Worker's own crate), proc macros and
10+# test harnesses are linked, and sccache compiles those as before.
11+#
12+# bash scripts/sccache.sh install # a step before the first cargo
13+# bash scripts/sccache.sh stats # the last step, with if: always()
14+#
15+# install downloads the pinned release and checks its sha256, starts the
16+# server (which reads and writes a check entry in the cache), and only then
17+# sets RUSTC_WRAPPER for the steps after it. If anything there fails, the
18+# job builds as it did without sccache, and the step says why.
19+# stats prints the server's hits and misses, and adds them to the job's
20+# summary.
21+#
22+# See docs/DEPLOYING.md ("Build speed") and the Actions guide's
23+# "Caching Rust builds".
24+set -euo pipefail
25+
26+VERSION=0.18.0
27+TARGET=x86_64-unknown-linux-musl
28+# sccache-v0.18.0-x86_64-unknown-linux-musl.tar.gz, from the release's
29+# .sha256 file.
30+SHA256=45f1447fbe231e3037bde351ef70677dd212216c8d62ae7ca409fecc4d6acc89
31+
32+BIN="$HOME/.cargo/bin"
33+SCCACHE="$BIN/sccache"
34+
35+# What sccache and Cargo read, here and in the steps after this one.
36+export SCCACHE_GHA_ENABLED=true
37+# One server for the whole job: its statistics are the job's.
38+export SCCACHE_IDLE_TIMEOUT=0
39+# If the server goes away mid-build, rustc runs without it.
40+export SCCACHE_IGNORE_SERVER_IO_ERROR=1
41+# sccache cannot cache incremental builds; a fresh checkout gains nothing
42+# from them anyway.
43+export CARGO_INCREMENTAL=0
44+
45+warn() { echo "::warning title=sccache::$1, so this job builds without it"; }
46+
47+fetch() {
48+ local name="sccache-v${VERSION}-${TARGET}"
49+ local dir="${RUNNER_TEMP:-/tmp}/sccache-${VERSION}"
50+ if [ -x "$SCCACHE" ] && [ "$("$SCCACHE" --version 2>/dev/null | awk '{print $2}')" = "$VERSION" ]; then
51+ return 0
52+ fi
53+ mkdir -p "$dir" "$BIN"
54+ curl -fsSL --retry 3 -o "$dir/$name.tar.gz" "https://github.com/mozilla/sccache/releases/download/v${VERSION}/${name}.tar.gz" || return 1
55+ echo "${SHA256} $dir/$name.tar.gz" | sha256sum -c --quiet - || return 1
56+ tar -xzf "$dir/$name.tar.gz" -C "$dir" || return 1
57+ install -m 0755 "$dir/$name/sccache" "$SCCACHE"
58+}
59+
60+install_sccache() {
61+ if [ -z "${ACTIONS_RUNTIME_TOKEN:-}" ] || [ -z "${ACTIONS_CACHE_URL:-}${ACTIONS_RESULTS_URL:-}" ]; then
62+ warn "The job has no Actions cache to use (ACTIONS_RUNTIME_TOKEN, ACTIONS_CACHE_URL)"
63+ return 0
64+ fi
65+ if ! fetch; then
66+ warn "sccache ${VERSION} could not be downloaded or did not match its checksum"
67+ return 0
68+ fi
69+ if ! "$SCCACHE" --start-server; then
70+ warn "sccache's server did not start (it could not read the cache)"
71+ return 0
72+ fi
73+ {
74+ echo "RUSTC_WRAPPER=$SCCACHE"
75+ echo "SCCACHE_GHA_ENABLED=$SCCACHE_GHA_ENABLED"
76+ echo "SCCACHE_IDLE_TIMEOUT=$SCCACHE_IDLE_TIMEOUT"
77+ echo "SCCACHE_IGNORE_SERVER_IO_ERROR=$SCCACHE_IGNORE_SERVER_IO_ERROR"
78+ echo "CARGO_INCREMENTAL=$CARGO_INCREMENTAL"
79+ } >> "$GITHUB_ENV"
80+ echo "sccache ${VERSION}: rustc goes through it from here on, cached in this repository's Actions cache."
81+}
82+
83+stats() {
84+ if [ -z "${RUSTC_WRAPPER:-}" ] || [ ! -x "$SCCACHE" ]; then
85+ echo "sccache was not used in this job."
86+ return 0
87+ fi
88+ local out
89+ out="$("$SCCACHE" --show-stats 2>&1)" || true
90+ echo "$out"
91+ if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
92+ {
93+ echo "### sccache"
94+ echo
95+ echo '```'
96+ echo "$out"
97+ echo '```'
98+ } >> "$GITHUB_STEP_SUMMARY"
99+ fi
100+}
101+
102+case "${1:-}" in
103+ install) install_sccache ;;
104+ stats) stats ;;
105+ *)
106+ echo "usage: bash scripts/sccache.sh install|stats" >&2
107+ exit 2
108+ ;;
109+esac
+33−0
8686 out
8787 }
8888
89+/// Whether a repository holding `total` bytes must evict to stay within
90+/// `quota`: what `to_evict` would take something from.
91+pub(crate) fn over_quota(total: u64, quota: u64) -> bool {
92+ total > quota
93+}
94+
8995 /// A month's GB-months from the bytes held each day so far: each day's
9096 /// bytes over 30 days.
9197 pub(crate) fn gb_months(days: &[u64]) -> f64 {
328334 if ready.is_none() {
329335 return Ok(fail(FailureCode::NotFound, "No upload of that entry is in progress."));
330336 }
337+ // What the repository holds, summed: only past the quota are its
338+ // entries listed to choose what to evict. A tool that saves an
339+ // entry per compiled file (sccache) commits thousands of small
340+ // entries a build, and listing them all on each was quadratic.
331341 #[derive(Deserialize)]
342+ struct Total {
343+ bytes: Option<f64>,
344+ }
345+ let total = self
346+ .db
347+ .prepare("SELECT SUM(size) AS bytes FROM cache_entries WHERE repo_id = ? AND status = 'ready'")
348+ .bind(&[job.repo_id.as_str().into()])?
349+ .first::<Total>(None)
350+ .await?
351+ .and_then(|t| t.bytes)
352+ .unwrap_or(0.0);
353+ if !over_quota(total as u64, CACHE_REPO_QUOTA_BYTES) {
354+ return Ok(Outcome::Ok(CacheCommitted { evicted: Vec::new() }));
355+ }
356+ #[derive(Deserialize)]
332357 struct Held {
333358 id: String,
334359 object: String,
510535 assert_eq!(to_evict(&held, "new", 7), ["old", "c"]);
511536 // The entry just saved stays, even when it alone is past the quota.
512537 assert_eq!(to_evict(&entries(&[("big", 20), ("a", 1)]), "big", 10), ["a"]);
538+ // Entries are listed only when the sum is over: at or under the
539+ // quota, nothing would be evicted.
540+ for (held, quota) in [(entries(&[("a", 4), ("b", 6)]), 10), (entries(&[("a", 1)]), 12)] {
541+ let total = held.iter().map(|(_, size)| size).sum();
542+ assert!(!over_quota(total, quota));
543+ assert!(to_evict(&held, "a", quota).is_empty());
544+ }
545+ assert!(over_quota(11, 10));
513546 }
514547
515548 #[test]
+19−0
203203 LIST_PRICES.iter().filter(|p| p.product == product && meter.starts_with(p.meter)).max_by_key(|p| p.meter.len())
204204 }
205205
206+/// What `quantity` of a meter comes to at its list price before any
207+/// included amount, and not in whole blocks: the price book costs every
208+/// unit, so this, not what Cloudflare billed past the included amounts, is
209+/// what it is checked against. None for a meter with no list price.
210+pub(crate) fn list_cost(product: &str, meter: &str, quantity: f64) -> Option<f64> {
211+ list_price(product, meter).map(|p| quantity.max(0.0) * p.usd / p.per)
212+}
213+
206214 /// Puts a cost on every billable-usage line, cycle by cycle (`anchor`):
207215 /// Cloudflare's own where it put one on any of the meter's lines in the
208216 /// cycle, else the list price past the included amount, landing on the
400408 }
401409
402410 #[test]
411+ fn a_list_cost_is_every_unit_at_the_list_price() {
412+ // 126.87k GiB-seconds is $0.32 at list, though only the 36.87k past
413+ // the included 90k were billed ($0.09).
414+ let memory = list_cost("containers", "container_memory_per_gib_second", 126_870.0).unwrap();
415+ assert!((memory - 0.317_175).abs() < 1e-9);
416+ // Per-million meters are not rounded up to a whole million.
417+ assert!((list_cost("workers", "workers_cpu_ms", 11_160_000.0).unwrap() - 0.2232).abs() < 1e-9);
418+ assert!(list_cost("email", "email_service_emails_sent", 7.0).is_none());
419+ }
420+
421+ #[test]
403422 fn subscriptions_accrue_by_the_cycles_days() {
404423 // A whole cycle is the month's price, whatever its length.
405424 assert_eq!(accrued(30_000_000, "2026-09-28", "2026-10-27", 28), 30_000_000);
+39−11
326326 seconds.max(0) as f64 * base_per_second + cpu_seconds.max(0.0) * per_vcpu_second
327327 }
328328
329−/// A unit's marginal rate from the bill: the median, over the days that
330−/// were charged, of cost over quantity. None while nothing was charged.
329+/// A unit's rate from the bill: the median, over the days with a list
330+/// cost, of list cost over quantity. Never what was billed: that is net of
331+/// the included amounts (nothing while a cycle is inside them, part of a
332+/// day's usage on the day it passes one), so over all of the quantity it
333+/// reads as a lower price when nothing changed. None without a list cost;
334+/// the published rates stand in.
331335 pub(crate) fn billed_rate(rows: &[&UsageRow]) -> Option<f64> {
332336 let mut rates: Vec<f64> = rows
333337 .iter()
334− .filter(|r| r.cost > 0.0 && r.quantity > 0.0)
335− .map(|r| r.cost / r.quantity)
338+ .filter(|r| r.list_cost > 0.0 && r.quantity > 0.0)
339+ .map(|r| r.list_cost / r.quantity)
336340 .collect();
337341 if rates.is_empty() {
338342 return None;
350354 unit: String,
351355 quantity: f64,
352356 cost: f64,
357+ /// The quantity at list price, before the included amounts.
358+ list_cost: f64,
353359 }
354360
355361 impl UsageRow {
376382 cost: Some(number(&["ContractedCost", "BilledCost", "contracted_cost"]))
377383 .filter(|cost| *cost > 0.0)
378384 .unwrap_or_else(|| number(&["ListCost", "list_cost"])),
385+ list_cost: number(&["ListCost", "list_cost"]),
379386 })
380387 }
381388 }
879886 .collect()
880887 };
881888
882− // Containers: each resource at what the bill shows it costs, or
883− // the published rate while the included amount still covers it,
889+ // Containers: each resource at the list cost the bill shows for it
890+ // (before the included amounts), or the published rate without one,
884891 // over how much CPU g1t's sandboxes really use per second.
885892 let memory = billed_rate(&named(&["container memory"]));
886893 let disk = billed_rate(&named(&["container disk"]));
894901 durable_object.unwrap_or(LIST_DO_GB_SECOND),
895902 );
896903 // The parts, for runs that report their own CPU.
897− let parts_reason = "Cloudflare's Containers and Durable Objects rates, as billed or published";
904+ let parts_reason = "Cloudflare's Containers and Durable Objects rates, as listed on the bill or published";
898905 self.measure("sandbox_base_second", sandbox_base_micros(rates.0, rates.1, rates.3), parts_reason).await?;
899906 self.measure("sandbox_cpu_second", rates.2 * MICROS_PER_DOLLAR as f64, parts_reason).await?;
900907 if let Some(per_second) = sandbox_second_micros(usage, rates.0, rates.1, rates.2, rates.3) {
911918 if billed.is_empty() {
912919 "rates are Cloudflare's published ones".to_owned()
913920 } else {
914− format!("{} at what Cloudflare billed", billed.join(", "))
921+ format!("{} at the list cost on Cloudflare's bill", billed.join(", "))
915922 },
916923 );
917924 for meter in ["sandbox_second", "build_second"] {
928935 ];
929936 for (meter, words, unit) in app_meters {
930937 if let Some(rate) = billed_rate(&named(words)) {
931− let reason = format!("Cloudflare billed Workers {unit} at ${:.2} per million", rate * 1e6);
938+ let reason = format!("Cloudflare's bill lists Workers {unit} at ${:.2} per million", rate * 1e6);
932939 self.measure(meter, rate * 1e6 * MICROS_PER_DOLLAR as f64, &reason).await?;
933940 }
934941 }
10711078
10721079 #[test]
10731080 fn a_billed_rate_is_the_median_of_the_charged_days() {
1074− let row = |quantity: f64, cost: f64| UsageRow {
1081+ let row = |quantity: f64, list_cost: f64| UsageRow {
10751082 period_start: String::new(),
10761083 period_end: String::new(),
10771084 service: "Containers / Container Memory".into(),
10781085 unit: "Count".into(),
10791086 quantity,
1080− cost,
1087+ cost: list_cost,
1088+ list_cost,
10811089 };
10821090 let rows = [row(100.0, 0.0), row(100.0, 0.0002), row(100.0, 0.00025), row(100.0, 0.00025)];
10831091 assert_eq!(billed_rate(&rows.iter().collect::<Vec<_>>()), Some(0.000_002_5));
10851093 }
10861094
10871095 #[test]
1096+ fn a_rate_is_the_list_price_not_what_was_billed_past_the_included_amount() {
1097+ // 126,870 GiB-seconds, of which the 36,870 past the included 90,000
1098+ // were billed: $0.09 billed, $0.32 at list. The rate is the list's.
1099+ let row = UsageRow {
1100+ period_start: String::new(),
1101+ period_end: String::new(),
1102+ service: "Containers / Container Memory".into(),
1103+ unit: "GiB-seconds".into(),
1104+ quantity: 126_870.0,
1105+ cost: 0.092_175,
1106+ list_cost: 0.317_175,
1107+ };
1108+ assert!((billed_rate(&[&row]).unwrap() - 0.000_002_5).abs() < 1e-15);
1109+ // Billed with no list cost says nothing about the price.
1110+ assert_eq!(billed_rate(&[&UsageRow { list_cost: 0.0, ..row }]), None);
1111+ }
1112+
1113+ #[test]
10881114 fn usage_rows_are_read_by_their_focus_names() {
10891115 let row = UsageRow::from_value(&json!({
10901116 "ServiceFamilyName": "Containers",
10921118 "PricingUnit": "GiB-seconds",
10931119 "PricingQuantity": "1200.5",
10941120 "ContractedCost": 0.003,
1121+ "ListCost": 0.0030,
10951122 "ChargePeriodStart": "2026-10-01",
10961123 }))
10971124 .unwrap();
10981125 assert_eq!(row.service, "Containers / Memory");
10991126 assert_eq!(row.quantity, 1200.5);
11001127 assert_eq!(row.cost, 0.003);
1128+ assert_eq!(row.list_cost, 0.003);
11011129 assert!(UsageRow::from_value(&json!({ "nothing": 1 })).is_none());
11021130 }
11031131 }
+164−36
495495 pub(crate) enum DriftKind {
496496 /// g1t counted a different number of units than Cloudflare did.
497497 Count,
498− /// What Cloudflare charged differs from what the price book says the
499− /// same usage cost.
498+ /// The price book's cost of a bucket's usage differs from what the
499+ /// same usage comes to at Cloudflare's list prices, before the included
500+ /// amounts (a price may be stale); on `models`, the ledger's model cost
501+ /// differs from what AI Gateway priced the same traffic at.
500502 Cost,
501503 /// Cloudflare charged for something nothing charges customers for.
502504 Leak,
526528 }
527529
528530 /// Drift over a window for one bucket: counts more than `threshold`
529−/// percent apart, a bill that far from the price book's cost of the same
530−/// usage, and cost with nothing charged for it. Under `min_cost_micros`
531−/// in all, cost says nothing.
532−pub(crate) fn drifts(bucket: &str, days: &[ProductDay], threshold: f64, counted: bool, min_cost_micros: i64) -> Vec<Drift> {
531+/// percent apart, the price book's cost of the usage that far from what the
532+/// same usage comes to at Cloudflare's list prices (`list_micros`, from
533+/// `list_costs`), and cost with nothing charged for it. Under
534+/// `min_cost_micros` in all, cost says nothing.
535+///
536+/// The price book is set against the list cost, never against what
537+/// Cloudflare billed: the bill is net of the included amounts and the price
538+/// book's cost is of every unit, so a bucket whose usage mostly fits in
539+/// them would read as a stale price when nothing changed. Without a list
540+/// cost (a meter in the bucket with no list price) the price book is not
541+/// checked. A leak is still what Cloudflare billed: money spent.
542+pub(crate) fn drifts(bucket: &str, days: &[ProductDay], threshold: f64, counted: bool, min_cost_micros: i64, list_micros: Option<f64>) -> Vec<Drift> {
533543 let overhead = OVERHEAD.contains(&bucket);
534544 let sum = |f: &dyn Fn(&ProductDay) -> f64| days.iter().map(f).sum::<f64>();
535545 let cf_cost = sum(&|d| d.cf_cost_micros as f64);
548558 out.push(Drift { bucket: bucket.into(), kind: DriftKind::Count, ours: own_quantity, cloudflare: cf_quantity, delta_percent: delta });
549559 }
550560 }
551− let enough = cf_cost.max(own_cost) >= min_cost_micros as f64;
552561 // Models: what AI Gateway priced g1t's own provider traffic at (its
553562 // lines, as "Cloudflare's" side) against the ledger's model cost. Only
554563 // once the gateway has been read; then the ledger having none of it is
555− // drift too (traffic no run was charged for).
556− let models = NOT_CLOUDFLARE.contains(&bucket) && cf_cost > 0.0;
557− if enough && !overhead && cf_cost > 0.0 && (own_cost > 0.0 || models) {
558− let delta = delta_percent(own_cost, cf_cost);
564+ // drift too (traffic no run was charged for). Every other bucket: its
565+ // usage at Cloudflare's list prices, before the included amounts.
566+ let not_cloudflare = NOT_CLOUDFLARE.contains(&bucket);
567+ let theirs = if not_cloudflare { cf_cost } else { list_micros.unwrap_or(0.0) };
568+ let enough = theirs.max(own_cost) >= min_cost_micros as f64;
569+ let models = not_cloudflare && cf_cost > 0.0;
570+ if enough && !overhead && theirs > 0.0 && (own_cost > 0.0 || models) {
571+ let delta = delta_percent(own_cost, theirs);
559572 if delta.is_some_and(|d| d.abs() > threshold) {
560− out.push(Drift { bucket: bucket.into(), kind: DriftKind::Cost, ours: own_cost, cloudflare: cf_cost, delta_percent: delta });
573+ out.push(Drift { bucket: bucket.into(), kind: DriftKind::Cost, ours: own_cost, cloudflare: theirs, delta_percent: delta });
561574 }
562575 }
563576 // The ledger has model cost and the gateway priced none of it: a token
572585 out
573586 }
574587
588+/// What each bucket's billable usage over a window comes to at
589+/// Cloudflare's list prices, before the included amounts (`cycle::list_cost`
590+/// line by line, in micros): what the price book's cost of the same usage
591+/// is checked against. None for a bucket with usage on a meter that has no
592+/// list price: its usage cannot be priced like for like, so it is not
593+/// checked. Other sources (Artifacts events, AI Gateway) are left out.
594+pub(crate) fn list_costs(rules: &[Rule], lines: &[LineRow]) -> BTreeMap<String, Option<f64>> {
595+ let mut out: BTreeMap<String, Option<f64>> = BTreeMap::new();
596+ for line in lines.iter().filter(|l| l.source == SOURCE_BILLABLE) {
597+ let bucket = costs::classify(rules, &line.product, &line.meter).map_or(UNMAPPED, |r| r.bucket.as_str());
598+ let entry = out.entry(bucket.to_owned()).or_insert(Some(0.0));
599+ if line.quantity <= 0.0 {
600+ continue;
601+ }
602+ *entry = match (*entry, crate::cycle::list_cost(&line.product, &line.meter, line.quantity)) {
603+ (Some(sum), Some(cost)) => Some(sum + cost * 1_000_000.0),
604+ _ => None,
605+ };
606+ }
607+ out
608+}
609+
575610 /// What can make AI Gateway's cost differ from what the providers bill,
576611 /// said for staff: cache tokens (priced by the gateway at its own rates for
577612 /// them, which may lag the provider's), requests Cloudflare billed itself,
893928 out
894929 }
895930
896−/// Cloudflare's marginal rate for one of its units: the median over the
897−/// charged days of cost over quantity, in dollars. None while the included
898−/// amounts still cover it. Each item is a day's (quantity, cost).
931+/// Cloudflare's rate for one of its units: the median over the costed
932+/// days of cost over quantity, in dollars. None while nothing is costed.
933+/// Each item is a day's (quantity, cost), from `rate_line`.
899934 pub(crate) fn billed_rate(days: &[(f64, f64)]) -> Option<f64> {
900935 let mut rates: Vec<f64> = days.iter().filter(|(q, c)| *q > 0.0 && *c > 0.0).map(|(q, c)| c / q).collect();
901936 if rates.is_empty() {
905940 Some(rates[rates.len() / 2])
906941 }
907942
943+/// A line's (quantity, cost) for `billed_rate`: at its list price for all
944+/// of the quantity where the meter has one, never what the cycle billed,
945+/// which is net of the included amounts (nothing until the cycle passes
946+/// them, part of a day's usage on the day it does, then whole millions) and
947+/// so says nothing about the price of each unit. A meter with no list
948+/// price has only what Cloudflare billed; the median over the charged days
949+/// leaves out the day its included amount ran out.
950+pub(crate) fn rate_line(product: &str, meter: &str, quantity: f64, cost_usd: f64) -> (f64, f64) {
951+ (quantity, crate::cycle::list_cost(product, meter, quantity).unwrap_or(cost_usd))
952+}
953+
908954 /// What one of g1t's units costs, from Cloudflare's rate per its own unit
909955 /// and how many of Cloudflare's units each of g1t's took: if Cloudflare
910956 /// counts three operations for every git operation g1t counts, a git
15681614 }
15691615 let caveats = self.gateway_caveats(&since, until).await?;
15701616 let resets = self.resets_since(&since, until).await?;
1617+ // The same days' usage at list prices, before the included amounts:
1618+ // what the price book is checked against (never the bill, which is
1619+ // net of them).
1620+ let lines = self
1621+ .db
1622+ .prepare("SELECT day, source, product, meter, quantity, cost_usd FROM cost_lines WHERE source = ?1 AND day >= ?2 AND day <= ?3")
1623+ .bind(&[SOURCE_BILLABLE.into(), since.as_str().into(), until.into()])?
1624+ .all()
1625+ .await?
1626+ .results::<LineRow>()?;
1627+ let list = list_costs(&rules, &lines);
15711628 let mut found = Vec::new();
15721629 if let Some(drift) = unpriced_drift(&caveats) {
15731630 found.push(drift);
15771634 let threshold = bucket_rules.iter().map(|r| r.drift_percent).fold(f64::INFINITY, f64::min);
15781635 let threshold = if threshold.is_finite() { threshold } else { 10.0 };
15791636 let counted = bucket_rules.iter().any(|r| r.own_meter.is_some());
1580− for drift in drifts(bucket, days, threshold, counted, settings.min_daily_cost_micros) {
1637+ let list_micros = list.get(bucket).copied().flatten();
1638+ for drift in drifts(bucket, days, threshold, counted, settings.min_daily_cost_micros, list_micros) {
15811639 if wiped_not_leaked(&drift, &resets) {
15821640 continue;
15831641 }
15911649 drift.delta_percent.unwrap_or(0.0)
15921650 ),
15931651 DriftKind::Cost => format!(
1594− "{title}: Cloudflare charged {} over the last {DRIFT_DAYS} days; the price book's cost of the same usage is {} ({:+.1}%). A price may be stale: see the proposals.",
1652+ "{title}: the last {DRIFT_DAYS} days' usage comes to {} at Cloudflare's list prices, before the included amounts; the price book's cost of the same usage is {} ({:+.1}%). A price may be stale: see the proposals.",
15951653 dollars(drift.cloudflare as i64),
15961654 dollars(drift.ours as i64),
15971655 drift.delta_percent.unwrap_or(0.0)
16761734 let mine: Vec<(f64, f64)> = lines
16771735 .iter()
16781736 .filter(|l| costs::classify(&rules, &l.product, &l.meter).is_some_and(|r| r.product == s.product && r.meter == s.meter))
1679− .map(|l| (l.quantity, l.cost_usd))
1737+ .map(|l| rate_line(&l.product, &l.meter, l.quantity, l.cost_usd))
16801738 .collect();
16811739 let Some(rate) = billed_rate(&mine) else { continue };
16821740 let cf_units: f64 = mine.iter().map(|(q, _)| q).sum();
25842642 fn counts_more_than_the_threshold_apart_are_drift() {
25852643 // Cloudflare counted 30,000 operations where g1t counted 10,000:
25862644 // binding reads, perhaps. -66.7%.
2587− let drift = drifts("git", &[day("git", 3_000_000, 1_500_000, 1_800_000, 30_000.0, 10_000.0)], 10.0, true, 100_000);
2645+ let drift = drifts("git", &[day("git", 3_000_000, 1_500_000, 1_800_000, 30_000.0, 10_000.0)], 10.0, true, 100_000, Some(3_000_000.0));
25882646 assert_eq!(drift.iter().map(|d| d.kind).collect::<Vec<_>>(), vec![DriftKind::Count, DriftKind::Cost]);
25892647 assert!((drift[0].delta_percent.unwrap() + 66.666).abs() < 0.01);
25902648 // 9% apart: within 10%.
2591− assert!(drifts("git", &[day("git", 1_000_000, 1_000_000, 1_200_000, 10_000.0, 10_900.0)], 10.0, true, 100_000).is_empty());
2649+ assert!(drifts("git", &[day("git", 1_000_000, 1_000_000, 1_200_000, 10_000.0, 10_900.0)], 10.0, true, 100_000, Some(1_000_000.0)).is_empty());
25922650 // Uncounted products have no count drift.
2593− assert!(drifts("sandboxes", &[day("sandboxes", 1_000_000, 1_050_000, 1_200_000, 5.0, 0.0)], 10.0, false, 100_000).is_empty());
2651+ assert!(drifts("sandboxes", &[day("sandboxes", 1_000_000, 1_050_000, 1_200_000, 5.0, 0.0)], 10.0, false, 100_000, Some(1_000_000.0)).is_empty());
25942652 }
25952653
25962654 #[test]
25972655 fn cost_with_no_revenue_is_a_leak_but_not_for_running_g1t() {
2598− let leak = drifts("actions_cache", &[day("actions_cache", 400_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000);
2656+ let leak = drifts("actions_cache", &[day("actions_cache", 400_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None);
25992657 assert_eq!(leak.len(), 1);
26002658 assert_eq!(leak[0].kind, DriftKind::Leak);
2601− assert!(drifts("platform", &[day("platform", 5_000_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000).is_empty());
2659+ assert!(drifts("platform", &[day("platform", 5_000_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty());
26022660 // Pennies say nothing.
2603− assert!(drifts("actions_cache", &[day("actions_cache", 50_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000).is_empty());
2604− assert!(drifts(UNMAPPED, &[day(UNMAPPED, 250_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000)[0].kind == DriftKind::Leak);
2661+ assert!(drifts("actions_cache", &[day("actions_cache", 50_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty());
2662+ assert!(drifts(UNMAPPED, &[day(UNMAPPED, 250_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None)[0].kind == DriftKind::Leak);
2663+ }
2664+
2665+ /// A week of sandboxes mostly inside the cycle's included amounts:
2666+ /// Cloudflare billed $0.0922 (the memory past 25 GiB-hours), and the
2667+ /// same usage is $1.70 at list prices.
2668+ fn sandbox_week() -> (Vec<Rule>, Vec<LineRow>) {
2669+ let mut rules = rules();
2670+ rules.push(rule("durable_objects", "durable_objects_compute_duration", "sandboxes", None));
2671+ let lines = vec![
2672+ line("2026-10-07", SOURCE_BILLABLE, "containers", "container_memory_per_gib_second", 126_870.0, 0.092_175),
2673+ line("2026-10-07", SOURCE_BILLABLE, "containers", "container_vcpu", 12_000.0, 0.0),
2674+ line("2026-10-07", SOURCE_BILLABLE, "containers", "container_disk_per_gb_second", 253_740.0, 0.0),
2675+ line("2026-10-07", SOURCE_BILLABLE, "durable_objects", "durable_objects_compute_duration", 90_000.0, 0.0),
2676+ // Not billable usage: no part of the list cost.
2677+ line("2026-10-07", SOURCE_ARTIFACTS, "artifacts", "events_push", 40.0, 0.0),
2678+ ];
2679+ (rules, lines)
2680+ }
2681+
2682+ #[test]
2683+ fn usage_inside_the_included_amounts_is_not_a_stale_price() {
2684+ let (rules, lines) = sandbox_week();
2685+ let list = list_costs(&rules, &lines);
2686+ let sandboxes = list["sandboxes"].unwrap();
2687+ assert!((sandboxes - 1_699_936.8).abs() < 1.0, "{sandboxes}");
2688+ // The price book's cost of the same usage is $1.69: within 1% of
2689+ // the list, though Cloudflare billed $0.0922 after the included
2690+ // amounts. Before, that read as +1733.6%.
2691+ let week = day("sandboxes", 92_175, 1_690_000, 2_028_000, 0.0, 0.0);
2692+ assert!(drifts("sandboxes", std::slice::from_ref(&week), 10.0, false, 100_000, Some(sandboxes)).is_empty());
2693+ let net = drifts("sandboxes", &[week], 10.0, false, 100_000, Some(92_175.0));
2694+ assert_eq!(net[0].kind, DriftKind::Cost, "billed against the price book was the false alarm");
2695+ }
2696+
2697+ #[test]
2698+ fn a_stale_price_is_still_drift() {
2699+ let (rules, lines) = sandbox_week();
2700+ let sandboxes = list_costs(&rules, &lines)["sandboxes"].unwrap();
2701+ // The price book still costs the same usage at $1.20: a list price
2702+ // rose and the book did not follow.
2703+ let found = drifts("sandboxes", &[day("sandboxes", 92_175, 1_200_000, 1_440_000, 0.0, 0.0)], 10.0, false, 100_000, Some(sandboxes));
2704+ assert_eq!(found.len(), 1);
2705+ assert_eq!((found[0].kind, found[0].ours, found[0].cloudflare.round()), (DriftKind::Cost, 1_200_000.0, 1_699_937.0));
2706+ assert!((found[0].delta_percent.unwrap() + 29.4).abs() < 0.1);
2707+ // Nothing billed at all (the whole week inside the included
2708+ // amounts) is checked all the same.
2709+ assert_eq!(drifts("sandboxes", &[day("sandboxes", 0, 1_200_000, 1_440_000, 0.0, 0.0)], 10.0, false, 100_000, Some(sandboxes)).len(), 1);
2710+ }
2711+
2712+ #[test]
2713+ fn a_bucket_with_a_meter_with_no_list_price_is_not_checked() {
2714+ let (rules, mut lines) = sandbox_week();
2715+ lines.push(line("2026-10-07", SOURCE_BILLABLE, "artifacts", "storage", 3.0, 0.4));
2716+ lines.push(line("2026-10-07", SOURCE_BILLABLE, "artifacts", "operations", 0.0, 0.0));
2717+ let list = list_costs(&rules, &lines);
2718+ assert_eq!(list["git"], None);
2719+ assert!(list["sandboxes"].is_some());
2720+ // No list cost: no cost drift, but a leak is still what was billed.
2721+ assert!(drifts("git", &[day("git", 3_000_000, 1_000_000, 1_200_000, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty());
2722+ assert_eq!(drifts("git", &[day("git", 3_000_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None)[0].kind, DriftKind::Leak);
2723+ }
2724+
2725+ #[test]
2726+ fn a_unit_rate_is_the_list_price_where_there_is_one() {
2727+ // Billed $0.20 for 11.16M CPU ms (9.16M past the included 30M, in
2728+ // whole millions): $0.02 a million at list, whatever was billed.
2729+ let (q, c) = rate_line("workers", "workers_cpu_ms", 11_160_000.0, 0.20);
2730+ assert!((billed_rate(&[(q, c)]).unwrap() * 1e6 - 0.02).abs() < 1e-12);
2731+ // No list price: what Cloudflare billed.
2732+ assert_eq!(rate_line("artifacts", "operations", 1_000.0, 0.5), (1_000.0, 0.5));
26052733 }
26062734
26072735 #[test]
26362764 let on = |day: &str, cf: f64, own: f64| ProductDay { day: day.into(), bucket: "git".into(), cf_quantity: cf, own_quantity: own, ..ProductDay::default() };
26372765 // Five days of Cloudflare's count before g1t's meter, then two that match.
26382766 let days = vec![on("2026-10-01", 500.0, 0.0), on("2026-10-05", 300.0, 0.0), on("2026-10-06", 210.0, 231.0), on("2026-10-07", 450.0, 458.0)];
2639− assert!(drifts("git", &days, 10.0, true, 0).iter().all(|d| d.kind != DriftKind::Count));
2767+ assert!(drifts("git", &days, 10.0, true, 0, None).iter().all(|d| d.kind != DriftKind::Count));
26402768 // A real gap on the days both counted still shows.
26412769 let days = vec![on("2026-10-01", 500.0, 0.0), on("2026-10-06", 400.0, 231.0), on("2026-10-07", 600.0, 300.0)];
2642− let found = drifts("git", &days, 10.0, true, 0);
2770+ let found = drifts("git", &days, 10.0, true, 0, None);
26432771 let count = found.iter().find(|d| d.kind == DriftKind::Count).unwrap();
26442772 assert_eq!((count.ours, count.cloudflare), (531.0, 1000.0));
26452773 // A meter that never counted is compared over every day.
26462774 let days = vec![on("2026-10-06", 400.0, 0.0)];
2647− assert!(drifts("git", &days, 10.0, true, 0).iter().any(|d| d.kind == DriftKind::Count));
2775+ assert!(drifts("git", &days, 10.0, true, 0, None).iter().any(|d| d.kind == DriftKind::Count));
26482776 }
26492777
26502778 #[test]
27522880 #[test]
27532881 fn the_gateways_total_against_the_ledgers_model_cost_is_drift() {
27542882 // The gateway priced $5 of g1t's own traffic; the ledger has $3.
2755− let short = drifts("models", &[day("models", 5_000_000, 3_000_000, 3_600_000, 0.0, 0.0)], 10.0, false, 100_000);
2883+ let short = drifts("models", &[day("models", 5_000_000, 3_000_000, 3_600_000, 0.0, 0.0)], 10.0, false, 100_000, None);
27562884 assert_eq!(short.iter().map(|d| d.kind).collect::<Vec<_>>(), vec![DriftKind::Cost]);
27572885 assert!((short[0].delta_percent.unwrap() + 40.0).abs() < 1e-9);
27582886 // Gateway traffic with nothing on the ledger at all: cost drift and a leak.
2759− let none = drifts("models", &[day("models", 2_000_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000);
2887+ let none = drifts("models", &[day("models", 2_000_000, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None);
27602888 assert_eq!(none.iter().map(|d| d.kind).collect::<Vec<_>>(), vec![DriftKind::Cost, DriftKind::Leak]);
27612889 // Within the threshold: nothing.
2762− assert!(drifts("models", &[day("models", 1_050_000, 1_000_000, 1_200_000, 0.0, 0.0)], 10.0, false, 100_000).is_empty());
2890+ assert!(drifts("models", &[day("models", 1_050_000, 1_000_000, 1_200_000, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty());
27632891 // The gateway priced nothing against a ledger that has model cost:
27642892 // not agreement (a token that cannot see AI Gateway reads as no
27652893 // rows), so it is said. Under the minimum, or no model cost: nothing.
2766− let silent = drifts("models", &[day("models", 0, 1_000_000, 1_200_000, 0.0, 0.0)], 10.0, false, 100_000);
2894+ let silent = drifts("models", &[day("models", 0, 1_000_000, 1_200_000, 0.0, 0.0)], 10.0, false, 100_000, None);
27672895 assert_eq!(silent, vec![Drift { bucket: "models".into(), kind: DriftKind::Cost, ours: 1_000_000.0, cloudflare: 0.0, delta_percent: None }]);
27682896 // Why it is empty, as far as the run could tell.
27692897 let why = |caveats: &costs::GatewayCaveats, read: costs::GatewayRead| models_detail(&silent[0], caveats, &[], &read);
27802908 let unpriced = costs::GatewayCaveats { requests: 42.0, unpriced: vec!["anthropic_claude_new_1".into()], ..Default::default() };
27812909 let said = why(&unpriced, costs::GatewayRead::Rows);
27822910 assert!(said.contains("logged 42 requests") && said.contains("no price for the models used (anthropic_claude_new_1)"), "{said}");
2783− assert!(drifts("models", &[day("models", 0, 50_000, 60_000, 0.0, 0.0)], 10.0, false, 100_000).is_empty());
2784− assert!(drifts("models", &[day("models", 0, 0, 0, 0.0, 0.0)], 10.0, false, 100_000).is_empty());
2911+ assert!(drifts("models", &[day("models", 0, 50_000, 60_000, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty());
2912+ assert!(drifts("models", &[day("models", 0, 0, 0, 0.0, 0.0)], 10.0, false, 100_000, None).is_empty());
27852913 // The detail says which way and why it may be off.
27862914 let caveats = costs::GatewayCaveats { cache_read_tokens: 3_000_000.0, unpriced: vec!["anthropic_claude_new_1".into()], ..Default::default() };
27872915 let detail = models_detail(&short[0], &caveats, &[], &costs::GatewayRead::default());
28512979 let models = |days: &[ProductDay]| days.iter().find(|d| d.bucket == "models").cloned().unwrap();
28522980 // Without it: AI Gateway's $11.11 against the ledger's $2.49.
28532981 let (days, _) = gateway_and_ledger(false);
2854− let drift = drifts("models", &[models(&days)], 10.0, false, 100_000);
2982+ let drift = drifts("models", &[models(&days)], 10.0, false, 100_000, None);
28552983 assert_eq!(drift.iter().map(|d| d.kind).collect::<Vec<_>>(), vec![DriftKind::Cost]);
28562984 assert_eq!((drift[0].ours, drift[0].cloudflare), (2_490_000.0, 11_110_000.0));
28572985 // With it: the ledger's model cost and the reset's add up to the gateway's.
28582986 let (days, _) = gateway_and_ledger(true);
28592987 let m = models(&days);
28602988 assert_eq!(m.own_cost_micros, 11_110_000);
2861− assert!(drifts("models", &[m], 10.0, false, 100_000).is_empty());
2989+ assert!(drifts("models", &[m], 10.0, false, 100_000, None).is_empty());
28622990 // The reset's own row makes no bucket of its own.
28632991 assert!(!days.iter().any(|d| d.bucket.is_empty()));
28642992 }
+65−8
9595 import { buildMentionPrompt, describeThread, handleMention, jobTokenRefusal, planMention } from "./mentions";
9696 import { instructionsFor, repoInstructions, withBlock } from "./repo-instructions";
9797 import { cancelTask, enqueueTask, handedOverStep, selfHostedRoute, taskEnv, taskRepo } from "./self-hosted";
98+import { answered, describeError, tellStopped, withinTimeCap } from "./lifecycle";
9899 import {
99100 ABUSE_EXIT_CODE,
100101 ABUSE_HOST,
412413 * it and cleans up if it dies without reporting.
413414 */
414415 export class AttemptSandbox extends Container<RunnerEnv> {
415− // Past the longest time cap (implement, 90 minutes) and its alarm, so a
416− // long run is never put to sleep before its own cap ends it. A finished
416+ // Past the longest default time cap (implement, 90 minutes) and its
417+ // alarm. A run whose guardrails allow longer (up to 240 minutes) is kept
418+ // past it by `onActivityExpired`, so only its own cap ends it. A finished
417419 // run's process exits and stops the sandbox well before this.
418420 sleepAfter = "100m";
419421 // A guarded sandbox's HTTPS goes through `egress` too (guard.ts).
439441 ? withPlanLimits(await buildGuardFor(this.env.WORK, build.repo, build.kind, build.minutes, build.repoId, build.job, build.hosts), limits)
440442 : null;
441443 } catch (error) {
444+ // Thrown to the caller, which says why the work did not start; logged
445+ // here too, so a sandbox that never started is traceable on its own.
446+ console.error("sandbox not started", run.kind, describeError(error));
442447 await this.settle(0);
443448 throw error;
444449 }
501506 await this.schedule(cap * 60 + ALARM_GRACE_SECONDS, "timeUp");
502507 }
503508 } catch (error) {
509+ console.error("sandbox not started", run.kind, describeError(error));
504510 await revokeCredentials(this.env.IDENTITY, this.ctx.storage, this.env.INTEGRATIONS);
505511 if (tracked) await this.closeRun("failed", `The sandbox could not start: ${String(error)}`);
506512 await this.settle(0);
632638 await this.remoteEnded(1, reason);
633639 return;
634640 }
635− await this.destroy();
641+ // A container already gone has nothing to stop; one that will not stop
642+ // is logged, and its time cap still ends it.
643+ await this.destroy().catch((error: unknown) => console.error("sandbox not destroyed", reason, describeError(error)));
644+ }
645+
646+ /**
647+ * The library's `sleepAfter` has passed. The runner never fetches its
648+ * container, so to the library every sandbox looks idle: a run inside its
649+ * time cap keeps going, and the cap's own alarm (`timeUp`) ends it. Only
650+ * a sandbox with no cap is stopped for inactivity.
651+ */
652+ override async onActivityExpired(): Promise<void> {
653+ const started = await this.ctx.storage.get<number>("started");
654+ const cap = await this.ctx.storage.get<number>("timeCap");
655+ if (withinTimeCap(started, cap, Date.now(), ALARM_GRACE_SECONDS)) return;
656+ await super.onActivityExpired();
657+ }
658+
659+ /**
660+ * A container that crashed or could not be reached, as the library tells
661+ * it. Logged at error level with the sandbox, never thrown: the library
662+ * ignores what this throws, and the stop that follows is handled by
663+ * `onStop` or by `run`, which says why the work did not start.
664+ */
665+ override onError(error: unknown): void {
666+ console.error("sandbox container error", this.ctx.id.toString(), describeError(error));
636667 }
637668
638669 /**
670+ * The library's alarm: scheduled callbacks, the container's keep-alive,
671+ * and `onStop` once it has stopped. A failure is logged with the retry it
672+ * was, then thrown so Cloudflare tries the alarm again.
673+ */
674+ override async alarm(alarmProps?: AlarmInvocationInfo): Promise<void> {
675+ try {
676+ await super.alarm(alarmProps);
677+ } catch (error) {
678+ console.error("sandbox alarm failed", this.ctx.id.toString(), `retry ${alarmProps?.retryCount ?? 0}`, describeError(error));
679+ throw error;
680+ }
681+ }
682+
683+ /**
639684 * A self-hosted runner's task ended (the actions service says so, or g1t
640685 * stopped it): everything a container's stop does, once.
641686 */
713758 if (ended === "stopped" && run && STOP_ENDS.has(run.kind)) return;
714759 console.log("sandbox stopped", run?.kind, "exit", exitCode, reason);
715760 if (!run) return;
761+ // Never thrown: this runs in the sandbox's alarm, which a throw would
762+ // fail, retry and count as an error, running all of the above again.
763+ // What could not be told is logged, and the sweep catches it up.
764+ await tellStopped(run.kind, () => this.reportStopped(run, why, exitCode));
765+ }
766+
767+ /**
768+ * Tells whoever is waiting on the sandbox's work that it stopped without
769+ * finishing it. Each is refused harmlessly when the sandbox reported its
770+ * end before it stopped. Throws when the service could not be reached.
771+ */
772+ private async reportStopped(run: Run, why: string | null, exitCode: number): Promise<void> {
716773 if (run.kind === "actions") {
717774 // Refused harmlessly if the job reported its end before it stopped.
718− await this.env.ACTIONS.fetch("https://actions/rpc/job_report", {
775+ const response = await this.env.ACTIONS.fetch("https://actions/rpc/job_report", {
719776 method: "POST",
720777 headers: { "content-type": "application/json" },
721778 body: JSON.stringify({
724781 report: { kind: "done", conclusion: "failure", reason: why ?? "The runner stopped before the job finished." },
725782 }),
726783 });
784+ await answered("job_report", response);
727785 return;
728786 }
729787 // Nothing was pushed, so no pull request opens; why is in its log.
731789 if (run.kind === "backup") {
732790 // Refused harmlessly if the sandbox reported before it stopped; the
733791 // job is otherwise tried again later tonight.
734− await reposClient(this.env.REPOS)
735− .failBackup(run.jobId, run.token, why ?? `The sandbox exited with ${exitCode}.`)
736− .catch((error: unknown) => console.log("backup failure not reported", run.jobId, String(error)));
792+ await reposClient(this.env.REPOS).failBackup(run.jobId, run.token, why ?? `The sandbox exited with ${exitCode}.`);
737793 return;
738794 }
739795 if (run.kind === "deploy") {
740796 // Refused harmlessly if the build reported its end before it stopped.
741− await this.env.DEPLOYMENTS.fetch(`https://deployments/jobs/${run.deployId}/fail`, {
797+ const response = await this.env.DEPLOYMENTS.fetch(`https://deployments/jobs/${run.deployId}/fail`, {
742798 method: "POST",
743799 headers: { "content-type": "application/json" },
744800 body: JSON.stringify({ token: run.token, message: why ?? "The build stopped before it finished." }),
745801 });
802+ await answered("deploy fail", response);
746803 return;
747804 }
748805 const work = workClient(this.env.WORK);
+86−0
1+import assert from "node:assert/strict";
2+import { test } from "node:test";
3+
4+import { answered, describeError, tellStopped, withinTimeCap } from "./lifecycle.ts";
5+
6+const quiet = { waitMs: 0 };
7+
8+test("a stop that is told the first time is told once and logs nothing", async () => {
9+ const logged: unknown[][] = [];
10+ let calls = 0;
11+ const told = await tellStopped("checks", async () => void calls++, { ...quiet, log: (...data) => logged.push(data) });
12+ assert.equal(told, true);
13+ assert.equal(calls, 1);
14+ assert.deepEqual(logged, []);
15+});
16+
17+test("a stop that fails once is tried again, and gets through", async () => {
18+ const logged: unknown[][] = [];
19+ let calls = 0;
20+ const told = await tellStopped(
21+ "queue",
22+ async () => {
23+ calls++;
24+ if (calls === 1) throw new Error("report_queue failed with status 500");
25+ },
26+ { ...quiet, log: (...data) => logged.push(data) },
27+ );
28+ assert.equal(told, true);
29+ assert.equal(calls, 2);
30+ assert.deepEqual(logged, []);
31+});
32+
33+test("a stop that cannot be told never throws, and is logged with what and why", async () => {
34+ const logged: unknown[][] = [];
35+ let calls = 0;
36+ const told = await tellStopped(
37+ "actions",
38+ async () => {
39+ calls++;
40+ throw new Error("job_report answered 503");
41+ },
42+ { ...quiet, log: (...data) => logged.push(data) },
43+ );
44+ assert.equal(told, false);
45+ assert.equal(calls, 2);
46+ assert.deepEqual(logged, [["sandbox stop not reported", "actions", "job_report answered 503"]]);
47+});
48+
49+test("whatever is thrown, even nothing, is logged as one line", async () => {
50+ const logged: unknown[][] = [];
51+ await tellStopped("plan", () => Promise.reject(undefined), { ...quiet, attempts: 1, log: (...data) => logged.push(data) });
52+ assert.deepEqual(logged, [["sandbox stop not reported", "plan", "no reason given"]]);
53+ assert.equal(describeError("gone"), "gone");
54+ assert.equal(describeError({ code: 7 }), '{"code":7}');
55+ assert.equal(describeError(new TypeError("bad")), "bad");
56+});
57+
58+test("a service that cannot answer is a failure; one that refuses is not", async () => {
59+ await answered("job_report", new Response("{}", { status: 200 }));
60+ // The deployments service's answer for a build that already finished.
61+ await answered("deploy fail", new Response("{}", { status: 404 }));
62+ await answered("job_report", new Response("{}", { status: 409 }));
63+ await assert.rejects(answered("job_report", new Response("worker threw", { status: 500 })), /job_report answered 500: worker threw/);
64+ await assert.rejects(answered("deploy fail", new Response("", { status: 503 })), /^Error: deploy fail answered 503$/);
65+ await assert.rejects(answered("job_report", new Response("slow down", { status: 429 })), /429/);
66+});
67+
68+test("a run inside its time cap is not stopped for inactivity", () => {
69+ const started = Date.UTC(2026, 9, 8, 12, 0, 0);
70+ const minute = 60_000;
71+ const grace = 180;
72+ // A 240-minute cap outlives the 100-minute sleepAfter.
73+ assert.equal(withinTimeCap(started, 240, started + 100 * minute, grace), true);
74+ assert.equal(withinTimeCap(started, 240, started + 242 * minute, grace), true);
75+ // Past the cap and its alarm's grace: the library may stop it.
76+ assert.equal(withinTimeCap(started, 240, started + 243 * minute, grace), false);
77+ assert.equal(withinTimeCap(started, 90, started + 100 * minute, grace), false);
78+});
79+
80+test("a sandbox with no cap or no start is stopped for inactivity as before", () => {
81+ const now = Date.now();
82+ assert.equal(withinTimeCap(undefined, 90, now, 180), false);
83+ assert.equal(withinTimeCap(now - 1000, undefined, now, 180), false);
84+ assert.equal(withinTimeCap(now - 1000, null, now, 180), false);
85+ assert.equal(withinTimeCap(now - 1000, 0, now, 180), false);
86+});
+80−0
1+/**
2+ * How a sandbox's Durable Object handles the end of its container without
3+ * failing the invocation it happens in. The containers library calls
4+ * `onStop` from the object's alarm: a stop hook that throws fails that
5+ * alarm, which Cloudflare retries and counts as an error each time, and
6+ * which runs the whole hook again. So the hook's calls to other services
7+ * go through `tellStopped`, which tries twice and logs at error level, and
8+ * never throws. Pure, so it is tested on its own.
9+ */
10+
11+/** How long `tellStopped` waits before its second try. */
12+export const RETRY_AFTER_MS = 500;
13+
14+/**
15+ * Tells another service that a sandbox stopped: `step` runs, and once more
16+ * if it throws. Never throws: a step that fails twice is logged with
17+ * `log` as `sandbox stop not reported`, with what and why, so it shows in
18+ * the runner's logs at error level and the five-minute sweep picks the
19+ * work up instead. Returns whether it got through.
20+ */
21+export async function tellStopped(
22+ what: string,
23+ step: () => Promise<unknown>,
24+ options: { log?: (...data: unknown[]) => void; attempts?: number; waitMs?: number } = {},
25+): Promise<boolean> {
26+ const log = options.log ?? console.error;
27+ const attempts = Math.max(1, options.attempts ?? 2);
28+ let last: unknown = null;
29+ for (let attempt = 1; attempt <= attempts; attempt++) {
30+ try {
31+ await step();
32+ return true;
33+ } catch (error) {
34+ last = error;
35+ if (attempt < attempts) await new Promise((resolve) => setTimeout(resolve, options.waitMs ?? RETRY_AFTER_MS));
36+ }
37+ }
38+ log("sandbox stop not reported", what, describeError(last));
39+ return false;
40+}
41+
42+/**
43+ * A service binding's answer as a step's outcome: a 5xx (or 429) throws, so
44+ * `tellStopped` tries again and logs it. A refusal is not a failure: the
45+ * work already reported its end, and services say so with `ok: false` or a
46+ * 4xx (the deployments service answers 404 for a build that has finished).
47+ */
48+export async function answered(what: string, response: Response): Promise<void> {
49+ if (response.status < 500 && response.status !== 429) return;
50+ const body = await response.text().catch(() => "");
51+ throw new Error(`${what} answered ${response.status}${body ? `: ${body.slice(0, 200)}` : ""}`);
52+}
53+
54+/**
55+ * Whether a sandbox whose `sleepAfter` has passed is still inside its run's
56+ * time cap, and so keeps running: the cap, not inactivity, ends a run (the
57+ * runner never fetches the container, so to the library every sandbox looks
58+ * idle). Without a cap or a start time, `sleepAfter` applies.
59+ */
60+export function withinTimeCap(
61+ started: number | null | undefined,
62+ capMinutes: number | null | undefined,
63+ now: number,
64+ graceSeconds: number,
65+): boolean {
66+ if (typeof started !== "number" || typeof capMinutes !== "number" || capMinutes <= 0) return false;
67+ return now < started + (capMinutes * 60 + graceSeconds) * 1000;
68+}
69+
70+/** An error as one line, for a log: its message, or what was thrown. */
71+export function describeError(error: unknown): string {
72+ if (error instanceof Error) return error.message || error.name;
73+ if (error === undefined) return "no reason given";
74+ if (typeof error === "string") return error;
75+ try {
76+ return JSON.stringify(error) ?? String(error);
77+ } catch {
78+ return String(error);
79+ }
80+}