Commit

Merge branch 'worktree-agent-ac5b181a013e54348'

# Conflicts: # deploy/stack.jsonc # docs/ARTIFACTS.md # docs/SELF_HOSTING.md

syntaqxcommitted Parents09230898cfa64aBrowse files
42 files+3365−7410/42 viewed
+15−0
941941 ]
942942
943943 [[package]]
944+name = "g1t-blobstore"
945+version = "0.1.0"
946+dependencies = [
947+ "g1t-contracts",
948+ "g1t-kit",
949+ "hex",
950+ "hmac 0.12.1",
951+ "serde",
952+ "sha2 0.10.9",
953+ "worker",
954+]
955+
956+[[package]]
944957 name = "g1t-contracts"
945958 version = "0.1.0"
946959 dependencies = [
10091022 dependencies = [
10101023 "base64 0.22.1",
10111024 "futures-util",
1025+ "g1t-blobstore",
10121026 "g1t-contracts",
10131027 "g1t-kit",
10141028 "hex",
10271041 dependencies = [
10281042 "base64 0.22.1",
10291043 "futures-util",
1044+ "g1t-blobstore",
10301045 "g1t-contracts",
10311046 "g1t-kit",
10321047 "g1t-scan",
+1−0
99
1010 [workspace.dependencies]
1111 g1t-actions = { path = "crates/actions" }
12+g1t-blobstore = { path = "crates/blobstore" }
1213 g1t-contracts = { path = "crates/contracts" }
1314 g1t-kit = { path = "crates/kit" }
1415 g1t-scan = { path = "crates/scan" }
+64−0
323323 }
324324 }
325325
326+/// A backup's sandbox, passed on to the repos service, which holds the
327+/// job (services/repos/src/backups.rs; the flow is in
328+/// `g1t_contracts::backups`):
329+///
330+/// - `POST /backups/{job}/spec`: what to cut, and a read-only git credential
331+/// - `PUT /backups/{job}/parts/{n}`: one part of the bundle, as bytes
332+/// - `POST /backups/{job}/complete` with `{ refs, size, sha256, parts, fetched_bytes }`
333+/// - `POST /backups/{job}/fail` with `{ error, fetched_bytes }`
334+///
335+/// Bodies are passed through as they are: snake_case already, and a
336+/// bundle's refs are keyed by ref names, which must not be converted.
337+async fn backup_job(request: &mut Request, services: &Services, method: &str, path: &str) -> Result<Response> {
338+ use g1t_contracts::backups::TOKEN_HEADER;
339+ let token = request.headers().get(TOKEN_HEADER)?.unwrap_or_default();
340+ let rest = path.trim_start_matches("/backups/");
341+ let (job, action) = rest.split_once('/').unwrap_or((rest, ""));
342+ if job.is_empty() || token.is_empty() {
343+ return fail(FailureCode::Unauthenticated, "A backup job's token is required.");
344+ }
345+ if method == "PUT" && action.starts_with("parts/") {
346+ let bytes = request.bytes().await?;
347+ if bytes.len() as u64 > g1t_contracts::backups::PART_BYTES {
348+ return fail(FailureCode::Invalid, "A part holds 32 MiB at most.");
349+ }
350+ let headers = worker::Headers::new();
351+ headers.set(TOKEN_HEADER, &token)?;
352+ let mut init = worker::RequestInit::new();
353+ init.with_method(Method::Put)
354+ .with_headers(headers)
355+ .with_body(Some(worker::js_sys::Uint8Array::from(bytes.as_slice()).into()));
356+ let forwarded = Request::new_with_init(&format!("https://repos/backups/{job}/{action}"), &init)?;
357+ let mut answered = services.repos.fetch_request(forwarded).await?;
358+ return outcome_as_given(answered.json().await?);
359+ }
360+ let rpc = match (method, action) {
361+ ("POST", "spec") => "backup_spec",
362+ ("POST", "complete") => "backup_complete",
363+ ("POST", "fail") => "backup_fail",
364+ _ => return fail(FailureCode::NotFound, "No such endpoint."),
365+ };
366+ let mut body = json_body(request).await;
367+ if !body.is_object() {
368+ body = json!({});
369+ }
370+ body["job_id"] = json!(job);
371+ body["token"] = json!(token);
372+ let answered: Value = g1t_kit::call(&services.repos, rpc, &body).await?;
373+ outcome_as_given(answered)
374+}
375+
376+/// An `Outcome` from a service whose keys are already the API's: the value,
377+/// or the failure in the shape every endpoint uses.
378+fn outcome_as_given(answered: Value) -> Result<Response> {
379+ match serde_json::from_value::<Outcome<Value>>(answered)? {
380+ Outcome::Ok(value) => Response::from_json(&value),
381+ Outcome::Fail(refused) => failure(&refused),
382+ }
383+}
384+
326385 /// A sandbox reporting the review its agent wrote. As with checks, the
327386 /// run's own token is the credential.
328387 async fn report_review(
565624 let pull_id = path.trim_start_matches("/mergechecks/").to_owned();
566625 return report_mergecheck(&mut request, &services, &pull_id).await;
567626 }
627+ // A sandbox making a repository's nightly backup. The job's own
628+ // token, in its header, is the credential.
629+ (method, path) if path.starts_with("/backups/") => {
630+ return backup_job(&mut request, &services, method, path).await;
631+ }
568632 ("POST", path) if path.starts_with("/queue/") => {
569633 let entry_id = path.trim_start_matches("/queue/").to_owned();
570634 return report_queue(&mut request, &services, &entry_id).await;
+2−1
108108 | `WAITLIST_NOTIFY_EMAIL` | (none) | Where a summary of new access requests goes, at most every 15 minutes. Empty sends none; requests still wait for you in the database. |
109109 | `S3_ENDPOINT`, `S3_BUCKET`, `S3_REGION`, `S3_ACCESS_KEY_ID`, `S3_SECRET_ACCESS_KEY` | the bundled MinIO, bucket `g1t-packages` | Where packages' files are kept: any S3-compatible store. Change the two keys before first start; MinIO is made with them. |
110110 | `S3_PUBLIC_ENDPOINT` | (none) | The store's address as clients reach it. When set, large layers are downloaded from it directly with a signed URL. |
111+| `BACKUP_S3_BUCKET` | `g1t-backups` | The bucket on the same store that nightly repository backups (a `git bundle` of each repository whose branches or tags changed) are kept in. The bundles are cut by g1t's runner, which this installation does not run yet, so the bucket stays empty for now: copy the volumes, as below. |
111112 | `STATUS_PORT` | `8788` | The port the status page is published on |
112113 | `STATUS_PROBE_REPO` | (none) | A public repository, `workspace/repo`, whose branches the status page lists every minute as a clone would. Empty: git is not checked. |
113114 | `INVITE_STAFF_WORKSPACES` | (none) | Workspace slugs, comma separated, whose owners can make invites without a limit. Set it to your own workspace before you switch to `invite`, so someone can invite the first people. |
186187 | --- | --- |
187188 | `g1t_g1t-data` | Accounts, workspaces, issues and every other record, as SQLite files; the keys that seal stored secrets (`keys.env`) |
188189 | `g1t_g1t-git` | Your repositories, one bare git repository each |
189−| `g1t_g1t-packages` | Container images' layers and other package files (MinIO) |
190+| `g1t_g1t-packages` | Container images' layers and other package files, and the `g1t-backups` bucket (MinIO) |
190191 | `g1t_g1t-secrets` | The key the site and the git store share |
191192
192193 To back up, stop g1t and copy the volumes:
+15−0
1+[package]
2+name = "g1t-blobstore"
3+version = "0.1.0"
4+edition.workspace = true
5+license.workspace = true
6+description = "Object storage behind one port: R2 on Cloudflare, any S3-compatible store (MinIO) when self-hosted."
7+
8+[dependencies]
9+g1t-contracts.workspace = true
10+g1t-kit.workspace = true
11+serde.workspace = true
12+worker.workspace = true
13+hex = "0.4"
14+hmac = "0.12"
15+sha2 = { version = "0.10", features = ["compress"] }
+192−0
1+//! Object storage behind one port, `BlobStore`: R2 on Cloudflare, and any
2+//! S3-compatible store (MinIO in the self-host compose file) elsewhere.
3+//!
4+//! Every service that keeps objects names its own [`Config`]: the variable
5+//! that chooses the store (`r2`, the default, or `s3`), the R2 bucket
6+//! binding, and the variable naming its S3 bucket. The S3 endpoint and its
7+//! credentials (S3_ENDPOINT, S3_REGION, S3_ACCESS_KEY_ID,
8+//! S3_SECRET_ACCESS_KEY) are the installation's, shared by every service.
9+//!
10+//! Large objects go up as multipart parts of one size, as R2 requires
11+//! (every part but the last the same size).
12+
13+mod r2;
14+mod s3;
15+pub mod sigv4;
16+
17+use serde::{Deserialize, Serialize};
18+use worker::{Env, Response, ResponseBody, Result};
19+
20+pub use r2::R2Store;
21+pub use s3::S3Store;
22+
23+/// A part of an object a read asks for: `length` bytes from `offset`.
24+#[derive(Clone, Copy, Debug, PartialEq, Eq)]
25+pub struct Wanted {
26+ pub offset: u64,
27+ pub length: u64,
28+}
29+
30+impl Wanted {
31+ /// `bytes <first>-<last>/<size>`.
32+ pub fn content_range(&self, size: u64) -> String {
33+ format!("bytes {}-{}/{size}", self.offset, self.offset + self.length - 1)
34+ }
35+}
36+
37+/// One part of a multipart upload, as completing it needs.
38+#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
39+pub struct Part {
40+ pub number: u16,
41+ pub etag: String,
42+}
43+
44+/// An object read back.
45+pub struct Got {
46+ /// The whole object's size, whatever range was read.
47+ pub size: u64,
48+ pub body: ResponseBody,
49+}
50+
51+impl Got {
52+ pub async fn bytes(self) -> Result<Vec<u8>> {
53+ match self.body {
54+ ResponseBody::Empty => Ok(Vec::new()),
55+ ResponseBody::Body(bytes) => Ok(bytes),
56+ stream => Response::from_body(stream)?.bytes().await,
57+ }
58+ }
59+}
60+
61+/// What a service needs of storage.
62+#[allow(async_fn_in_trait)]
63+pub trait BlobStore {
64+ async fn put(&self, key: &str, bytes: Vec<u8>) -> Result<()>;
65+ /// The object, or the part of it `range` asks for.
66+ async fn get(&self, key: &str, range: Option<Wanted>) -> Result<Option<Got>>;
67+ /// The object's size, if it is there.
68+ async fn head(&self, key: &str) -> Result<Option<u64>>;
69+ async fn delete(&self, key: &str) -> Result<()>;
70+ /// Starts a multipart upload to `key`, and says its id.
71+ async fn create_multipart(&self, key: &str) -> Result<String>;
72+ async fn upload_part(&self, key: &str, upload_id: &str, number: u16, bytes: Vec<u8>) -> Result<Part>;
73+ async fn complete_multipart(&self, key: &str, upload_id: &str, parts: &[Part]) -> Result<()>;
74+ async fn abort_multipart(&self, key: &str, upload_id: &str) -> Result<()>;
75+ /// A URL that downloads the object for `expires` seconds without
76+ /// passing through this Worker, when the store can sign one.
77+ fn presign_get(&self, key: &str, expires: u32, now_ms: u64) -> Option<String>;
78+
79+ /// The whole object, read into memory: for small ones only.
80+ async fn read(&self, key: &str) -> Result<Option<Vec<u8>>> {
81+ match self.get(key, None).await? {
82+ Some(got) => Ok(Some(got.bytes().await?)),
83+ None => Ok(None),
84+ }
85+ }
86+}
87+
88+/// Where one service's objects are, by the names of its bindings and
89+/// variables.
90+#[derive(Clone, Copy, Debug)]
91+pub struct Config {
92+ /// The variable that chooses the store: `r2` (or unset) or `s3`.
93+ pub kind: &'static str,
94+ /// The R2 bucket binding.
95+ pub binding: &'static str,
96+ /// The variables that let R2's S3 endpoint sign download URLs: access
97+ /// key id, secret, account id and bucket name. None: never signed.
98+ pub r2_signer: Option<[&'static str; 4]>,
99+ /// The variable naming the S3 bucket.
100+ pub s3_bucket: &'static str,
101+ /// The variable naming where clients reach the S3 store, for signed
102+ /// downloads. None: never signed.
103+ pub s3_public_endpoint: Option<&'static str>,
104+}
105+
106+/// The store a service is configured with.
107+pub enum Store {
108+ R2(R2Store),
109+ S3(S3Store),
110+}
111+
112+impl Store {
113+ pub fn from_env(env: &Env, config: &Config) -> Result<Store> {
114+ if var(env, config.kind) == "s3" {
115+ return Ok(Store::S3(S3Store::from_env(env, config)?));
116+ }
117+ Ok(Store::R2(R2Store::from_env(env, config)?))
118+ }
119+}
120+
121+impl BlobStore for Store {
122+ async fn put(&self, key: &str, bytes: Vec<u8>) -> Result<()> {
123+ match self {
124+ Store::R2(s) => s.put(key, bytes).await,
125+ Store::S3(s) => s.put(key, bytes).await,
126+ }
127+ }
128+
129+ async fn get(&self, key: &str, range: Option<Wanted>) -> Result<Option<Got>> {
130+ match self {
131+ Store::R2(s) => s.get(key, range).await,
132+ Store::S3(s) => s.get(key, range).await,
133+ }
134+ }
135+
136+ async fn head(&self, key: &str) -> Result<Option<u64>> {
137+ match self {
138+ Store::R2(s) => s.head(key).await,
139+ Store::S3(s) => s.head(key).await,
140+ }
141+ }
142+
143+ async fn delete(&self, key: &str) -> Result<()> {
144+ match self {
145+ Store::R2(s) => s.delete(key).await,
146+ Store::S3(s) => s.delete(key).await,
147+ }
148+ }
149+
150+ async fn create_multipart(&self, key: &str) -> Result<String> {
151+ match self {
152+ Store::R2(s) => s.create_multipart(key).await,
153+ Store::S3(s) => s.create_multipart(key).await,
154+ }
155+ }
156+
157+ async fn upload_part(&self, key: &str, upload_id: &str, number: u16, bytes: Vec<u8>) -> Result<Part> {
158+ match self {
159+ Store::R2(s) => s.upload_part(key, upload_id, number, bytes).await,
160+ Store::S3(s) => s.upload_part(key, upload_id, number, bytes).await,
161+ }
162+ }
163+
164+ async fn complete_multipart(&self, key: &str, upload_id: &str, parts: &[Part]) -> Result<()> {
165+ match self {
166+ Store::R2(s) => s.complete_multipart(key, upload_id, parts).await,
167+ Store::S3(s) => s.complete_multipart(key, upload_id, parts).await,
168+ }
169+ }
170+
171+ async fn abort_multipart(&self, key: &str, upload_id: &str) -> Result<()> {
172+ match self {
173+ Store::R2(s) => s.abort_multipart(key, upload_id).await,
174+ Store::S3(s) => s.abort_multipart(key, upload_id).await,
175+ }
176+ }
177+
178+ fn presign_get(&self, key: &str, expires: u32, now_ms: u64) -> Option<String> {
179+ match self {
180+ Store::R2(s) => s.presign_get(key, expires, now_ms),
181+ Store::S3(s) => s.presign_get(key, expires, now_ms),
182+ }
183+ }
184+}
185+
186+/// A configuration variable or secret, or the empty string.
187+pub fn var(env: &Env, name: &str) -> String {
188+ env.var(name)
189+ .map(|v| v.to_string())
190+ .or_else(|_| env.secret(name).map(|v| v.to_string()))
191+ .unwrap_or_default()
192+}
+90−0
1+//! The R2 adapter: the service's bucket binding for everything, and R2's S3
2+//! endpoint only to sign download URLs, when the service names signing
3+//! variables (`Config::r2_signer`) and they are set. Without them, large
4+//! objects stream through the Worker like small ones.
5+
6+use worker::{Bucket, Env, Range, Result, UploadedPart};
7+
8+use crate::sigv4::{Credentials, amz_date};
9+use crate::{BlobStore, Config, Got, Part, Wanted, var};
10+
11+pub struct R2Store {
12+ bucket: Bucket,
13+ signer: Option<(Credentials, String, String)>,
14+}
15+
16+impl R2Store {
17+ pub fn from_env(env: &Env, config: &Config) -> Result<R2Store> {
18+ let [key, secret, account, bucket] = config
19+ .r2_signer
20+ .map(|names| names.map(|name| var(env, name)))
21+ .unwrap_or_default();
22+ let signer = (!key.is_empty() && !secret.is_empty() && !account.is_empty() && !bucket.is_empty()).then(|| {
23+ (
24+ Credentials { access_key_id: key, secret_access_key: secret, region: "auto".to_owned() },
25+ format!("{account}.r2.cloudflarestorage.com"),
26+ bucket,
27+ )
28+ });
29+ Ok(R2Store { bucket: env.bucket(config.binding)?, signer })
30+ }
31+}
32+
33+impl BlobStore for R2Store {
34+ async fn put(&self, key: &str, bytes: Vec<u8>) -> Result<()> {
35+ self.bucket.put(key, bytes).execute().await?;
36+ Ok(())
37+ }
38+
39+ async fn get(&self, key: &str, range: Option<Wanted>) -> Result<Option<Got>> {
40+ let mut get = self.bucket.get(key);
41+ if let Some(range) = range {
42+ get = get.range(Range::OffsetWithLength { offset: range.offset, length: range.length });
43+ }
44+ let Some(object) = get.execute().await? else {
45+ return Ok(None);
46+ };
47+ let size = object.size();
48+ let Some(body) = object.body() else {
49+ return Ok(None);
50+ };
51+ Ok(Some(Got { size, body: body.response_body()? }))
52+ }
53+
54+ async fn head(&self, key: &str) -> Result<Option<u64>> {
55+ Ok(self.bucket.head(key).await?.map(|object| object.size()))
56+ }
57+
58+ async fn delete(&self, key: &str) -> Result<()> {
59+ self.bucket.delete(key).await
60+ }
61+
62+ async fn create_multipart(&self, key: &str) -> Result<String> {
63+ let upload = self.bucket.create_multipart_upload(key).execute().await?;
64+ Ok(upload.upload_id().await)
65+ }
66+
67+ async fn upload_part(&self, key: &str, upload_id: &str, number: u16, bytes: Vec<u8>) -> Result<Part> {
68+ let upload = self.bucket.resume_multipart_upload(key, upload_id)?;
69+ let part = upload.upload_part(number, bytes).await?;
70+ Ok(Part { number: part.part_number(), etag: part.etag() })
71+ }
72+
73+ async fn complete_multipart(&self, key: &str, upload_id: &str, parts: &[Part]) -> Result<()> {
74+ let upload = self.bucket.resume_multipart_upload(key, upload_id)?;
75+ upload
76+ .complete(parts.iter().map(|part| UploadedPart::new(part.number, part.etag.clone())))
77+ .await?;
78+ Ok(())
79+ }
80+
81+ async fn abort_multipart(&self, key: &str, upload_id: &str) -> Result<()> {
82+ self.bucket.resume_multipart_upload(key, upload_id)?.abort().await
83+ }
84+
85+ fn presign_get(&self, key: &str, expires: u32, now_ms: u64) -> Option<String> {
86+ let (credentials, host, bucket) = self.signer.as_ref()?;
87+ let path = format!("/{bucket}/{key}");
88+ Some(credentials.presign_get(&format!("https://{host}"), host, &path, &amz_date(now_ms), expires))
89+ }
90+}
+255−0
1+//! The S3 adapter, for self-hosted installations: any S3-compatible store
2+//! (MinIO, Ceph, Garage, AWS) over fetch, signed with SigV4, path-style.
3+//! S3_ENDPOINT, S3_ACCESS_KEY_ID, S3_SECRET_ACCESS_KEY and S3_REGION say
4+//! where, and the service's own variable (`Config::s3_bucket`) which
5+//! bucket. Its public endpoint variable, when it names one and that is
6+//! set, is the address clients reach the store at, and large downloads are
7+//! then sent there with a signed URL instead of through the Worker.
8+
9+use worker::wasm_bindgen::JsValue;
10+use worker::{Env, Fetch, Headers, Method, Request, RequestInit, Response, Result, Url};
11+
12+use crate::sigv4::{Credentials, UNSIGNED, amz_date};
13+use crate::{BlobStore, Config, Got, Part, Wanted, var};
14+
15+pub struct S3Store {
16+ /// `http://minio:9000`, without a trailing slash.
17+ endpoint: String,
18+ /// Where clients reach the same store, for signed URLs.
19+ public_endpoint: Option<String>,
20+ bucket: String,
21+ credentials: Credentials,
22+}
23+
24+fn failed(what: &str, status: u16, body: &str) -> worker::Error {
25+ let said: String = body.chars().take(300).collect();
26+ worker::Error::RustError(format!("storage {what} failed with status {status}: {said}"))
27+}
28+
29+/// The text of the first `<tag>` in an XML answer.
30+fn xml_value<'a>(xml: &'a str, tag: &str) -> Option<&'a str> {
31+ let open = format!("<{tag}>");
32+ let start = xml.find(&open)? + open.len();
33+ let end = xml[start..].find(&format!("</{tag}>"))? + start;
34+ Some(&xml[start..end])
35+}
36+
37+fn host_of(endpoint: &str) -> String {
38+ endpoint
39+ .split_once("://")
40+ .map_or(endpoint, |(_, rest)| rest)
41+ .split('/')
42+ .next()
43+ .unwrap_or_default()
44+ .to_owned()
45+}
46+
47+impl S3Store {
48+ pub fn from_env(env: &Env, config: &Config) -> Result<S3Store> {
49+ let endpoint = var(env, "S3_ENDPOINT").trim_end_matches('/').to_owned();
50+ let bucket = var(env, config.s3_bucket);
51+ if endpoint.is_empty() || bucket.is_empty() {
52+ return Err(worker::Error::RustError(format!(
53+ "{} is s3, but S3_ENDPOINT or {} is not set",
54+ config.kind, config.s3_bucket
55+ )));
56+ }
57+ let region = var(env, "S3_REGION");
58+ let public = config
59+ .s3_public_endpoint
60+ .map(|name| var(env, name).trim_end_matches('/').to_owned())
61+ .unwrap_or_default();
62+ Ok(S3Store {
63+ endpoint,
64+ public_endpoint: (!public.is_empty()).then_some(public),
65+ bucket,
66+ credentials: Credentials {
67+ access_key_id: var(env, "S3_ACCESS_KEY_ID"),
68+ secret_access_key: var(env, "S3_SECRET_ACCESS_KEY"),
69+ region: if region.is_empty() { "us-east-1".to_owned() } else { region },
70+ },
71+ })
72+ }
73+
74+ fn path(&self, key: &str) -> String {
75+ format!("/{}/{key}", self.bucket)
76+ }
77+
78+ /// Sends one signed request, and answers with the response whatever
79+ /// its status.
80+ async fn send(
81+ &self,
82+ method: Method,
83+ key: &str,
84+ query: &[(String, String)],
85+ extra: &[(&str, String)],
86+ body: Option<Vec<u8>>,
87+ ) -> Result<Response> {
88+ let path = self.path(key);
89+ let date = amz_date(g1t_kit::now_ms());
90+ let mut signed = vec![
91+ ("host".to_owned(), host_of(&self.endpoint)),
92+ ("x-amz-content-sha256".to_owned(), UNSIGNED.to_owned()),
93+ ("x-amz-date".to_owned(), date),
94+ ];
95+ for (name, value) in extra {
96+ signed.push(((*name).to_owned(), value.clone()));
97+ }
98+ let authorization = self
99+ .credentials
100+ .authorization(method.as_ref(), &path, query, &signed, UNSIGNED);
101+ let headers = Headers::new();
102+ for (name, value) in &signed {
103+ if name != "host" {
104+ headers.set(name, value)?;
105+ }
106+ }
107+ headers.set("authorization", &authorization)?;
108+ let mut url = Url::parse(&format!("{}{}", self.endpoint, crate::sigv4::uri_encode(&path, true)))?;
109+ if !query.is_empty() {
110+ let text: Vec<String> = query
111+ .iter()
112+ .map(|(k, v)| {
113+ let (k, v) = (crate::sigv4::uri_encode(k, false), crate::sigv4::uri_encode(v, false));
114+ if v.is_empty() { format!("{k}=") } else { format!("{k}={v}") }
115+ })
116+ .collect();
117+ url.set_query(Some(&text.join("&")));
118+ }
119+ let mut init = RequestInit::new();
120+ init.with_method(method).with_headers(headers);
121+ if let Some(body) = body {
122+ init.with_body(Some(JsValue::from(worker::js_sys::Uint8Array::from(body.as_slice()))));
123+ }
124+ Fetch::Request(Request::new_with_init(url.as_str(), &init)?).send().await
125+ }
126+
127+ async fn ok(&self, what: &str, mut response: Response) -> Result<Response> {
128+ let status = response.status_code();
129+ if (200..300).contains(&status) {
130+ return Ok(response);
131+ }
132+ let body = response.text().await.unwrap_or_default();
133+ Err(failed(what, status, &body))
134+ }
135+}
136+
137+fn query(pairs: &[(&str, &str)]) -> Vec<(String, String)> {
138+ pairs.iter().map(|(k, v)| ((*k).to_owned(), (*v).to_owned())).collect()
139+}
140+
141+impl BlobStore for S3Store {
142+ async fn put(&self, key: &str, bytes: Vec<u8>) -> Result<()> {
143+ let length = bytes.len().to_string();
144+ let response = self
145+ .send(Method::Put, key, &[], &[("content-length", length)], Some(bytes))
146+ .await?;
147+ self.ok("put", response).await.map(|_| ())
148+ }
149+
150+ async fn get(&self, key: &str, range: Option<Wanted>) -> Result<Option<Got>> {
151+ let extra: Vec<(&str, String)> = range
152+ .map(|r| ("range", format!("bytes={}-{}", r.offset, r.offset + r.length - 1)))
153+ .into_iter()
154+ .collect();
155+ let response = self.send(Method::Get, key, &[], &extra, None).await?;
156+ if response.status_code() == 404 {
157+ return Ok(None);
158+ }
159+ let response = self.ok("get", response).await?;
160+ let size = match response.headers().get("content-range")? {
161+ // `bytes 0-9/100`: the whole object's size is after the slash.
162+ Some(range) => range.rsplit('/').next().and_then(|n| n.parse().ok()).unwrap_or(0),
163+ None => response.headers().get("content-length")?.and_then(|n| n.parse().ok()).unwrap_or(0),
164+ };
165+ let (_, body) = response.into_parts();
166+ Ok(Some(Got { size, body }))
167+ }
168+
169+ async fn head(&self, key: &str) -> Result<Option<u64>> {
170+ let response = self.send(Method::Head, key, &[], &[], None).await?;
171+ if response.status_code() == 404 {
172+ return Ok(None);
173+ }
174+ let response = self.ok("head", response).await?;
175+ Ok(response.headers().get("content-length")?.and_then(|n| n.parse().ok()))
176+ }
177+
178+ async fn delete(&self, key: &str) -> Result<()> {
179+ let response = self.send(Method::Delete, key, &[], &[], None).await?;
180+ if response.status_code() == 404 {
181+ return Ok(());
182+ }
183+ self.ok("delete", response).await.map(|_| ())
184+ }
185+
186+ async fn create_multipart(&self, key: &str) -> Result<String> {
187+ let response = self.send(Method::Post, key, &query(&[("uploads", "")]), &[], None).await?;
188+ let text = self.ok("create multipart", response).await?.text().await?;
189+ xml_value(&text, "UploadId")
190+ .map(str::to_owned)
191+ .ok_or_else(|| failed("create multipart", 200, &text))
192+ }
193+
194+ async fn upload_part(&self, key: &str, upload_id: &str, number: u16, bytes: Vec<u8>) -> Result<Part> {
195+ let number_text = number.to_string();
196+ let length = bytes.len().to_string();
197+ let response = self
198+ .send(
199+ Method::Put,
200+ key,
201+ &query(&[("partNumber", &number_text), ("uploadId", upload_id)]),
202+ &[("content-length", length)],
203+ Some(bytes),
204+ )
205+ .await?;
206+ let response = self.ok("upload part", response).await?;
207+ let etag = response.headers().get("etag")?.unwrap_or_default();
208+ Ok(Part { number, etag })
209+ }
210+
211+ async fn complete_multipart(&self, key: &str, upload_id: &str, parts: &[Part]) -> Result<()> {
212+ let mut xml = String::from("<CompleteMultipartUpload>");
213+ for part in parts {
214+ xml.push_str(&format!("<Part><PartNumber>{}</PartNumber><ETag>{}</ETag></Part>", part.number, part.etag));
215+ }
216+ xml.push_str("</CompleteMultipartUpload>");
217+ let length = xml.len().to_string();
218+ let response = self
219+ .send(Method::Post, key, &query(&[("uploadId", upload_id)]), &[("content-length", length)], Some(xml.into_bytes()))
220+ .await?;
221+ // S3 may answer 200 and still have failed, saying so in the body.
222+ let text = self.ok("complete multipart", response).await?.text().await?;
223+ if text.contains("<Error>") {
224+ return Err(failed("complete multipart", 200, &text));
225+ }
226+ Ok(())
227+ }
228+
229+ async fn abort_multipart(&self, key: &str, upload_id: &str) -> Result<()> {
230+ let response = self.send(Method::Delete, key, &query(&[("uploadId", upload_id)]), &[], None).await?;
231+ if response.status_code() == 404 {
232+ return Ok(());
233+ }
234+ self.ok("abort multipart", response).await.map(|_| ())
235+ }
236+
237+ fn presign_get(&self, key: &str, expires: u32, now_ms: u64) -> Option<String> {
238+ let base = self.public_endpoint.as_ref()?;
239+ Some(self.credentials.presign_get(base, &host_of(base), &self.path(key), &amz_date(now_ms), expires))
240+ }
241+}
242+
243+#[cfg(test)]
244+mod tests {
245+ use super::*;
246+
247+ #[test]
248+ fn answers_are_read_from_their_xml() {
249+ let xml = "<InitiateMultipartUploadResult><Bucket>b</Bucket><UploadId>abc-123</UploadId></InitiateMultipartUploadResult>";
250+ assert_eq!(xml_value(xml, "UploadId"), Some("abc-123"));
251+ assert_eq!(xml_value(xml, "Key"), None);
252+ assert_eq!(host_of("http://minio:9000"), "minio:9000");
253+ assert_eq!(host_of("https://s3.example.com/base"), "s3.example.com");
254+ }
255+}
+201−0
1+//! AWS Signature Version 4, for S3-compatible storage: signing a request's
2+//! headers, and signing a URL that lets its holder download one object for
3+//! a while. R2's S3 endpoint takes the same signatures, which is how large
4+//! downloads are sent straight to it.
5+
6+use hmac::{Hmac, Mac};
7+use sha2::{Digest as _, Sha256};
8+
9+type HmacSha256 = Hmac<Sha256>;
10+
11+/// The hash of a payload that is not signed: bodies stream as they are.
12+pub const UNSIGNED: &str = "UNSIGNED-PAYLOAD";
13+
14+/// Who signs, and for which region.
15+#[derive(Clone, Debug)]
16+pub struct Credentials {
17+ pub access_key_id: String,
18+ pub secret_access_key: String,
19+ pub region: String,
20+}
21+
22+/// `20130524T000000Z` from milliseconds since the epoch.
23+pub fn amz_date(now_ms: u64) -> String {
24+ let text = g1t_contracts::time::rfc3339(now_ms);
25+ let whole = text.split('.').next().unwrap_or(&text);
26+ format!("{}Z", whole.replace(['-', ':'], ""))
27+}
28+
29+/// Percent-encodes everything but the unreserved characters, and `/` too
30+/// unless `path`.
31+pub fn uri_encode(text: &str, path: bool) -> String {
32+ let mut out = String::with_capacity(text.len());
33+ for byte in text.bytes() {
34+ match byte {
35+ b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'.' | b'_' | b'~' => out.push(byte as char),
36+ b'/' if path => out.push('/'),
37+ _ => out.push_str(&format!("%{byte:02X}")),
38+ }
39+ }
40+ out
41+}
42+
43+fn hmac(key: &[u8], data: &str) -> Vec<u8> {
44+ let mut mac = HmacSha256::new_from_slice(key).expect("HMAC takes a key of any length");
45+ mac.update(data.as_bytes());
46+ mac.finalize().into_bytes().to_vec()
47+}
48+
49+fn sha256_hex(data: &[u8]) -> String {
50+ hex::encode(Sha256::digest(data))
51+}
52+
53+/// The query string, sorted and encoded as signing needs it.
54+fn canonical_query(query: &[(String, String)]) -> String {
55+ let mut pairs: Vec<(String, String)> = query
56+ .iter()
57+ .map(|(key, value)| (uri_encode(key, false), uri_encode(value, false)))
58+ .collect();
59+ pairs.sort();
60+ pairs
61+ .iter()
62+ .map(|(key, value)| format!("{key}={value}"))
63+ .collect::<Vec<_>>()
64+ .join("&")
65+}
66+
67+impl Credentials {
68+ fn scope(&self, date: &str) -> String {
69+ format!("{}/{}/s3/aws4_request", &date[..8], self.region)
70+ }
71+
72+ fn signature(&self, date: &str, canonical_request: &str) -> String {
73+ let to_sign = format!(
74+ "AWS4-HMAC-SHA256\n{date}\n{}\n{}",
75+ self.scope(date),
76+ sha256_hex(canonical_request.as_bytes())
77+ );
78+ let key = hmac(format!("AWS4{}", self.secret_access_key).as_bytes(), &date[..8]);
79+ let key = hmac(&key, &self.region);
80+ let key = hmac(&key, "s3");
81+ let key = hmac(&key, "aws4_request");
82+ hex::encode(hmac(&key, &to_sign))
83+ }
84+
85+ /// The `Authorization` header for a request. `headers` must include
86+ /// `host`, `x-amz-date` and `x-amz-content-sha256`, lowercase; every
87+ /// one given is signed.
88+ pub fn authorization(
89+ &self,
90+ method: &str,
91+ path: &str,
92+ query: &[(String, String)],
93+ headers: &[(String, String)],
94+ payload_hash: &str,
95+ ) -> String {
96+ let mut headers: Vec<(String, String)> = headers
97+ .iter()
98+ .map(|(name, value)| (name.to_ascii_lowercase(), value.trim().to_owned()))
99+ .collect();
100+ headers.sort();
101+ let date = headers
102+ .iter()
103+ .find(|(name, _)| name == "x-amz-date")
104+ .map(|(_, value)| value.clone())
105+ .unwrap_or_default();
106+ let signed: Vec<&str> = headers.iter().map(|(name, _)| name.as_str()).collect();
107+ let signed = signed.join(";");
108+ let canonical_headers: String = headers.iter().map(|(name, value)| format!("{name}:{value}\n")).collect();
109+ let canonical = format!(
110+ "{method}\n{}\n{}\n{canonical_headers}\n{signed}\n{payload_hash}",
111+ uri_encode(path, true),
112+ canonical_query(query)
113+ );
114+ format!(
115+ "AWS4-HMAC-SHA256 Credential={}/{}, SignedHeaders={signed}, Signature={}",
116+ self.access_key_id,
117+ self.scope(&date),
118+ self.signature(&date, &canonical)
119+ )
120+ }
121+
122+ /// A URL that lets anyone `GET` the object at `path` on `host` for
123+ /// `expires` seconds from `date`. `base` is the scheme and host the
124+ /// URL starts with.
125+ pub fn presign_get(&self, base: &str, host: &str, path: &str, date: &str, expires: u32) -> String {
126+ let mut query = vec![
127+ ("X-Amz-Algorithm".to_owned(), "AWS4-HMAC-SHA256".to_owned()),
128+ ("X-Amz-Credential".to_owned(), format!("{}/{}", self.access_key_id, self.scope(date))),
129+ ("X-Amz-Date".to_owned(), date.to_owned()),
130+ ("X-Amz-Expires".to_owned(), expires.to_string()),
131+ ("X-Amz-SignedHeaders".to_owned(), "host".to_owned()),
132+ ];
133+ let canonical = format!(
134+ "GET\n{}\n{}\nhost:{host}\n\nhost\n{UNSIGNED}",
135+ uri_encode(path, true),
136+ canonical_query(&query)
137+ );
138+ query.push(("X-Amz-Signature".to_owned(), self.signature(date, &canonical)));
139+ format!("{base}{}?{}", uri_encode(path, true), canonical_query(&query))
140+ }
141+}
142+
143+#[cfg(test)]
144+mod tests {
145+ use super::*;
146+
147+ fn example() -> Credentials {
148+ Credentials {
149+ access_key_id: "AKIAIOSFODNN7EXAMPLE".into(),
150+ secret_access_key: "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY".into(),
151+ region: "us-east-1".into(),
152+ }
153+ }
154+
155+ /// AWS's own example of a presigned URL (Authenticating Requests:
156+ /// Using Query Parameters).
157+ #[test]
158+ fn a_presigned_url_matches_the_aws_example() {
159+ let url = example().presign_get(
160+ "https://examplebucket.s3.amazonaws.com",
161+ "examplebucket.s3.amazonaws.com",
162+ "/test.txt",
163+ "20130524T000000Z",
164+ 86400,
165+ );
166+ assert!(url.starts_with("https://examplebucket.s3.amazonaws.com/test.txt?X-Amz-Algorithm=AWS4-HMAC-SHA256"));
167+ assert!(url.contains("X-Amz-Credential=AKIAIOSFODNN7EXAMPLE%2F20130524%2Fus-east-1%2Fs3%2Faws4_request"));
168+ assert!(url.contains("&X-Amz-Signature=aeeed9bbccd4d02ee5c0109b86d86835f995330da4c265957d157751f604d404&"), "{url}");
169+ }
170+
171+ /// AWS's own example of a signed GET with a range (Authenticating
172+ /// Requests: Using the Authorization Header).
173+ #[test]
174+ fn a_signed_request_matches_the_aws_example() {
175+ let empty = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855";
176+ let headers = [
177+ ("Host", "examplebucket.s3.amazonaws.com"),
178+ ("Range", "bytes=0-9"),
179+ ("x-amz-content-sha256", empty),
180+ ("x-amz-date", "20130524T000000Z"),
181+ ]
182+ .map(|(name, value)| (name.to_owned(), value.to_owned()));
183+ let authorization = example().authorization("GET", "/test.txt", &[], &headers, empty);
184+ assert_eq!(
185+ authorization,
186+ "AWS4-HMAC-SHA256 Credential=AKIAIOSFODNN7EXAMPLE/20130524/us-east-1/s3/aws4_request, \
187+ SignedHeaders=host;range;x-amz-content-sha256;x-amz-date, \
188+ Signature=f0e8bdb87c964420e857bd35b5d6ed310bd44f0170aba48dd91039c6036bdb41"
189+ );
190+ }
191+
192+ #[test]
193+ fn dates_and_encoding() {
194+ assert_eq!(amz_date(1_369_353_600_000), "20130524T000000Z");
195+ assert_eq!(uri_encode("a b/c+d~", true), "a%20b/c%2Bd~");
196+ assert_eq!(uri_encode("a/b", false), "a%2Fb");
197+ let query = [("uploadId".to_owned(), "x y".to_owned()), ("partNumber".to_owned(), "2".to_owned())];
198+ assert_eq!(canonical_query(&query), "partNumber=2&uploadId=x%20y");
199+ assert_eq!(canonical_query(&[("uploads".to_owned(), String::new())]), "uploads=");
200+ }
201+}
+138−0
1+//! Repository backups: a nightly `git bundle` of every repository whose
2+//! refs changed, kept outside the git store (docs/ARTIFACTS.md, R11).
3+//!
4+//! The repos service decides what is due and keeps the bundles and their
5+//! manifests (services/repos/src/backups.rs). It cannot run git, so the
6+//! bundle is cut where git runs: a sandbox the runner starts, which only
7+//! ever calls out, as a merge check does.
8+//!
9+//! 1. Each night the repos service queues the repositories whose refs
10+//! moved since their last backup.
11+//! 2. The runner's sweep claims a few at a time (`claim_backups`) and starts
12+//! a sandbox for each, with the job's id and token and nothing else.
13+//! 3. The sandbox asks for its job (`POST api.g1t.sh/backups/{job}/spec`):
14+//! a read-only git credential for the repository, minutes long, and the
15+//! commits the last bundle ended at. It clones, cuts the bundle, sends
16+//! it in parts (`PUT .../parts/{n}`), and says what it holds
17+//! (`POST .../complete`), or why it could not (`POST .../fail`).
18+//!
19+//! The job's token, in the `x-g1t-backup-token` header, is the only
20+//! credential the sandbox holds for g1t; it lasts as long as the job.
21+//!
22+//! What the runner and the repos service exchange is camelCase, as between
23+//! every service. What the sandbox sends and is sent is snake_case: it is
24+//! the API's.
25+
26+use std::collections::BTreeMap;
27+
28+use serde::{Deserialize, Serialize};
29+
30+use crate::repos::RepoPath;
31+
32+/// A bundle is sent in parts of this size; the last may be smaller.
33+pub const PART_BYTES: u64 = 32 * 1024 * 1024;
34+/// The header the sandbox sends its job's token in.
35+pub const TOKEN_HEADER: &str = "x-g1t-backup-token";
36+
37+/// `claim_backups`: up to `limit` queued backups, so long as no more than
38+/// `max_running` are then running. Returns `Vec<BackupClaim>`.
39+#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)]
40+#[serde(rename_all = "camelCase")]
41+pub struct ClaimBackupsArgs {
42+ pub limit: u32,
43+ pub max_running: u32,
44+}
45+
46+/// One backup to start.
47+#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
48+#[serde(rename_all = "camelCase")]
49+pub struct BackupClaim {
50+ pub job_id: String,
51+ /// Lets the sandbox, and nothing else, do this job.
52+ pub token: String,
53+ pub repo_id: String,
54+ /// Where the repository is now: for the sandbox's name and the logs.
55+ pub path: RepoPath,
56+}
57+
58+/// What kind of bundle a job cuts.
59+#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
60+#[serde(rename_all = "snake_case")]
61+pub enum BackupKind {
62+ /// Everything the repository has.
63+ Full,
64+ /// What is new since the last bundle: its prerequisites are the commits
65+ /// the last bundle's refs pointed to.
66+ Incremental,
67+}
68+
69+impl BackupKind {
70+ /// How a bundle's file name says what it is.
71+ pub fn suffix(self) -> &'static str {
72+ match self {
73+ BackupKind::Full => "full",
74+ BackupKind::Incremental => "incr",
75+ }
76+ }
77+}
78+
79+/// `backup_spec`, `backup_part` and the job's other calls: which job, and
80+/// its token.
81+#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
82+pub struct BackupJobArgs {
83+ pub job_id: String,
84+ pub token: String,
85+}
86+
87+/// The job, as the sandbox is given it. `Outcome<BackupSpec>`.
88+#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
89+pub struct BackupSpec {
90+ pub kind: BackupKind,
91+ /// The repository in the git store, and a read-only credential for it
92+ /// that lasts minutes (sent as `Authorization: Bearer`).
93+ pub remote: String,
94+ pub git_token: String,
95+ /// For an incremental bundle: the commits it may leave out, and every
96+ /// commit they reach. Empty for a full one.
97+ pub prerequisites: Vec<String>,
98+ /// The refs the last bundle held. When the clone has exactly these,
99+ /// nothing has changed and no bundle is cut.
100+ pub previous_refs: BTreeMap<String, String>,
101+ pub part_bytes: u64,
102+}
103+
104+/// A part the repos service has kept: what completing the upload needs.
105+#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
106+pub struct BackupPart {
107+ pub number: u16,
108+ pub etag: String,
109+}
110+
111+/// `backup_complete`: the bundle is cut and sent. `Outcome<bool>`.
112+#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
113+pub struct BackupComplete {
114+ pub job_id: String,
115+ pub token: String,
116+ /// Every ref the bundle holds (`git for-each-ref` of the clone, and
117+ /// `HEAD`), by name.
118+ pub refs: BTreeMap<String, String>,
119+ /// The bundle's size; 0 when there was nothing new to bundle.
120+ pub size: u64,
121+ #[serde(default)]
122+ pub sha256: Option<String>,
123+ #[serde(default)]
124+ pub parts: Vec<BackupPart>,
125+ /// What the clone read from the git store, for its meters.
126+ #[serde(default)]
127+ pub fetched_bytes: u64,
128+}
129+
130+/// `backup_fail`: the job could not be done. `Outcome<bool>`.
131+#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
132+pub struct BackupFail {
133+ pub job_id: String,
134+ pub token: String,
135+ pub error: String,
136+ #[serde(default)]
137+ pub fetched_bytes: u64,
138+}
+1−0
99 pub mod actions;
1010 pub mod agents;
1111 pub mod audit;
12+pub mod backups;
1213 pub mod billing;
1314 pub mod capture;
1415 pub mod credentials;
+445−0
1+//! Cuts one repository's nightly backup: a `git bundle` of everything it
2+//! has, or of what is new since the last one, sent to g1t in parts
3+//! (`g1t_contracts::backups` has the flow; services/repos/src/backups.rs
4+//! keeps the chain).
5+//!
6+//! Nothing here is an agent and nothing is pushed. The sandbox holds the
7+//! job's token and nothing else: it asks for the job, which comes with a
8+//! read-only credential for the repository in the git store that lasts
9+//! minutes, clones every ref (`--mirror`, never shallow), and bundles
10+//! `--all` but the commits the last bundle ended at, and everything they
11+//! reach. Those are written as refs of their own first, so a repository
12+//! with thousands of refs never makes a command line too long.
13+//!
14+//! Configuration comes from the environment:
15+//!
16+//! - `G1T_API`: where to report.
17+//! - `BACKUP_JOB`, `BACKUP_TOKEN`: the job, and the token that does it.
18+
19+use std::collections::BTreeMap;
20+use std::fs::File;
21+use std::io::{Read, Write};
22+use std::path::{Path, PathBuf};
23+use std::process::{Command, Stdio};
24+use std::time::Duration;
25+
26+use anyhow::{Context, Result, bail};
27+use serde::Deserialize;
28+use serde_json::{Value, json};
29+use sha2::{Digest, Sha256};
30+
31+use crate::checks::redact;
32+use crate::env;
33+
34+/// The header the job's token goes in (`g1t_contracts::backups::TOKEN_HEADER`).
35+const TOKEN_HEADER: &str = "x-g1t-backup-token";
36+/// Where the prerequisites are written as refs while the bundle is cut.
37+const PREREQ_REFS: &str = "refs/g1t-backup-prerequisites";
38+/// A transfer slower than this many bytes a second for `LOW_SPEED_SECONDS`
39+/// is given up, so a stalled clone does not hold the sandbox for hours.
40+const LOW_SPEED_BYTES: &str = "1000";
41+const LOW_SPEED_SECONDS: &str = "120";
42+const PART_TRIES: u32 = 3;
43+
44+/// The job, as the API gives it (`BackupSpec`).
45+#[derive(Debug, Deserialize)]
46+struct Spec {
47+ kind: String,
48+ remote: String,
49+ git_token: String,
50+ #[serde(default)]
51+ prerequisites: Vec<String>,
52+ #[serde(default)]
53+ previous_refs: BTreeMap<String, String>,
54+ part_bytes: u64,
55+}
56+
57+/// What was cut.
58+#[derive(Debug, PartialEq, Eq)]
59+pub(crate) struct Cut {
60+ /// Every ref the clone has, and `HEAD`.
61+ pub refs: BTreeMap<String, String>,
62+ /// The bundle, or None when there was nothing new to put in one.
63+ pub bundle: Option<PathBuf>,
64+}
65+
66+/// Runs git in `dir`, with `stdin` given to it, and returns its trimmed
67+/// output, failing on a non-zero exit with what it said.
68+fn git_in(dir: &Path, args: &[&str], stdin: Option<&str>) -> Result<String> {
69+ let mut child = Command::new("git")
70+ .current_dir(dir)
71+ .args(args)
72+ .stdin(if stdin.is_some() { Stdio::piped() } else { Stdio::null() })
73+ .stdout(Stdio::piped())
74+ .stderr(Stdio::piped())
75+ .spawn()
76+ .context("could not run git")?;
77+ if let (Some(input), Some(mut pipe)) = (stdin, child.stdin.take()) {
78+ pipe.write_all(input.as_bytes())?;
79+ }
80+ let output = child.wait_with_output()?;
81+ if !output.status.success() {
82+ bail!(
83+ "git {} failed: {}",
84+ args.iter().find(|arg| !arg.starts_with('-') && !arg.contains('=')).unwrap_or(&""),
85+ String::from_utf8_lossy(&output.stderr).trim()
86+ );
87+ }
88+ Ok(String::from_utf8_lossy(&output.stdout).trim().to_owned())
89+}
90+
91+/// `git for-each-ref` output, `<hash> <name>` a line, as a map.
92+pub(crate) fn parse_refs(listing: &str) -> BTreeMap<String, String> {
93+ listing
94+ .lines()
95+ .filter_map(|line| line.trim().split_once(' '))
96+ .filter(|(_, name)| !name.starts_with(PREREQ_REFS))
97+ .map(|(hash, name)| (name.trim().to_owned(), hash.trim().to_owned()))
98+ .collect()
99+}
100+
101+/// Of `git cat-file --batch-check` output, the objects it found.
102+pub(crate) fn present(check: &str) -> Vec<String> {
103+ check
104+ .lines()
105+ .filter(|line| !line.ends_with(" missing"))
106+ .filter_map(|line| line.split(' ').next())
107+ .filter(|hash| !hash.is_empty())
108+ .map(str::to_owned)
109+ .collect()
110+}
111+
112+/// Whether git refused because the bundle would hold no objects: every
113+/// ref still points where the last bundle left it, or at what it reaches.
114+pub(crate) fn is_empty_bundle(error: &str) -> bool {
115+ error.contains("empty bundle")
116+}
117+
118+/// Every ref of the clone in `dir`, and `HEAD` when it points somewhere.
119+pub(crate) fn refs_of(dir: &Path) -> Result<BTreeMap<String, String>> {
120+ let mut refs = parse_refs(&git_in(dir, &["for-each-ref", "--format=%(objectname) %(refname)"], None)?);
121+ if let Ok(head) = git_in(dir, &["rev-parse", "--verify", "--quiet", "HEAD"], None)
122+ && !head.is_empty()
123+ {
124+ refs.insert("HEAD".to_owned(), head);
125+ }
126+ Ok(refs)
127+}
128+
129+/// Cuts the bundle of the clone in `dir` into `out`: every ref, leaving
130+/// out `prerequisites` (those the clone still has) and all they reach.
131+/// Nothing is cut when the refs are `previous` exactly, or there are none,
132+/// or nothing new is there.
133+pub(crate) fn cut(dir: &Path, prerequisites: &[String], previous: &BTreeMap<String, String>, out: &Path) -> Result<Cut> {
134+ let refs = refs_of(dir)?;
135+ if refs.is_empty() || (&refs == previous && !previous.is_empty() && !prerequisites.is_empty()) {
136+ return Ok(Cut { refs, bundle: None });
137+ }
138+ // Only those the clone has: a commit force-pushed away is no longer
139+ // there to leave out, and the bundle then carries a little more.
140+ let kept = if prerequisites.is_empty() {
141+ Vec::new()
142+ } else {
143+ present(&git_in(dir, &["cat-file", "--batch-check=%(objectname) %(objecttype)"], Some(&format!("{}\n", prerequisites.join("\n"))))?)
144+ };
145+ if !kept.is_empty() {
146+ let updates: String = kept.iter().map(|hash| format!("create {PREREQ_REFS}/{hash} {hash}\n")).collect();
147+ git_in(dir, &["update-ref", "--stdin"], Some(&updates))?;
148+ }
149+ let out_text = out.to_str().context("the bundle's path is not text")?;
150+ let glob = format!("--glob={PREREQ_REFS}/*");
151+ let exclude = format!("--exclude={PREREQ_REFS}/*");
152+ let mut args = vec!["bundle", "create", "--quiet", out_text, exclude.as_str(), "--all"];
153+ if !kept.is_empty() {
154+ args.extend(["--not", glob.as_str()]);
155+ }
156+ let made = git_in(dir, &args, None);
157+ if !kept.is_empty() {
158+ let deletes: String = kept.iter().map(|hash| format!("delete {PREREQ_REFS}/{hash}\n")).collect();
159+ git_in(dir, &["update-ref", "--stdin"], Some(&deletes))?;
160+ }
161+ match made {
162+ Ok(_) => {
163+ git_in(dir, &["bundle", "verify", "--quiet", out_text], None).context("the bundle does not verify")?;
164+ Ok(Cut { refs, bundle: Some(out.to_owned()) })
165+ }
166+ Err(error) if is_empty_bundle(&error.to_string()) => Ok(Cut { refs, bundle: None }),
167+ Err(error) => Err(error),
168+ }
169+}
170+
171+/// The bytes of the clone's packs, which is what it read from the store.
172+fn pack_bytes(dir: &Path) -> u64 {
173+ std::fs::read_dir(dir.join("objects/pack"))
174+ .map(|entries| entries.filter_map(|entry| entry.ok()?.metadata().ok()).map(|meta| meta.len()).sum())
175+ .unwrap_or(0)
176+}
177+
178+/// Talks to the API about one job.
179+struct Job {
180+ api: String,
181+ id: String,
182+ token: String,
183+ agent: ureq::Agent,
184+}
185+
186+impl Job {
187+ fn url(&self, action: &str) -> String {
188+ format!("{}/backups/{}/{action}", self.api, self.id)
189+ }
190+
191+ fn answer(result: std::result::Result<ureq::Response, ureq::Error>) -> Result<Value> {
192+ match result {
193+ Ok(response) => Ok(response.into_json()?),
194+ Err(ureq::Error::Status(status, response)) => {
195+ let body: Value = response.into_json().unwrap_or(Value::Null);
196+ let said = body["error"]["message"].as_str().unwrap_or("no reason given").to_owned();
197+ bail!("g1t answered {status}: {said}")
198+ }
199+ Err(error) => Err(error.into()),
200+ }
201+ }
202+
203+ fn post(&self, action: &str, body: Value) -> Result<Value> {
204+ Job::answer(self.agent.post(&self.url(action)).set(TOKEN_HEADER, &self.token).send_json(body))
205+ }
206+
207+ fn put_part(&self, number: u16, bytes: &[u8]) -> Result<Value> {
208+ let mut last = None;
209+ for _ in 0..PART_TRIES {
210+ let sent = self
211+ .agent
212+ .put(&self.url(&format!("parts/{number}")))
213+ .set(TOKEN_HEADER, &self.token)
214+ .set("content-type", "application/octet-stream")
215+ .send_bytes(bytes);
216+ match Job::answer(sent) {
217+ Ok(part) => return Ok(part),
218+ Err(error) => last = Some(error),
219+ }
220+ }
221+ Err(last.unwrap_or_else(|| anyhow::anyhow!("the part was not sent")))
222+ }
223+}
224+
225+/// Sends the bundle in parts of `part_bytes`, hashing it on the way, and
226+/// returns its size, SHA-256 and the parts as g1t kept them.
227+fn send(job: &Job, bundle: &Path, part_bytes: u64) -> Result<(u64, String, Vec<Value>)> {
228+ let mut file = File::open(bundle)?;
229+ let mut hasher = Sha256::new();
230+ let mut parts = Vec::new();
231+ let mut size = 0u64;
232+ let mut buffer = vec![0u8; part_bytes as usize];
233+ loop {
234+ let mut filled = 0;
235+ while filled < buffer.len() {
236+ let read = file.read(&mut buffer[filled..])?;
237+ if read == 0 {
238+ break;
239+ }
240+ filled += read;
241+ }
242+ if filled == 0 {
243+ break;
244+ }
245+ hasher.update(&buffer[..filled]);
246+ size += filled as u64;
247+ let number = u16::try_from(parts.len() + 1).context("the bundle has too many parts")?;
248+ parts.push(job.put_part(number, &buffer[..filled]).with_context(|| format!("could not send part {number}"))?);
249+ crate::abuse::touch();
250+ if filled < buffer.len() {
251+ break;
252+ }
253+ }
254+ Ok((size, hex::encode(hasher.finalize()), parts))
255+}
256+
257+fn back_up(job: &Job, fetched: &mut u64) -> Result<String> {
258+ let spec: Spec = serde_json::from_value(job.post("spec", json!({}))?).context("the job's spec could not be read")?;
259+ let work = Path::new("/work");
260+ std::fs::create_dir_all(work)?;
261+ let mirror = work.join("backup.git");
262+ let auth = format!("http.extraHeader=Authorization: Bearer {}", spec.git_token);
263+ let mirror_text = mirror.to_str().context("the clone's path is not text")?;
264+ git_in(
265+ work,
266+ &[
267+ "-c",
268+ &auth,
269+ "-c",
270+ &format!("http.lowSpeedLimit={LOW_SPEED_BYTES}"),
271+ "-c",
272+ &format!("http.lowSpeedTime={LOW_SPEED_SECONDS}"),
273+ "clone",
274+ "--mirror",
275+ "--quiet",
276+ &spec.remote,
277+ mirror_text,
278+ ],
279+ None,
280+ )
281+ .context("could not clone the repository")?;
282+ *fetched = pack_bytes(&mirror).max(1);
283+ let cut = cut(&mirror, &spec.prerequisites, &spec.previous_refs, &work.join("backup.bundle"))?;
284+ let (size, sha256, parts) = match &cut.bundle {
285+ Some(bundle) => {
286+ let (size, sha256, parts) = send(job, bundle, spec.part_bytes)?;
287+ (size, Some(sha256), parts)
288+ }
289+ None => (0, None, Vec::new()),
290+ };
291+ job.post(
292+ "complete",
293+ json!({ "refs": cut.refs, "size": size, "sha256": sha256, "parts": parts, "fetched_bytes": *fetched }),
294+ )
295+ .context("could not report the backup")?;
296+ Ok(match cut.bundle {
297+ Some(_) => format!("{} bundle of {} refs, {size} bytes", spec.kind, cut.refs.len()),
298+ None => format!("nothing new to bundle in {} refs", cut.refs.len()),
299+ })
300+}
301+
302+pub fn main() -> i32 {
303+ let (api, id, token) = match (env("G1T_API"), env("BACKUP_JOB"), env("BACKUP_TOKEN")) {
304+ (Ok(api), Ok(id), Ok(token)) => (api, id, token),
305+ _ => {
306+ eprintln!("g1t-runner: G1T_API, BACKUP_JOB and BACKUP_TOKEN must be set");
307+ return 2;
308+ }
309+ };
310+ let agent = ureq::AgentBuilder::new().timeout_connect(Duration::from_secs(30)).timeout(Duration::from_secs(600)).build();
311+ let job = Job { api, id, token: token.clone(), agent };
312+ let mut fetched = 0;
313+ match back_up(&job, &mut fetched) {
314+ Ok(said) => {
315+ println!("g1t-runner: backed up: {said}");
316+ 0
317+ }
318+ Err(error) => {
319+ let said = redact(&format!("{error:#}"), &[token]);
320+ eprintln!("g1t-runner: the backup failed: {said}");
321+ if let Err(error) = job.post("fail", json!({ "error": said, "fetched_bytes": fetched })) {
322+ eprintln!("g1t-runner: could not report the failure: {error:#}");
323+ }
324+ 1
325+ }
326+ }
327+}
328+
329+#[cfg(test)]
330+mod tests {
331+ use super::*;
332+
333+ #[test]
334+ fn refs_are_read_and_the_prerequisites_own_left_out() {
335+ let listing = "c71546fcd893ef8b0f57388b65e620d759705dda refs/heads/main\n\
336+ 4807077b296e6edbf410d55e72749d3e1170c291 refs/pull/pr_1/head\n\
337+ 4807077b296e6edbf410d55e72749d3e1170c291 refs/g1t-backup-prerequisites/4807077b296e6edbf410d55e72749d3e1170c291\n";
338+ let refs = parse_refs(listing);
339+ assert_eq!(refs.len(), 2);
340+ assert_eq!(refs["refs/heads/main"], "c71546fcd893ef8b0f57388b65e620d759705dda");
341+ assert!(parse_refs("").is_empty());
342+ }
343+
344+ #[test]
345+ fn missing_prerequisites_are_dropped() {
346+ let check = "c71546fcd893ef8b0f57388b65e620d759705dda commit\n4807077b296e6edbf410d55e72749d3e1170c291 missing\n";
347+ assert_eq!(present(check), ["c71546fcd893ef8b0f57388b65e620d759705dda"]);
348+ }
349+
350+ #[test]
351+ fn an_empty_bundle_is_told_apart_from_a_failure() {
352+ assert!(is_empty_bundle("git bundle failed: fatal: Refusing to create empty bundle."));
353+ assert!(!is_empty_bundle("git bundle failed: fatal: bad revision"));
354+ }
355+
356+ // With git itself: a repository backed up full, then incrementally,
357+ // then restored from the chain the way the restore drill does it.
358+
359+ fn scratch(name: &str) -> PathBuf {
360+ let dir = std::env::temp_dir().join(format!("g1t-backup-test-{name}-{}", std::process::id()));
361+ let _ = std::fs::remove_dir_all(&dir);
362+ std::fs::create_dir_all(&dir).unwrap();
363+ dir
364+ }
365+
366+ fn commit(dir: &Path, file: &str, text: &str) {
367+ std::fs::write(dir.join(file), text).unwrap();
368+ git_in(dir, &["add", "--all"], None).unwrap();
369+ git_in(dir, &["-c", "user.name=t", "-c", "user.email=t@example.com", "commit", "--quiet", "-m", text], None).unwrap();
370+ }
371+
372+ fn mirror_of(origin: &Path, into: &Path) {
373+ let _ = std::fs::remove_dir_all(into);
374+ let parent = into.parent().unwrap();
375+ git_in(parent, &["clone", "--mirror", "--quiet", origin.to_str().unwrap(), into.to_str().unwrap()], None).unwrap();
376+ }
377+
378+ fn prerequisites(refs: &BTreeMap<String, String>) -> Vec<String> {
379+ let unique: std::collections::BTreeSet<&String> = refs.values().collect();
380+ unique.into_iter().cloned().collect()
381+ }
382+
383+ #[test]
384+ fn a_chain_of_bundles_restores_every_ref() {
385+ let root = scratch("chain");
386+ let origin = root.join("origin");
387+ std::fs::create_dir_all(&origin).unwrap();
388+ git_in(&origin, &["init", "--quiet", "--initial-branch=main"], None).unwrap();
389+ commit(&origin, "a.txt", "one");
390+ git_in(&origin, &["tag", "-a", "v1", "-m", "v1"], None).unwrap();
391+ let mirror = root.join("mirror.git");
392+
393+ // Full.
394+ mirror_of(&origin, &mirror);
395+ let full = cut(&mirror, &[], &BTreeMap::new(), &root.join("0-full.bundle")).unwrap();
396+ assert!(full.bundle.is_some());
397+ assert!(full.refs.contains_key("refs/tags/v1") && full.refs.contains_key("HEAD"));
398+
399+ // Nothing changed: nothing is cut.
400+ mirror_of(&origin, &mirror);
401+ let same = cut(&mirror, &prerequisites(&full.refs), &full.refs, &root.join("x.bundle")).unwrap();
402+ assert_eq!(same.bundle, None);
403+
404+ // New work, a new branch and a deleted tag: incremental.
405+ commit(&origin, "b.txt", "two");
406+ git_in(&origin, &["branch", "feature"], None).unwrap();
407+ git_in(&origin, &["tag", "-d", "v1"], None).unwrap();
408+ mirror_of(&origin, &mirror);
409+ let incr = cut(&mirror, &prerequisites(&full.refs), &full.refs, &root.join("1-incr.bundle")).unwrap();
410+ let incremental = incr.bundle.clone().expect("new commits make a bundle");
411+ assert!(!incr.refs.contains_key("refs/tags/v1"));
412+ // It needs the full one: alone, it does not verify.
413+ let lone = root.join("lone");
414+ git_in(&root, &["init", "--quiet", "--bare", lone.to_str().unwrap()], None).unwrap();
415+ assert!(git_in(&lone, &["bundle", "verify", incremental.to_str().unwrap()], None).is_err());
416+
417+ // Only a ref moved to a commit already kept: no objects, no bundle.
418+ git_in(&origin, &["branch", "-f", "feature", "HEAD~1"], None).unwrap();
419+ mirror_of(&origin, &mirror);
420+ let moved = cut(&mirror, &prerequisites(&incr.refs), &incr.refs, &root.join("2-incr.bundle")).unwrap();
421+ assert_eq!(moved.bundle, None);
422+ assert_ne!(moved.refs, incr.refs);
423+
424+ // Restored: each bundle in order, then the refs the last entry says.
425+ let restored = root.join("restored.git");
426+ git_in(&root, &["init", "--quiet", "--bare", restored.to_str().unwrap()], None).unwrap();
427+ for bundle in [full.bundle.unwrap(), incremental] {
428+ git_in(&restored, &["bundle", "verify", "--quiet", bundle.to_str().unwrap()], None).unwrap();
429+ git_in(&restored, &["fetch", "--quiet", "--no-tags", bundle.to_str().unwrap(), "+refs/*:refs/backup-staging/*"], None).unwrap();
430+ }
431+ let updates: String = moved
432+ .refs
433+ .iter()
434+ .filter(|(name, _)| name.as_str() != "HEAD")
435+ .map(|(name, hash)| format!("update {name} {hash}\n"))
436+ .collect();
437+ git_in(&restored, &["update-ref", "--stdin"], Some(&updates)).unwrap();
438+ let staging: String = git_in(&restored, &["for-each-ref", "--format=delete %(refname)", "refs/backup-staging/"], None).unwrap();
439+ git_in(&restored, &["update-ref", "--stdin"], Some(&format!("{staging}\n"))).unwrap();
440+ git_in(&restored, &["symbolic-ref", "HEAD", "refs/heads/main"], None).unwrap();
441+ assert_eq!(refs_of(&restored).unwrap(), moved.refs);
442+ git_in(&restored, &["fsck", "--no-progress", "--connectivity-only"], None).unwrap();
443+ let _ = std::fs::remove_dir_all(&root);
444+ }
445+}
+4−1
1212 //! address what the checks or a review found, `plan` turns an outcome
1313 //! into issues, `queue` builds and checks a state of the merge queue,
1414 //! `mergecheck` finds out whether a pull request merges cleanly,
15−//! `actions` runs one job of a GitHub Actions workflow, and `bump` makes a
15+//! `actions` runs one job of a GitHub Actions workflow, `backup` cuts a
16+//! repository's nightly backup bundle, and `bump` makes a
1617 //! security update: one package raised in its lockfiles, pushed as g1t.
1718 //! See the modules of those names.
1819 //!
3334
3435 mod abuse;
3536 mod actions;
37+mod backup;
3638 mod bump;
3739 mod checks;
3840 mod clone;
221223 // The same image does the other jobs a sandbox is started for.
222224 match std::env::var("MODE").as_deref() {
223225 Ok("actions") => std::process::exit(actions::main()),
226+ Ok("backup") => std::process::exit(backup::main()),
224227 Ok("bump") => std::process::exit(bump::main()),
225228 Ok("checks") => std::process::exit(checks::main()),
226229 Ok("deploy") => std::process::exit(deploy::main()),
+15−2
1010 // which keeps repositories in the git store (gitstore/server.mjs);
1111 // - EMAIL (Email Sending) becomes a service binding to workers/mail;
1212 // - the packages service keeps files in S3-compatible storage (MinIO)
13−// instead of R2;
13+// instead of R2, and the repos service its nightly backups (a bucket of
14+// their own, BACKUP_S3_BUCKET);
1415 // - services that are off in this phase (agents, the context hub, the
1516 // g1t.page dispatcher, model proxy) are bound to workers/off instead, and
1617 // events stop queueing work for them;
1920 // Usage: node configs.mjs [outDir]
2021 // Environment: PUBLIC_URL, GITSTORE_URL, GITSTORE_SECRET, MAIL_URL,
2122 // ACTIONS_KEY, INTEGRATIONS_KEY, WEBHOOKS_KEY, IDENTITY_KEY,
22−// PACKAGES_TOKEN_SECRET, S3_ENDPOINT, S3_BUCKET, S3_REGION,
23+// PACKAGES_TOKEN_SECRET, S3_ENDPOINT, S3_BUCKET, BACKUP_S3_BUCKET, S3_REGION,
2324 // S3_ACCESS_KEY_ID, S3_SECRET_ACCESS_KEY, S3_PUBLIC_ENDPOINT, and optionally
2425 // your own GitHub App: GITHUB_APP_ID, GITHUB_APP_SLUG, GITHUB_APP_CLIENT_ID,
2526 // GITHUB_APP_CLIENT_SECRET, GITHUB_APP_PRIVATE_KEY, GITHUB_APP_WEBHOOK_SECRET.
216217 delete config.vars.R2_ACCOUNT_ID;
217218 delete config.vars.R2_BUCKET;
218219 }
220+ // Nightly backups' bundles go to a bucket of their own on the same
221+ // S3-compatible store, instead of the BACKUPS R2 bucket.
222+ if (hosted.name === "g1t-repos") {
223+ Object.assign(config.vars, {
224+ BACKUP_STORE: "s3",
225+ BACKUP_S3_BUCKET: process.env.BACKUP_S3_BUCKET ?? "g1t-backups",
226+ S3_ENDPOINT: process.env.S3_ENDPOINT ?? "http://minio:9000",
227+ S3_REGION: process.env.S3_REGION ?? "us-east-1",
228+ S3_ACCESS_KEY_ID: process.env.S3_ACCESS_KEY_ID ?? "",
229+ S3_SECRET_ACCESS_KEY: process.env.S3_SECRET_ACCESS_KEY ?? "",
230+ });
231+ }
219232 // Nothing to deploy to: deployments are off (no Cloudflare API token).
220233 if (hosted.name === "g1t-deployments") delete config.vars.CUSTOM_HOSTNAMES_ZONE_ID;
221234
+6−2
5151 S3_ACCESS_KEY_ID: ${S3_ACCESS_KEY_ID:-g1t}
5252 S3_SECRET_ACCESS_KEY: ${S3_SECRET_ACCESS_KEY:-g1t-packages-secret}
5353 S3_PUBLIC_ENDPOINT: ${S3_PUBLIC_ENDPOINT:-}
54+ # Nightly backups' bundles and manifests, in a bucket of their own on
55+ # the same store (docs/SELF_HOSTING.md, "Backups").
56+ BACKUP_S3_BUCKET: ${BACKUP_S3_BUCKET:-g1t-backups}
5457 volumes:
5558 - g1t-data:/data
5659 - g1t-secrets:/secrets:ro
111114 retries: 20
112115 restart: unless-stopped
113116
114− # Makes the bucket once, then exits.
117+ # Makes the buckets once, then exits: packages' files, and backups.
115118 minio-setup:
116119 image: minio/mc:latest
117120 depends_on:
120123 entrypoint:
121124 - sh
122125 - -c
123− - mc alias set local http://minio:9000 "$$MINIO_ROOT_USER" "$$MINIO_ROOT_PASSWORD" && mc mb --ignore-existing "local/$$S3_BUCKET"
126+ - mc alias set local http://minio:9000 "$$MINIO_ROOT_USER" "$$MINIO_ROOT_PASSWORD" && mc mb --ignore-existing "local/$$S3_BUCKET" && mc mb --ignore-existing "local/$$BACKUP_S3_BUCKET"
124127 environment:
125128 MINIO_ROOT_USER: ${S3_ACCESS_KEY_ID:-g1t}
126129 MINIO_ROOT_PASSWORD: ${S3_SECRET_ACCESS_KEY:-g1t-packages-secret}
127130 S3_BUCKET: ${S3_BUCKET:-g1t-packages}
131+ BACKUP_S3_BUCKET: ${BACKUP_S3_BUCKET:-g1t-backups}
128132
129133 mailpit:
130134 image: axllent/mailpit:latest
+2−1
7979 "secrets": ["REPOS_KEY"],
8080 "setup": [
8181 "The Artifacts namespace `g1t` (the ARTIFACTS binding)",
82− "The R2 bucket `g1t-git-packs` (GIT_PACKS) with its lifecycle rule: `npx wrangler r2 bucket create g1t-git-packs`, then `npx wrangler r2 bucket lifecycle add g1t-git-packs expire-packs packs/ --expire-days 7 --abort-multipart-days 1`"
82+ "The R2 bucket `g1t-git-packs` (GIT_PACKS) with its lifecycle rule: `npx wrangler r2 bucket create g1t-git-packs`, then `npx wrangler r2 bucket lifecycle add g1t-git-packs expire-packs packs/ --expire-days 7 --abort-multipart-days 1`",
83+ "The R2 bucket for nightly backups: npx wrangler r2 bucket create g1t-backups"
8384 ],
8485 "self_host": "run"
8586 },
+127−2
2828 128 MB Worker isolate that buffers each push body twice. Large pushes and imports fail late, without a
2929 message git can show.
3030 5. **No backup, no exit drill.** Cloudflare replicates data, but there is no SLA, no documented export
31− besides git itself, and the self-host git store is not a production fallback yet.
31+ besides git itself, and the self-host git store is not a production fallback yet. Nightly bundles
32+ to R2 and a restore drill are now built (R11, section 9); the fallback store is not (R12).
3233
3334 None of these blocks an invite-only launch. Items 1 and 2 must be answered before billing starts on
3435 2026-10-14, and the fork cleanup must ship before agent pull requests reach thousands a day.
350351
351352 Code in `services/repos` unless named; one migration,
352353 `migrations/0011_artifacts_meters_forks_health.sql` (new columns on `repos`, new tables
353−`artifacts_meters`, `operation_mapping`, `store_health`; additive, no backfill).
354+`artifacts_meters`, `operation_mapping`, `store_health`; additive, no backfill). R11 added
355+`migrations/0013_backups.sql` (a new table, `repo_backups`, and one `operation_mapping` row;
356+additive).
354357
355358 | # | Status | What |
356359 | --- | --- | --- |
365368 | R7 | Groundwork | `shards.rs`: bindings named in `ARTIFACTS_NAMESPACES` (JSON, binding → namespace; `ARTIFACTS` → `g1t` always there), a repository's namespace kept in its `store` column as `<namespace>/<key>` (no prefix means the `ARTIFACTS` namespace, so every existing key reads the same), new repositories placed by `ARTIFACTS_NEW_REPOS` (comma-separated, spread by an FNV hash of the repository id; names not bound are skipped), forks always in their repository's namespace, `ARTIFACTS_EU_NAMESPACE` reserved for EU residency (no workspace setting yet). Works with only `ARTIFACTS` bound, as today. |
366369 | R8 | Built | `crates/runner/src/clone.rs`: every sandbox clones at `--depth=1` (a full g1t clone took 5.4 s, depth 1 took 3.8 s). Work that merges (catch-up, the merge queue, merge checks, a review's diff) deepens 50, 500, then 5000 commits until the two sides share one, and fetches everything only as the last resort (`share_history`). `G1T_CLONE_DEPTH` (0 or `full` for everything) and `G1T_CLONE_FILTER=blob:none` change it per runner. |
367370 | R6 | Built; the bucket must exist before it deploys | `pack_cache.rs`: an upload-pack POST with wants and no `have` or `shallow` lines (a fresh clone, the sandboxes' `deepen 1` ones included), uncompressed and at most 1 MiB, is keyed `packs/<repo id>/<refs_version>/<sha256>` over the request normalized: protocol v2 capabilities without `agent=`/`session-id=` and its arguments, each sorted and deduplicated; v0/v1 wants sorted, the first want's capabilities split off, sorted and without `agent=`, then `deepen`/`filter` lines, a flush and `done`. Only while `refs_cache::usable` (the version known, and no push credential out of g1t's hands), so never across a refs change. Looked up after authorization, alongside the free-workspace limits and the kept refs answer; a hit streams from the bucket (`Server-Timing` `pack;desc=hit`). A miss streams the store's 200 to git through a tee that copies it to a fill in `ctx.wait_until` (at most 5 MiB queued between them, 2 fills per isolate, one per key): under 5 MiB it is one `put` once it all arrived; larger, 5 MiB multipart parts completed only after the last part and a check that it is one whole side-band pack (well-formed pkt-lines, `PACK` on channel 1, no `ERR` or channel 3, a closing flush). Over 200 MB, a queue that falls behind, git going away or the store's stream failing lets the fill go and aborts the upload; nothing partial can be read. Meters `pack_cache.hit` (with the bytes served) and `pack_cache.miss` (counted with `record`, bytes added at the end), neither an operation by default; a hit records no `git.fetch`. Storage is behind the `PackStore` port with an R2 adapter (`GIT_PACKS`, bucket `g1t-git-packs`, lifecycle: packs deleted after 7 days, unfinished uploads after 1); without the binding (self-hosted) nothing is kept. |
371+| R11 | Built; not yet deployed | Nightly `git bundle` backups to the `g1t-backups` R2 bucket, and a restore drill. Migration `0013_backups.sql` (`repo_backups`, and an `operation_mapping` row). See "R11: backups and the restore drill" below. |
368372
369373 ### R1: reading `scripts/ops/artifacts-usage.mjs`
370374
499503 Moving an existing repository between namespaces is not built (a clone and push, then a `store`
500504 update).
501505
506+### R11: backups and the restore drill
507+
508+Every repository whose refs moved is bundled once a night and kept outside the git store, so a
509+repository can be rebuilt without Artifacts. The flow is in `crates/contracts/src/backups.rs`;
510+the chain, the manifest and the record are in `services/repos/src/backups.rs`.
511+
512+1. **Queued.** At 02:53 UTC (`53 2 * * *` in `services/repos/wrangler.jsonc`) the repos service
513+ queues the repositories that are due, at most `BACKUPS_PER_NIGHT` (200), the longest since
514+ their last backup first. A repository is due when it has never been backed up, when its
515+ `refs_version` went past the one its last backup was cut at, or when a credential that can
516+ push was handed out (`refs_open_until`) after that backup's clone began: a push with such a
517+ credential does not move `refs_version`. Deleted repositories, retired working copies and
518+ pull request working copies (`pulls/…`, whose heads end up in their repository as
519+ `refs/pull/<id>/head`) are not backed up.
520+2. **Claimed.** The runner's five-minute sweep claims `BACKUPS_PER_SWEEP` (4) at a time, with at
521+ most `BACKUPS_RUNNING` (6) running (`claim_backups`), and starts a sandbox for each in
522+ `MODE=backup` (`crates/runner/src/backup.rs`). The sandbox is given the job's id and a token
523+ for it, nothing else; the repos service keeps only the token's hash. It has a 60-minute time
524+ cap. Its time is g1t's: it is not metered to the workspace.
525+3. **Cut.** The sandbox asks for its job (`POST api.g1t.sh/backups/{job}/spec`, the token in
526+ `x-g1t-backup-token`) and gets a read-only credential for the repository in the store (a
527+ `git_access`-style handout, 5 minutes), the bundle's kind, and the commits the last bundle
528+ ended at. It clones with `--mirror` (every ref, never shallow), writes those commits as refs
529+ of its own, and runs `git bundle create --all --not <them>`, then `git bundle verify`.
530+ When the clone has exactly the refs of the last backup, or git finds nothing new to bundle
531+ (a branch deleted, a ref moved to a commit already kept), no bundle is cut and only the refs
532+ are recorded.
533+4. **Sent.** The bundle goes in 32 MiB parts (`PUT /backups/{job}/parts/{n}`), which the API
534+ passes to the repos service and the repos service to an R2 multipart upload; then
535+ `POST /backups/{job}/complete` with every ref, the size, the SHA-256 and the parts. A failure
536+ is `POST /backups/{job}/fail`; a sandbox that dies is failed by the runner. A job is tried 3
537+ times a night; one running past 3 hours is queued again.
538+5. **Recorded.** The manifest gains the entry, and `repo_backups` the refs version the clone began
539+ at, so a push during the backup leaves the repository due the next night.
540+
541+Storage, through the `BlobStore` port in `crates/blobstore` (the adapters packages already used):
542+the `BACKUPS` binding (bucket `g1t-backups`) with `BACKUP_STORE=r2`; any S3-compatible store with
543+`BACKUP_STORE=s3` and `BACKUP_S3_BUCKET` (self-hosted: MinIO). Without either, backups are off and
544+the nightly cron does nothing.
545+
546+```text
547+backups/<repo id>/manifest.json
548+backups/<repo id>/20261006T025300Z-full.bundle
549+backups/<repo id>/20261007T025302Z-incr.bundle
550+```
551+
552+The manifest (version 1) lists `chain`, oldest first, and `previous`, the chain before it. Each
553+entry has `id`, `kind` (`full` or `incremental`), `key` (null when only refs moved),
554+`created_at`, `refs_version`, `refs` (every ref and `HEAD` once it is applied), `prerequisites`,
555+`size` and `sha256`. The first backup is full; the next ones are incremental, their
556+prerequisites the last entry's tips, until the chain holds `BACKUP_FULL_EVERY` (30) incremental
557+ones, when a full one starts a new chain. The chain before that is kept until the next full one
558+replaces it, so the oldest backup kept is about two chains old. Backups of purged repositories
559+are removed the night after (50 a night).
560+
561+Meters: the clone counts as `internal.git.info_refs` and `internal.git.backup_fetch` with the
562+bytes it read, on the repository (they show in `artifacts_usage` and
563+`scripts/ops/artifacts-usage.mjs`). `operation_mapping` has `internal.git.backup_fetch` at 1 for
564+`cost_operations` and 0 for `billable_operations`: an operation on g1t's bill, never on the
565+workspace's. The credential's `binding.create_token` is metered as before.
566+
567+**The restore drill** (read-only against production: SELECTs on `g1t-repos`, reads of
568+`g1t-backups` through Wrangler, `git ls-remote` of the live repository):
569+
570+```sh
571+node scripts/ops/backup-restore-drill.mjs # a repository unchanged since its last backup
572+node scripts/ops/backup-restore-drill.mjs --repo acme/rocket # this one
573+G1T_USER=you G1T_TOKEN=g1t_... node scripts/ops/backup-restore-drill.mjs --repo acme/private-thing
574+```
575+
576+It downloads the manifest and each bundle of the chain, checks each against its size and SHA-256,
577+`git bundle verify`s it, fetches it into a new bare repository without following tags, sets every
578+ref to what the last entry says (and removes the rest), points `HEAD` at the branch at its commit,
579+and runs `git fsck --connectivity-only`. Then it compares every ref with the manifest and with
580+`git ls-remote` of the live repository and prints each difference. Exit 0: every ref matches;
581+1: a difference; 2: it could not run (a bundle that does not match its manifest is this). Picked
582+at random, the repository is one whose refs have not moved since its last backup, so any
583+difference is the backup's. Run it after the first night, then monthly, and after any change to
584+`backups.rs` or `backup.rs`. `--bundles <dir>` reads a local copy of the bucket instead
585+(self-hosted: `mc mirror local/g1t-backups <dir>`), with `--repo-id` and `--live <url or path>`.
586+`npm run test:ops` runs it against bundles cut with git.
587+
588+**A real restore into the store**, as it can be done today:
589+
590+1. Run the drill for the repository with `--keep`. It prints where the restored copy is
591+ (`…/restored.git`). Go on only if every ref matches the manifest; differences from the live
592+ repository are what the restore is for.
593+2. Tell the workspace, and stop the repository's agents and merge queue for the time.
594+3. If its default branch is protected, turn protection off in the repository's settings for the
595+ push: a push that changes a protected branch is declined.
596+4. From the restored copy, push every ref as an owner, with an access token that has
597+ `code:write`:
598+
599+ ```sh
600+ cd /tmp/g1t-drill-…/restored.git
601+ git -c "http.extraHeader=Authorization: Basic $(printf 'you:g1t_...' | base64)" \
602+ push --force https://g1t.sh/acme/rocket.git 'refs/*:refs/*'
603+ ```
604+
605+ It goes through the git door like any push: size limits, push protection (pushes over 24 MiB
606+ per `LARGE_PUSHES`) and the audit log apply, and the refs version moves, so the next night
607+ backs the repository up again. `--force` rewinds refs that went wrong; refs the live
608+ repository has that the backup does not are left alone (`git push --mirror` would delete
609+ them).
610+5. Turn protection back on, and run the drill again: every ref now matches the live repository.
611+
612+When the repository is gone from the store itself (its key answers not found), there is no
613+operator call yet to make an empty repository under an existing row's key; that is part of R12.
614+
615+To deploy: make the bucket (`npx wrangler r2 bucket create g1t-backups`, a setup step of `repos`
616+in `deploy/stack.jsonc`), then migration 0013, then `g1t-repos` (the `BACKUPS` binding and the
617+new cron), `g1t-api` (the `/backups/` door), and `g1t-runner` (a new image: the `backup` mode).
618+Until the runner is out, queued backups wait; nothing fails. Set `BACKUPS_PER_SWEEP` to `0` on the
619+runner to stop starting them.
620+
621+What is not covered: a pull request's working copy while its pull request is open (its head is
622+kept in the repository only once the working copy is retired), and anything that is not a git
623+ref (issues, pull requests and the rest live in D1, which has its own Time Travel). Every backup
624+clones the whole repository, so a night reads each changed repository in full from the store;
625+incremental bundles save storage, not reads.
626+
502627 ### Deploy order and what to watch
503628
504629 1. Migration 0011 (the deploy tool applies migrations first). `forks_of` reads `retired_at`, so
+10−4
8484 | **Static Assets** | `apps/web` (Vite plugin build), `apps/docs`, `apps/sudo` (`run_worker_first`) | thin | workerd serves them. |
8585 | **`placement`, `observability`, routes, custom domains** | every `wrangler.jsonc` | config only | Dropped by `deploy/self-host/configs.mjs`. |
8686 | **`cf-ray`** | Used as an audit request id, with a fallback: `services/repos/src/run_access.rs:131`, `apps/api/src/audit.rs:37` | thin | Falls back already. |
87−| **R2** | `services/packages` (`BLOBS`: container layers and other package files, `src/store/r2.rs`), the API's Actions cache (`ACTIONS_CACHE`), the runner's downloads | thin | **S3-compatible storage**: the packages service's `BlobStore` port has an S3 adapter (`src/store/s3.rs`, SigV4 over fetch), run against MinIO in the compose file. |
87+| **R2** | `services/packages` (`BLOBS`: container layers and other package files), `services/repos` (`BACKUPS`: nightly backup bundles), the API's Actions cache (`ACTIONS_CACHE`), the runner's downloads | thin | **S3-compatible storage**: the `BlobStore` port in `crates/blobstore` has an R2 adapter and an S3 one (`s3.rs`, SigV4 over fetch); each service names its own bucket (`BLOB_STORE`/`S3_BUCKET` for packages, `BACKUP_STORE`/`BACKUP_S3_BUCKET` for backups), run against MinIO in the compose file. |
8888 | **Not used** | Hyperdrive, Workflows, Analytics Engine, Browser Rendering, Images, Turnstile, Secrets Store, `connect()`, HTMLRewriter, `request.cf` | — | — |
8989
9090 ### By service
102102 | `apps/docs` | Static | — | Not run (docs.g1t.sh serves them) |
103103 | `apps/status` | TS Worker | Email Sending, cron; bound only to billing | Runs in a process of its own (`status.sh`), so it stays up when the site does not |
104104 | `services/identity` | Rust | Email Sending, KV `AVATARS` | Runs unchanged; `EMAIL` goes to the mail shim |
105−| `services/repos` | Rust | **Artifacts**, Cache API, optional KV `GIT_CACHE` with `REPOS_KEY`, optional R2 `GIT_PACKS` | Runs unchanged; `ARTIFACTS` goes to the git store. Without `GIT_CACHE` and `REPOS_KEY`, credentials and ref listings are kept per isolate only. `GIT_PACKS` (the clone pack cache, behind the `PackStore` port in `src/pack_cache.rs`) is not given, so every clone goes to the git store; an S3 adapter like packages' would turn it on |
105+| `services/repos` | Rust | **Artifacts**, **R2** (`BACKUPS`), Cache API, optional KV `GIT_CACHE` with `REPOS_KEY`, optional R2 `GIT_PACKS` | Runs unchanged; `ARTIFACTS` goes to the git store, and backups to MinIO's `g1t-backups` bucket (`BACKUP_STORE=s3`). Without `GIT_CACHE` and `REPOS_KEY`, credentials and ref listings are kept per isolate only. Its nightly cron queues backups, but bundles are cut by the runner, which is off in phase 1: none are made yet. `GIT_PACKS` (the clone pack cache, behind the `PackStore` port in `src/pack_cache.rs`) is not given, so every clone goes to the git store; an S3 adapter like packages' would turn it on |
106106 | `services/work` | Rust | Queue consumer | Runs unchanged |
107107 | `services/events` | Rust | Queues (producer and fan-out) | Runs unchanged; the off services' queues are not produced to |
108108 | `services/projects` | TS | Queue consumer | Runs unchanged |
381381 a backfill a self-hoster's start can run.
382382 - **Backups.** Phase 1: stop, then tar the `g1t-data` and `g1t-git`
383383 volumes (documented in the guide). Phase 2: online backups with
384− `sqlite3 .backup` per database and `git bundle` or rsync of the bare
385− repositories, or Litestream for continuous replication.
384+ `sqlite3 .backup` per database, or Litestream for continuous
385+ replication. The repositories get hosted g1t's nightly bundles
386+ (docs/ARTIFACTS.md, R11) once the runner runs: the storage is already
387+ configured (`BACKUP_STORE=s3`, the `g1t-backups` bucket that
388+ `minio-setup` makes, `BACKUP_S3_BUCKET` to choose another), and the
389+ restore drill reads a copy of that bucket
390+ (`mc mirror local/g1t-backups ./copy`, then
391+ `node scripts/ops/backup-restore-drill.mjs --bundles ./copy --repo-id <id> --live <bare repository>`).
386392
387393 ## 3. Phase 1: what works today
388394
+2−1
1313 "scripts": {
1414 "typecheck": "npm run typecheck --workspaces --if-present",
1515 "deploy": "npm run deploy -w @g1t/web",
16− "test:deploy": "node --test \"scripts/deploy/*.test.mjs\""
16+ "test:deploy": "node --test \"scripts/deploy/*.test.mjs\"",
17+ "test:ops": "node --test \"scripts/ops/*.test.mjs\""
1718 },
1819 "devDependencies": {
1920 "typescript": "^5.9.3",
+3−0
292292 commitFile: (repo, actor, file) => call("commit_file", { repo, actor, ...file }),
293293 land: (sourceId, actor, branch) => call("land", { sourceId, actor, branch }),
294294 compare: (repoId, viewer, base, head) => call("compare", { repoId, viewer, base, head }),
295+ claimBackups: (limit, maxRunning) => call("claim_backups", { limit, maxRunning }),
296+ // The sandbox's own calls are snake_case (they come through the API).
297+ failBackup: (jobId, token, error) => call("backup_fail", { job_id: jobId, token, error }),
295298 };
296299 }
297300
+24−0
277277 * branch with the point where it left the default branch.
278278 */
279279 compare(repoId: string, viewer: Viewer, base?: string | null, head?: string | null): Promise<Result<Comparison>>;
280+
281+ /**
282+ * Services only, for the runner's sweep: up to `limit` queued nightly
283+ * backups, each now running with a token of its own, so long as no more
284+ * than `maxRunning` are then running. Empty when backups are off.
285+ */
286+ claimBackups(limit: number, maxRunning: number): Promise<BackupClaim[]>;
287+
288+ /**
289+ * Services only: a backup's sandbox stopped before it reported, so the
290+ * job is tried again later. Refused harmlessly once it has reported.
291+ */
292+ failBackup(jobId: string, token: string, error: string): Promise<Result<boolean>>;
280293 }
281294
295+/**
296+ * A nightly backup to start (`g1t_contracts::backups`): the sandbox is
297+ * given the job's id and token, and nothing else.
298+ */
299+export type BackupClaim = {
300+ jobId: string;
301+ token: string;
302+ repoId: string;
303+ path: RepoPath;
304+};
305+
282306 /** Lines `start` to `end` (inclusive, from 1) last changed by `commit`. */
283307 export type BlameRange = { start: number; end: number; commit: string };
284308
+1−1
9898
9999 test("shared crates and packages are read from workspace metadata", () => {
100100 assert.deepEqual(unit("events").dependsOn, ["crates/contracts", "crates/kit"]);
101− assert.deepEqual(unit("repos").dependsOn, ["crates/contracts", "crates/kit", "crates/scan", "crates/secrets"]);
101+ assert.deepEqual(unit("repos").dependsOn, ["crates/blobstore", "crates/contracts", "crates/kit", "crates/scan", "crates/secrets"]);
102102 assert.ok(unit("actions").dependsOn.includes("crates/actions"));
103103 assert.ok(!unit("events").dependsOn.includes("crates/scan"));
104104 assert.deepEqual(unit("web").dependsOn, ["packages/contracts", "packages/theme"]);
+259−0
1+#!/usr/bin/env node
2+// The restore drill for nightly backups (docs/ARTIFACTS.md, R11;
3+// services/repos/src/backups.rs). Picks a repository, downloads its
4+// manifest and bundle chain from the g1t-backups bucket, rebuilds the
5+// repository from them in a temporary directory, and compares every ref
6+// with the live repository. Exits 1 on any difference, 2 when it could not
7+// run.
8+//
9+// Read-only: SELECTs against the g1t-repos database, reads of the bucket,
10+// and `git ls-remote` of the live repository. Nothing is written anywhere
11+// but the temporary directory, which is removed unless you pass --keep.
12+//
13+// node scripts/ops/backup-restore-drill.mjs # a repository unchanged since its last backup
14+// node scripts/ops/backup-restore-drill.mjs --repo acme/rocket
15+// node scripts/ops/backup-restore-drill.mjs --repo-id repo_... --bundles ./copy --live /srv/git/acme--rocket.git
16+//
17+// Options:
18+// --repo <workspace/name> the repository; default: one picked at random
19+// among those whose refs have not moved since
20+// their last backup, so any difference is the
21+// backup's.
22+// --repo-id <id> the repository by id (no database needed with --bundles).
23+// --bundles <dir> read the bucket from a local copy (`backups/<id>/...`
24+// under it, as `mc mirror` or `rclone copy` leave it)
25+// instead of R2, e.g. a self-hosted MinIO's.
26+// --live <url or path> the live repository to compare with; default
27+// https://g1t.sh/<workspace>/<name>.git.
28+// --keep keep the temporary directory, and say where it is.
29+//
30+// The live repository is read as G1T_USER with G1T_TOKEN (an access token
31+// with code:read) when they are set, which private repositories need. The
32+// database and the bucket are read through Wrangler, as you are logged in
33+// (`npx wrangler login`), or with CLOUDFLARE_DEPLOY_TOKEN when that is set.
34+
35+import { createHash } from "node:crypto";
36+import { createReadStream, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
37+import { tmpdir } from "node:os";
38+import { join } from "node:path";
39+
40+import { exec, jsonFrom, wranglerEnv } from "../deploy/cloudflare.mjs";
41+import { ROOT } from "../deploy/stack.mjs";
42+
43+const WRANGLER = join(ROOT, "node_modules/wrangler/bin/wrangler.js");
44+const DATABASE = "g1t-repos";
45+const BUCKET = process.env.BACKUP_BUCKET || "g1t-backups";
46+const MANIFEST_VERSION = 1;
47+const STAGING = "refs/drill-staging";
48+
49+/** Runs git; resolves with its output, or throws with what it said. */
50+async function git(args, { cwd, input } = {}) {
51+ const { code, out } = await exec("git", args, { cwd, input });
52+ if (code !== 0) throw new Error(`git ${args.find((arg) => !arg.startsWith("-")) ?? ""} failed: ${out.trim().slice(-600)}`);
53+ return out.trim();
54+}
55+
56+/** A file's SHA-256, read as a stream: a bundle can be a gigabyte. */
57+async function sha256Of(file) {
58+ const hash = createHash("sha256");
59+ for await (const chunk of createReadStream(file)) hash.update(chunk);
60+ return hash.digest("hex");
61+}
62+
63+/** `<hash> <name>` lines (for-each-ref) or `<hash>\t<name>` (ls-remote), as a map. Peeled tags are left out. */
64+export function parseRefs(listing) {
65+ const refs = {};
66+ for (const line of listing.split(/\r?\n/)) {
67+ const match = /^([0-9a-f]{40,64})\s+(\S+)$/.exec(line.trim());
68+ if (!match || match[2].endsWith("^{}")) continue;
69+ refs[match[2]] = match[1];
70+ }
71+ return refs;
72+}
73+
74+/** Every ref of the repository in `dir`, and HEAD. */
75+async function refsOf(dir) {
76+ const refs = parseRefs(await git(["for-each-ref", "--format=%(objectname) %(refname)"], { cwd: dir }));
77+ const head = await git(["rev-parse", "--verify", "--quiet", "HEAD"], { cwd: dir }).catch(() => "");
78+ if (head) refs.HEAD = head;
79+ return refs;
80+}
81+
82+/** The refs that differ between `want` and `have`: [{ ref, want, have }], `null` for absent. */
83+export function compareRefs(want, have) {
84+ const names = [...new Set([...Object.keys(want), ...Object.keys(have)])].sort();
85+ return names
86+ .filter((ref) => want[ref] !== have[ref])
87+ .map((ref) => ({ ref, want: want[ref] ?? null, have: have[ref] ?? null }));
88+}
89+
90+/** The branch HEAD should name: one at HEAD's commit, `main` or `master` first. */
91+export function headBranch(refs) {
92+ if (!refs.HEAD) return null;
93+ const branches = Object.keys(refs).filter((ref) => ref.startsWith("refs/heads/") && refs[ref] === refs.HEAD);
94+ return ["refs/heads/main", "refs/heads/master"].find((ref) => branches.includes(ref)) ?? branches.sort()[0] ?? null;
95+}
96+
97+/** Whether a manifest can be read by this drill. */
98+export function readManifest(text) {
99+ const manifest = JSON.parse(text);
100+ if (manifest.version !== MANIFEST_VERSION) throw new Error(`manifest version ${manifest.version}; this drill reads ${MANIFEST_VERSION}`);
101+ if (!Array.isArray(manifest.chain) || manifest.chain.length === 0) throw new Error("the manifest lists no backups");
102+ if (manifest.chain[0].kind !== "full") throw new Error("the chain does not start with a full backup");
103+ return manifest;
104+}
105+
106+/**
107+ * Rebuilds the repository the manifest's chain describes into `dir`, a
108+ * new bare repository: each bundle in order, checked against its size and
109+ * SHA-256 and verified by git, fetched without following tags; then every
110+ * ref set to what the last entry says, and nothing else kept. `fetchObject`
111+ * saves one object of the bucket to a file and resolves with its path.
112+ */
113+export async function restore(manifest, fetchObject, dir, work) {
114+ await git(["init", "--quiet", "--bare", dir]);
115+ for (const entry of manifest.chain) {
116+ if (!entry.key) continue;
117+ const file = await fetchObject(entry.key, join(work, `${entry.id}.bundle`));
118+ const size = statSync(file).size;
119+ if (size !== entry.size) throw new Error(`${entry.key}: ${size} bytes, the manifest says ${entry.size}`);
120+ if (entry.sha256) {
121+ const sha256 = await sha256Of(file);
122+ if (sha256 !== entry.sha256) throw new Error(`${entry.key}: SHA-256 ${sha256}, the manifest says ${entry.sha256}`);
123+ }
124+ await git(["bundle", "verify", "--quiet", file], { cwd: dir });
125+ await git(["fetch", "--quiet", "--no-tags", file, `+refs/*:${STAGING}/${entry.id}/*`], { cwd: dir });
126+ }
127+ const last = manifest.chain.at(-1).refs;
128+ const updates = Object.entries(last)
129+ .filter(([ref]) => ref !== "HEAD")
130+ .map(([ref, hash]) => `update ${ref} ${hash}\n`)
131+ .join("");
132+ const staged = await git(["for-each-ref", "--format=delete %(refname)", `${STAGING}/`], { cwd: dir });
133+ const commands = [updates.trimEnd(), staged].filter(Boolean).join("\n");
134+ if (commands) await git(["update-ref", "--stdin"], { cwd: dir, input: `${commands}\n` });
135+ const head = headBranch(last);
136+ if (head) await git(["symbolic-ref", "HEAD", head], { cwd: dir });
137+ // Every object every ref reaches is there.
138+ await git(["fsck", "--no-progress", "--connectivity-only"], { cwd: dir });
139+ return refsOf(dir);
140+}
141+
142+// ---------------------------------------------------------------------
143+
144+async function d1(sql) {
145+ const env = wranglerEnv();
146+ const { code, out } = await exec(process.execPath, [WRANGLER, "d1", "execute", DATABASE, "--remote", "--json", "--command", sql], {
147+ cwd: join(ROOT, "services/repos"),
148+ env,
149+ });
150+ if (code !== 0) throw new Error(out.slice(-600));
151+ return jsonFrom(out)[0]?.results ?? [];
152+}
153+
154+const quoted = (text) => `'${String(text).replaceAll("'", "''")}'`;
155+
156+/** The repository to drill, and whether its refs moved since its last backup. */
157+async function pick({ repo, repoId }) {
158+ const select = `SELECT r.id, r.namespace, r.name, r.refs_version, b.refs_version AS backed_version,
159+ coalesce(r.refs_open_until, 0) > coalesce(b.backed_up_ms, 0) AS opened
160+ FROM repo_backups b JOIN repos r ON r.id = b.repo_id
161+ WHERE b.last_entry IS NOT NULL AND r.deleted_at IS NULL`;
162+ let rows;
163+ if (repoId) rows = await d1(`${select} AND r.id = ${quoted(repoId)}`);
164+ else if (repo) {
165+ const [namespace, name] = repo.toLowerCase().split("/");
166+ rows = await d1(`${select} AND r.namespace = ${quoted(namespace)} AND r.name = ${quoted(name)}`);
167+ } else {
168+ rows = await d1(`${select} AND b.refs_version = r.refs_version AND coalesce(r.refs_open_until, 0) <= coalesce(b.backed_up_ms, 0)
169+ ORDER BY random() LIMIT 1`);
170+ }
171+ const row = rows[0];
172+ if (!row) throw new Error(repo || repoId ? `no backup of ${repo ?? repoId}` : "no repository has a backup yet");
173+ return {
174+ id: row.id,
175+ path: `${row.namespace}/${row.name}`,
176+ moved: row.refs_version !== row.backed_version || Boolean(row.opened),
177+ };
178+}
179+
180+/** Saves one object of the bucket to `file`. */
181+function bucketReader(localCopy) {
182+ if (localCopy) return async (key) => join(localCopy, key);
183+ return async (key, file) => {
184+ const env = wranglerEnv();
185+ const { code, out } = await exec(process.execPath, [WRANGLER, "r2", "object", "get", `${BUCKET}/${key}`, "--remote", "--file", file], { env });
186+ if (code !== 0) throw new Error(`${key} could not be read: ${out.slice(-400)}`);
187+ return file;
188+ };
189+}
190+
191+/** The live repository's refs, as a clone would see them. */
192+async function liveRefs(live) {
193+ const args = [];
194+ if (process.env.G1T_TOKEN && /^https?:/.test(live)) {
195+ const user = process.env.G1T_USER || "g1t";
196+ const basic = Buffer.from(`${user}:${process.env.G1T_TOKEN}`).toString("base64");
197+ args.push("-c", `http.extraHeader=Authorization: Basic ${basic}`);
198+ }
199+ return parseRefs(await git([...args, "ls-remote", live]));
200+}
201+
202+async function main() {
203+ const args = process.argv.slice(2);
204+ const option = (name) => {
205+ const at = args.indexOf(name);
206+ return at >= 0 ? args[at + 1] : undefined;
207+ };
208+ const keep = args.includes("--keep");
209+ const localCopy = option("--bundles");
210+ const target =
211+ localCopy && option("--repo-id")
212+ ? { id: option("--repo-id"), path: option("--repo") ?? null, moved: false }
213+ : await pick({ repo: option("--repo"), repoId: option("--repo-id") });
214+ const live = option("--live") ?? (target.path ? `https://g1t.sh/${target.path}.git` : null);
215+ if (!live) throw new Error("say which live repository to compare with: --live, or --repo");
216+
217+ const work = mkdtempSync(join(tmpdir(), "g1t-drill-"));
218+ try {
219+ const read = bucketReader(localCopy);
220+ const manifest = readManifest(readFileSync(await read(`backups/${target.id}/manifest.json`, join(work, "manifest.json")), "utf8"));
221+ const started = Date.now();
222+ const restored = await restore(manifest, read, join(work, "restored.git"), work);
223+ const seconds = ((Date.now() - started) / 1000).toFixed(1);
224+ const last = manifest.chain.at(-1);
225+ const bytes = manifest.chain.reduce((sum, entry) => sum + (entry.size ?? 0), 0);
226+ console.log(`${target.path ?? target.id}: ${manifest.chain.length} backups (${bytes} bytes), the last ${last.created_at}, restored in ${seconds}s`);
227+
228+ const fromChain = compareRefs(last.refs, restored);
229+ const fromLive = compareRefs(await liveRefs(live), restored);
230+ for (const [what, differences] of [
231+ ["the manifest", fromChain],
232+ ["the live repository", fromLive],
233+ ]) {
234+ if (differences.length === 0) {
235+ console.log(` every ref matches ${what} (${Object.keys(restored).length} refs)`);
236+ continue;
237+ }
238+ console.log(` ${differences.length} refs differ from ${what}:`);
239+ for (const { ref, want, have } of differences) console.log(` ${ref}: ${what} ${want ?? "(none)"}, restored ${have ?? "(none)"}`);
240+ }
241+ if (target.moved && fromLive.length > 0) {
242+ console.log(" The repository's refs moved since its last backup, so differences from it may be new work, not a fault.");
243+ }
244+ if (keep) console.log(` kept: ${work}`);
245+ return fromChain.length === 0 && fromLive.length === 0 ? 0 : 1;
246+ } finally {
247+ if (!keep) rmSync(work, { recursive: true, force: true });
248+ }
249+}
250+
251+if (process.argv[1]?.replaceAll("\\", "/").endsWith("scripts/ops/backup-restore-drill.mjs")) {
252+ main().then(
253+ (code) => process.exit(code),
254+ (error) => {
255+ console.error(`drill: ${error.message}`);
256+ process.exit(2);
257+ },
258+ );
259+}
+142−0
1+// The restore drill against a chain of real bundles: a full one and an
2+// incremental one, cut the way the runner's backup mode cuts them
3+// (crates/runner/src/backup.rs), from a local copy of the bucket.
4+
5+import assert from "node:assert/strict";
6+import { execFileSync, spawnSync } from "node:child_process";
7+import { createHash } from "node:crypto";
8+import { mkdirSync, mkdtempSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs";
9+import { tmpdir } from "node:os";
10+import { join } from "node:path";
11+import { test } from "node:test";
12+import { fileURLToPath } from "node:url";
13+
14+import { compareRefs, headBranch, parseRefs, readManifest, restore } from "./backup-restore-drill.mjs";
15+
16+const DRILL = fileURLToPath(new URL("./backup-restore-drill.mjs", import.meta.url));
17+const git = (cwd, ...args) => execFileSync("git", args, { cwd, encoding: "utf8" }).trim();
18+
19+function commit(dir, file, text) {
20+ writeFileSync(join(dir, file), text);
21+ git(dir, "add", "--all");
22+ git(dir, "-c", "user.name=t", "-c", "user.email=t@example.com", "commit", "--quiet", "-m", text);
23+}
24+
25+function refsOf(dir) {
26+ const refs = parseRefs(git(dir, "for-each-ref", "--format=%(objectname) %(refname)"));
27+ refs.HEAD = git(dir, "rev-parse", "HEAD");
28+ return refs;
29+}
30+
31+/** A bundle of the mirror, leaving out `prerequisites`, as the runner cuts one. */
32+function cut(mirror, prerequisites, out) {
33+ for (const hash of prerequisites) git(mirror, "update-ref", `refs/g1t-backup-prerequisites/${hash}`, hash);
34+ const not = prerequisites.length ? ["--not", "--glob=refs/g1t-backup-prerequisites/*"] : [];
35+ git(mirror, "bundle", "create", "--quiet", out, "--exclude=refs/g1t-backup-prerequisites/*", "--all", ...not);
36+ for (const hash of prerequisites) git(mirror, "update-ref", "-d", `refs/g1t-backup-prerequisites/${hash}`);
37+}
38+
39+function entry(id, kind, key, file, refs, prerequisites) {
40+ return {
41+ id,
42+ kind,
43+ key,
44+ created_at: "2026-10-06T02:53:00.000Z",
45+ refs_version: 1,
46+ refs,
47+ prerequisites,
48+ size: statSync(file).size,
49+ sha256: createHash("sha256").update(readFileSync(file)).digest("hex"),
50+ };
51+}
52+
53+test("refs, HEAD and differences are read as git writes them", () => {
54+ const a = "c71546fcd893ef8b0f57388b65e620d759705dda";
55+ const b = "4807077b296e6edbf410d55e72749d3e1170c291";
56+ assert.deepEqual(parseRefs(`${a}\tHEAD\n${a}\trefs/heads/main\n${b}\trefs/tags/v1\n${a}\trefs/tags/v1^{}\n`), {
57+ HEAD: a,
58+ "refs/heads/main": a,
59+ "refs/tags/v1": b,
60+ });
61+ assert.equal(headBranch({ HEAD: a, "refs/heads/dev": a, "refs/heads/main": a }), "refs/heads/main");
62+ assert.equal(headBranch({ HEAD: a, "refs/heads/dev": a }), "refs/heads/dev");
63+ assert.deepEqual(compareRefs({ x: a, y: b }, { x: a, z: b }), [
64+ { ref: "y", want: b, have: null },
65+ { ref: "z", want: null, have: b },
66+ ]);
67+ assert.throws(() => readManifest(JSON.stringify({ version: 2, chain: [] })), /version 2/);
68+ assert.throws(() => readManifest(JSON.stringify({ version: 1, chain: [{ kind: "incremental" }] })), /full/);
69+});
70+
71+test("a chain restores every ref, and the drill fails once the live repository differs", () => {
72+ const root = mkdtempSync(join(tmpdir(), "g1t-drill-test-"));
73+ try {
74+ const origin = join(root, "origin");
75+ mkdirSync(origin);
76+ git(origin, "init", "--quiet", "--initial-branch=main");
77+ commit(origin, "a.txt", "one");
78+ git(origin, "tag", "-a", "v1", "-m", "v1");
79+ const mirror = join(root, "mirror.git");
80+ const bucket = join(root, "bucket");
81+ const dir = join(bucket, "backups", "repo_1");
82+ mkdirSync(dir, { recursive: true });
83+
84+ git(root, "clone", "--mirror", "--quiet", origin, mirror);
85+ const fullRefs = refsOf(mirror);
86+ cut(mirror, [], join(dir, "1-full.bundle"));
87+ const full = entry("1", "full", "backups/repo_1/1-full.bundle", join(dir, "1-full.bundle"), fullRefs, []);
88+
89+ commit(origin, "b.txt", "two");
90+ git(origin, "branch", "feature");
91+ git(origin, "tag", "-d", "v1");
92+ rmSync(mirror, { recursive: true, force: true });
93+ git(root, "clone", "--mirror", "--quiet", origin, mirror);
94+ const prerequisites = [...new Set(Object.values(fullRefs))].sort();
95+ cut(mirror, prerequisites, join(dir, "2-incr.bundle"));
96+ const incr = entry("2", "incremental", "backups/repo_1/2-incr.bundle", join(dir, "2-incr.bundle"), refsOf(mirror), prerequisites);
97+
98+ const manifest = { version: 1, repo_id: "repo_1", store_key: "acme--rocket", path: null, updated_at: "", chain: [full, incr], previous: [] };
99+ writeFileSync(join(dir, "manifest.json"), JSON.stringify(manifest, null, 2));
100+
101+ const run = () =>
102+ spawnSync(process.execPath, [DRILL, "--repo-id", "repo_1", "--bundles", bucket, "--live", origin], { encoding: "utf8" });
103+ const passed = run();
104+ assert.equal(passed.status, 0, passed.stdout + passed.stderr);
105+ assert.match(passed.stdout, /every ref matches the live repository/);
106+
107+ commit(origin, "c.txt", "three");
108+ const failed = run();
109+ assert.equal(failed.status, 1, failed.stdout + failed.stderr);
110+ assert.match(failed.stdout, /refs differ from the live repository/);
111+
112+ // A bundle that is not what the manifest says is refused.
113+ writeFileSync(join(dir, "2-incr.bundle"), "not a bundle");
114+ const broken = run();
115+ assert.equal(broken.status, 2, broken.stdout + broken.stderr);
116+ assert.match(broken.stderr, /bytes, the manifest says|SHA-256/);
117+ } finally {
118+ rmSync(root, { recursive: true, force: true });
119+ }
120+});
121+
122+test("restore keeps only the refs the last entry names", async () => {
123+ const root = mkdtempSync(join(tmpdir(), "g1t-drill-restore-"));
124+ try {
125+ const origin = join(root, "origin");
126+ mkdirSync(origin);
127+ git(origin, "init", "--quiet", "--initial-branch=main");
128+ commit(origin, "a.txt", "one");
129+ git(origin, "branch", "gone");
130+ const mirror = join(root, "mirror.git");
131+ git(root, "clone", "--mirror", "--quiet", origin, mirror);
132+ const bundle = join(root, "1-full.bundle");
133+ cut(mirror, [], bundle);
134+ const refs = refsOf(mirror);
135+ delete refs["refs/heads/gone"];
136+ const manifest = { version: 1, chain: [entry("1", "full", "k", bundle, refs, [])] };
137+ const restored = await restore(manifest, async () => bundle, join(root, "restored.git"), root);
138+ assert.deepEqual(compareRefs(refs, restored), []);
139+ } finally {
140+ rmSync(root, { recursive: true, force: true });
141+ }
142+});
+1−0
1111 [dependencies]
1212 g1t-contracts.workspace = true
1313 g1t-kit.workspace = true
14+g1t-blobstore.workspace = true
1415 serde.workspace = true
1516 serde_json.workspace = true
1617 worker.workspace = true
+1−2
2222 mod oci;
2323 mod quota;
2424 mod range;
25−mod sigv4;
2625 mod store;
2726 mod token;
2827 mod upload;
166165 let host = store::var(env, "REGISTRY_HOST");
167166 Ok(Packages {
168167 db: Db { db: env.d1("DB")? },
169− store: Store::from_env(env)?,
168+ store: store::from_env(env)?,
170169 identity: env.service("IDENTITY")?,
171170 repos: env.service("REPOS")?,
172171 events: env.service("EVENTS")?,
+1−12
2727 }
2828
2929 /// A part of a blob a download asks for, resolved against its size.
30−#[derive(Clone, Copy, Debug, PartialEq, Eq)]
31−pub struct Wanted {
32− pub offset: u64,
33− pub length: u64,
34−}
35−
36−impl Wanted {
37− /// `bytes <first>-<last>/<size>`.
38− pub fn content_range(&self, size: u64) -> String {
39− format!("bytes {}-{}/{size}", self.offset, self.offset + self.length - 1)
40− }
41−}
30+pub use g1t_blobstore::Wanted;
4231
4332 /// What a download's `Range` header asks for, against a blob of `size`
4433 /// bytes. `Ok(None)`: the whole blob (no header, or one this does not
+0−201
1−//! AWS Signature Version 4, for S3-compatible storage: signing a request's
2−//! headers, and signing a URL that lets its holder download one object for
3−//! a while. R2's S3 endpoint takes the same signatures, which is how large
4−//! downloads are sent straight to it.
5−
6−use hmac::{Hmac, Mac};
7−use sha2::{Digest as _, Sha256};
8−
9−type HmacSha256 = Hmac<Sha256>;
10−
11−/// The hash of a payload that is not signed: bodies stream as they are.
12−pub const UNSIGNED: &str = "UNSIGNED-PAYLOAD";
13−
14−/// Who signs, and for which region.
15−#[derive(Clone, Debug)]
16−pub struct Credentials {
17− pub access_key_id: String,
18− pub secret_access_key: String,
19− pub region: String,
20−}
21−
22−/// `20130524T000000Z` from milliseconds since the epoch.
23−pub fn amz_date(now_ms: u64) -> String {
24− let text = g1t_contracts::time::rfc3339(now_ms);
25− let whole = text.split('.').next().unwrap_or(&text);
26− format!("{}Z", whole.replace(['-', ':'], ""))
27−}
28−
29−/// Percent-encodes everything but the unreserved characters, and `/` too
30−/// unless `path`.
31−pub fn uri_encode(text: &str, path: bool) -> String {
32− let mut out = String::with_capacity(text.len());
33− for byte in text.bytes() {
34− match byte {
35− b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'.' | b'_' | b'~' => out.push(byte as char),
36− b'/' if path => out.push('/'),
37− _ => out.push_str(&format!("%{byte:02X}")),
38− }
39− }
40− out
41−}
42−
43−fn hmac(key: &[u8], data: &str) -> Vec<u8> {
44− let mut mac = HmacSha256::new_from_slice(key).expect("HMAC takes a key of any length");
45− mac.update(data.as_bytes());
46− mac.finalize().into_bytes().to_vec()
47−}
48−
49−fn sha256_hex(data: &[u8]) -> String {
50− hex::encode(Sha256::digest(data))
51−}
52−
53−/// The query string, sorted and encoded as signing needs it.
54−fn canonical_query(query: &[(String, String)]) -> String {
55− let mut pairs: Vec<(String, String)> = query
56− .iter()
57− .map(|(key, value)| (uri_encode(key, false), uri_encode(value, false)))
58− .collect();
59− pairs.sort();
60− pairs
61− .iter()
62− .map(|(key, value)| format!("{key}={value}"))
63− .collect::<Vec<_>>()
64− .join("&")
65−}
66−
67−impl Credentials {
68− fn scope(&self, date: &str) -> String {
69− format!("{}/{}/s3/aws4_request", &date[..8], self.region)
70− }
71−
72− fn signature(&self, date: &str, canonical_request: &str) -> String {
73− let to_sign = format!(
74− "AWS4-HMAC-SHA256\n{date}\n{}\n{}",
75− self.scope(date),
76− sha256_hex(canonical_request.as_bytes())
77− );
78− let key = hmac(format!("AWS4{}", self.secret_access_key).as_bytes(), &date[..8]);
79− let key = hmac(&key, &self.region);
80− let key = hmac(&key, "s3");
81− let key = hmac(&key, "aws4_request");
82− hex::encode(hmac(&key, &to_sign))
83− }
84−
85− /// The `Authorization` header for a request. `headers` must include
86− /// `host`, `x-amz-date` and `x-amz-content-sha256`, lowercase; every
87− /// one given is signed.
88− pub fn authorization(
89− &self,
90− method: &str,
91− path: &str,
92− query: &[(String, String)],
93− headers: &[(String, String)],
94− payload_hash: &str,
95− ) -> String {
96− let mut headers: Vec<(String, String)> = headers
97− .iter()
98− .map(|(name, value)| (name.to_ascii_lowercase(), value.trim().to_owned()))
99− .collect();
100− headers.sort();
101− let date = headers
102− .iter()
103− .find(|(name, _)| name == "x-amz-date")
104− .map(|(_, value)| value.clone())
105− .unwrap_or_default();
106− let signed: Vec<&str> = headers.iter().map(|(name, _)| name.as_str()).collect();
107− let signed = signed.join(";");
108− let canonical_headers: String = headers.iter().map(|(name, value)| format!("{name}:{value}\n")).collect();
109− let canonical = format!(
110− "{method}\n{}\n{}\n{canonical_headers}\n{signed}\n{payload_hash}",
111− uri_encode(path, true),
112− canonical_query(query)
113− );
114− format!(
115− "AWS4-HMAC-SHA256 Credential={}/{}, SignedHeaders={signed}, Signature={}",
116− self.access_key_id,
117− self.scope(&date),
118− self.signature(&date, &canonical)
119− )
120− }
121−
122− /// A URL that lets anyone `GET` the object at `path` on `host` for
123− /// `expires` seconds from `date`. `base` is the scheme and host the
124− /// URL starts with.
125− pub fn presign_get(&self, base: &str, host: &str, path: &str, date: &str, expires: u32) -> String {
126− let mut query = vec![
127− ("X-Amz-Algorithm".to_owned(), "AWS4-HMAC-SHA256".to_owned()),
128− ("X-Amz-Credential".to_owned(), format!("{}/{}", self.access_key_id, self.scope(date))),
129− ("X-Amz-Date".to_owned(), date.to_owned()),
130− ("X-Amz-Expires".to_owned(), expires.to_string()),
131− ("X-Amz-SignedHeaders".to_owned(), "host".to_owned()),
132− ];
133− let canonical = format!(
134− "GET\n{}\n{}\nhost:{host}\n\nhost\n{UNSIGNED}",
135− uri_encode(path, true),
136− canonical_query(&query)
137− );
138− query.push(("X-Amz-Signature".to_owned(), self.signature(date, &canonical)));
139− format!("{base}{}?{}", uri_encode(path, true), canonical_query(&query))
140− }
141−}
142−
143−#[cfg(test)]
144−mod tests {
145− use super::*;
146−
147− fn example() -> Credentials {
148− Credentials {
149− access_key_id: "AKIAIOSFODNN7EXAMPLE".into(),
150− secret_access_key: "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY".into(),
151− region: "us-east-1".into(),
152− }
153− }
154−
155− /// AWS's own example of a presigned URL (Authenticating Requests:
156− /// Using Query Parameters).
157− #[test]
158− fn a_presigned_url_matches_the_aws_example() {
159− let url = example().presign_get(
160− "https://examplebucket.s3.amazonaws.com",
161− "examplebucket.s3.amazonaws.com",
162− "/test.txt",
163− "20130524T000000Z",
164− 86400,
165− );
166− assert!(url.starts_with("https://examplebucket.s3.amazonaws.com/test.txt?X-Amz-Algorithm=AWS4-HMAC-SHA256"));
167− assert!(url.contains("X-Amz-Credential=AKIAIOSFODNN7EXAMPLE%2F20130524%2Fus-east-1%2Fs3%2Faws4_request"));
168− assert!(url.contains("&X-Amz-Signature=aeeed9bbccd4d02ee5c0109b86d86835f995330da4c265957d157751f604d404&"), "{url}");
169− }
170−
171− /// AWS's own example of a signed GET with a range (Authenticating
172− /// Requests: Using the Authorization Header).
173− #[test]
174− fn a_signed_request_matches_the_aws_example() {
175− let empty = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855";
176− let headers = [
177− ("Host", "examplebucket.s3.amazonaws.com"),
178− ("Range", "bytes=0-9"),
179− ("x-amz-content-sha256", empty),
180− ("x-amz-date", "20130524T000000Z"),
181− ]
182− .map(|(name, value)| (name.to_owned(), value.to_owned()));
183− let authorization = example().authorization("GET", "/test.txt", &[], &headers, empty);
184− assert_eq!(
185− authorization,
186− "AWS4-HMAC-SHA256 Credential=AKIAIOSFODNN7EXAMPLE/20130524/us-east-1/s3/aws4_request, \
187− SignedHeaders=host;range;x-amz-content-sha256;x-amz-date, \
188− Signature=f0e8bdb87c964420e857bd35b5d6ed310bd44f0170aba48dd91039c6036bdb41"
189− );
190− }
191−
192− #[test]
193− fn dates_and_encoding() {
194− assert_eq!(amz_date(1_369_353_600_000), "20130524T000000Z");
195− assert_eq!(uri_encode("a b/c+d~", true), "a%20b/c%2Bd~");
196− assert_eq!(uri_encode("a/b", false), "a%2Fb");
197− let query = [("uploadId".to_owned(), "x y".to_owned()), ("partNumber".to_owned(), "2".to_owned())];
198− assert_eq!(canonical_query(&query), "partNumber=2&uploadId=x%20y");
199− assert_eq!(canonical_query(&[("uploads".to_owned(), String::new())]), "uploads=");
200− }
201−}
+32−0
1+//! Where packages' files are kept: the `BlobStore` port (crates/blobstore),
2+//! with R2 behind it on Cloudflare and any S3-compatible storage (MinIO in
3+//! the compose file) when self-hosted. BLOB_STORE chooses: `r2` (the
4+//! default) or `s3`.
5+//!
6+//! Files are content-addressed: a blob stored whole is at
7+//! `blobs/sha256/<hex>`, and one that came in parts at the key its upload
8+//! started with, which the `blobs` table records. Large uploads go up as
9+//! multipart parts of one size, as R2 requires (every part but the last
10+//! the same size), however the client cut its chunks.
11+
12+use worker::{Env, Result};
13+
14+#[cfg(test)]
15+pub use g1t_blobstore::Got;
16+pub use g1t_blobstore::{BlobStore, Part, Store, var};
17+
18+/// The `BLOBS` bucket; R2's S3 endpoint signs downloads when
19+/// R2_ACCESS_KEY_ID, R2_SECRET_ACCESS_KEY, R2_ACCOUNT_ID and R2_BUCKET are
20+/// set. Self-hosted: S3_BUCKET, and S3_PUBLIC_ENDPOINT for signed downloads.
21+const CONFIG: g1t_blobstore::Config = g1t_blobstore::Config {
22+ kind: "BLOB_STORE",
23+ binding: "BLOBS",
24+ r2_signer: Some(["R2_ACCESS_KEY_ID", "R2_SECRET_ACCESS_KEY", "R2_ACCOUNT_ID", "R2_BUCKET"]),
25+ s3_bucket: "S3_BUCKET",
26+ s3_public_endpoint: Some("S3_PUBLIC_ENDPOINT"),
27+};
28+
29+/// The store this installation keeps packages' files in.
30+pub fn from_env(env: &Env) -> Result<Store> {
31+ Store::from_env(env, &CONFIG)
32+}
+0−159
1−//! Where packages' files are kept: the `BlobStore` port, with R2 behind it
2−//! on Cloudflare and any S3-compatible storage (MinIO in the compose file)
3−//! when self-hosted. BLOB_STORE chooses: `r2` (the default) or `s3`.
4−//!
5−//! Files are content-addressed: a blob stored whole is at
6−//! `blobs/sha256/<hex>`, and one that came in parts at the key its upload
7−//! started with, which the `blobs` table records. Large uploads go up as
8−//! multipart parts of one size, as R2 requires (every part but the last
9−//! the same size), however the client cut its chunks.
10−
11−mod r2;
12−mod s3;
13−
14−use serde::{Deserialize, Serialize};
15−use worker::{Env, Response, ResponseBody, Result};
16−
17−use crate::range::Wanted;
18−
19−pub use r2::R2Store;
20−pub use s3::S3Store;
21−
22−/// One part of a multipart upload, as completing it needs.
23−#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
24−pub struct Part {
25− pub number: u16,
26− pub etag: String,
27−}
28−
29−/// An object read back.
30−pub struct Got {
31− /// The whole object's size, whatever range was read.
32− pub size: u64,
33− pub body: ResponseBody,
34−}
35−
36−impl Got {
37− pub async fn bytes(self) -> Result<Vec<u8>> {
38− match self.body {
39− ResponseBody::Empty => Ok(Vec::new()),
40− ResponseBody::Body(bytes) => Ok(bytes),
41− stream => Response::from_body(stream)?.bytes().await,
42− }
43− }
44−}
45−
46−/// What the registry needs of storage.
47−pub trait BlobStore {
48− async fn put(&self, key: &str, bytes: Vec<u8>) -> Result<()>;
49− /// The object, or the part of it `range` asks for.
50− async fn get(&self, key: &str, range: Option<Wanted>) -> Result<Option<Got>>;
51− /// The object's size, if it is there.
52− async fn head(&self, key: &str) -> Result<Option<u64>>;
53− async fn delete(&self, key: &str) -> Result<()>;
54− /// Starts a multipart upload to `key`, and says its id.
55− async fn create_multipart(&self, key: &str) -> Result<String>;
56− async fn upload_part(&self, key: &str, upload_id: &str, number: u16, bytes: Vec<u8>) -> Result<Part>;
57− async fn complete_multipart(&self, key: &str, upload_id: &str, parts: &[Part]) -> Result<()>;
58− async fn abort_multipart(&self, key: &str, upload_id: &str) -> Result<()>;
59− /// A URL that downloads the object for `expires` seconds without
60− /// passing through this Worker, when the store can sign one.
61− fn presign_get(&self, key: &str, expires: u32, now_ms: u64) -> Option<String>;
62−
63− /// The whole object, read into memory: for small ones only.
64− async fn read(&self, key: &str) -> Result<Option<Vec<u8>>> {
65− match self.get(key, None).await? {
66− Some(got) => Ok(Some(got.bytes().await?)),
67− None => Ok(None),
68− }
69− }
70−}
71−
72−/// The store this installation is configured with.
73−pub enum Store {
74− R2(R2Store),
75− S3(S3Store),
76−}
77−
78−impl Store {
79− pub fn from_env(env: &Env) -> Result<Store> {
80− let kind = env.var("BLOB_STORE").map(|v| v.to_string()).unwrap_or_default();
81− if kind == "s3" {
82− return Ok(Store::S3(S3Store::from_env(env)?));
83− }
84− Ok(Store::R2(R2Store::from_env(env)?))
85− }
86−}
87−
88−impl BlobStore for Store {
89− async fn put(&self, key: &str, bytes: Vec<u8>) -> Result<()> {
90− match self {
91− Store::R2(s) => s.put(key, bytes).await,
92− Store::S3(s) => s.put(key, bytes).await,
93− }
94− }
95−
96− async fn get(&self, key: &str, range: Option<Wanted>) -> Result<Option<Got>> {
97− match self {
98− Store::R2(s) => s.get(key, range).await,
99− Store::S3(s) => s.get(key, range).await,
100− }
101− }
102−
103− async fn head(&self, key: &str) -> Result<Option<u64>> {
104− match self {
105− Store::R2(s) => s.head(key).await,
106− Store::S3(s) => s.head(key).await,
107− }
108− }
109−
110− async fn delete(&self, key: &str) -> Result<()> {
111− match self {
112− Store::R2(s) => s.delete(key).await,
113− Store::S3(s) => s.delete(key).await,
114− }
115− }
116−
117− async fn create_multipart(&self, key: &str) -> Result<String> {
118− match self {
119− Store::R2(s) => s.create_multipart(key).await,
120− Store::S3(s) => s.create_multipart(key).await,
121− }
122− }
123−
124− async fn upload_part(&self, key: &str, upload_id: &str, number: u16, bytes: Vec<u8>) -> Result<Part> {
125− match self {
126− Store::R2(s) => s.upload_part(key, upload_id, number, bytes).await,
127− Store::S3(s) => s.upload_part(key, upload_id, number, bytes).await,
128− }
129− }
130−
131− async fn complete_multipart(&self, key: &str, upload_id: &str, parts: &[Part]) -> Result<()> {
132− match self {
133− Store::R2(s) => s.complete_multipart(key, upload_id, parts).await,
134− Store::S3(s) => s.complete_multipart(key, upload_id, parts).await,
135− }
136− }
137−
138− async fn abort_multipart(&self, key: &str, upload_id: &str) -> Result<()> {
139− match self {
140− Store::R2(s) => s.abort_multipart(key, upload_id).await,
141− Store::S3(s) => s.abort_multipart(key, upload_id).await,
142− }
143− }
144−
145− fn presign_get(&self, key: &str, expires: u32, now_ms: u64) -> Option<String> {
146− match self {
147− Store::R2(s) => s.presign_get(key, expires, now_ms),
148− Store::S3(s) => s.presign_get(key, expires, now_ms),
149− }
150− }
151−}
152−
153−/// A configuration variable, or the empty string.
154−pub(crate) fn var(env: &Env, name: &str) -> String {
155− env.var(name)
156− .map(|v| v.to_string())
157− .or_else(|_| env.secret(name).map(|v| v.to_string()))
158− .unwrap_or_default()
159−}
+0−93
1−//! The R2 adapter: the `BLOBS` bucket binding for everything, and R2's S3
2−//! endpoint only to sign download URLs, when R2_ACCESS_KEY_ID,
3−//! R2_SECRET_ACCESS_KEY, R2_ACCOUNT_ID and R2_BUCKET are set. Without
4−//! them, large blobs stream through the Worker like small ones.
5−
6−use worker::{Bucket, Env, Range, Result, UploadedPart};
7−
8−use super::{BlobStore, Got, Part, var};
9−use crate::range::Wanted;
10−use crate::sigv4::{Credentials, amz_date};
11−
12−pub struct R2Store {
13− bucket: Bucket,
14− signer: Option<(Credentials, String, String)>,
15−}
16−
17−impl R2Store {
18− pub fn from_env(env: &Env) -> Result<R2Store> {
19− let (key, secret, account, bucket) = (
20− var(env, "R2_ACCESS_KEY_ID"),
21− var(env, "R2_SECRET_ACCESS_KEY"),
22− var(env, "R2_ACCOUNT_ID"),
23− var(env, "R2_BUCKET"),
24− );
25− let signer = (!key.is_empty() && !secret.is_empty() && !account.is_empty() && !bucket.is_empty()).then(|| {
26− (
27− Credentials { access_key_id: key, secret_access_key: secret, region: "auto".to_owned() },
28− format!("{account}.r2.cloudflarestorage.com"),
29− bucket,
30− )
31− });
32− Ok(R2Store { bucket: env.bucket("BLOBS")?, signer })
33− }
34−}
35−
36−impl BlobStore for R2Store {
37− async fn put(&self, key: &str, bytes: Vec<u8>) -> Result<()> {
38− self.bucket.put(key, bytes).execute().await?;
39− Ok(())
40− }
41−
42− async fn get(&self, key: &str, range: Option<Wanted>) -> Result<Option<Got>> {
43− let mut get = self.bucket.get(key);
44− if let Some(range) = range {
45− get = get.range(Range::OffsetWithLength { offset: range.offset, length: range.length });
46− }
47− let Some(object) = get.execute().await? else {
48− return Ok(None);
49− };
50− let size = object.size();
51− let Some(body) = object.body() else {
52− return Ok(None);
53− };
54− Ok(Some(Got { size, body: body.response_body()? }))
55− }
56−
57− async fn head(&self, key: &str) -> Result<Option<u64>> {
58− Ok(self.bucket.head(key).await?.map(|object| object.size()))
59− }
60−
61− async fn delete(&self, key: &str) -> Result<()> {
62− self.bucket.delete(key).await
63− }
64−
65− async fn create_multipart(&self, key: &str) -> Result<String> {
66− let upload = self.bucket.create_multipart_upload(key).execute().await?;
67− Ok(upload.upload_id().await)
68− }
69−
70− async fn upload_part(&self, key: &str, upload_id: &str, number: u16, bytes: Vec<u8>) -> Result<Part> {
71− let upload = self.bucket.resume_multipart_upload(key, upload_id)?;
72− let part = upload.upload_part(number, bytes).await?;
73− Ok(Part { number: part.part_number(), etag: part.etag() })
74− }
75−
76− async fn complete_multipart(&self, key: &str, upload_id: &str, parts: &[Part]) -> Result<()> {
77− let upload = self.bucket.resume_multipart_upload(key, upload_id)?;
78− upload
79− .complete(parts.iter().map(|part| UploadedPart::new(part.number, part.etag.clone())))
80− .await?;
81− Ok(())
82− }
83−
84− async fn abort_multipart(&self, key: &str, upload_id: &str) -> Result<()> {
85− self.bucket.resume_multipart_upload(key, upload_id)?.abort().await
86− }
87−
88− fn presign_get(&self, key: &str, expires: u32, now_ms: u64) -> Option<String> {
89− let (credentials, host, bucket) = self.signer.as_ref()?;
90− let path = format!("/{bucket}/{key}");
91− Some(credentials.presign_get(&format!("https://{host}"), host, &path, &amz_date(now_ms), expires))
92− }
93−}
+0−249
1−//! The S3 adapter, for self-hosted installations: any S3-compatible store
2−//! (MinIO, Ceph, Garage, AWS) over fetch, signed with SigV4, path-style.
3−//! S3_ENDPOINT, S3_BUCKET, S3_ACCESS_KEY_ID, S3_SECRET_ACCESS_KEY and
4−//! S3_REGION say where; S3_PUBLIC_ENDPOINT, when set, is the address
5−//! clients reach the store at, and large downloads are then sent there
6−//! with a signed URL instead of through the Worker.
7−
8−use worker::wasm_bindgen::JsValue;
9−use worker::{Env, Fetch, Headers, Method, Request, RequestInit, Response, Result, Url};
10−
11−use super::{BlobStore, Got, Part, var};
12−use crate::range::Wanted;
13−use crate::sigv4::{Credentials, UNSIGNED, amz_date};
14−
15−pub struct S3Store {
16− /// `http://minio:9000`, without a trailing slash.
17− endpoint: String,
18− /// Where clients reach the same store, for signed URLs.
19− public_endpoint: Option<String>,
20− bucket: String,
21− credentials: Credentials,
22−}
23−
24−fn failed(what: &str, status: u16, body: &str) -> worker::Error {
25− let said: String = body.chars().take(300).collect();
26− worker::Error::RustError(format!("storage {what} failed with status {status}: {said}"))
27−}
28−
29−/// The text of the first `<tag>` in an XML answer.
30−fn xml_value<'a>(xml: &'a str, tag: &str) -> Option<&'a str> {
31− let open = format!("<{tag}>");
32− let start = xml.find(&open)? + open.len();
33− let end = xml[start..].find(&format!("</{tag}>"))? + start;
34− Some(&xml[start..end])
35−}
36−
37−fn host_of(endpoint: &str) -> String {
38− endpoint
39− .split_once("://")
40− .map_or(endpoint, |(_, rest)| rest)
41− .split('/')
42− .next()
43− .unwrap_or_default()
44− .to_owned()
45−}
46−
47−impl S3Store {
48− pub fn from_env(env: &Env) -> Result<S3Store> {
49− let endpoint = var(env, "S3_ENDPOINT").trim_end_matches('/').to_owned();
50− let bucket = var(env, "S3_BUCKET");
51− if endpoint.is_empty() || bucket.is_empty() {
52− return Err(worker::Error::RustError("BLOB_STORE is s3, but S3_ENDPOINT or S3_BUCKET is not set".into()));
53− }
54− let region = var(env, "S3_REGION");
55− let public = var(env, "S3_PUBLIC_ENDPOINT").trim_end_matches('/').to_owned();
56− Ok(S3Store {
57− endpoint,
58− public_endpoint: (!public.is_empty()).then_some(public),
59− bucket,
60− credentials: Credentials {
61− access_key_id: var(env, "S3_ACCESS_KEY_ID"),
62− secret_access_key: var(env, "S3_SECRET_ACCESS_KEY"),
63− region: if region.is_empty() { "us-east-1".to_owned() } else { region },
64− },
65− })
66− }
67−
68− fn path(&self, key: &str) -> String {
69− format!("/{}/{key}", self.bucket)
70− }
71−
72− /// Sends one signed request, and answers with the response whatever
73− /// its status.
74− async fn send(
75− &self,
76− method: Method,
77− key: &str,
78− query: &[(String, String)],
79− extra: &[(&str, String)],
80− body: Option<Vec<u8>>,
81− ) -> Result<Response> {
82− let path = self.path(key);
83− let date = amz_date(g1t_kit::now_ms());
84− let mut signed = vec![
85− ("host".to_owned(), host_of(&self.endpoint)),
86− ("x-amz-content-sha256".to_owned(), UNSIGNED.to_owned()),
87− ("x-amz-date".to_owned(), date),
88− ];
89− for (name, value) in extra {
90− signed.push(((*name).to_owned(), value.clone()));
91− }
92− let authorization = self
93− .credentials
94− .authorization(method.as_ref(), &path, query, &signed, UNSIGNED);
95− let headers = Headers::new();
96− for (name, value) in &signed {
97− if name != "host" {
98− headers.set(name, value)?;
99− }
100− }
101− headers.set("authorization", &authorization)?;
102− let mut url = Url::parse(&format!("{}{}", self.endpoint, crate::sigv4::uri_encode(&path, true)))?;
103− if !query.is_empty() {
104− let text: Vec<String> = query
105− .iter()
106− .map(|(k, v)| {
107− let (k, v) = (crate::sigv4::uri_encode(k, false), crate::sigv4::uri_encode(v, false));
108− if v.is_empty() { format!("{k}=") } else { format!("{k}={v}") }
109− })
110− .collect();
111− url.set_query(Some(&text.join("&")));
112− }
113− let mut init = RequestInit::new();
114− init.with_method(method).with_headers(headers);
115− if let Some(body) = body {
116− init.with_body(Some(JsValue::from(worker::js_sys::Uint8Array::from(body.as_slice()))));
117− }
118− Fetch::Request(Request::new_with_init(url.as_str(), &init)?).send().await
119− }
120−
121− async fn ok(&self, what: &str, mut response: Response) -> Result<Response> {
122− let status = response.status_code();
123− if (200..300).contains(&status) {
124− return Ok(response);
125− }
126− let body = response.text().await.unwrap_or_default();
127− Err(failed(what, status, &body))
128− }
129−}
130−
131−fn query(pairs: &[(&str, &str)]) -> Vec<(String, String)> {
132− pairs.iter().map(|(k, v)| ((*k).to_owned(), (*v).to_owned())).collect()
133−}
134−
135−impl BlobStore for S3Store {
136− async fn put(&self, key: &str, bytes: Vec<u8>) -> Result<()> {
137− let length = bytes.len().to_string();
138− let response = self
139− .send(Method::Put, key, &[], &[("content-length", length)], Some(bytes))
140− .await?;
141− self.ok("put", response).await.map(|_| ())
142− }
143−
144− async fn get(&self, key: &str, range: Option<Wanted>) -> Result<Option<Got>> {
145− let extra: Vec<(&str, String)> = range
146− .map(|r| ("range", format!("bytes={}-{}", r.offset, r.offset + r.length - 1)))
147− .into_iter()
148− .collect();
149− let response = self.send(Method::Get, key, &[], &extra, None).await?;
150− if response.status_code() == 404 {
151− return Ok(None);
152− }
153− let response = self.ok("get", response).await?;
154− let size = match response.headers().get("content-range")? {
155− // `bytes 0-9/100`: the whole object's size is after the slash.
156− Some(range) => range.rsplit('/').next().and_then(|n| n.parse().ok()).unwrap_or(0),
157− None => response.headers().get("content-length")?.and_then(|n| n.parse().ok()).unwrap_or(0),
158− };
159− let (_, body) = response.into_parts();
160− Ok(Some(Got { size, body }))
161− }
162−
163− async fn head(&self, key: &str) -> Result<Option<u64>> {
164− let response = self.send(Method::Head, key, &[], &[], None).await?;
165− if response.status_code() == 404 {
166− return Ok(None);
167− }
168− let response = self.ok("head", response).await?;
169− Ok(response.headers().get("content-length")?.and_then(|n| n.parse().ok()))
170− }
171−
172− async fn delete(&self, key: &str) -> Result<()> {
173− let response = self.send(Method::Delete, key, &[], &[], None).await?;
174− if response.status_code() == 404 {
175− return Ok(());
176− }
177− self.ok("delete", response).await.map(|_| ())
178− }
179−
180− async fn create_multipart(&self, key: &str) -> Result<String> {
181− let response = self.send(Method::Post, key, &query(&[("uploads", "")]), &[], None).await?;
182− let text = self.ok("create multipart", response).await?.text().await?;
183− xml_value(&text, "UploadId")
184− .map(str::to_owned)
185− .ok_or_else(|| failed("create multipart", 200, &text))
186− }
187−
188− async fn upload_part(&self, key: &str, upload_id: &str, number: u16, bytes: Vec<u8>) -> Result<Part> {
189− let number_text = number.to_string();
190− let length = bytes.len().to_string();
191− let response = self
192− .send(
193− Method::Put,
194− key,
195− &query(&[("partNumber", &number_text), ("uploadId", upload_id)]),
196− &[("content-length", length)],
197− Some(bytes),
198− )
199− .await?;
200− let response = self.ok("upload part", response).await?;
201− let etag = response.headers().get("etag")?.unwrap_or_default();
202− Ok(Part { number, etag })
203− }
204−
205− async fn complete_multipart(&self, key: &str, upload_id: &str, parts: &[Part]) -> Result<()> {
206− let mut xml = String::from("<CompleteMultipartUpload>");
207− for part in parts {
208− xml.push_str(&format!("<Part><PartNumber>{}</PartNumber><ETag>{}</ETag></Part>", part.number, part.etag));
209− }
210− xml.push_str("</CompleteMultipartUpload>");
211− let length = xml.len().to_string();
212− let response = self
213− .send(Method::Post, key, &query(&[("uploadId", upload_id)]), &[("content-length", length)], Some(xml.into_bytes()))
214− .await?;
215− // S3 may answer 200 and still have failed, saying so in the body.
216− let text = self.ok("complete multipart", response).await?.text().await?;
217− if text.contains("<Error>") {
218− return Err(failed("complete multipart", 200, &text));
219− }
220− Ok(())
221− }
222−
223− async fn abort_multipart(&self, key: &str, upload_id: &str) -> Result<()> {
224− let response = self.send(Method::Delete, key, &query(&[("uploadId", upload_id)]), &[], None).await?;
225− if response.status_code() == 404 {
226− return Ok(());
227− }
228− self.ok("abort multipart", response).await.map(|_| ())
229− }
230−
231− fn presign_get(&self, key: &str, expires: u32, now_ms: u64) -> Option<String> {
232− let base = self.public_endpoint.as_ref()?;
233− Some(self.credentials.presign_get(base, &host_of(base), &self.path(key), &amz_date(now_ms), expires))
234− }
235−}
236−
237−#[cfg(test)]
238−mod tests {
239− use super::*;
240−
241− #[test]
242− fn answers_are_read_from_their_xml() {
243− let xml = "<InitiateMultipartUploadResult><Bucket>b</Bucket><UploadId>abc-123</UploadId></InitiateMultipartUploadResult>";
244− assert_eq!(xml_value(xml, "UploadId"), Some("abc-123"));
245− assert_eq!(xml_value(xml, "Key"), None);
246− assert_eq!(host_of("http://minio:9000"), "minio:9000");
247− assert_eq!(host_of("https://s3.example.com/base"), "s3.example.com");
248− }
249−}
+1−0
1313 g1t-kit.workspace = true
1414 g1t-scan.workspace = true
1515 g1t-secrets.workspace = true
16+g1t-blobstore.workspace = true
1617 serde.workspace = true
1718 serde_json.workspace = true
1819 worker.workspace = true
+65−0
1+-- Nightly backups of every repository, outside the git store
2+-- (src/backups.rs; docs/ARTIFACTS.md, R11). One row per repository that
3+-- has been queued at least once: its last backup, and the job in hand.
4+--
5+-- status: idle | queued | running. The nightly cron queues the
6+-- repositories whose refs moved since their last backup; the runner's
7+-- sweep claims queued ones (`claim_backups`) and starts a sandbox for
8+-- each; the sandbox's `backup_complete` or `backup_fail` makes it idle
9+-- again, or queued for another try.
10+-- queued_ms, claimed_ms: milliseconds since the epoch. A job running past
11+-- its lease (3 hours) goes back in the queue.
12+-- attempts: tries tonight; past 3 it waits for the next night.
13+-- last_error: why the last try failed.
14+--
15+-- The job in hand, while running:
16+-- job_id, token_hash: the job, and the SHA-256 of the token its sandbox
17+-- holds (the only credential it has for g1t).
18+-- target_version: the repository's refs_version when the clone began,
19+-- which the backup is recorded at once done.
20+-- backed_from_ms: when that was.
21+-- store_key: the repository's name in the git store, for its meters.
22+-- upload_key, upload_id: the bundle's key in storage, and its multipart
23+-- upload. upload_kind: full | incr. upload_entry: the entry's id in the
24+-- manifest (`20261006T025300Z`).
25+-- prerequisites: JSON array, the commits the bundle leaves out.
26+--
27+-- The last backup:
28+-- refs_version: the refs_version it was cut at. The repository is due
29+-- again once its own goes past this, or once a credential that can push
30+-- is handed out after backed_up_ms (repos.refs_open_until).
31+-- backed_up_ms: when its clone began.
32+-- tips: JSON object, every ref it held by name: the next bundle's
33+-- prerequisites. The manifest in storage says the same, and is what a
34+-- restore reads.
35+-- last_entry: its id in the manifest. When the manifest's last entry is
36+-- another, the next backup is full.
37+CREATE TABLE repo_backups (
38+ repo_id TEXT PRIMARY KEY,
39+ status TEXT NOT NULL DEFAULT 'idle',
40+ queued_ms INTEGER,
41+ claimed_ms INTEGER,
42+ attempts INTEGER NOT NULL DEFAULT 0,
43+ last_error TEXT,
44+ job_id TEXT,
45+ token_hash TEXT,
46+ target_version INTEGER,
47+ backed_from_ms INTEGER,
48+ store_key TEXT,
49+ upload_key TEXT,
50+ upload_id TEXT,
51+ upload_kind TEXT,
52+ upload_entry TEXT,
53+ prerequisites TEXT,
54+ refs_version INTEGER,
55+ backed_up_ms INTEGER,
56+ tips TEXT,
57+ last_entry TEXT
58+);
59+CREATE INDEX repo_backups_queue ON repo_backups (status, queued_ms);
60+CREATE UNIQUE INDEX repo_backups_job ON repo_backups (job_id) WHERE job_id IS NOT NULL;
61+
62+-- A backup's clone is metered as `internal.git.backup_fetch`: an operation
63+-- on g1t's own bill, never on a workspace's.
64+INSERT INTO operation_mapping (meter, cost_operations, billable_operations, note, updated_at) VALUES
65+ ('internal.git.backup_fetch', 1, 0, 'Nightly backup clone (R11): g1t''s cost, not the workspace''s', '2026-10-06T00:00:00Z');
+992−0
1+//! Nightly backups of every repository, outside the git store
2+//! (docs/ARTIFACTS.md, R11; the flow is in `g1t_contracts::backups`).
3+//!
4+//! Each repository whose refs moved since its last backup gets a
5+//! `git bundle`: a full one first, then incremental ones whose
6+//! prerequisites are the commits the one before ended at, and a full one
7+//! again after [`Settings::full_every`] incremental ones, so a restore
8+//! never reads a long chain. Bundles and a manifest that lists the chain
9+//! are kept in object storage through the `BlobStore` port: the BACKUPS R2
10+//! bucket hosted, any S3-compatible store (MinIO in the compose file)
11+//! self-hosted, as BACKUP_STORE says.
12+//!
13+//! ```text
14+//! backups/<repo id>/manifest.json
15+//! backups/<repo id>/<20261006T025300Z>-full.bundle
16+//! backups/<repo id>/<20261007T025300Z>-incr.bundle
17+//! ```
18+//!
19+//! Restoring is fetching each bundle of `chain` in order into an empty
20+//! repository, then setting every ref to what the last entry says
21+//! (scripts/ops/backup-restore-drill.mjs does it and compares).
22+//!
23+//! `repo_backups` keeps, per repository, the last backup (the refs version
24+//! it was cut at, its tips, when) and the job in hand, if any:
25+//! `idle` → `queued` (the nightly cron) → `running` (claimed by the
26+//! runner's sweep) → `idle` again, done or failed. A job that has been
27+//! running longer than [`LEASE_MS`] is queued again; one that failed
28+//! [`MAX_ATTEMPTS`] times waits for the next night.
29+
30+use std::collections::{BTreeMap, BTreeSet};
31+
32+use g1t_blobstore::{BlobStore, Config, Part, Store};
33+use g1t_contracts::backups::{
34+ BackupClaim, BackupComplete, BackupFail, BackupJobArgs, BackupKind, BackupPart, BackupSpec, ClaimBackupsArgs, PART_BYTES,
35+};
36+use g1t_contracts::repos::RepoPath;
37+use g1t_contracts::time::rfc3339;
38+use g1t_contracts::{FailureCode, Outcome, new_id};
39+use serde::{Deserialize, Serialize};
40+use worker::wasm_bindgen::JsValue;
41+use worker::{D1Database, Env, Result};
42+
43+use crate::PULLS_NAMESPACE;
44+use crate::meters;
45+use crate::registry::{Registry, store_key};
46+use crate::store::{GitStore, Scope};
47+
48+/// Where backups are kept: the BACKUPS bucket, or, when BACKUP_STORE is
49+/// `s3`, the bucket BACKUP_S3_BUCKET names on the installation's S3 store.
50+pub const STORAGE: Config = Config {
51+ kind: "BACKUP_STORE",
52+ binding: "BACKUPS",
53+ r2_signer: None,
54+ s3_bucket: "BACKUP_S3_BUCKET",
55+ s3_public_endpoint: None,
56+};
57+
58+/// The meter a backup's clone is counted under: an operation for g1t's own
59+/// bill, never for the workspace's (migrations/0013).
60+pub const FETCH_METER: &str = "internal.git.backup_fetch";
61+
62+/// How long a claimed job may run before it is given to another sandbox.
63+pub const LEASE_MS: u64 = 3 * 60 * 60 * 1000;
64+/// How many times a night a backup is tried.
65+pub const MAX_ATTEMPTS: u32 = 3;
66+/// How many backups of deleted repositories one night removes.
67+const PRUNES_PER_NIGHT: u32 = 50;
68+
69+/// How backups are paced, from the service's variables.
70+#[derive(Clone, Copy, Debug, PartialEq, Eq)]
71+pub struct Settings {
72+ /// BACKUPS_PER_NIGHT: how many repositories one night queues.
73+ pub per_night: u32,
74+ /// BACKUP_FULL_EVERY: incremental bundles before the next full one.
75+ pub full_every: u32,
76+}
77+
78+impl Default for Settings {
79+ fn default() -> Self {
80+ Settings { per_night: 200, full_every: 30 }
81+ }
82+}
83+
84+impl Settings {
85+ pub fn from_env(env: &Env) -> Settings {
86+ let number = |name: &str| env.var(name).ok().and_then(|value| value.to_string().parse::<u32>().ok());
87+ let defaults = Settings::default();
88+ Settings {
89+ per_night: number("BACKUPS_PER_NIGHT").unwrap_or(defaults.per_night),
90+ full_every: number("BACKUP_FULL_EVERY").unwrap_or(defaults.full_every).max(1),
91+ }
92+ }
93+}
94+
95+/// The storage backups go to, or None when this installation has none
96+/// (no BACKUPS binding and BACKUP_STORE is not `s3`): backups are then off.
97+pub fn storage(env: &Env) -> Option<Store> {
98+ match Store::from_env(env, &STORAGE) {
99+ Ok(store) => Some(store),
100+ Err(error) => {
101+ worker::console_log!("repos: backups are off: {error}");
102+ None
103+ }
104+ }
105+}
106+
107+// ---------------------------------------------------------------------
108+// Which repositories are due
109+// ---------------------------------------------------------------------
110+
111+/// A repository and its last backup, as the nightly query reads them.
112+#[derive(Clone, Debug, Default, PartialEq, Eq, Deserialize)]
113+pub struct Candidate {
114+ pub repo_id: String,
115+ pub namespace: String,
116+ pub created_at: String,
117+ pub refs_version: u64,
118+ /// Until when a credential that can push was out of g1t's hands
119+ /// (`git_access`); a push with it does not move `refs_version`.
120+ pub refs_open_until: u64,
121+ pub deleted: bool,
122+ pub retired: bool,
123+ /// None: never backed up, and no row.
124+ pub status: Option<String>,
125+ pub backed_version: Option<u64>,
126+ pub backed_up_ms: u64,
127+}
128+
129+/// Whether a repository needs a backup tonight: it is live, it is not a
130+/// pull request's working copy (whose work lands in its repository, and
131+/// whose head is kept there once it goes), nothing is already queued or
132+/// running for it, and its refs moved since the last backup: its
133+/// `refs_version` went past the one backed up, or a credential that could
134+/// push was handed out after the last backup started.
135+pub fn is_due(c: &Candidate) -> bool {
136+ if c.deleted || c.retired || c.namespace == PULLS_NAMESPACE {
137+ return false;
138+ }
139+ match c.status.as_deref() {
140+ None => true,
141+ Some("idle") => match c.backed_version {
142+ None => true,
143+ Some(version) => version < c.refs_version || c.refs_open_until > c.backed_up_ms,
144+ },
145+ _ => false,
146+ }
147+}
148+
149+/// The repositories to queue: those due, the longest since their last
150+/// backup first (never backed up first of all, oldest repository first),
151+/// `limit` at most.
152+pub fn pick_due(candidates: &[Candidate], limit: usize) -> Vec<String> {
153+ let mut due: Vec<&Candidate> = candidates.iter().filter(|c| is_due(c)).collect();
154+ due.sort_by(|a, b| {
155+ a.backed_up_ms
156+ .cmp(&b.backed_up_ms)
157+ .then_with(|| a.created_at.cmp(&b.created_at))
158+ .then_with(|| a.repo_id.cmp(&b.repo_id))
159+ });
160+ due.into_iter().take(limit).map(|c| c.repo_id.clone()).collect()
161+}
162+
163+/// The same choice in SQL, so a night reads only what it queues. `?1`:
164+/// the working copies' namespace, `?2`: how many.
165+const DUE_SQL: &str = "
166+SELECT r.id AS repo_id, r.namespace, r.created_at,
167+ coalesce(r.refs_version, 0) AS refs_version,
168+ coalesce(r.refs_open_until, 0) AS refs_open_until,
169+ (r.deleted_at IS NOT NULL) AS deleted,
170+ (r.retired_at IS NOT NULL) AS retired,
171+ b.status, b.refs_version AS backed_version,
172+ coalesce(b.backed_up_ms, 0) AS backed_up_ms
173+FROM repos r LEFT JOIN repo_backups b ON b.repo_id = r.id
174+WHERE r.deleted_at IS NULL AND r.retired_at IS NULL AND r.namespace != ?1
175+ AND (b.repo_id IS NULL
176+ OR (b.status = 'idle'
177+ AND (b.refs_version IS NULL
178+ OR b.refs_version < coalesce(r.refs_version, 0)
179+ OR coalesce(r.refs_open_until, 0) > coalesce(b.backed_up_ms, 0))))
180+ORDER BY coalesce(b.backed_up_ms, 0), r.created_at, r.id
181+LIMIT ?2";
182+
183+// ---------------------------------------------------------------------
184+// The chain and its manifest
185+// ---------------------------------------------------------------------
186+
187+/// One backup in a chain: a bundle, or, when nothing new was there to
188+/// bundle (a branch deleted, a ref pointed at a commit already kept), only
189+/// the refs it ended with.
190+#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
191+pub struct Entry {
192+ /// When it was cut, `20261006T025300Z`: also its name in the chain.
193+ pub id: String,
194+ pub kind: BackupKind,
195+ /// The bundle's key in storage; None when only the refs moved.
196+ pub key: Option<String>,
197+ pub created_at: String,
198+ /// The repository's `refs_version` when the clone began.
199+ pub refs_version: u64,
200+ /// Every ref, by name, once this backup is applied.
201+ pub refs: BTreeMap<String, String>,
202+ /// The commits the bundle leaves out: the previous entry's tips.
203+ pub prerequisites: Vec<String>,
204+ pub size: u64,
205+ pub sha256: Option<String>,
206+}
207+
208+/// What restoring a repository reads first: its chain, oldest first, and
209+/// the chain before it, kept until the next full backup replaces it.
210+#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
211+pub struct Manifest {
212+ pub version: u32,
213+ pub repo_id: String,
214+ /// The repository's name in the git store, and its path, when last cut.
215+ pub store_key: String,
216+ pub path: Option<RepoPath>,
217+ pub updated_at: String,
218+ pub chain: Vec<Entry>,
219+ #[serde(default)]
220+ pub previous: Vec<Entry>,
221+}
222+
223+pub const MANIFEST_VERSION: u32 = 1;
224+
225+impl Manifest {
226+ pub fn new(repo_id: &str, store_key: &str) -> Manifest {
227+ Manifest {
228+ version: MANIFEST_VERSION,
229+ repo_id: repo_id.to_owned(),
230+ store_key: store_key.to_owned(),
231+ path: None,
232+ updated_at: String::new(),
233+ chain: Vec::new(),
234+ previous: Vec::new(),
235+ }
236+ }
237+
238+ pub fn last(&self) -> Option<&Entry> {
239+ self.chain.last()
240+ }
241+
242+ /// Adds `entry` to the chain. A full one starts a new chain, and the
243+ /// current one becomes `previous`. Returns the bundles no longer kept.
244+ pub fn add(&mut self, entry: Entry) -> Vec<String> {
245+ if entry.kind == BackupKind::Full {
246+ let dropped = std::mem::take(&mut self.previous);
247+ self.previous = std::mem::replace(&mut self.chain, vec![entry]);
248+ return dropped.into_iter().filter_map(|entry| entry.key).collect();
249+ }
250+ self.chain.push(entry);
251+ Vec::new()
252+ }
253+
254+ /// Every bundle it lists.
255+ pub fn keys(&self) -> Vec<String> {
256+ self.chain.iter().chain(&self.previous).filter_map(|entry| entry.key.clone()).collect()
257+ }
258+
259+ pub fn to_bytes(&self) -> Vec<u8> {
260+ serde_json::to_vec_pretty(self).unwrap_or_default()
261+ }
262+
263+ pub fn from_bytes(bytes: &[u8]) -> Option<Manifest> {
264+ serde_json::from_slice::<Manifest>(bytes).ok().filter(|manifest| manifest.version == MANIFEST_VERSION)
265+ }
266+}
267+
268+/// What the next backup of a repository cuts.
269+#[derive(Clone, Debug, PartialEq, Eq)]
270+pub struct Plan {
271+ pub kind: BackupKind,
272+ pub prerequisites: Vec<String>,
273+ pub previous_refs: BTreeMap<String, String>,
274+}
275+
276+/// The next backup, from the manifest (what a restore reads) and the
277+/// entry the repository's row says was last (`last_entry`). Full when
278+/// there is no chain, the two disagree, the last entry had no refs, or the
279+/// chain already holds `full_every` incremental backups. Otherwise
280+/// incremental, leaving out every commit the last entry's refs reach.
281+pub fn plan(manifest: Option<&Manifest>, last_entry: Option<&str>, full_every: u32) -> Plan {
282+ let full = Plan { kind: BackupKind::Full, prerequisites: Vec::new(), previous_refs: BTreeMap::new() };
283+ let Some(manifest) = manifest else { return full };
284+ let Some(last) = manifest.last() else { return full };
285+ let previous_refs = last.refs.clone();
286+ if last_entry != Some(last.id.as_str()) || last.refs.is_empty() {
287+ return full;
288+ }
289+ let incrementals = manifest.chain.len().saturating_sub(1);
290+ if incrementals >= full_every as usize {
291+ return Plan { previous_refs, ..full };
292+ }
293+ let prerequisites: BTreeSet<&String> = last.refs.values().collect();
294+ Plan {
295+ kind: BackupKind::Incremental,
296+ prerequisites: prerequisites.into_iter().cloned().collect(),
297+ previous_refs,
298+ }
299+}
300+
301+/// `20261006T025300Z`, from milliseconds since the epoch.
302+pub fn stamp(now_ms: u64) -> String {
303+ let text = rfc3339(now_ms);
304+ let whole = text.split('.').next().unwrap_or(&text).trim_end_matches('Z');
305+ format!("{}Z", whole.replace(['-', ':'], ""))
306+}
307+
308+pub fn manifest_key(repo_id: &str) -> String {
309+ format!("backups/{repo_id}/manifest.json")
310+}
311+
312+pub fn bundle_key(repo_id: &str, id: &str, kind: BackupKind) -> String {
313+ format!("backups/{repo_id}/{id}-{}.bundle", kind.suffix())
314+}
315+
316+fn is_hash(text: &str) -> bool {
317+ text.len() == 40 && text.bytes().all(|b| b.is_ascii_digit() || (b'a'..=b'f').contains(&b))
318+}
319+
320+/// Whether what a sandbox says a bundle holds can be: ref names git would
321+/// make (`HEAD` or under `refs/`) pointing at full commit hashes.
322+pub fn valid_refs(refs: &BTreeMap<String, String>) -> bool {
323+ refs.iter().all(|(name, hash)| {
324+ let named = name == "HEAD"
325+ || (name.starts_with("refs/")
326+ && name.len() <= 1024
327+ && !name.contains("..")
328+ && !name.ends_with('/')
329+ && !name.bytes().any(|b| b <= b' ' || b"~^:?*[\\".contains(&b)));
330+ named && is_hash(hash)
331+ })
332+}
333+
334+/// Whether the parts a sandbox says it sent are the ones a bundle of
335+/// `size` bytes makes, numbered from 1 with none missing.
336+pub fn parts_fit(parts: &[BackupPart], size: u64, part_bytes: u64) -> bool {
337+ let wanted = size.div_ceil(part_bytes.max(1));
338+ parts.len() as u64 == wanted && parts.iter().enumerate().all(|(index, part)| part.number as usize == index + 1)
339+}
340+
341+// ---------------------------------------------------------------------
342+// The record in D1, and the job in hand
343+// ---------------------------------------------------------------------
344+
345+/// A repository's row in `repo_backups`, as a job reads it.
346+#[derive(Clone, Debug, Default, Deserialize)]
347+struct JobRow {
348+ repo_id: String,
349+ token_hash: Option<String>,
350+ attempts: Option<f64>,
351+ target_version: Option<f64>,
352+ store_key: Option<String>,
353+ upload_key: Option<String>,
354+ upload_id: Option<String>,
355+ upload_kind: Option<String>,
356+ upload_entry: Option<String>,
357+ prerequisites: Option<String>,
358+ last_entry: Option<String>,
359+}
360+
361+fn refused<T>(message: &str) -> Outcome<T> {
362+ Outcome::fail(FailureCode::NotFound, message)
363+}
364+
365+fn n(value: u64) -> JsValue {
366+ JsValue::from_f64(value as f64)
367+}
368+
369+fn text(value: Option<&str>) -> JsValue {
370+ value.map_or(JsValue::NULL, JsValue::from)
371+}
372+
373+/// What a nightly run did.
374+#[derive(Debug, Default)]
375+pub struct Night {
376+ pub queued: u32,
377+ pub pruned: u32,
378+}
379+
380+/// Queues tonight's backups: the repositories due, `settings.per_night`
381+/// at most. Also removes the backups of repositories that were purged.
382+pub async fn nightly(db: &D1Database, blobs: &Store, settings: Settings, now: u64) -> Result<Night> {
383+ let candidates = db
384+ .prepare(DUE_SQL)
385+ .bind(&[PULLS_NAMESPACE.into(), n(settings.per_night as u64)])?
386+ .all()
387+ .await?
388+ .results::<CandidateRow>()?
389+ .into_iter()
390+ .map(Candidate::from)
391+ .collect::<Vec<_>>();
392+ let due = pick_due(&candidates, settings.per_night as usize);
393+ let mut statements = Vec::new();
394+ for repo_id in &due {
395+ statements.push(
396+ db.prepare(
397+ "INSERT INTO repo_backups (repo_id, status, queued_ms, attempts) VALUES (?1, 'queued', ?2, 0)
398+ ON CONFLICT (repo_id) DO UPDATE SET status = 'queued', queued_ms = ?2, attempts = 0, last_error = NULL
399+ WHERE repo_backups.status = 'idle'",
400+ )
401+ .bind(&[repo_id.as_str().into(), n(now)])?,
402+ );
403+ }
404+ if !statements.is_empty() {
405+ db.batch(statements).await?;
406+ }
407+ let pruned = prune(db, blobs).await?;
408+ Ok(Night { queued: due.len() as u32, pruned })
409+}
410+
411+/// The row the due query reads; D1 gives numbers as floats.
412+#[derive(Deserialize)]
413+struct CandidateRow {
414+ repo_id: String,
415+ namespace: String,
416+ created_at: Option<serde_json::Value>,
417+ refs_version: Option<f64>,
418+ refs_open_until: Option<f64>,
419+ deleted: Option<f64>,
420+ retired: Option<f64>,
421+ status: Option<String>,
422+ backed_version: Option<f64>,
423+ backed_up_ms: Option<f64>,
424+}
425+
426+impl From<CandidateRow> for Candidate {
427+ fn from(row: CandidateRow) -> Candidate {
428+ Candidate {
429+ repo_id: row.repo_id,
430+ namespace: row.namespace,
431+ created_at: match row.created_at {
432+ Some(serde_json::Value::String(text)) => text,
433+ Some(other) => other.to_string(),
434+ None => String::new(),
435+ },
436+ refs_version: row.refs_version.unwrap_or(0.0) as u64,
437+ refs_open_until: row.refs_open_until.unwrap_or(0.0) as u64,
438+ deleted: row.deleted.unwrap_or(0.0) != 0.0,
439+ retired: row.retired.unwrap_or(0.0) != 0.0,
440+ status: row.status,
441+ backed_version: row.backed_version.map(|v| v as u64),
442+ backed_up_ms: row.backed_up_ms.unwrap_or(0.0) as u64,
443+ }
444+ }
445+}
446+
447+/// Removes the backups of repositories that no longer exist (purged, so
448+/// their data is gone for good): every bundle the manifest lists, the
449+/// manifest, and the row.
450+async fn prune(db: &D1Database, blobs: &Store) -> Result<u32> {
451+ #[derive(Deserialize)]
452+ struct Gone {
453+ repo_id: String,
454+ }
455+ let gone = db
456+ .prepare(
457+ "SELECT b.repo_id FROM repo_backups b LEFT JOIN repos r ON r.id = b.repo_id
458+ WHERE r.id IS NULL LIMIT ?1",
459+ )
460+ .bind(&[n(PRUNES_PER_NIGHT as u64)])?
461+ .all()
462+ .await?
463+ .results::<Gone>()?;
464+ for row in &gone {
465+ let key = manifest_key(&row.repo_id);
466+ if let Some(manifest) = blobs.read(&key).await?.as_deref().and_then(Manifest::from_bytes) {
467+ for bundle in manifest.keys() {
468+ blobs.delete(&bundle).await?;
469+ }
470+ }
471+ blobs.delete(&key).await?;
472+ db.prepare("DELETE FROM repo_backups WHERE repo_id = ?1")
473+ .bind(&[row.repo_id.as_str().into()])?
474+ .run()
475+ .await?;
476+ }
477+ Ok(gone.len() as u32)
478+}
479+
480+/// For the runner's sweep: up to `limit` queued backups, each marked
481+/// running with a token of its own, so long as no more than `max_running`
482+/// are then running. Jobs past their lease go back in the queue first.
483+pub async fn claim(db: &D1Database, blobs: Option<&Store>, a: &ClaimBackupsArgs, now: u64) -> Result<Vec<BackupClaim>> {
484+ // Past their lease: the sandbox died without saying so.
485+ let stale = db
486+ .prepare("SELECT * FROM repo_backups WHERE status = 'running' AND claimed_ms < ?1")
487+ .bind(&[n(now.saturating_sub(LEASE_MS))])?
488+ .all()
489+ .await?
490+ .results::<JobRow>()?;
491+ for row in &stale {
492+ give_up_upload(blobs, row).await;
493+ settle_failure(db, row, "The backup ran past its time and was started again.").await?;
494+ }
495+ if blobs.is_none() {
496+ return Ok(Vec::new());
497+ }
498+
499+ #[derive(Deserialize)]
500+ struct Count {
501+ running: f64,
502+ }
503+ let running = db
504+ .prepare("SELECT count(*) AS running FROM repo_backups WHERE status = 'running'")
505+ .first::<Count>(None)
506+ .await?
507+ .map_or(0, |row| row.running as u32);
508+ let room = a.max_running.saturating_sub(running).min(a.limit);
509+ if room == 0 {
510+ return Ok(Vec::new());
511+ }
512+
513+ #[derive(Deserialize)]
514+ struct Queued {
515+ repo_id: String,
516+ namespace: Option<String>,
517+ name: Option<String>,
518+ deleted: Option<f64>,
519+ }
520+ let queued = db
521+ .prepare(
522+ "SELECT b.repo_id, r.namespace, r.name, (r.id IS NULL OR r.deleted_at IS NOT NULL) AS deleted
523+ FROM repo_backups b LEFT JOIN repos r ON r.id = b.repo_id
524+ WHERE b.status = 'queued' ORDER BY b.queued_ms, b.repo_id LIMIT ?1",
525+ )
526+ .bind(&[n(room as u64)])?
527+ .all()
528+ .await?
529+ .results::<Queued>()?;
530+ let mut claims = Vec::new();
531+ for row in queued {
532+ let (Some(namespace), Some(name)) = (row.namespace, row.name) else { continue };
533+ if row.deleted.unwrap_or(0.0) != 0.0 {
534+ // Deleted since it was queued: nothing to back up until it is
535+ // restored, if it is.
536+ db.prepare("UPDATE repo_backups SET status = 'idle' WHERE repo_id = ?1 AND status = 'queued'")
537+ .bind(&[row.repo_id.as_str().into()])?
538+ .run()
539+ .await?;
540+ continue;
541+ }
542+ let job_id = new_id("bkp", now);
543+ let token = g1t_secrets::random_hex(32);
544+ let claimed = db
545+ .prepare(
546+ "UPDATE repo_backups SET status = 'running', job_id = ?2, token_hash = ?3, claimed_ms = ?4,
547+ attempts = attempts + 1, target_version = NULL, store_key = NULL, upload_key = NULL,
548+ upload_id = NULL, upload_kind = NULL, upload_entry = NULL, prerequisites = NULL
549+ WHERE repo_id = ?1 AND status = 'queued' RETURNING repo_id",
550+ )
551+ .bind(&[
552+ row.repo_id.as_str().into(),
553+ job_id.as_str().into(),
554+ g1t_secrets::sha256_hex(&token).into(),
555+ n(now),
556+ ])?
557+ .first::<serde_json::Value>(None)
558+ .await?;
559+ if claimed.is_some() {
560+ claims.push(BackupClaim { job_id, token, repo_id: row.repo_id, path: RepoPath { namespace, name } });
561+ }
562+ }
563+ Ok(claims)
564+}
565+
566+/// The running job `a` names, if its token is the one it was given.
567+async fn job(db: &D1Database, a: &BackupJobArgs) -> Result<Option<JobRow>> {
568+ let row = db
569+ .prepare("SELECT * FROM repo_backups WHERE job_id = ?1 AND status = 'running'")
570+ .bind(&[a.job_id.as_str().into()])?
571+ .first::<JobRow>(None)
572+ .await?;
573+ Ok(row.filter(|row| {
574+ row.token_hash
575+ .as_deref()
576+ .is_some_and(|hash| g1t_secrets::same(hash, &g1t_secrets::sha256_hex(&a.token)))
577+ }))
578+}
579+
580+const NO_JOB: &str = "No such backup job, or it is not running.";
581+
582+/// The job, for its sandbox: what to cut, and a read-only credential for
583+/// the repository that lasts minutes. Asked again, the same bundle with a
584+/// new credential.
585+pub async fn spec<S: GitStore>(
586+ registry: &Registry,
587+ blobs: &Store,
588+ store: &S,
589+ a: &BackupJobArgs,
590+ full_every: u32,
591+ now: u64,
592+) -> Result<Outcome<BackupSpec>> {
593+ let db = &registry.db;
594+ let Some(row) = job(db, a).await? else { return Ok(refused(NO_JOB)) };
595+ #[derive(Deserialize)]
596+ struct Live {
597+ refs_version: Option<f64>,
598+ }
599+ let Some(repo) = registry.by_id(&row.repo_id).await? else {
600+ settle_failure(db, &row, "The repository was deleted.").await?;
601+ return Ok(refused("The repository was deleted."));
602+ };
603+ let key = store_key(&repo);
604+ let access = store.handout(&key, Scope::Read).await?;
605+ // Asked before: the same bundle, so parts already sent still fit.
606+ if let (Some(kind), Some(_)) = (row.upload_kind.as_deref(), row.upload_id.as_deref()) {
607+ let manifest = blobs.read(&manifest_key(&row.repo_id)).await?.as_deref().and_then(Manifest::from_bytes);
608+ let prerequisites: Vec<String> = row.prerequisites.as_deref().and_then(|p| serde_json::from_str(p).ok()).unwrap_or_default();
609+ let previous_refs = manifest.as_ref().and_then(|m| m.last()).map(|e| e.refs.clone()).unwrap_or_default();
610+ return Ok(Outcome::Ok(BackupSpec {
611+ kind: if kind == "full" { BackupKind::Full } else { BackupKind::Incremental },
612+ remote: access.remote,
613+ git_token: access.token,
614+ prerequisites,
615+ previous_refs,
616+ part_bytes: PART_BYTES,
617+ }));
618+ }
619+ let version = db
620+ .prepare("SELECT refs_version FROM repos WHERE id = ?1")
621+ .bind(&[row.repo_id.as_str().into()])?
622+ .first::<Live>(None)
623+ .await?
624+ .and_then(|live| live.refs_version)
625+ .unwrap_or(0.0) as u64;
626+ let manifest = blobs.read(&manifest_key(&row.repo_id)).await?.as_deref().and_then(Manifest::from_bytes);
627+ let next = plan(manifest.as_ref(), row.last_entry.as_deref(), full_every);
628+ let entry = stamp(now);
629+ let upload_key = bundle_key(&row.repo_id, &entry, next.kind);
630+ let upload_id = blobs.create_multipart(&upload_key).await?;
631+ db.prepare(
632+ "UPDATE repo_backups SET target_version = ?2, store_key = ?3, upload_key = ?4, upload_id = ?5,
633+ upload_kind = ?6, upload_entry = ?7, prerequisites = ?8, backed_from_ms = ?9
634+ WHERE job_id = ?1",
635+ )
636+ .bind(&[
637+ a.job_id.as_str().into(),
638+ n(version),
639+ key.as_str().into(),
640+ upload_key.as_str().into(),
641+ upload_id.as_str().into(),
642+ next.kind.suffix().into(),
643+ entry.as_str().into(),
644+ serde_json::to_string(&next.prerequisites)?.into(),
645+ n(now),
646+ ])?
647+ .run()
648+ .await?;
649+ Ok(Outcome::Ok(BackupSpec {
650+ kind: next.kind,
651+ remote: access.remote,
652+ git_token: access.token,
653+ prerequisites: next.prerequisites,
654+ previous_refs: next.previous_refs,
655+ part_bytes: PART_BYTES,
656+ }))
657+}
658+
659+/// One part of the job's bundle, kept.
660+pub async fn part(db: &D1Database, blobs: &Store, a: &BackupJobArgs, number: u16, bytes: Vec<u8>) -> Result<Outcome<BackupPart>> {
661+ let Some(row) = job(db, a).await? else { return Ok(refused(NO_JOB)) };
662+ let (Some(key), Some(upload)) = (row.upload_key.as_deref(), row.upload_id.as_deref()) else {
663+ return Ok(Outcome::fail(FailureCode::Conflict, "Ask for the job's spec first."));
664+ };
665+ if number == 0 || bytes.len() as u64 > PART_BYTES {
666+ return Ok(Outcome::fail(FailureCode::Invalid, "Parts are numbered from 1, and hold 32 MiB at most."));
667+ }
668+ let Part { number, etag } = blobs.upload_part(key, upload, number, bytes).await?;
669+ Ok(Outcome::Ok(BackupPart { number, etag }))
670+}
671+
672+/// The bundle is cut and sent: the upload is completed, the chain gains
673+/// an entry (unless nothing changed), bundles of the chain before last
674+/// are removed, and the repository is backed up as of the refs version it
675+/// had when the clone began.
676+pub async fn complete(registry: &Registry, blobs: &Store, a: &BackupComplete, now: u64) -> Result<Outcome<bool>> {
677+ let db = &registry.db;
678+ let args = BackupJobArgs { job_id: a.job_id.clone(), token: a.token.clone() };
679+ let Some(row) = job(db, &args).await? else { return Ok(refused(NO_JOB)) };
680+ let (Some(upload_key), Some(upload_id), Some(entry_id), Some(store_key)) =
681+ (row.upload_key.clone(), row.upload_id.clone(), row.upload_entry.clone(), row.store_key.clone())
682+ else {
683+ return Ok(Outcome::fail(FailureCode::Conflict, "Ask for the job's spec first."));
684+ };
685+ if !valid_refs(&a.refs) {
686+ return Ok(Outcome::fail(FailureCode::Invalid, "A ref name or commit is not one git makes."));
687+ }
688+ if !parts_fit(&a.parts, a.size, PART_BYTES) {
689+ return Ok(Outcome::fail(FailureCode::Invalid, "The parts do not add up to the bundle's size."));
690+ }
691+ meter_fetch(&store_key, a.fetched_bytes);
692+ let kind = if row.upload_kind.as_deref() == Some("full") { BackupKind::Full } else { BackupKind::Incremental };
693+ let mut manifest = blobs
694+ .read(&manifest_key(&row.repo_id))
695+ .await?
696+ .as_deref()
697+ .and_then(Manifest::from_bytes)
698+ .unwrap_or_else(|| Manifest::new(&row.repo_id, &store_key));
699+ let unchanged = a.size == 0 && kind == BackupKind::Incremental && manifest.last().is_some_and(|last| last.refs == a.refs);
700+ let version = row.target_version.unwrap_or(0.0) as u64;
701+ let mut last_entry = row.last_entry.clone();
702+ if a.size == 0 {
703+ blobs.abort_multipart(&upload_key, &upload_id).await?;
704+ } else {
705+ let parts: Vec<Part> = a.parts.iter().map(|p| Part { number: p.number, etag: p.etag.clone() }).collect();
706+ blobs.complete_multipart(&upload_key, &upload_id, &parts).await?;
707+ }
708+ if !unchanged {
709+ let prerequisites: Vec<String> = row.prerequisites.as_deref().and_then(|p| serde_json::from_str(p).ok()).unwrap_or_default();
710+ let dropped = manifest.add(Entry {
711+ id: entry_id.clone(),
712+ kind,
713+ key: (a.size > 0).then(|| upload_key.clone()),
714+ created_at: rfc3339(now),
715+ refs_version: version,
716+ refs: a.refs.clone(),
717+ prerequisites,
718+ size: a.size,
719+ sha256: a.sha256.clone(),
720+ });
721+ manifest.store_key = store_key.clone();
722+ manifest.updated_at = rfc3339(now);
723+ manifest.path = registry
724+ .by_id(&row.repo_id)
725+ .await?
726+ .map(|repo| RepoPath { namespace: repo.namespace, name: repo.name });
727+ blobs.put(&manifest_key(&row.repo_id), manifest.to_bytes()).await?;
728+ for key in dropped {
729+ if let Err(error) = blobs.delete(&key).await {
730+ worker::console_error!("repos: backup {key} not removed: {error}");
731+ }
732+ }
733+ last_entry = Some(entry_id);
734+ }
735+ db.prepare(
736+ "UPDATE repo_backups SET status = 'idle', job_id = NULL, token_hash = NULL, claimed_ms = NULL,
737+ attempts = 0, last_error = NULL, refs_version = ?2, backed_up_ms = backed_from_ms,
738+ tips = ?3, last_entry = ?4, upload_key = NULL, upload_id = NULL, upload_kind = NULL,
739+ upload_entry = NULL, prerequisites = NULL, target_version = NULL
740+ WHERE job_id = ?1",
741+ )
742+ .bind(&[a.job_id.as_str().into(), n(version), serde_json::to_string(&a.refs)?.into(), text(last_entry.as_deref())])?
743+ .run()
744+ .await?;
745+ Ok(Outcome::Ok(true))
746+}
747+
748+/// The sandbox could not do the job: tried again later tonight, up to
749+/// [`MAX_ATTEMPTS`] times, and tomorrow night after that.
750+pub async fn fail(db: &D1Database, blobs: &Store, a: &BackupFail) -> Result<Outcome<bool>> {
751+ let args = BackupJobArgs { job_id: a.job_id.clone(), token: a.token.clone() };
752+ let Some(row) = job(db, &args).await? else { return Ok(refused(NO_JOB)) };
753+ if let Some(key) = row.store_key.as_deref() {
754+ meter_fetch(key, a.fetched_bytes);
755+ }
756+ give_up_upload(Some(blobs), &row).await;
757+ let said: String = a.error.chars().take(500).collect();
758+ settle_failure(db, &row, &said).await?;
759+ Ok(Outcome::Ok(true))
760+}
761+
762+/// A clone counts once, with what it read, whenever it got as far as
763+/// reading anything.
764+fn meter_fetch(store_key: &str, fetched_bytes: u64) {
765+ if fetched_bytes > 0 {
766+ meters::record("internal.git.info_refs", store_key, 0, 0);
767+ meters::record(FETCH_METER, store_key, 0, fetched_bytes);
768+ }
769+}
770+
771+async fn give_up_upload(blobs: Option<&Store>, row: &JobRow) {
772+ if let (Some(blobs), Some(key), Some(upload)) = (blobs, row.upload_key.as_deref(), row.upload_id.as_deref())
773+ && let Err(error) = blobs.abort_multipart(key, upload).await
774+ {
775+ worker::console_error!("repos: backup upload {key} not given up: {error}");
776+ }
777+}
778+
779+/// Back in the queue for another try, or idle until tomorrow night once
780+/// it has been tried enough.
781+async fn settle_failure(db: &D1Database, row: &JobRow, error: &str) -> Result<()> {
782+ let attempts = row.attempts.unwrap_or(0.0) as u32;
783+ let next = if attempts >= MAX_ATTEMPTS { "idle" } else { "queued" };
784+ worker::console_error!("repos: backup of {} failed ({attempts} of {MAX_ATTEMPTS}): {error}", row.repo_id);
785+ db.prepare(
786+ "UPDATE repo_backups SET status = ?2, last_error = ?3, job_id = NULL, token_hash = NULL,
787+ claimed_ms = NULL, upload_key = NULL, upload_id = NULL, upload_kind = NULL,
788+ upload_entry = NULL, prerequisites = NULL, target_version = NULL
789+ WHERE repo_id = ?1",
790+ )
791+ .bind(&[row.repo_id.as_str().into(), next.into(), error.into()])?
792+ .run()
793+ .await?;
794+ Ok(())
795+}
796+
797+#[cfg(test)]
798+mod tests {
799+ use super::*;
800+
801+ const A: &str = "c71546fcd893ef8b0f57388b65e620d759705dda";
802+ const B: &str = "4807077b296e6edbf410d55e72749d3e1170c291";
803+ const C: &str = "0000000000000000000000000000000000000abc";
804+
805+ fn candidate(id: &str) -> Candidate {
806+ Candidate {
807+ repo_id: id.into(),
808+ namespace: "acme".into(),
809+ created_at: "2026-01-01T00:00:00.000Z".into(),
810+ refs_version: 3,
811+ ..Candidate::default()
812+ }
813+ }
814+
815+ fn backed(id: &str, version: u64, at: u64) -> Candidate {
816+ Candidate { status: Some("idle".into()), backed_version: Some(version), backed_up_ms: at, ..candidate(id) }
817+ }
818+
819+ #[test]
820+ fn a_repository_never_backed_up_is_due() {
821+ assert!(is_due(&candidate("r1")));
822+ // Even one whose refs never moved: rows from before refs_version.
823+ assert!(is_due(&Candidate { refs_version: 0, ..candidate("r1") }));
824+ }
825+
826+ #[test]
827+ fn a_repository_is_due_only_once_its_refs_moved() {
828+ assert!(!is_due(&backed("r1", 3, 1_000)));
829+ assert!(is_due(&backed("r1", 2, 1_000)));
830+ // A credential that could push went out after the last backup began.
831+ assert!(is_due(&Candidate { refs_open_until: 2_000, ..backed("r1", 3, 1_000) }));
832+ assert!(!is_due(&Candidate { refs_open_until: 900, ..backed("r1", 3, 1_000) }));
833+ // Tried and failed: no version yet.
834+ assert!(is_due(&Candidate { backed_version: None, ..backed("r1", 0, 0) }));
835+ }
836+
837+ #[test]
838+ fn deleted_retired_working_copies_and_jobs_in_hand_are_not_due() {
839+ assert!(!is_due(&Candidate { deleted: true, ..candidate("r1") }));
840+ assert!(!is_due(&Candidate { retired: true, ..candidate("r1") }));
841+ assert!(!is_due(&Candidate { namespace: PULLS_NAMESPACE.into(), ..candidate("r1") }));
842+ assert!(!is_due(&Candidate { status: Some("queued".into()), ..backed("r1", 1, 0) }));
843+ assert!(!is_due(&Candidate { status: Some("running".into()), ..backed("r1", 1, 0) }));
844+ }
845+
846+ #[test]
847+ fn the_longest_waiting_go_first_and_no_more_than_the_limit() {
848+ let candidates = vec![
849+ backed("recent", 1, 5_000),
850+ backed("older", 1, 1_000),
851+ backed("current", 3, 0),
852+ Candidate { created_at: "2026-02-01T00:00:00.000Z".into(), ..candidate("new-b") },
853+ Candidate { created_at: "2025-02-01T00:00:00.000Z".into(), ..candidate("new-a") },
854+ ];
855+ assert_eq!(pick_due(&candidates, 10), ["new-a", "new-b", "older", "recent"]);
856+ assert_eq!(pick_due(&candidates, 2), ["new-a", "new-b"]);
857+ assert!(pick_due(&candidates, 0).is_empty());
858+ }
859+
860+ fn refs(pairs: &[(&str, &str)]) -> BTreeMap<String, String> {
861+ pairs.iter().map(|(name, hash)| ((*name).to_owned(), (*hash).to_owned())).collect()
862+ }
863+
864+ fn entry(id: &str, kind: BackupKind, tips: &[(&str, &str)]) -> Entry {
865+ Entry {
866+ id: id.into(),
867+ kind,
868+ key: Some(format!("backups/r1/{id}-{}.bundle", kind.suffix())),
869+ created_at: String::new(),
870+ refs_version: 1,
871+ refs: refs(tips),
872+ prerequisites: Vec::new(),
873+ size: 10,
874+ sha256: None,
875+ }
876+ }
877+
878+ fn manifest(incrementals: usize) -> Manifest {
879+ let mut manifest = Manifest::new("r1", "acme--rocket");
880+ manifest.add(entry("e0", BackupKind::Full, &[("HEAD", A), ("refs/heads/main", A)]));
881+ for index in 0..incrementals {
882+ manifest.add(entry(&format!("e{}", index + 1), BackupKind::Incremental, &[("HEAD", A), ("refs/heads/main", A), ("refs/tags/v1", B)]));
883+ }
884+ manifest
885+ }
886+
887+ #[test]
888+ fn the_first_backup_is_full() {
889+ let next = plan(None, None, 30);
890+ assert_eq!(next.kind, BackupKind::Full);
891+ assert!(next.prerequisites.is_empty() && next.previous_refs.is_empty());
892+ assert_eq!(plan(Some(&Manifest::new("r1", "k")), None, 30).kind, BackupKind::Full);
893+ }
894+
895+ #[test]
896+ fn the_next_is_incremental_from_the_last_tips_each_once() {
897+ let manifest = manifest(1);
898+ let next = plan(Some(&manifest), Some("e1"), 30);
899+ assert_eq!(next.kind, BackupKind::Incremental);
900+ assert_eq!(next.prerequisites, [B, A]);
901+ assert_eq!(next.previous_refs, manifest.last().unwrap().refs);
902+ }
903+
904+ #[test]
905+ fn a_full_bundle_is_cut_again_after_enough_incrementals() {
906+ assert_eq!(plan(Some(&manifest(29)), Some("e29"), 30).kind, BackupKind::Incremental);
907+ let next = plan(Some(&manifest(30)), Some("e30"), 30);
908+ assert_eq!(next.kind, BackupKind::Full);
909+ assert!(next.prerequisites.is_empty());
910+ // What it last held is still said, so an unchanged clone is seen.
911+ assert!(!next.previous_refs.is_empty());
912+ assert_eq!(plan(Some(&manifest(1)), Some("e1"), 1).kind, BackupKind::Full);
913+ }
914+
915+ #[test]
916+ fn a_chain_the_row_does_not_know_or_an_empty_one_starts_again() {
917+ assert_eq!(plan(Some(&manifest(2)), Some("e1"), 30).kind, BackupKind::Full);
918+ assert_eq!(plan(Some(&manifest(2)), None, 30).kind, BackupKind::Full);
919+ let mut empty = Manifest::new("r1", "k");
920+ empty.add(entry("e0", BackupKind::Full, &[]));
921+ assert_eq!(plan(Some(&empty), Some("e0"), 30).kind, BackupKind::Full);
922+ }
923+
924+ #[test]
925+ fn a_full_backup_starts_a_chain_and_the_one_before_last_goes() {
926+ let mut manifest = manifest(2);
927+ assert!(manifest.previous.is_empty());
928+ let dropped = manifest.add(entry("f1", BackupKind::Full, &[("refs/heads/main", C)]));
929+ assert!(dropped.is_empty(), "the chain before is kept until the next full one");
930+ assert_eq!(manifest.chain.len(), 1);
931+ assert_eq!(manifest.previous.len(), 3);
932+ manifest.add(entry("f1a", BackupKind::Incremental, &[("refs/heads/main", C)]));
933+ let dropped = manifest.add(entry("f2", BackupKind::Full, &[("refs/heads/main", C)]));
934+ assert_eq!(dropped, ["backups/r1/e0-full.bundle", "backups/r1/e1-incr.bundle", "backups/r1/e2-incr.bundle"]);
935+ assert_eq!(manifest.previous.iter().map(|e| e.id.as_str()).collect::<Vec<_>>(), ["f1", "f1a"]);
936+ assert_eq!(manifest.keys().len(), 3);
937+ }
938+
939+ #[test]
940+ fn a_manifest_round_trips_and_an_unknown_version_is_not_read() {
941+ let mut manifest = manifest(2);
942+ manifest.path = Some(RepoPath { namespace: "acme".into(), name: "rocket".into() });
943+ manifest.add(Entry { key: None, size: 0, ..entry("e3", BackupKind::Incremental, &[("refs/heads/main", A)]) });
944+ let bytes = manifest.to_bytes();
945+ let text = String::from_utf8(bytes.clone()).unwrap();
946+ // What the restore drill reads: snake_case, the kind in words.
947+ assert!(text.contains("\"repo_id\": \"r1\"") && text.contains("\"kind\": \"incremental\"") && text.contains("\"key\": null"));
948+ assert_eq!(Manifest::from_bytes(&bytes), Some(manifest.clone()));
949+ let mut later = manifest;
950+ later.version = 2;
951+ assert_eq!(Manifest::from_bytes(&later.to_bytes()), None);
952+ assert_eq!(Manifest::from_bytes(b"not json"), None);
953+ }
954+
955+ #[test]
956+ fn a_backup_clone_is_g1ts_operation_not_the_workspaces() {
957+ let mapping = meters::Mapping::defaults();
958+ assert_eq!(mapping.cost(FETCH_METER), 1.0);
959+ assert_eq!(mapping.billable(FETCH_METER), 0.0);
960+ assert_eq!(mapping.billable("internal.git.fetch"), 1.0);
961+ }
962+
963+ #[test]
964+ fn keys_and_stamps() {
965+ assert_eq!(stamp(1_369_353_600_123), "20130524T000000Z");
966+ assert_eq!(manifest_key("r1"), "backups/r1/manifest.json");
967+ assert_eq!(bundle_key("r1", "20130524T000000Z", BackupKind::Incremental), "backups/r1/20130524T000000Z-incr.bundle");
968+ assert_eq!(bundle_key("r1", "20130524T000000Z", BackupKind::Full), "backups/r1/20130524T000000Z-full.bundle");
969+ }
970+
971+ #[test]
972+ fn only_refs_git_makes_are_taken() {
973+ assert!(valid_refs(&refs(&[("HEAD", A), ("refs/heads/main", A), ("refs/pull/pr_1/head", B)])));
974+ assert!(valid_refs(&BTreeMap::new()));
975+ assert!(!valid_refs(&refs(&[("main", A)])));
976+ assert!(!valid_refs(&refs(&[("refs/heads/../x", A)])));
977+ assert!(!valid_refs(&refs(&[("refs/heads/a b", A)])));
978+ assert!(!valid_refs(&refs(&[("refs/heads/main", "abc")])));
979+ assert!(!valid_refs(&refs(&[("refs/heads/main", &A.to_uppercase())])));
980+ }
981+
982+ #[test]
983+ fn parts_must_cover_the_bundle_in_order() {
984+ let part = |number| BackupPart { number, etag: "e".into() };
985+ assert!(parts_fit(&[], 0, 10));
986+ assert!(parts_fit(&[part(1)], 10, 10));
987+ assert!(parts_fit(&[part(1), part(2)], 11, 10));
988+ assert!(!parts_fit(&[part(1)], 11, 10));
989+ assert!(!parts_fit(&[part(2), part(1)], 11, 10));
990+ assert!(!parts_fit(&[part(1)], 0, 10));
991+ }
992+}
+83−2
55 //! `g1t_contracts::repos` for the methods and their arguments. Any other
66 //! request is treated as git's smart HTTP protocol.
77
8+mod backups;
89 mod blame;
910 mod catch_up;
1011 mod coalesce;
19611962 }
19621963 }
19631964
1965+/// `/backups/<job id>/parts/<number>`: the job and the part's number.
1966+fn backup_part_path(path: &str) -> Option<(String, u16)> {
1967+ let rest = path.strip_prefix("/backups/")?;
1968+ let (job, number) = rest.split_once("/parts/")?;
1969+ let number = number.parse::<u16>().ok()?;
1970+ (!job.is_empty() && !job.contains('/')).then(|| (job.to_owned(), number))
1971+}
1972+
1973+fn backups_off<T>() -> Outcome<T> {
1974+ Outcome::fail(FailureCode::Conflict, "Backups are off on this installation: it has no storage for them.")
1975+}
1976+
1977+/// One part of a backup's bundle, with the job's token in its header.
1978+async fn backup_part(request: &mut Request, env: &Env, repos: &Repos<ArtifactsStore>, job_id: String, number: u16) -> Result<Response> {
1979+ let Some(blobs) = backups::storage(env) else {
1980+ return reply(&backups_off::<()>());
1981+ };
1982+ let token = request.headers().get(g1t_contracts::backups::TOKEN_HEADER)?.unwrap_or_default();
1983+ let bytes = request.bytes().await?;
1984+ let job = g1t_contracts::backups::BackupJobArgs { job_id, token };
1985+ reply(&backups::part(&repos.registry.db, &blobs, &job, number, bytes).await?)
1986+}
1987+
1988+#[cfg(test)]
1989+mod backup_path_tests {
1990+ use super::backup_part_path;
1991+
1992+ #[test]
1993+ fn a_part_is_named_by_its_job_and_number() {
1994+ assert_eq!(backup_part_path("/backups/bkp_1/parts/3"), Some(("bkp_1".to_owned(), 3)));
1995+ assert_eq!(backup_part_path("/backups/bkp_1/parts/x"), None);
1996+ assert_eq!(backup_part_path("/backups//parts/1"), None);
1997+ assert_eq!(backup_part_path("/acme/rocket.git/info/refs"), None);
1998+ }
1999+}
2000+
19642001 /// Read methods whose answer is an `Outcome`: when the git store is busy,
19652002 /// the site is told so in words instead of failing the page.
19662003 const OUTCOME_READS: [&str; 6] = ["tree", "blob", "log", "branches", "blame", "compare"];
19682005 #[event(fetch)]
19692006 async fn fetch(mut request: Request, env: Env, ctx: Context) -> Result<Response> {
19702007 let mut repos = service(&env)?;
2008+ // A part of a backup's bundle, as the API passes it on from the
2009+ // sandbox: bytes, not JSON (backups.rs).
2010+ if request.method() == Method::Put
2011+ && let Some((job_id, number)) = backup_part_path(&request.path())
2012+ {
2013+ let answered = backup_part(&mut request, &env, &repos, job_id, number).await;
2014+ flush_later(&env, &ctx);
2015+ return answered;
2016+ }
19712017 let Some(method) = rpc_method(&request) else {
19722018 let answered = repos.git_http(request, &env, &ctx).await;
19732019 flush_later(&env, &ctx);
20992145 meters::set_mapping(&repos.registry.db, &row, &rfc3339(now_ms())).await?;
21002146 reply(&meters::read_mapping(&repos.registry.db).await?)
21012147 }
2148+ // Backups (backups.rs): the runner's sweep claims queued ones, and
2149+ // each sandbox, through the API, asks for its job and says how it went.
2150+ "claim_backups" => {
2151+ let a: g1t_contracts::backups::ClaimBackupsArgs = args(body)?;
2152+ let blobs = backups::storage(&env);
2153+ reply(&backups::claim(&repos.registry.db, blobs.as_ref(), &a, now_ms()).await?)
2154+ }
2155+ "backup_spec" => match backups::storage(&env) {
2156+ Some(blobs) => {
2157+ let a: g1t_contracts::backups::BackupJobArgs = args(body)?;
2158+ let every = backups::Settings::from_env(&env).full_every;
2159+ reply(&backups::spec(&repos.registry, &blobs, &repos.store, &a, every, now_ms()).await?)
2160+ }
2161+ None => reply(&backups_off::<bool>()),
2162+ },
2163+ "backup_complete" => match backups::storage(&env) {
2164+ Some(blobs) => reply(&backups::complete(&repos.registry, &blobs, &args(body)?, now_ms()).await?),
2165+ None => reply(&backups_off::<bool>()),
2166+ },
2167+ "backup_fail" => match backups::storage(&env) {
2168+ Some(blobs) => reply(&backups::fail(&repos.registry.db, &blobs, &args(body)?).await?),
2169+ None => reply(&backups_off::<bool>()),
2170+ },
21022171 // How the git store has been answering, for the status page.
21032172 "store_health" => {
21042173 let a: meters::HealthArgs = args(body)?;
21262195 served.finish(answered)
21272196 }
21282197
2198+/// The nightly cron in wrangler.jsonc: tonight's backups are queued.
2199+const BACKUP_CRON: &str = "53 2 * * *";
2200+
21292201 /// The hourly sweep: deleted repositories whose time to be restored has
2130−/// passed are purged. See lifecycle.rs.
2202+/// passed are purged. See lifecycle.rs. And, at [`BACKUP_CRON`], the
2203+/// repositories whose refs moved are queued for a backup (backups.rs).
21312204 #[event(scheduled)]
2132−async fn scheduled(_event: ScheduledEvent, env: Env, _ctx: ScheduleContext) {
2205+async fn scheduled(event: ScheduledEvent, env: Env, _ctx: ScheduleContext) {
21332206 let repos = match service(&env) {
21342207 Ok(repos) => repos,
21352208 Err(error) => {
21372210 return;
21382211 }
21392212 };
2213+ if event.cron() == BACKUP_CRON {
2214+ let Some(blobs) = backups::storage(&env) else { return };
2215+ match backups::nightly(&repos.registry.db, &blobs, backups::Settings::from_env(&env), now_ms()).await {
2216+ Ok(night) => worker::console_log!("repos: queued {} backups, removed {} of purged repositories", night.queued, night.pruned),
2217+ Err(error) => worker::console_error!("repos: backups could not be queued: {error}"),
2218+ }
2219+ return;
2220+ }
21402221 match repos.purge_due(PurgeDueArgs::default()).await {
21412222 Ok(0) => {}
21422223 Ok(count) => worker::console_log!("repos: purged {count} deleted repositories"),
+9−0
168168 let rows = DEFAULT_OPERATIONS
169169 .iter()
170170 .map(|meter| MappingRow { meter: (*meter).to_owned(), cost_operations: 1.0, billable_operations: 1.0, ..MappingRow::default() })
171+ .chain(COST_ONLY_OPERATIONS.iter().map(|meter| MappingRow {
172+ meter: (*meter).to_owned(),
173+ cost_operations: 1.0,
174+ ..MappingRow::default()
175+ }))
171176 .collect::<Vec<_>>();
172177 Mapping::from_rows(&rows)
173178 }
194199 "binding.delete",
195200 ];
196201
202+/// Meters that are an operation on g1t's own bill and on no workspace's:
203+/// a nightly backup's clone (backups.rs, migrations/0013).
204+pub const COST_ONLY_OPERATIONS: [&str; 1] = [crate::backups::FETCH_METER];
205+
197206 /// How long a read of `operation_mapping` is used for.
198207 const MAPPING_TTL_MS: u64 = 5 * 60 * 1000;
199208
+18−2
6868 // ARTIFACTS_REMOTE_BASE (optional): where remotes start,
6969 // https://<account>.artifacts.cloudflare.net/git, so a new isolate
7070 // needs no info() call before its first credential (src/store.rs).
71+ //
72+ // Backups (src/backups.rs):
73+ // BACKUP_STORE: r2 (the BACKUPS bucket) or s3 (BACKUP_S3_BUCKET on the
74+ // installation's S3_ENDPOINT, as self-hosting uses).
75+ // BACKUPS_PER_NIGHT: how many repositories one night queues, those
76+ // longest since their last backup first.
77+ // BACKUP_FULL_EVERY: incremental bundles before a full one again.
7178 "vars": {
79+ "BACKUP_STORE": "r2",
80+ "BACKUPS_PER_NIGHT": "200",
81+ "BACKUP_FULL_EVERY": "30",
7282 "GIT_OPERATIONS_FREE_CAP": "50000",
7383 "GIT_OPERATIONS_FREE_HOURLY": "60",
7484 "FREE_PRIVATE_STORAGE_BYTES": "1000000000",
8191 // Every hour, deleted repositories whose 30 days to be restored have
8292 // passed are purged, their git data with them (src/lifecycle.rs), and
8393 // pull request working copies past FORK_RETENTION_DAYS are removed
84− // (src/forks.rs).
85− "triggers": { "crons": ["23 * * * *"] },
94+ // (src/forks.rs). At 02:53 UTC, the repositories whose refs moved since
95+ // their last backup are queued for one (src/backups.rs, BACKUP_CRON);
96+ // the runner's sweep starts them a few at a time.
97+ "triggers": { "crons": ["23 * * * *", "53 2 * * *"] },
98+ // Nightly backups: a git bundle per repository and a manifest of each
99+ // chain (src/backups.rs). Made by
100+ // `npx wrangler r2 bucket create g1t-backups`.
101+ "r2_buckets": [{ "binding": "BACKUPS", "bucket_name": "g1t-backups" }],
86102 // A workspace renamed moves its repositories to the new slug; one
87103 // deleted purges the repositories it left in Recently deleted.
88104 "queues": {
+36−0
1+import assert from "node:assert/strict";
2+import { test } from "node:test";
3+
4+import type { BackupClaim } from "@g1t/contracts";
5+
6+import { backupEnv, backupPace, backupSandboxName } from "./backup.ts";
7+
8+const claim: BackupClaim = {
9+ jobId: "bkp_01",
10+ token: "secret-token",
11+ repoId: "repo_1",
12+ path: { namespace: "acme", name: "rocket" },
13+};
14+
15+test("the sandbox gets its job and token, and no git credential", () => {
16+ const env = backupEnv(claim, "https://api.g1t.sh");
17+ assert.deepEqual(env, {
18+ MODE: "backup",
19+ G1T_API: "https://api.g1t.sh",
20+ BACKUP_JOB: "bkp_01",
21+ BACKUP_TOKEN: "secret-token",
22+ });
23+ assert.ok(!("G1T_TOKEN" in env) && !("GIT_REMOTE" in env));
24+});
25+
26+test("one sandbox per job", () => {
27+ assert.equal(backupSandboxName(claim), "backup-bkp_01");
28+ assert.notEqual(backupSandboxName({ ...claim, jobId: "bkp_02" }), backupSandboxName(claim));
29+});
30+
31+test("the pace comes from the variables, and 0 turns backups off", () => {
32+ assert.deepEqual(backupPace(undefined, undefined), { perSweep: 4, running: 6 });
33+ assert.deepEqual(backupPace("2", "3"), { perSweep: 2, running: 3 });
34+ assert.deepEqual(backupPace("0", "6"), { perSweep: 0, running: 6 });
35+ assert.deepEqual(backupPace("many", "-1"), { perSweep: 4, running: 6 });
36+});
+48−0
1+/**
2+ * Nightly backups (`RunnerService.startBackups`): each sweep claims a few
3+ * of the backups the repos service queued and starts a sandbox for each,
4+ * which runs the runner's `backup` mode (crates/runner backup.rs). The
5+ * flow is in `crates/contracts/src/backups.rs`. Pure, so it is tested on
6+ * its own.
7+ */
8+import type { BackupClaim } from "@g1t/contracts";
9+
10+/** A backup's time cap: a clone and a bundle of at most 1 GB. */
11+export const BACKUP_MINUTES = 60;
12+
13+/** How many backups one sweep starts, and how many may run at once. */
14+export type BackupPace = { perSweep: number; running: number };
15+
16+const DEFAULT_PACE: BackupPace = { perSweep: 4, running: 6 };
17+
18+function count(value: string | undefined, fallback: number): number {
19+ if (value === undefined || value.trim() === "") return fallback;
20+ const parsed = Number(value);
21+ return Number.isInteger(parsed) && parsed >= 0 ? parsed : fallback;
22+}
23+
24+/**
25+ * BACKUPS_PER_SWEEP and BACKUPS_RUNNING, or their defaults. `0` per sweep
26+ * starts none: backups are off.
27+ */
28+export function backupPace(perSweep: string | undefined, running: string | undefined): BackupPace {
29+ return {
30+ perSweep: count(perSweep, DEFAULT_PACE.perSweep),
31+ running: count(running, DEFAULT_PACE.running),
32+ };
33+}
34+
35+/** One sandbox per job: asking twice starts nothing twice. */
36+export function backupSandboxName(claim: BackupClaim): string {
37+ return `backup-${claim.jobId}`;
38+}
39+
40+/** What the sandbox is started with: its job, and nothing that reads git. */
41+export function backupEnv(claim: BackupClaim, api: string): Record<string, string> {
42+ return {
43+ MODE: "backup",
44+ G1T_API: api,
45+ BACKUP_JOB: claim.jobId,
46+ BACKUP_TOKEN: claim.token,
47+ };
48+}
+57−4
7777 import { hostedOpen } from "./hosted";
7878 import { delegateInput, noModelMessage, notStarted, queued, started } from "./delegate";
7979 import { BUMP_MINUTES, BUMP_TOKEN_TTL_SECONDS, bumpEnv, bumpProblem, bumpSandboxName, systemActor } from "./bump";
80+import { BACKUP_MINUTES, backupEnv, backupPace, backupSandboxName } from "./backup";
8081 import { type ProjectSurroundings, readableSurroundings } from "./surroundings";
8182 import { holdCredentials, pushGrant, remotePath, revokeCredentials, runCredential } from "./credentials";
8283 import { buildMentionPrompt, describeThread, handleMention, planMention } from "./mentions";
180181 * either way.
181182 */
182183 ABUSE_WATCH?: string;
184+ /**
185+ * Nightly backups (backup.ts): how many queued backups one sweep starts
186+ * (`0`: none, backups off here), and how many may run at once.
187+ */
188+ BACKUPS_PER_SWEEP?: string;
189+ BACKUPS_RUNNING?: string;
183190 }
184191
185192 /**
226233 * its branch. The security service opens the pull request when it hears
227234 * the push, so a failure has no one to tell.
228235 */
229− | { kind: "bump"; repo: RepoPath; branch: string };
236+ | { kind: "bump"; repo: RepoPath; branch: string }
237+ /**
238+ * A repository's nightly backup: a bundle cut and sent to the repos
239+ * service. g1t's own work, never charged to the workspace.
240+ */
241+ | { kind: "backup"; jobId: string; token: string };
230242 /**
231243 * Whose sandbox time it is, reported when the sandbox stops, and the
232244 * machine it ran on when it was not the standard one.
284296 return "workflow";
285297 case "deploy":
286298 return "deploy";
299+ // Not metered: a backup is g1t's own cost.
300+ case "backup":
301+ return null;
287302 default:
288303 return "agent";
289304 }
449464 if (build) delete harness.GUARDRAILS;
450465 const watch: Record<string, string> = this.env.ABUSE_WATCH === "off" ? { G1T_ABUSE: "off" } : {};
451466 await this.start({ envVars: { ...vars, ...harness, ...watch }, enableInternet: !restricted });
452− if (guard) {
453− await this.ctx.storage.put("timeCap", guard.minutes);
454− await this.schedule(guard.minutes * 60 + ALARM_GRACE_SECONDS, "timeUp");
467+ // A backup has no guardrails, but still a time cap.
468+ const cap = guard?.minutes ?? (run.kind === "backup" ? BACKUP_MINUTES : null);
469+ if (cap) {
470+ await this.ctx.storage.put("timeCap", cap);
471+ await this.schedule(cap * 60 + ALARM_GRACE_SECONDS, "timeUp");
455472 }
456473 } catch (error) {
457474 await revokeCredentials(this.env.IDENTITY, this.ctx.storage, this.env.INTEGRATIONS);
677694 }
678695 // Nothing was pushed, so no pull request opens; why is in its log.
679696 if (run.kind === "bump") return;
697+ if (run.kind === "backup") {
698+ // Refused harmlessly if the sandbox reported before it stopped; the
699+ // job is otherwise tried again later tonight.
700+ await reposClient(this.env.REPOS)
701+ .failBackup(run.jobId, run.token, why ?? `The sandbox exited with ${exitCode}.`)
702+ .catch((error: unknown) => console.log("backup failure not reported", run.jobId, String(error)));
703+ return;
704+ }
680705 if (run.kind === "deploy") {
681706 // Refused harmlessly if the build reported its end before it stopped.
682707 await this.env.DEPLOYMENTS.fetch(`https://deployments/jobs/${run.deployId}/fail`, {
18801905 await this.drainWaits();
18811906 await this.advanceAll();
18821907 await this.startReady();
1908+ await this.startBackups().catch((error: unknown) => console.log("backups not started", String(error)));
1909+ }
1910+
1911+ /**
1912+ * Starts a few of the nightly backups the repos service queued, each in
1913+ * a sandbox of its own that holds only its job's token: the sandbox asks
1914+ * for a read-only git credential itself, when it is ready to clone. No
1915+ * plan is asked and nothing is metered: backups are g1t's own work.
1916+ */
1917+ private async startBackups(): Promise<void> {
1918+ const pace = backupPace(this.env.BACKUPS_PER_SWEEP, this.env.BACKUPS_RUNNING);
1919+ if (pace.perSweep === 0) return;
1920+ const repos = reposClient(this.env.REPOS);
1921+ for (const claim of await repos.claimBackups(pace.perSweep, pace.running)) {
1922+ try {
1923+ const sandbox = this.env.SANDBOX.get(this.env.SANDBOX.idFromName(backupSandboxName(claim)));
1924+ await sandbox.run({
1925+ kind: "backup",
1926+ jobId: claim.jobId,
1927+ token: claim.token,
1928+ // For `abuse.flagged`: whose repository it was.
1929+ owner: { workspace: claim.path.namespace, repo: `${claim.path.namespace}/${claim.path.name}` },
1930+ envVars: backupEnv(claim, "https://api.g1t.sh"),
1931+ });
1932+ } catch (error) {
1933+ await repos.failBackup(claim.jobId, claim.token, `The sandbox could not start: ${String(error)}`).catch(() => null);
1934+ }
1935+ }
18831936 }
18841937
18851938 /**
+7−2
6363 { "binding": "EVENTS", "service": "g1t-events" }
6464 ],
6565 // A sweep for lifecycle steps whose trigger was missed or whose sandbox
66− // died before reporting.
66+ // died before reporting, which also starts queued nightly backups.
6767 "triggers": { "crons": ["*/5 * * * *"] },
6868 // Events it reacts to: a pull request ready for review, or its head moving.
6969 "queues": {
9898 // Sandboxes stop themselves when they look like they are mining
9999 // (crates/runner abuse.rs). "off" turns the CPU watch off; miners
100100 // named in commands are refused either way.
101− "ABUSE_WATCH": "on"
101+ "ABUSE_WATCH": "on",
102+ // Nightly backups (src/backup.ts): each sweep starts this many of the
103+ // backups the repos service queued, with at most BACKUPS_RUNNING at
104+ // once. "0" starts none.
105+ "BACKUPS_PER_SWEEP": "4",
106+ "BACKUPS_RUNNING": "6"
102107 },
103108 "observability": { "enabled": true }
104109 }