Public surfaces are rate limited: anonymous git, archives, pages, the API, MCP, social cards and status subscriptions
Anonymous git clones, archive downloads, run pages, the REST API, MCP, og renders and status subscriptions had no limit, so one client could run up a repository owner's git operations, R2 writes and Worker time, or send confirmation emails to any address. Each is now behind a Workers Rate Limiting binding, over 60 seconds: - Front door (apps/web): git without credentials 120 per address, with them 1,200 per credential hash; signed-out pages 600 per address and archives, run pages, logs and search 30; signed-in 1,200 per session hash; every request 3,000 per address. Static files are never counted. - Repos: anonymous fetches the store answers, 120 per repository (cache hits never count), and pack-cache writes, 30 per repository; past that the pack streams to git without being kept. - API: 1,000 per token hash, 60 per address without one or with a wrong one; REST and MCP count apart. Sandbox and runner reports are not counted. 429 with Retry-After and {"error": {"code": "rate_limited"}}. - og: renders of uncached cards, 60 per address; past it, the brand card. - Status: POST /subscribe, 3 per address and 2 per email address. Every limit fails open when the binding is missing (self-hosted) or throws. RATE_LIMITS in packages/contracts is the table of record, and a test checks every wrangler.jsonc against it. Namespace ids are allocated in blocks per Worker (docs/RATE-LIMITS.md). The docs gain a rate limits reference page, linked from the API, MCP and git pages and llms.txt.
| 14 | 14 | mod checks; | |
| 15 | 15 | mod deployments; | |
| 16 | 16 | mod deploy_keys; | |
| 17 | + | mod limits; | |
| 17 | 18 | mod logs; | |
| 18 | 19 | mod mcp; | |
| 19 | 20 | mod notifications; | |
| ⋯ | |||
| 621 | 622 | return Ok(response); | |
| 622 | 623 | } | |
| 623 | 624 | ||
| 625 | + | // Per token, or per address without one (limits.rs). | |
| 626 | + | if let Some(limited) = limits::limited(&request, env, method, &path, on_mcp).await? { | |
| 627 | + | return Ok(limited); | |
| 628 | + | } | |
| 624 | 629 | let viewer = match authenticate(&request, &services).await? { | |
| 625 | 630 | Ok(viewer) => viewer, | |
| 626 | − | Err(refused) => return Ok(refused), | |
| 631 | + | Err(refused) => return Ok(limits::wrong_token(&request, env, on_mcp).await?.unwrap_or(refused)), | |
| 627 | 632 | }; | |
| 628 | 633 | // A person who has not confirmed their email address: who they are, | |
| 629 | 634 | // their addresses, and confirming one, nothing else (REST or MCP). | |
| ⋯ | |||
| 946 | 951 | "authorization, content-type", | |
| 947 | 952 | )?; | |
| 948 | 953 | headers.set("access-control-allow-methods", "GET, POST, PATCH, OPTIONS")?; | |
| 949 | − | headers.set("access-control-expose-headers", "www-authenticate")?; | |
| 954 | + | headers.set("access-control-expose-headers", "www-authenticate, retry-after")?; | |
| 950 | 955 | Ok(response) | |
| 951 | 956 | } | |
| 1 | + | //! How often a client may call the API and the MCP server. | |
| 2 | + | //! | |
| 3 | + | //! A request with a token counts against API_TOKEN_LIMIT under a hash of | |
| 4 | + | //! the token (never the token itself); one without counts against | |
| 5 | + | //! API_ANONYMOUS_LIMIT under the client's address (`CF-Connecting-IP`), as | |
| 6 | + | //! does one whose token turns out wrong, so guessing is limited by address. | |
| 7 | + | //! REST and MCP count apart. The limits are in `RATE_LIMITS` | |
| 8 | + | //! (packages/contracts/src/rate-limits.ts) and the docs' rate limits page. | |
| 9 | + | //! | |
| 10 | + | //! Not counted: what sandboxes, runners and outside systems send with | |
| 11 | + | //! credentials of their own (Stripe, connections' hooks, job tokens, | |
| 12 | + | //! report routes), which are answered before this or listed in | |
| 13 | + | //! [`counts`]. Without the bindings (self-hosted) nothing is limited, and | |
| 14 | + | //! a binding that fails lets the request through. | |
| 15 | + | ||
| 16 | + | use g1t_kit::limits::{self, PERIOD_SECONDS}; | |
| 17 | + | use serde_json::json; | |
| 18 | + | use sha2::{Digest, Sha256}; | |
| 19 | + | use worker::{Env, Request, Response, Result}; | |
| 20 | + | ||
| 21 | + | pub const ANONYMOUS: &str = "API_ANONYMOUS_LIMIT"; | |
| 22 | + | pub const TOKEN: &str = "API_TOKEN_LIMIT"; | |
| 23 | + | ||
| 24 | + | /// Paths a sandbox or runner reports to with its own token in the body: | |
| 25 | + | /// many sandboxes share an address, and none of them is a person. | |
| 26 | + | const REPORTS: &[&str] = &[ | |
| 27 | + | "/mergechecks/", | |
| 28 | + | "/backups/", | |
| 29 | + | "/queue/", | |
| 30 | + | "/actions/jobs/", | |
| 31 | + | "/agent-runs/", | |
| 32 | + | "/checks/", | |
| 33 | + | "/runs/", | |
| 34 | + | "/plans/", | |
| 35 | + | "/reviews/", | |
| 36 | + | "/runners/", | |
| 37 | + | ]; | |
| 38 | + | ||
| 39 | + | /// Whether a request counts against a limit at all. | |
| 40 | + | pub fn counts(method: &str, path: &str, has_token: bool) -> bool { | |
| 41 | + | has_token || method != "POST" || !REPORTS.iter().any(|prefix| path.starts_with(prefix)) | |
| 42 | + | } | |
| 43 | + | ||
| 44 | + | /// The binding a request counts against and its key there. | |
| 45 | + | pub fn key(token: Option<&str>, address: Option<&str>, on_mcp: bool) -> (&'static str, String) { | |
| 46 | + | let surface = if on_mcp { "mcp" } else { "rest" }; | |
| 47 | + | match token.filter(|token| !token.is_empty()) { | |
| 48 | + | Some(token) => (TOKEN, format!("{surface}:tok:{}", token_hash(token))), | |
| 49 | + | None => (ANONYMOUS, format!("{surface}:{}", limits::address_key(address))), | |
| 50 | + | } | |
| 51 | + | } | |
| 52 | + | ||
| 53 | + | /// The first 16 hex digits of the token's SHA-256. | |
| 54 | + | fn token_hash(token: &str) -> String { | |
| 55 | + | Sha256::digest(token.as_bytes())[..8].iter().map(|byte| format!("{byte:02x}")).collect() | |
| 56 | + | } | |
| 57 | + | ||
| 58 | + | /// The bearer token of an `Authorization` header, if it has one. | |
| 59 | + | pub fn bearer(header: &str) -> Option<&str> { | |
| 60 | + | match header.split_once(' ') { | |
| 61 | + | Some((scheme, token)) if scheme.eq_ignore_ascii_case("bearer") => Some(token.trim()).filter(|t| !t.is_empty()), | |
| 62 | + | _ => None, | |
| 63 | + | } | |
| 64 | + | } | |
| 65 | + | ||
| 66 | + | /// The 429 every limit answers, in the shape every error takes. | |
| 67 | + | pub fn too_many(signed_in: bool) -> Result<Response> { | |
| 68 | + | let message = if signed_in { | |
| 69 | + | "Too many requests with this token. Wait a minute and try again: https://docs.g1t.sh/reference/rate-limits/" | |
| 70 | + | } else { | |
| 71 | + | "Too many requests from this address. Wait a minute, or use an access token for a higher limit: https://docs.g1t.sh/reference/rate-limits/" | |
| 72 | + | }; | |
| 73 | + | let mut response = Response::from_json(&json!({ "error": { "code": "rate_limited", "message": message } }))?.with_status(429); | |
| 74 | + | response.headers_mut().set("retry-after", &PERIOD_SECONDS.to_string())?; | |
| 75 | + | Ok(response) | |
| 76 | + | } | |
| 77 | + | ||
| 78 | + | /// The 429 for a request past its limit, or `None` to go on. | |
| 79 | + | pub async fn limited(request: &Request, env: &Env, method: &str, path: &str, on_mcp: bool) -> Result<Option<Response>> { | |
| 80 | + | let header = request.headers().get("authorization")?.unwrap_or_default(); | |
| 81 | + | let token = bearer(&header); | |
| 82 | + | if !counts(method, path, token.is_some()) { | |
| 83 | + | return Ok(None); | |
| 84 | + | } | |
| 85 | + | let address = request.headers().get("cf-connecting-ip")?; | |
| 86 | + | let (binding, key) = key(token, address.as_deref(), on_mcp); | |
| 87 | + | if limits::check(env, binding, key).await.limited() { | |
| 88 | + | return too_many(token.is_some()).map(Some); | |
| 89 | + | } | |
| 90 | + | Ok(None) | |
| 91 | + | } | |
| 92 | + | ||
| 93 | + | /// A request whose token was wrong also counts against its address. | |
| 94 | + | pub async fn wrong_token(request: &Request, env: &Env, on_mcp: bool) -> Result<Option<Response>> { | |
| 95 | + | let address = request.headers().get("cf-connecting-ip")?; | |
| 96 | + | let (binding, key) = key(None, address.as_deref(), on_mcp); | |
| 97 | + | if limits::check(env, binding, key).await.limited() { | |
| 98 | + | return too_many(false).map(Some); | |
| 99 | + | } | |
| 100 | + | Ok(None) | |
| 101 | + | } | |
| 102 | + | ||
| 103 | + | #[cfg(test)] | |
| 104 | + | mod tests { | |
| 105 | + | use super::*; | |
| 106 | + | ||
| 107 | + | #[test] | |
| 108 | + | fn tokens_are_counted_by_their_hash_and_others_by_address() { | |
| 109 | + | let (binding, tok) = key(Some("g1t_secret"), Some("203.0.113.9"), false); | |
| 110 | + | assert_eq!(binding, TOKEN); | |
| 111 | + | assert!(tok.starts_with("rest:tok:") && tok.len() == "rest:tok:".len() + 16); | |
| 112 | + | assert!(!tok.contains("g1t_secret"), "the token never reaches the limiter"); | |
| 113 | + | assert_eq!(key(Some("g1t_secret"), None, false).1, tok, "the same token, the same key"); | |
| 114 | + | assert_ne!(key(Some("g1t_other"), None, false).1, tok); | |
| 115 | + | assert_eq!(key(None, Some("203.0.113.9"), false), (ANONYMOUS, "rest:ip:203.0.113.9".to_owned())); | |
| 116 | + | assert_eq!(key(Some(""), None, false), (ANONYMOUS, "rest:ip:unknown".to_owned())); | |
| 117 | + | } | |
| 118 | + | ||
| 119 | + | #[test] | |
| 120 | + | fn rest_and_mcp_count_apart() { | |
| 121 | + | assert_eq!(key(None, Some("203.0.113.9"), true).1, "mcp:ip:203.0.113.9"); | |
| 122 | + | assert!(key(Some("g1t_secret"), None, true).1.starts_with("mcp:tok:")); | |
| 123 | + | } | |
| 124 | + | ||
| 125 | + | #[test] | |
| 126 | + | fn sandbox_reports_are_not_counted_but_everything_else_is() { | |
| 127 | + | assert!(!counts("POST", "/checks/run_1", false)); | |
| 128 | + | assert!(!counts("POST", "/agent-runs/run_1/report", false)); | |
| 129 | + | assert!(!counts("POST", "/runs/run_1/usage", false)); | |
| 130 | + | assert!(counts("GET", "/repos/acme/rocket", false)); | |
| 131 | + | assert!(counts("POST", "/device/code", false)); | |
| 132 | + | assert!(counts("POST", "/checks/run_1", true), "with a bearer token it counts as that token"); | |
| 133 | + | } | |
| 134 | + | ||
| 135 | + | #[test] | |
| 136 | + | fn only_bearer_tokens_are_read() { | |
| 137 | + | assert_eq!(bearer("Bearer g1t_abc "), Some("g1t_abc")); | |
| 138 | + | assert_eq!(bearer("bearer g1t_abc"), Some("g1t_abc")); | |
| 139 | + | assert_eq!(bearer("Basic dXNlcjpwYXNz"), None); | |
| 140 | + | assert_eq!(bearer("Bearer "), None); | |
| 141 | + | assert_eq!(bearer(""), None); | |
| 142 | + | } | |
| 143 | + | } |
| 42 | 42 | // actions/cache entries (`c/`) and artifacts (`a/`), uploaded in parts. The actions | |
| 43 | 43 | // service lists them and decides what is kept (services/actions/src/cache.rs). | |
| 44 | 44 | "r2_buckets": [{ "binding": "ACTIONS_CACHE", "bucket_name": "g1t-actions-cache" }], | |
| 45 | + | // REST and MCP requests (src/limits.rs): per token, or per address | |
| 46 | + | // without one. Ids and limits: RATE_LIMITS in packages/contracts. | |
| 47 | + | "ratelimits": [ | |
| 48 | + | { "name": "API_ANONYMOUS_LIMIT", "namespace_id": "4401", "simple": { "limit": 60, "period": 60 } }, | |
| 49 | + | { "name": "API_TOKEN_LIMIT", "namespace_id": "4402", "simple": { "limit": 1000, "period": 60 } } | |
| 50 | + | ], | |
| 45 | 51 | "observability": { "enabled": true } | |
| 46 | 52 | } |
| 159 | 159 | items: [ | |
| 160 | 160 | { label: 'API overview', slug: 'reference/api' }, | |
| 161 | 161 | { label: 'MCP tools', slug: 'reference/mcp' }, | |
| 162 | + | { label: 'Rate limits', slug: 'reference/rate-limits' }, | |
| 162 | 163 | { label: 'Try it in the explorer', link: '/api/reference/', attrs: { target: '_self' } }, | |
| 163 | 164 | { label: 'OpenAPI document', link: 'https://api.g1t.sh/openapi.json' }, | |
| 164 | 165 | { label: 'llms.txt', link: 'https://g1t.sh/llms.txt' }, |
| 204 | 204 | until the month turns. Counting starts on 2026-10-14. See | |
| 205 | 205 | [git operations](/guides/usage-and-billing/#git-operations). | |
| 206 | 206 | ||
| 207 | + | ### Request limits | |
| 208 | + | ||
| 209 | + | Git requests without credentials are limited to 120 a minute from each IP | |
| 210 | + | address, about 40 clones; with credentials, 1,200 a minute for each set of | |
| 211 | + | credentials. Anonymous clones of one repository that g1t has not cached | |
| 212 | + | are limited to 120 a minute, whoever makes them. Past a limit, git is | |
| 213 | + | answered `429` with a message saying to wait a minute. Clone with | |
| 214 | + | [credentials](#authentication) to count against your own limit. See | |
| 215 | + | [rate limits](/reference/rate-limits/). | |
| 216 | + | ||
| 207 | 217 | What these limits mean in practice, and what to do instead, is on | |
| 208 | 218 | [What g1t can't do yet](/about/limitations/#git). | |
| 209 | 219 |
| 142 | 142 | | 404 | `not_found` | It does not exist, or you cannot see it. A path that is not an endpoint answers this too. | | |
| 143 | 143 | | 409 | `conflict` | The request conflicts with the current state. | | |
| 144 | 144 | | 422 | `invalid` | The input is not valid. | | |
| 145 | + | | 429 | `rate_limited` | Too many requests in the last minute. Wait the seconds `Retry-After` says. See [rate limits](#rate-limits). | | |
| 145 | 146 | ||
| 146 | 147 | Branch on `code`, not on `message`: messages are written for people and | |
| 147 | 148 | may change. | |
| ⋯ | |||
| 163 | 164 | [Settings → Access tokens](https://g1t.sh/settings/tokens), or use another | |
| 164 | 165 | token. A `403` for any other reason has no `needed_scope`. | |
| 165 | 166 | ||
| 167 | + | ## Rate limits | |
| 168 | + | ||
| 169 | + | Each token may make 1,000 requests a minute. Requests without a token are | |
| 170 | + | limited to 60 a minute for each IP address. Past either, the API answers | |
| 171 | + | `429` with `rate_limited` and a `Retry-After` header saying how many | |
| 172 | + | seconds to wait: | |
| 173 | + | ||
| 174 | + | ```json | |
| 175 | + | { "error": { "code": "rate_limited", "message": "Too many requests with this token. Wait a minute and try again: https://docs.g1t.sh/reference/rate-limits/" } } | |
| 176 | + | ``` | |
| 177 | + | ||
| 178 | + | Every limit, and how to stay under them, is on [rate limits](/reference/rate-limits/). | |
| 179 | + | ||
| 166 | 180 | ## Lists | |
| 167 | 181 | ||
| 168 | 182 | Lists come newest first, unless an endpoint says otherwise. Most return | |
| 203 | 203 | the REST API returns them. | |
| 204 | 204 | - Reading a public repository needs no sign-in through the API. Through MCP, | |
| 205 | 205 | every call needs to be signed in. | |
| 206 | + | - Each token may make 1,000 requests a minute to the MCP server, apart from | |
| 207 | + | its REST API calls. Past that, the request is answered `429` with a | |
| 208 | + | `Retry-After` header and a `rate_limited` error. See | |
| 209 | + | [rate limits](/reference/rate-limits/). | |
| 206 | 210 | ||
| 207 | 211 | The tables below list each action's required inputs. Optional inputs are | |
| 208 | 212 | in the tool's schema, which `tools/list` returns, and on the action's page |
| 1 | + | --- | |
| 2 | + | title: Rate limits | |
| 3 | + | description: How many requests the API, the MCP server, git over HTTPS and the site take a minute, what a limited request is answered with, and how to stay under the limits. | |
| 4 | + | --- | |
| 5 | + | ||
| 6 | + | g1t limits how many requests a client can make in a minute, so that one | |
| 7 | + | client cannot slow g1t down for everyone else or run up a repository | |
| 8 | + | owner's bill. The limits are well above what a person or an agent working | |
| 9 | + | normally reaches. Each counts requests over 60 seconds, approximately: a | |
| 10 | + | client can sometimes get a few more through before it is limited. | |
| 11 | + | ||
| 12 | + | ## Limits | |
| 13 | + | ||
| 14 | + | | Where | Counted by | Requests a minute | | |
| 15 | + | | --- | --- | --- | | |
| 16 | + | | REST API, with a token | Token | 1,000 | | |
| 17 | + | | REST API, without a token | Client IP address | 60 | | |
| 18 | + | | MCP server | Token | 1,000 | | |
| 19 | + | | Git over HTTPS, with credentials | Credentials | 1,200 | | |
| 20 | + | | Git over HTTPS, without credentials | Client IP address | 120 | | |
| 21 | + | | Anonymous clones of one repository that are not cached | Repository | 120 | | |
| 22 | + | | Pages on g1t.sh, signed in | Session | 1,200 | | |
| 23 | + | | Pages on g1t.sh, signed out | Client IP address | 600 | | |
| 24 | + | | Archive downloads, workflow run pages, logs and search, signed out | Client IP address | 30 | | |
| 25 | + | | Container and package registries | See [storage and pull limits](/guides/containers/#storage-and-pull-limits) | | | |
| 26 | + | ||
| 27 | + | The REST API and the MCP server count apart: calls to one do not use up | |
| 28 | + | the other's limit. A request with a token that is not valid counts against | |
| 29 | + | its IP address, as a request without a token does. On g1t.sh, every | |
| 30 | + | request from one IP address, signed in or not, also counts toward 3,000 a | |
| 31 | + | minute. | |
| 32 | + | ||
| 33 | + | Agents' sandboxes and workflow jobs report to g1t with their own | |
| 34 | + | credentials. Those reports are not rate limited. | |
| 35 | + | ||
| 36 | + | ## When you are limited | |
| 37 | + | ||
| 38 | + | A request past a limit is answered `429 Too Many Requests` with a | |
| 39 | + | `Retry-After` header: the number of seconds to wait before trying again. | |
| 40 | + | ||
| 41 | + | ```http | |
| 42 | + | HTTP/1.1 429 Too Many Requests | |
| 43 | + | Retry-After: 60 | |
| 44 | + | Content-Type: application/json | |
| 45 | + | ||
| 46 | + | { "error": { "code": "rate_limited", "message": "Too many requests with this token. Wait a minute and try again: https://docs.g1t.sh/reference/rate-limits/" } } | |
| 47 | + | ``` | |
| 48 | + | ||
| 49 | + | | Where | Body | | |
| 50 | + | | --- | --- | | |
| 51 | + | | REST API and MCP server | JSON in the [error shape](/reference/api/#errors) every endpoint uses, with `code` `rate_limited`. Through MCP it comes back as the HTTP answer to the request, not as a tool result. | | |
| 52 | + | | Git over HTTPS | Plain text, which git prints after `remote:` or in its error. | | |
| 53 | + | | Pages on g1t.sh | Plain text. | | |
| 54 | + | ||
| 55 | + | Branch on the `429` status or the `rate_limited` code, never on the | |
| 56 | + | message. The API sends `Access-Control-Expose-Headers: retry-after`, so a | |
| 57 | + | browser app can read the header too. | |
| 58 | + | ||
| 59 | + | ## Staying under the limits | |
| 60 | + | ||
| 61 | + | - **Send a token.** Signed-in limits are much higher than anonymous ones, | |
| 62 | + | and they follow the token rather than the network you are on, so people | |
| 63 | + | sharing an office or a VPN do not share a limit. | |
| 64 | + | - **Clone with credentials.** A clone is about three git requests. Use | |
| 65 | + | [a token as the password](/guides/git/#authentication) to count against your own | |
| 66 | + | limit rather than your network's. | |
| 67 | + | - **Wait for `Retry-After`.** Retrying sooner is answered `429` again and | |
| 68 | + | counts against the limit. | |
| 69 | + | - **Cache what does not change.** A commit's contents never change: read | |
| 70 | + | a commit by its SHA once and keep it. | |
| 71 | + | - **Use webhooks instead of polling.** A [webhook](/guides/webhooks/) | |
| 72 | + | tells you when something changes, without a request a minute. | |
| 73 | + | ||
| 74 | + | Repeated anonymous clones of the same commit are answered from a cache and | |
| 75 | + | do not count against the repository's limit. If a limit gets in the way of | |
| 76 | + | something you need to do, [contact support](https://g1t.sh/support). | |
| 77 | + | ||
| 78 | + | ## Running g1t yourself | |
| 79 | + | ||
| 80 | + | An installation you run yourself has no rate limits. See | |
| 81 | + | [run g1t yourself](/guides/self-hosting/). |
| 47 | 47 | fail, | |
| 48 | 48 | ok, | |
| 49 | 49 | } from "@g1t/contracts"; | |
| 50 | + | import { LIMIT_PERIOD_SECONDS, type RateLimitBinding, clientAddress, isLimited, secretKey } from "@g1t/contracts/rate-limits"; | |
| 50 | 51 | import bricolage from "@g1t/theme/fonts/bricolage-grotesque-latin.woff2"; | |
| 51 | 52 | import hanken from "@g1t/theme/fonts/hanken-grotesk-latin.woff2"; | |
| 52 | 53 | import plexMono from "@g1t/theme/fonts/ibm-plex-mono-latin-400.woff2"; | |
| ⋯ | |||
| 186 | 187 | * it, deploys are not announced and detection does not hold off for them. | |
| 187 | 188 | */ | |
| 188 | 189 | STATUS_DEPLOY_TOKEN?: string; | |
| 190 | + | /** | |
| 191 | + | * Asking for a subscription, per client address and per email address | |
| 192 | + | * (RATE_LIMITS in packages/contracts). Without them, only the resend | |
| 193 | + | * window (RESEND_AFTER_MS) holds back repeated emails to one address. | |
| 194 | + | */ | |
| 195 | + | STATUS_SUBSCRIBE_LIMIT?: RateLimitBinding; | |
| 196 | + | STATUS_EMAIL_LIMIT?: RateLimitBinding; | |
| 189 | 197 | } | |
| 190 | 198 | ||
| 191 | 199 | /** How long the edge keeps a page or the JSON. */ | |
| ⋯ | |||
| 539 | 547 | ||
| 540 | 548 | if (path === "/subscribe" && post) { | |
| 541 | 549 | if (!emailOn(env)) return message("Email updates are not available", "Follow the Atom or JSON feed instead.", 503); | |
| 550 | + | // Each request can send an email: limited per client, then per address. | |
| 551 | + | const tooMany = () => { | |
| 552 | + | const answer = message("Too many requests", "Wait a minute and try again.", 429); | |
| 553 | + | answer.headers.set("retry-after", String(LIMIT_PERIOD_SECONDS)); | |
| 554 | + | return answer; | |
| 555 | + | }; | |
| 556 | + | if (await isLimited(env.STATUS_SUBSCRIBE_LIMIT, `ip:${clientAddress(request)}`)) return tooMany(); | |
| 542 | 557 | const data = await form(request); | |
| 543 | 558 | if (String(data.get("website") ?? "")) return message("Check your inbox", "If the address is right, a confirmation link is on its way."); | |
| 544 | 559 | const email = normalizeEmail(data.get("email")); | |
| 545 | 560 | if (!email) return message("That is not an email address", "Go back and check it.", 400); | |
| 561 | + | if (await isLimited(env.STATUS_EMAIL_LIMIT, await secretKey("email", email))) return tooMany(); | |
| 546 | 562 | const chosen = chosenParts(data.getAll("components").map(String), parts(env).map((p) => p.key)); | |
| 547 | 563 | const token = newToken(); | |
| 548 | 564 | const { send } = await requestSubscription(env.DB, email, chosen, await hashToken(token), new Date(), CONFIRM_TTL_MS, RESEND_AFTER_MS); | |
| 69 | 69 | "STATUS_FROM": "g1t status <noreply@g1t.sh>", | |
| 70 | 70 | "OG_IMAGE": "https://og.g1t.sh/image?path=%2Fstatus&v=2" | |
| 71 | 71 | }, | |
| 72 | + | // Asking for a subscription sends an email: limited per client address | |
| 73 | + | // and per email address (src/index.ts). Ids and limits: RATE_LIMITS in | |
| 74 | + | // packages/contracts. | |
| 75 | + | "ratelimits": [ | |
| 76 | + | { "name": "STATUS_SUBSCRIBE_LIMIT", "namespace_id": "4601", "simple": { "limit": 3, "period": 60 } }, | |
| 77 | + | { "name": "STATUS_EMAIL_LIMIT", "namespace_id": "4602", "simple": { "limit": 2, "period": 60 } } | |
| 78 | + | ], | |
| 72 | 79 | "observability": { "enabled": true } | |
| 73 | 80 | } |
| 1 | + | import assert from "node:assert/strict"; | |
| 2 | + | import { readFileSync } from "node:fs"; | |
| 3 | + | import { test } from "node:test"; | |
| 4 | + | ||
| 5 | + | import { | |
| 6 | + | LIMIT_PERIOD_SECONDS, | |
| 7 | + | RATE_LIMITS, | |
| 8 | + | type RateLimitBinding, | |
| 9 | + | checkLimit, | |
| 10 | + | isLimited, | |
| 11 | + | secretKey, | |
| 12 | + | } from "@g1t/contracts/rate-limits"; | |
| 13 | + | ||
| 14 | + | import { gitLimited, heavy, pageLimited, sessionCookie, unlimited } from "./front-door-limits.ts"; | |
| 15 | + | ||
| 16 | + | /** A binding that lets `allow` requests through per key, recording each key it was asked. */ | |
| 17 | + | function binding(allow: number): RateLimitBinding & { keys: string[] } { | |
| 18 | + | const counts = new Map<string, number>(); | |
| 19 | + | const keys: string[] = []; | |
| 20 | + | return { | |
| 21 | + | keys, | |
| 22 | + | async limit({ key }) { | |
| 23 | + | keys.push(key); | |
| 24 | + | const count = (counts.get(key) ?? 0) + 1; | |
| 25 | + | counts.set(key, count); | |
| 26 | + | return { success: count <= allow }; | |
| 27 | + | }, | |
| 28 | + | }; | |
| 29 | + | } | |
| 30 | + | ||
| 31 | + | const broken: RateLimitBinding = { | |
| 32 | + | async limit() { | |
| 33 | + | throw new Error("the binding is down"); | |
| 34 | + | }, | |
| 35 | + | }; | |
| 36 | + | ||
| 37 | + | function request(path: string, headers: Record<string, string> = {}): Request { | |
| 38 | + | return new Request(`https://g1t.sh${path}`, { headers: { "cf-connecting-ip": "203.0.113.9", ...headers } }); | |
| 39 | + | } | |
| 40 | + | ||
| 41 | + | test("a limit lets requests through until it is reached", async () => { | |
| 42 | + | const limit = binding(2); | |
| 43 | + | assert.equal(await checkLimit(limit, "ip:1"), "allowed"); | |
| 44 | + | assert.equal(await checkLimit(limit, "ip:1"), "allowed"); | |
| 45 | + | assert.equal(await checkLimit(limit, "ip:1"), "limited"); | |
| 46 | + | assert.equal(await checkLimit(limit, "ip:2"), "allowed", "each key counts apart"); | |
| 47 | + | }); | |
| 48 | + | ||
| 49 | + | test("a missing or failing binding fails open", async () => { | |
| 50 | + | const logged = console.error; | |
| 51 | + | console.error = () => {}; | |
| 52 | + | try { | |
| 53 | + | assert.equal(await checkLimit(undefined, "ip:1"), "unavailable"); | |
| 54 | + | assert.equal(await checkLimit(broken, "ip:1"), "unavailable"); | |
| 55 | + | assert.equal(await isLimited(broken, "ip:1"), false); | |
| 56 | + | assert.equal(await gitLimited({ GIT_ANONYMOUS_LIMIT: broken }, request("/acme/rocket.git/info/refs")), null); | |
| 57 | + | assert.equal(await pageLimited({ WEB_ADDRESS_LIMIT: broken, WEB_ANONYMOUS_LIMIT: broken }, request("/acme/rocket"), "/acme/rocket"), null); | |
| 58 | + | } finally { | |
| 59 | + | console.error = logged; | |
| 60 | + | } | |
| 61 | + | }); | |
| 62 | + | ||
| 63 | + | test("secrets are keyed by a short hash, never as they are", async () => { | |
| 64 | + | const key = await secretKey("session", "s3cret-session"); | |
| 65 | + | assert.match(key, /^session:[0-9a-f]{16}$/); | |
| 66 | + | assert.equal(await secretKey("session", "s3cret-session"), key); | |
| 67 | + | assert.notEqual(await secretKey("session", "another"), key); | |
| 68 | + | }); | |
| 69 | + | ||
| 70 | + | test("git without credentials is limited by address, with a message git shows", async () => { | |
| 71 | + | const env = { GIT_ANONYMOUS_LIMIT: binding(1), GIT_SIGNED_LIMIT: binding(1) }; | |
| 72 | + | const clone = () => request("/acme/rocket.git/info/refs"); | |
| 73 | + | assert.equal(await gitLimited(env, clone()), null); | |
| 74 | + | const refused = await gitLimited(env, clone()); | |
| 75 | + | assert.equal(refused?.status, 429); | |
| 76 | + | assert.equal(refused?.headers.get("retry-after"), String(LIMIT_PERIOD_SECONDS)); | |
| 77 | + | assert.match(refused?.headers.get("content-type") ?? "", /^text\/plain/); | |
| 78 | + | assert.match((await refused?.text()) ?? "", /Too many git requests/); | |
| 79 | + | assert.deepEqual(env.GIT_ANONYMOUS_LIMIT.keys, ["ip:203.0.113.9", "ip:203.0.113.9"]); | |
| 80 | + | }); | |
| 81 | + | ||
| 82 | + | test("git with credentials counts by a hash of them, apart from the address", async () => { | |
| 83 | + | const env = { GIT_ANONYMOUS_LIMIT: binding(0), GIT_SIGNED_LIMIT: binding(5) }; | |
| 84 | + | const authorization = `Basic ${btoa("ada:g1t_token")}`; | |
| 85 | + | assert.equal(await gitLimited(env, request("/acme/rocket.git/git-upload-pack", { authorization })), null); | |
| 86 | + | assert.equal(env.GIT_ANONYMOUS_LIMIT.keys.length, 0); | |
| 87 | + | assert.match(env.GIT_SIGNED_LIMIT.keys[0]!, /^git:[0-9a-f]{16}$/); | |
| 88 | + | assert.ok(!env.GIT_SIGNED_LIMIT.keys[0]!.includes("g1t_token")); | |
| 89 | + | }); | |
| 90 | + | ||
| 91 | + | test("signed-out pages count by address, and costly ones against a tighter limit too", async () => { | |
| 92 | + | const env = { WEB_ADDRESS_LIMIT: binding(100), WEB_ANONYMOUS_LIMIT: binding(100), WEB_HEAVY_LIMIT: binding(1) }; | |
| 93 | + | assert.equal(await pageLimited(env, request("/acme/rocket"), "/acme/rocket"), null); | |
| 94 | + | assert.equal(env.WEB_HEAVY_LIMIT.keys.length, 0); | |
| 95 | + | const archive = "/acme/rocket/archive/main.zip"; | |
| 96 | + | assert.equal(await pageLimited(env, request(archive), archive), null); | |
| 97 | + | const refused = await pageLimited(env, request(archive), archive); | |
| 98 | + | assert.equal(refused?.status, 429); | |
| 99 | + | assert.match((await refused?.text()) ?? "", /Signed-in accounts have a higher limit/); | |
| 100 | + | assert.equal(await pageLimited(env, request("/acme/rocket/issues"), "/acme/rocket/issues"), null, "other pages go on"); | |
| 101 | + | }); | |
| 102 | + | ||
| 103 | + | test("signed-in requests count by session, and every request by address", async () => { | |
| 104 | + | const env = { WEB_ADDRESS_LIMIT: binding(1), WEB_SESSION_LIMIT: binding(100), WEB_ANONYMOUS_LIMIT: binding(0) }; | |
| 105 | + | const signedIn = () => request("/acme/rocket/archive/main.zip", { cookie: "theme=dark; g1t_session=abc123" }); | |
| 106 | + | assert.equal(await pageLimited(env, signedIn(), "/acme/rocket/archive/main.zip"), null); | |
| 107 | + | assert.equal(env.WEB_ANONYMOUS_LIMIT.keys.length, 0); | |
| 108 | + | assert.match(env.WEB_SESSION_LIMIT.keys[0]!, /^session:[0-9a-f]{16}$/); | |
| 109 | + | // Made-up cookies still meet the ceiling per address. | |
| 110 | + | const refused = await pageLimited(env, request("/", { cookie: "g1t_session=made-up" }), "/"); | |
| 111 | + | assert.equal(refused?.status, 429); | |
| 112 | + | }); | |
| 113 | + | ||
| 114 | + | test("files the Worker serves itself are never limited", async () => { | |
| 115 | + | const env = { WEB_ADDRESS_LIMIT: binding(0), WEB_ANONYMOUS_LIMIT: binding(0) }; | |
| 116 | + | for (const path of ["/assets/app-1a2b.js", "/fonts/hanken.woff2", "/favicon.ico", "/robots.txt", "/llms.txt", "/sitemap.xml"]) { | |
| 117 | + | assert.ok(unlimited(path), path); | |
| 118 | + | assert.equal(await pageLimited(env, request(path), path), null, path); | |
| 119 | + | } | |
| 120 | + | assert.ok(!unlimited("/acme/rocket/blob/main/logo.png"), "a file in a repository is a page"); | |
| 121 | + | }); | |
| 122 | + | ||
| 123 | + | test("costly pages are recognised", () => { | |
| 124 | + | for (const path of [ | |
| 125 | + | "/search", | |
| 126 | + | "/search.data", | |
| 127 | + | "/acme/rocket/archive/main.zip", | |
| 128 | + | "/acme/rocket/actions/runs/run_1", | |
| 129 | + | "/acme/rocket/actions/runs/run_1.data", | |
| 130 | + | "/acme/rocket/actions/runs/run_1/logs.zip", | |
| 131 | + | "/acme/rocket/actions/runs/run_1/artifacts/dist", | |
| 132 | + | "/acme/rocket/actions/jobs/job_1/log.txt", | |
| 133 | + | ]) { | |
| 134 | + | assert.ok(heavy(path), path); | |
| 135 | + | } | |
| 136 | + | for (const path of ["/", "/acme/rocket", "/acme/rocket/actions", "/acme/rocket/issues/1", "/acme/search"]) { | |
| 137 | + | assert.ok(!heavy(path), path); | |
| 138 | + | } | |
| 139 | + | }); | |
| 140 | + | ||
| 141 | + | test("the session cookie is read from among others", () => { | |
| 142 | + | assert.equal(sessionCookie("theme=dark; g1t_session=abc; x=1"), "abc"); | |
| 143 | + | assert.equal(sessionCookie("g1t_session=abc"), "abc"); | |
| 144 | + | assert.equal(sessionCookie("not_g1t_session=abc"), null); | |
| 145 | + | assert.equal(sessionCookie(null), null); | |
| 146 | + | }); | |
| 147 | + | ||
| 148 | + | /** A wrangler.jsonc as JSON: comments and trailing commas out, strings kept. */ | |
| 149 | + | function readJsonc(path: string): { ratelimits?: { name: string; namespace_id: string; simple: { limit: number; period: number } }[] } { | |
| 150 | + | const text = readFileSync(new URL(path, import.meta.url), "utf8") | |
| 151 | + | .replace(/("(?:\\.|[^"\\])*")|\/\/[^\n]*|\/\*[\s\S]*?\*\//g, (_, string: string | undefined) => string ?? "") | |
| 152 | + | .replace(/,(\s*[}\]])/g, "$1"); | |
| 153 | + | return JSON.parse(text); | |
| 154 | + | } | |
| 155 | + | ||
| 156 | + | test("every wrangler.jsonc declares the rate limits RATE_LIMITS lists, and only those", () => { | |
| 157 | + | const ids = new Set<number>(); | |
| 158 | + | const workers = new Set(Object.values(RATE_LIMITS).map((spec) => spec.worker)); | |
| 159 | + | for (const worker of workers) { | |
| 160 | + | const declared = readJsonc(`../../../../${worker}/wrangler.jsonc`).ratelimits ?? []; | |
| 161 | + | const listed = Object.entries(RATE_LIMITS).filter(([, spec]) => spec.worker === worker); | |
| 162 | + | assert.deepEqual( | |
| 163 | + | declared.map((binding) => binding.name).sort(), | |
| 164 | + | listed.map(([name]) => name).sort(), | |
| 165 | + | `${worker}'s bindings`, | |
| 166 | + | ); | |
| 167 | + | for (const [name, spec] of listed) { | |
| 168 | + | const binding = declared.find((b) => b.name === name)!; | |
| 169 | + | assert.equal(Number(binding.namespace_id), spec.namespaceId, `${name}'s namespace id`); | |
| 170 | + | assert.equal(binding.simple.limit, spec.limit, `${name}'s limit`); | |
| 171 | + | assert.equal(binding.simple.period, LIMIT_PERIOD_SECONDS, `${name}'s period`); | |
| 172 | + | } | |
| 173 | + | } | |
| 174 | + | for (const spec of Object.values(RATE_LIMITS)) { | |
| 175 | + | assert.ok(!ids.has(spec.namespaceId), `namespace id ${spec.namespaceId} is used once`); | |
| 176 | + | ids.add(spec.namespaceId); | |
| 177 | + | } | |
| 178 | + | }); |
| 1 | + | /** | |
| 2 | + | * The front door's rate limits (workers/app.ts), per request before it is | |
| 3 | + | * answered. The bindings and their limits are in `RATE_LIMITS` | |
| 4 | + | * (packages/contracts/src/rate-limits.ts); docs/RATE-LIMITS.md says why. | |
| 5 | + | * | |
| 6 | + | * - Git over HTTPS: without credentials, by address; with them, by a hash | |
| 7 | + | * of the credential, much higher. A clone is about three requests. | |
| 8 | + | * - Pages: every request that reaches the Worker counts against a ceiling | |
| 9 | + | * per address. Signed out, by address, with a tighter limit on what is | |
| 10 | + | * costly to answer (archives, run pages, logs, search). Signed in, by a | |
| 11 | + | * hash of the session cookie, higher: the session is not checked here, | |
| 12 | + | * which would cost a call to identity, and the ceiling per address keeps | |
| 13 | + | * made-up cookies from getting round the signed-out limit. | |
| 14 | + | * | |
| 15 | + | * Static assets never reach the Worker (the assets binding answers them), | |
| 16 | + | * and the few files it serves itself are left out here too. Every limit | |
| 17 | + | * fails open. | |
| 18 | + | */ | |
| 19 | + | import { | |
| 20 | + | type RateLimitBinding, | |
| 21 | + | checkLimit, | |
| 22 | + | clientAddress, | |
| 23 | + | secretKey, | |
| 24 | + | tooManyRequests, | |
| 25 | + | } from "@g1t/contracts/rate-limits"; | |
| 26 | + | ||
| 27 | + | export type FrontDoorLimits = { | |
| 28 | + | WEB_ANONYMOUS_LIMIT?: RateLimitBinding; | |
| 29 | + | WEB_HEAVY_LIMIT?: RateLimitBinding; | |
| 30 | + | WEB_SESSION_LIMIT?: RateLimitBinding; | |
| 31 | + | WEB_ADDRESS_LIMIT?: RateLimitBinding; | |
| 32 | + | GIT_ANONYMOUS_LIMIT?: RateLimitBinding; | |
| 33 | + | GIT_SIGNED_LIMIT?: RateLimitBinding; | |
| 34 | + | }; | |
| 35 | + | ||
| 36 | + | /** Files the Worker serves that are never limited: build output, fonts, and top-level files such as robots.txt. */ | |
| 37 | + | const UNLIMITED = /^\/(?:assets\/|fonts\/|favicon|[^/]+\.(?:ico|png|svg|txt|xml|webmanifest)$)/; | |
| 38 | + | ||
| 39 | + | /** What is costly to answer for a signed-out visitor: a repository's archives, a run's page, logs and artifacts, and search. */ | |
| 40 | + | const HEAVY = | |
| 41 | + | /^\/(?:search(?:\.data)?$|[^/]+\/[^/]+\/(?:archive\/|actions\/runs\/[^/]+(?:\.data|\/logs\.zip|\/artifacts\/[^/]+)?$|actions\/jobs\/[^/]+\/log))/; | |
| 42 | + | ||
| 43 | + | export function unlimited(pathname: string): boolean { | |
| 44 | + | return UNLIMITED.test(pathname); | |
| 45 | + | } | |
| 46 | + | ||
| 47 | + | export function heavy(pathname: string): boolean { | |
| 48 | + | return HEAVY.test(pathname); | |
| 49 | + | } | |
| 50 | + | ||
| 51 | + | /** The session cookie's value, or null when signed out. */ | |
| 52 | + | export function sessionCookie(cookie: string | null): string | null { | |
| 53 | + | const match = /(?:^|;\s*)g1t_session=([^;]+)/.exec(cookie ?? ""); | |
| 54 | + | return match?.[1] ?? null; | |
| 55 | + | } | |
| 56 | + | ||
| 57 | + | const GIT_MESSAGE_ANONYMOUS = | |
| 58 | + | "Too many git requests from your network. Wait a minute and try again, or use credentials for a higher limit: https://docs.g1t.sh/reference/rate-limits/\n"; | |
| 59 | + | const GIT_MESSAGE_SIGNED = "Too many git requests with these credentials. Wait a minute and try again: https://docs.g1t.sh/reference/rate-limits/\n"; | |
| 60 | + | const PAGE_MESSAGE = "Too many requests from your network. Wait a minute and try again.\n"; | |
| 61 | + | ||
| 62 | + | /** | |
| 63 | + | * The 429 for a git request past its limit, or null to go on. Git shows a | |
| 64 | + | * plain-text answer's body to the person running it. | |
| 65 | + | */ | |
| 66 | + | export async function gitLimited(env: FrontDoorLimits, request: Request): Promise<Response | null> { | |
| 67 | + | const credentials = request.headers.get("authorization"); | |
| 68 | + | if (credentials) { | |
| 69 | + | const verdict = await checkLimit(env.GIT_SIGNED_LIMIT, await secretKey("git", credentials)); | |
| 70 | + | return verdict === "limited" ? tooManyRequests(GIT_MESSAGE_SIGNED) : null; | |
| 71 | + | } | |
| 72 | + | const verdict = await checkLimit(env.GIT_ANONYMOUS_LIMIT, `ip:${clientAddress(request)}`); | |
| 73 | + | return verdict === "limited" ? tooManyRequests(GIT_MESSAGE_ANONYMOUS) : null; | |
| 74 | + | } | |
| 75 | + | ||
| 76 | + | /** The 429 for a page or data request past its limit, or null to go on. */ | |
| 77 | + | export async function pageLimited(env: FrontDoorLimits, request: Request, pathname: string): Promise<Response | null> { | |
| 78 | + | if (unlimited(pathname)) return null; | |
| 79 | + | const address = `ip:${clientAddress(request)}`; | |
| 80 | + | const session = sessionCookie(request.headers.get("cookie")); | |
| 81 | + | const checks: Promise<string>[] = [checkLimit(env.WEB_ADDRESS_LIMIT, address)]; | |
| 82 | + | if (session) { | |
| 83 | + | checks.push(secretKey("session", session).then((key) => checkLimit(env.WEB_SESSION_LIMIT, key))); | |
| 84 | + | } else { | |
| 85 | + | checks.push(checkLimit(env.WEB_ANONYMOUS_LIMIT, address)); | |
| 86 | + | if (heavy(pathname)) checks.push(checkLimit(env.WEB_HEAVY_LIMIT, address)); | |
| 87 | + | } | |
| 88 | + | const verdicts = await Promise.all(checks); | |
| 89 | + | if (!verdicts.includes("limited")) return null; | |
| 90 | + | return tooManyRequests(session ? PAGE_MESSAGE : `${PAGE_MESSAGE.trimEnd()} Signed-in accounts have a higher limit.\n`); | |
| 91 | + | } |
| 684 | 684 | `{"error": {"code": "…", "message": "…"}}` with codes `unauthenticated` | |
| 685 | 685 | (401), `payment_required` (402, when the plan or a limit refuses compute), | |
| 686 | 686 | `forbidden` (403, with `needed_scope` when the token lacks a | |
| 687 | − | scope), `not_found` (404), `conflict` (409), `invalid` (422). Each | |
| 687 | + | scope), `not_found` (404), `conflict` (409), `invalid` (422), | |
| 688 | + | `rate_limited` (429, with `Retry-After`: 1,000 requests a minute per | |
| 689 | + | token, 60 per IP address without one; wait, then retry). Each | |
| 688 | 690 | operation's scope is `x-scope` in the OpenAPI document. The full description is at https://api.g1t.sh/openapi.json. | |
| 689 | 691 | - Git remote: `https://g1t.sh/{workspace}/{repo}.git`. In API paths, | |
| 690 | 692 | `{owner}` is the workspace. Pull request forks: | |
| ⋯ | |||
| 752 | 754 | - [Git](https://docs.g1t.sh/guides/git/) | |
| 753 | 755 | - [MCP tools](https://docs.g1t.sh/reference/mcp/) | |
| 754 | 756 | - [API reference](https://docs.g1t.sh/reference/api/) | |
| 757 | + | - [Rate limits](https://docs.g1t.sh/reference/rate-limits/) | |
| 755 | 758 | - [What g1t can't do yet](https://docs.g1t.sh/about/limitations/) | |
| 756 | 759 | - [An open letter to Cloudflare](https://docs.g1t.sh/about/open-letter-to-cloudflare/) | |
| 757 | 760 | - [Source](https://g1t.sh/flagon-io/g1t), MIT licensed | |
| 3 | 3 | import { identityClient, isNamespaceShaped } from "@g1t/contracts"; | |
| 4 | 4 | ||
| 5 | 5 | import { finishResponse, withRequestPerf } from "../app/lib/perf.server"; | |
| 6 | + | import { gitLimited, pageLimited } from "../app/lib/front-door-limits"; | |
| 6 | 7 | import { goImport } from "../app/lib/go-get"; | |
| 7 | 8 | import { repositoryOfPage, stillPublic } from "../app/lib/public-cache"; | |
| 8 | 9 | import { registryWorkspace, servicePath } from "../app/lib/registry-paths"; | |
| ⋯ | |||
| 50 | 51 | } | |
| 51 | 52 | const service = servicePath(pathname); | |
| 52 | 53 | if (service === "git") { | |
| 53 | − | return proxyGit(env, request); | |
| 54 | + | // Per address without credentials, per credential with them | |
| 55 | + | // (app/lib/front-door-limits.ts): anonymous clones are not free to | |
| 56 | + | // the repository's owner. | |
| 57 | + | return (await gitLimited(env, request)) ?? proxyGit(env, request); | |
| 54 | 58 | } | |
| 55 | 59 | if (service === "packages") { | |
| 56 | 60 | return proxyPackages(env, request); | |
| ⋯ | |||
| 65 | 69 | const target = MOVED_DOCS[page] ?? "/"; | |
| 66 | 70 | return Response.redirect(DOCS + target, 301); | |
| 67 | 71 | } | |
| 72 | + | // Pages and data requests, limited per address signed out and per | |
| 73 | + | // session signed in (app/lib/front-door-limits.ts). | |
| 74 | + | const limited = await pageLimited(env, request, pathname); | |
| 75 | + | if (limited) return limited; | |
| 68 | 76 | // Every page and data request says where its time went (Server-Timing) | |
| 69 | 77 | // and keeps the reader's D1 bookmarks (app/lib/perf.server.ts). | |
| 70 | 78 | const render = () => withRequestPerf(request, async () => finishResponse(request, await requestHandler(request))); | |
| 1 | − | import type { RunnerApi, ServiceBinding } from "@g1t/contracts"; | |
| 1 | + | import type { RateLimitBinding, RunnerApi, ServiceBinding } from "@g1t/contracts"; | |
| 2 | 2 | ||
| 3 | 3 | declare global { | |
| 4 | 4 | namespace Cloudflare { | |
| ⋯ | |||
| 44 | 44 | MCP_URL?: string; | |
| 45 | 45 | /** The social-card image service; an empty string for none. Unset on g1t.sh. */ | |
| 46 | 46 | OG_URL?: string; | |
| 47 | + | /** The front door's rate limits (app/lib/front-door-limits.ts). Absent when self-hosted. */ | |
| 48 | + | WEB_ANONYMOUS_LIMIT?: RateLimitBinding; | |
| 49 | + | WEB_HEAVY_LIMIT?: RateLimitBinding; | |
| 50 | + | WEB_SESSION_LIMIT?: RateLimitBinding; | |
| 51 | + | WEB_ADDRESS_LIMIT?: RateLimitBinding; | |
| 52 | + | GIT_ANONYMOUS_LIMIT?: RateLimitBinding; | |
| 53 | + | GIT_SIGNED_LIMIT?: RateLimitBinding; | |
| 47 | 54 | } | |
| 48 | 55 | } | |
| 49 | 56 | interface Env extends Cloudflare.Env {} | |
| 43 | 43 | // Production screenshots for a project's overview (services/og). | |
| 44 | 44 | { "binding": "SCREENSHOTS", "service": "g1t-og", "entrypoint": "Screenshots" } | |
| 45 | 45 | ], | |
| 46 | + | // The front door's limits (app/lib/front-door-limits.ts). Every id and | |
| 47 | + | // limit is in RATE_LIMITS (packages/contracts/src/rate-limits.ts), which | |
| 48 | + | // the tests check this against. Self-hosted, without them, nothing is limited. | |
| 49 | + | "ratelimits": [ | |
| 50 | + | { "name": "WEB_ANONYMOUS_LIMIT", "namespace_id": "4201", "simple": { "limit": 600, "period": 60 } }, | |
| 51 | + | { "name": "WEB_HEAVY_LIMIT", "namespace_id": "4202", "simple": { "limit": 30, "period": 60 } }, | |
| 52 | + | { "name": "WEB_SESSION_LIMIT", "namespace_id": "4203", "simple": { "limit": 1200, "period": 60 } }, | |
| 53 | + | { "name": "WEB_ADDRESS_LIMIT", "namespace_id": "4204", "simple": { "limit": 3000, "period": 60 } }, | |
| 54 | + | { "name": "GIT_ANONYMOUS_LIMIT", "namespace_id": "4205", "simple": { "limit": 120, "period": 60 } }, | |
| 55 | + | { "name": "GIT_SIGNED_LIMIT", "namespace_id": "4206", "simple": { "limit": 1200, "period": 60 } } | |
| 56 | + | ], | |
| 46 | 57 | "observability": { "enabled": true }, | |
| 47 | 58 | "upload_source_maps": true | |
| 48 | 59 | } |
| 61 | 61 | } | |
| 62 | 62 | ||
| 63 | 63 | pub mod d1; | |
| 64 | + | pub mod limits; | |
| 64 | 65 | pub mod wire; | |
| 65 | 66 | ||
| 66 | 67 | /// Helpers for bindings that workers-rs has no typed wrapper for, such as |
| 1 | + | //! Rate limits: Workers Rate Limiting bindings, asked once per request with | |
| 2 | + | //! a key (a client's address, a repository's id, a hash of a token). | |
| 3 | + | //! | |
| 4 | + | //! Every binding, its namespace id and its limit is listed in `RATE_LIMITS` | |
| 5 | + | //! (packages/contracts/src/rate-limits.ts), which apps/web's tests check | |
| 6 | + | //! each wrangler.jsonc against; docs/RATE-LIMITS.md says why each is what | |
| 7 | + | //! it is. | |
| 8 | + | //! | |
| 9 | + | //! A limit fails open: a binding that is not there (self-hosted) or that | |
| 10 | + | //! fails lets the request through. A limit guards against floods; it is | |
| 11 | + | //! never a reason for g1t to stop answering. | |
| 12 | + | ||
| 13 | + | use worker::Env; | |
| 14 | + | ||
| 15 | + | /// Every limit's window, in seconds, and how long a client past one is | |
| 16 | + | /// told to wait (`Retry-After`). Workers Rate Limiting counts over 10 or 60. | |
| 17 | + | pub const PERIOD_SECONDS: u32 = 60; | |
| 18 | + | ||
| 19 | + | /// What asking a limit said. Only `Limited` refuses the request. | |
| 20 | + | #[derive(Clone, Copy, Debug, PartialEq, Eq)] | |
| 21 | + | pub enum Verdict { | |
| 22 | + | Allowed, | |
| 23 | + | Limited, | |
| 24 | + | /// No binding, or it failed: let through. | |
| 25 | + | Unavailable, | |
| 26 | + | } | |
| 27 | + | ||
| 28 | + | impl Verdict { | |
| 29 | + | pub fn limited(self) -> bool { | |
| 30 | + | self == Verdict::Limited | |
| 31 | + | } | |
| 32 | + | } | |
| 33 | + | ||
| 34 | + | /// The verdict from what the binding answered: `None` when there is no | |
| 35 | + | /// binding, `Some(Err)` when asking it failed, else whether it let the | |
| 36 | + | /// request through. | |
| 37 | + | pub fn verdict<E>(answered: Option<std::result::Result<bool, E>>) -> Verdict { | |
| 38 | + | match answered { | |
| 39 | + | Some(Ok(true)) => Verdict::Allowed, | |
| 40 | + | Some(Ok(false)) => Verdict::Limited, | |
| 41 | + | Some(Err(_)) | None => Verdict::Unavailable, | |
| 42 | + | } | |
| 43 | + | } | |
| 44 | + | ||
| 45 | + | /// Counts one request against the binding named `binding` under `key`. | |
| 46 | + | /// Never fails; a failure to ask is logged and lets the request through. | |
| 47 | + | pub async fn check(env: &Env, binding: &str, key: String) -> Verdict { | |
| 48 | + | let Ok(limiter) = env.rate_limiter(binding) else { | |
| 49 | + | return Verdict::Unavailable; | |
| 50 | + | }; | |
| 51 | + | let answered = limiter.limit(key).await.map(|outcome| outcome.success); | |
| 52 | + | if let Err(problem) = &answered { | |
| 53 | + | worker::console_error!("rate limit {binding} could not be asked: {problem}"); | |
| 54 | + | } | |
| 55 | + | verdict(Some(answered)) | |
| 56 | + | } | |
| 57 | + | ||
| 58 | + | /// A client's address for a key: `ip:` and `CF-Connecting-IP`, or | |
| 59 | + | /// `ip:unknown` when there is none (local development). | |
| 60 | + | pub fn address_key(address: Option<&str>) -> String { | |
| 61 | + | format!("ip:{}", address.map(str::trim).filter(|a| !a.is_empty()).unwrap_or("unknown")) | |
| 62 | + | } | |
| 63 | + | ||
| 64 | + | #[cfg(test)] | |
| 65 | + | mod tests { | |
| 66 | + | use super::*; | |
| 67 | + | ||
| 68 | + | #[test] | |
| 69 | + | fn only_a_limit_the_binding_says_is_reached_refuses() { | |
| 70 | + | assert_eq!(verdict::<()>(Some(Ok(true))), Verdict::Allowed); | |
| 71 | + | assert_eq!(verdict::<()>(Some(Ok(false))), Verdict::Limited); | |
| 72 | + | assert!(verdict::<()>(Some(Ok(false))).limited()); | |
| 73 | + | } | |
| 74 | + | ||
| 75 | + | #[test] | |
| 76 | + | fn no_binding_or_a_failing_one_lets_requests_through() { | |
| 77 | + | assert_eq!(verdict::<()>(None), Verdict::Unavailable); | |
| 78 | + | assert_eq!(verdict(Some(Err("binding threw"))), Verdict::Unavailable); | |
| 79 | + | assert!(!verdict::<()>(None).limited()); | |
| 80 | + | assert!(!verdict(Some(Err("binding threw"))).limited()); | |
| 81 | + | } | |
| 82 | + | ||
| 83 | + | #[test] | |
| 84 | + | fn addresses_are_keyed_as_given_or_unknown() { | |
| 85 | + | assert_eq!(address_key(Some(" 203.0.113.9 ")), "ip:203.0.113.9"); | |
| 86 | + | assert_eq!(address_key(Some("")), "ip:unknown"); | |
| 87 | + | assert_eq!(address_key(None), "ip:unknown"); | |
| 88 | + | } | |
| 89 | + | } |
| 1 | + | # Rate limits and log sampling | |
| 2 | + | ||
| 3 | + | Internal notes. The public limits are on the docs' rate limits page | |
| 4 | + | (`apps/docs/src/content/docs/reference/rate-limits.md`); change both, and | |
| 5 | + | `RATE_LIMITS`, together. | |
| 6 | + | ||
| 7 | + | ## Rate limits | |
| 8 | + | ||
| 9 | + | Every public surface that costs money or sends something is behind a | |
| 10 | + | [Workers Rate Limiting](https://developers.cloudflare.com/workers/runtime-apis/bindings/rate-limit/) | |
| 11 | + | binding (`ratelimits` in its wrangler.jsonc). The table of record is | |
| 12 | + | `RATE_LIMITS` in `packages/contracts/src/rate-limits.ts`; | |
| 13 | + | `apps/web/app/lib/front-door-limits.test.ts` fails when a wrangler.jsonc | |
| 14 | + | and the table disagree, or a namespace id is used twice. | |
| 15 | + | ||
| 16 | + | Rules every limit follows: | |
| 17 | + | ||
| 18 | + | - **Fails open.** No binding (self-hosted: `deploy/self-host/configs.mjs` | |
| 19 | + | does not copy `ratelimits`) or a binding that throws lets the request | |
| 20 | + | through, logged as `rate_limit.unavailable` (TS) or | |
| 21 | + | `rate limit … could not be asked` (Rust). Helpers: `checkLimit` in | |
| 22 | + | packages/contracts, `g1t_kit::limits::check`. | |
| 23 | + | - **Keys never hold a secret.** Addresses are `ip:<CF-Connecting-IP>`; | |
| 24 | + | sessions, tokens, git credentials and email addresses are the first 16 hex | |
| 25 | + | digits of their SHA-256. | |
| 26 | + | - **60-second window.** The binding only counts over 10 or 60 seconds, so | |
| 27 | + | "per hour" limits are not possible here; D1 throttles cover those | |
| 28 | + | (identity's `throttle.rs`, the waitlist's `rate_limits`, status's resend | |
| 29 | + | window). | |
| 30 | + | - **429 with `Retry-After: 60`.** JSON `{"error": {"code": "rate_limited", …}}` | |
| 31 | + | on the API and MCP; plain text for git (git prints it) and pages. | |
| 32 | + | - Counts are per Cloudflare location and approximate: a limit is a guard | |
| 33 | + | against floods, not a quota. | |
| 34 | + | ||
| 35 | + | ### Namespace ids | |
| 36 | + | ||
| 37 | + | Unique per Cloudflare account, in blocks of a hundred per Worker. Take the | |
| 38 | + | next free id in the Worker's block; a new Worker takes the next block. | |
| 39 | + | ||
| 40 | + | | Block | Worker | | |
| 41 | + | | --- | --- | | |
| 42 | + | | 41xx | services/packages | | |
| 43 | + | | 42xx | apps/web | | |
| 44 | + | | 43xx | services/repos | | |
| 45 | + | | 44xx | apps/api | | |
| 46 | + | | 45xx | services/og | | |
| 47 | + | | 46xx | apps/status | | |
| 48 | + | ||
| 49 | + | ### Bindings | |
| 50 | + | ||
| 51 | + | | Binding | Id | Limit / 60 s | Key | Where enforced | | |
| 52 | + | | --- | --- | --- | --- | --- | | |
| 53 | + | | `ANONYMOUS_LIMIT` | 4101 | 300 | address | packages `src/limits.rs`: registry pulls and token requests | | |
| 54 | + | | `SIGNED_LIMIT` | 4102 | 5,000 | person, workspace or agent | packages, the same | | |
| 55 | + | | `WEB_ANONYMOUS_LIMIT` | 4201 | 600 | address | web `app/lib/front-door-limits.ts`: signed-out pages and data | | |
| 56 | + | | `WEB_HEAVY_LIMIT` | 4202 | 30 | address | web: signed-out archives, run pages, logs, artifacts, search | | |
| 57 | + | | `WEB_SESSION_LIMIT` | 4203 | 1,200 | session cookie hash | web: requests with a session cookie (not checked there) | | |
| 58 | + | | `WEB_ADDRESS_LIMIT` | 4204 | 3,000 | address | web: every request that reaches the Worker; stops made-up cookies getting round the signed-out limit | | |
| 59 | + | | `GIT_ANONYMOUS_LIMIT` | 4205 | 120 | address | web: git smart HTTP without `Authorization` (~40 clones) | | |
| 60 | + | | `GIT_SIGNED_LIMIT` | 4206 | 1,200 | `Authorization` hash | web: git smart HTTP with credentials, sandboxes' included | | |
| 61 | + | | `PACK_FILL_LIMIT` | 4301 | 30 | repository id | repos `src/limits.rs`: packs written to `GIT_PACKS`; past it the pack is streamed, not kept | | |
| 62 | + | | `ANONYMOUS_FETCH_LIMIT` | 4302 | 120 | repository id | repos: anonymous fetches the git store answers (cache hits never count) | | |
| 63 | + | | `API_ANONYMOUS_LIMIT` | 4401 | 60 | `rest:`/`mcp:` + address | api `src/limits.rs`: no token, or a wrong one | | |
| 64 | + | | `API_TOKEN_LIMIT` | 4402 | 1,000 | `rest:`/`mcp:` + token hash | api: with a bearer token | | |
| 65 | + | | `OG_RENDER_LIMIT` | 4501 | 60 | address | og `src/index.ts`: cache misses only; past it, the brand card (brief cache) | | |
| 66 | + | | `STATUS_SUBSCRIBE_LIMIT` | 4601 | 3 | address | status `src/index.ts`: `POST /subscribe` | | |
| 67 | + | | `STATUS_EMAIL_LIMIT` | 4602 | 2 | email hash | status: the same, per address asked for (the D1 resend window also allows one email per 10 minutes) | | |
| 68 | + | ||
| 69 | + | What is deliberately not limited: | |
| 70 | + | ||
| 71 | + | - Static assets (served by the assets binding before the Worker), avatars, | |
| 72 | + | `go get` answers and the docs redirect on the front door. | |
| 73 | + | - Packages through the front door: the packages service limits itself. | |
| 74 | + | - On the API: Stripe and connection webhooks, sandbox and runner reports | |
| 75 | + | (job, run, check, plan, review, backup and queue tokens; `REPORTS` in | |
| 76 | + | `apps/api/src/limits.rs`), the Actions toolkit, OIDC and deployment | |
| 77 | + | build reports. Many sandboxes share an egress address. | |
| 78 | + | - Status has no Turnstile: nothing in g1t uses Turnstile yet. The honeypot, | |
| 79 | + | both limits and the resend window are the guard. | |
| 80 | + | ||
| 81 | + | Why anonymous git matters: each clone or fetch the store answers is an | |
| 82 | + | operation billed to the repository's workspace (`services/repos/src/git_ops.rs`) | |
| 83 | + | and a miss writes a pack to R2 (`pack_cache.rs`). The address limit stops | |
| 84 | + | one client; `ANONYMOUS_FETCH_LIMIT` bounds many addresses against one | |
| 85 | + | repository (at most 120 × 60 × 24 ≈ 173k operations a day, about $26 at | |
| 86 | + | cost), and `PACK_FILL_LIMIT` bounds R2 writes. | |
| 87 | + | ||
| 88 | + | ## Log sampling | |
| 89 | + | ||
| 90 | + | Workers Logs keeps `head_sampling_rate` of invocations (`observability` | |
| 91 | + | in each wrangler.jsonc). The decision is made when the request starts, so | |
| 92 | + | an unsampled request's `console.error` and uncaught exception are dropped | |
| 93 | + | with the rest of its logs. Workers metrics (requests, errors, CPU) in the | |
| 94 | + | dashboard are not sampled; only the log lines are. | |
| 95 | + | ||
| 96 | + | | Rate | Workers | Why | | |
| 97 | + | | --- | --- | --- | | |
| 98 | + | | 1 | billing, identity, runner, actions, deployments, status, sudo | Money, sign-in or a run someone will ask about, or too few requests to matter. Every error is kept. | | |
| 99 | + | | 0.1 | web, api, pages, repos, og, models, search, context, events, work, webhooks, integrations, packages, projects, security, docs | Every page view, git request or event; a tenth is enough to see a pattern. | | |
| 100 | + | ||
| 101 | + | The tradeoff: on a 0.1 Worker, a one-off error has a 90% chance of leaving | |
| 102 | + | no log line. Error counts in the dashboard still show it, and a recurring | |
| 103 | + | one shows up within a few occurrences. To chase a rare error on one of | |
| 104 | + | those Workers, raise its rate to 1 for the investigation and lower it after. |
| 4 | 4 | "private": true, | |
| 5 | 5 | "type": "module", | |
| 6 | 6 | "license": "MIT", | |
| 7 | − | "exports": { ".": "./src/index.ts", "./og": "./src/og.ts", "./status": "./src/status.ts", "./scopes": "./src/scopes.ts", "./fine-grained": "./src/fine-grained.ts" }, | |
| 7 | + | "exports": { ".": "./src/index.ts", "./og": "./src/og.ts", "./status": "./src/status.ts", "./scopes": "./src/scopes.ts", "./fine-grained": "./src/fine-grained.ts", "./rate-limits": "./src/rate-limits.ts" }, | |
| 8 | 8 | "scripts": { "typecheck": "tsc -p tsconfig.json" } | |
| 9 | 9 | } |
| 30 | 30 | export * from "./og"; | |
| 31 | 31 | export * from "./packages"; | |
| 32 | 32 | export * from "./projects"; | |
| 33 | + | export * from "./rate-limits"; | |
| 33 | 34 | export * from "./repos"; | |
| 34 | 35 | export * from "./result"; | |
| 35 | 36 | export * from "./rules"; |
| 1 | + | /** | |
| 2 | + | * Rate limits on g1t's public surfaces: Workers Rate Limiting bindings | |
| 3 | + | * (`ratelimits` in each wrangler.jsonc), asked once per request with a key: | |
| 4 | + | * the client's address, or a hash of its session or token, never the | |
| 5 | + | * secret itself. | |
| 6 | + | * | |
| 7 | + | * Every binding is listed in `RATE_LIMITS` with its namespace id and limit, | |
| 8 | + | * and apps/web's tests check each wrangler.jsonc against it, so this table | |
| 9 | + | * is the one place to read or change them. docs/RATE-LIMITS.md explains | |
| 10 | + | * the choices; the docs' rate limits page shows the public ones. | |
| 11 | + | * | |
| 12 | + | * A limit fails open: no binding (self-hosted, `wrangler dev` without one) | |
| 13 | + | * or a binding that throws lets the request through. A limit guards | |
| 14 | + | * against floods; it is never a reason for g1t to stop answering. | |
| 15 | + | */ | |
| 16 | + | ||
| 17 | + | /** A Workers Rate Limiting binding, as the runtime hands it over. */ | |
| 18 | + | export type RateLimitBinding = { limit(options: { key: string }): Promise<{ success: boolean }> }; | |
| 19 | + | ||
| 20 | + | /** What asking a limit said. `unavailable` lets the request through. */ | |
| 21 | + | export type LimitVerdict = "allowed" | "limited" | "unavailable"; | |
| 22 | + | ||
| 23 | + | /** | |
| 24 | + | * Every limit's window, in seconds. Workers Rate Limiting counts over 10 | |
| 25 | + | * or 60 seconds; every g1t limit uses 60, and a client past one is told to | |
| 26 | + | * wait that long. | |
| 27 | + | */ | |
| 28 | + | export const LIMIT_PERIOD_SECONDS = 60; | |
| 29 | + | ||
| 30 | + | export type RateLimitSpec = { | |
| 31 | + | /** The Worker whose wrangler.jsonc declares it. */ | |
| 32 | + | worker: string; | |
| 33 | + | /** Unique per Cloudflare account; allocated in blocks per Worker (docs/RATE-LIMITS.md). */ | |
| 34 | + | namespaceId: number; | |
| 35 | + | /** Requests per `LIMIT_PERIOD_SECONDS` per key. */ | |
| 36 | + | limit: number; | |
| 37 | + | /** What a key is. */ | |
| 38 | + | per: string; | |
| 39 | + | }; | |
| 40 | + | ||
| 41 | + | /** | |
| 42 | + | * Every rate limit binding, by name. Generous for people, tight for floods: | |
| 43 | + | * a person browsing makes a few requests a second at most, a clone is three | |
| 44 | + | * git requests, and nothing a person does needs dozens of archives a minute. | |
| 45 | + | * Namespace ids go in blocks of a hundred per Worker: 41xx packages, | |
| 46 | + | * 42xx web, 43xx repos, 44xx api, 45xx og, 46xx status. | |
| 47 | + | */ | |
| 48 | + | export const RATE_LIMITS = { | |
| 49 | + | // services/packages (src/limits.rs): registry pulls and token requests. | |
| 50 | + | ANONYMOUS_LIMIT: { worker: "services/packages", namespaceId: 4101, limit: 300, per: "address" }, | |
| 51 | + | SIGNED_LIMIT: { worker: "services/packages", namespaceId: 4102, limit: 5000, per: "person, workspace or agent" }, | |
| 52 | + | // apps/web (workers/app.ts): the front door. | |
| 53 | + | WEB_ANONYMOUS_LIMIT: { worker: "apps/web", namespaceId: 4201, limit: 600, per: "address, signed-out pages" }, | |
| 54 | + | WEB_HEAVY_LIMIT: { worker: "apps/web", namespaceId: 4202, limit: 30, per: "address, signed-out archives, run pages, logs and search" }, | |
| 55 | + | WEB_SESSION_LIMIT: { worker: "apps/web", namespaceId: 4203, limit: 1200, per: "session, signed-in requests" }, | |
| 56 | + | WEB_ADDRESS_LIMIT: { worker: "apps/web", namespaceId: 4204, limit: 3000, per: "address, every request that reaches the Worker" }, | |
| 57 | + | GIT_ANONYMOUS_LIMIT: { worker: "apps/web", namespaceId: 4205, limit: 120, per: "address, git requests without credentials" }, | |
| 58 | + | GIT_SIGNED_LIMIT: { worker: "apps/web", namespaceId: 4206, limit: 1200, per: "credential, git requests with credentials" }, | |
| 59 | + | // services/repos (src/lib.rs): what an anonymous clone can cost a repository's owner. | |
| 60 | + | PACK_FILL_LIMIT: { worker: "services/repos", namespaceId: 4301, limit: 30, per: "repository, packs written to the pack cache" }, | |
| 61 | + | ANONYMOUS_FETCH_LIMIT: { worker: "services/repos", namespaceId: 4302, limit: 120, per: "repository, anonymous fetches the store answers" }, | |
| 62 | + | // apps/api (src/limits.rs): REST and MCP, which count apart. | |
| 63 | + | API_ANONYMOUS_LIMIT: { worker: "apps/api", namespaceId: 4401, limit: 60, per: "address, requests without a token" }, | |
| 64 | + | API_TOKEN_LIMIT: { worker: "apps/api", namespaceId: 4402, limit: 1000, per: "token" }, | |
| 65 | + | // services/og: drawing a card that is not in the cache. | |
| 66 | + | OG_RENDER_LIMIT: { worker: "services/og", namespaceId: 4501, limit: 60, per: "address, cards drawn" }, | |
| 67 | + | // apps/status: asking for a subscription, which sends an email. | |
| 68 | + | STATUS_SUBSCRIBE_LIMIT: { worker: "apps/status", namespaceId: 4601, limit: 3, per: "address" }, | |
| 69 | + | STATUS_EMAIL_LIMIT: { worker: "apps/status", namespaceId: 4602, limit: 2, per: "email address" }, | |
| 70 | + | } as const satisfies Record<string, RateLimitSpec>; | |
| 71 | + | ||
| 72 | + | export type RateLimitName = keyof typeof RATE_LIMITS; | |
| 73 | + | ||
| 74 | + | /** | |
| 75 | + | * Counts one request against `binding` under `key`. Never throws: a missing | |
| 76 | + | * binding or one that fails is `unavailable`, which callers let through. | |
| 77 | + | */ | |
| 78 | + | export async function checkLimit(binding: RateLimitBinding | undefined, key: string): Promise<LimitVerdict> { | |
| 79 | + | if (!binding) return "unavailable"; | |
| 80 | + | try { | |
| 81 | + | const { success } = await binding.limit({ key }); | |
| 82 | + | return success ? "allowed" : "limited"; | |
| 83 | + | } catch (error) { | |
| 84 | + | console.error(JSON.stringify({ event: "rate_limit.unavailable", error: String(error) })); | |
| 85 | + | return "unavailable"; | |
| 86 | + | } | |
| 87 | + | } | |
| 88 | + | ||
| 89 | + | /** Whether the request is past its limit. Fails open, like `checkLimit`. */ | |
| 90 | + | export async function isLimited(binding: RateLimitBinding | undefined, key: string): Promise<boolean> { | |
| 91 | + | return (await checkLimit(binding, key)) === "limited"; | |
| 92 | + | } | |
| 93 | + | ||
| 94 | + | /** The client's address as Cloudflare saw it, for keys; `unknown` when there is none (local dev). */ | |
| 95 | + | export function clientAddress(request: Request): string { | |
| 96 | + | return request.headers.get("cf-connecting-ip")?.trim() || "unknown"; | |
| 97 | + | } | |
| 98 | + | ||
| 99 | + | /** | |
| 100 | + | * A key for a secret (a session cookie, a token, a credential): `prefix:` | |
| 101 | + | * and the first 16 hex digits of its SHA-256, so the secret itself never | |
| 102 | + | * reaches the rate limiter. | |
| 103 | + | */ | |
| 104 | + | export async function secretKey(prefix: string, secret: string): Promise<string> { | |
| 105 | + | const digest = new Uint8Array(await crypto.subtle.digest("SHA-256", new TextEncoder().encode(secret))); | |
| 106 | + | let hex = ""; | |
| 107 | + | for (const byte of digest.subarray(0, 8)) hex += byte.toString(16).padStart(2, "0"); | |
| 108 | + | return `${prefix}:${hex}`; | |
| 109 | + | } | |
| 110 | + | ||
| 111 | + | /** A 429 with `Retry-After`, as plain text unless `headers` says otherwise. */ | |
| 112 | + | export function tooManyRequests(body: string, headers: Record<string, string> = {}): Response { | |
| 113 | + | return new Response(body, { | |
| 114 | + | status: 429, | |
| 115 | + | headers: { | |
| 116 | + | "content-type": "text/plain; charset=utf-8", | |
| 117 | + | "retry-after": String(LIMIT_PERIOD_SECONDS), | |
| 118 | + | "cache-control": "no-store", | |
| 119 | + | ...headers, | |
| 120 | + | }, | |
| 121 | + | }); | |
| 122 | + | } |
| 18 | 18 | */ | |
| 19 | 19 | import { WorkerEntrypoint } from "cloudflare:workers"; | |
| 20 | 20 | import { type ServiceBinding, identityClient, projectsClient, reposClient, workClient } from "@g1t/contracts"; | |
| 21 | + | import { type RateLimitBinding, clientAddress, isLimited } from "@g1t/contracts/rate-limits"; | |
| 21 | 22 | import type { Font } from "satori/standalone"; | |
| 22 | 23 | import resvgWasm from "@resvg/resvg-wasm/index_bg.wasm"; | |
| 23 | 24 | import yogaWasm from "satori/yoga.wasm"; | |
| ⋯ | |||
| 47 | 48 | BROWSER: Fetcher; | |
| 48 | 49 | /** Production screenshots, by app hostname. */ | |
| 49 | 50 | SCREENSHOTS: R2Bucket; | |
| 51 | + | /** Cards drawn per address (RATE_LIMITS in packages/contracts). Absent: no limit. */ | |
| 52 | + | OG_RENDER_LIMIT?: RateLimitBinding; | |
| 50 | 53 | } | |
| 51 | 54 | ||
| 52 | 55 | /** | |
| ⋯ | |||
| 122 | 125 | const hit = await cache.match(key); | |
| 123 | 126 | if (hit) return hit; | |
| 124 | 127 | ||
| 125 | − | const card = await cardFor(url, env); | |
| 126 | − | const failed = card.kind === "brand" && card.failed === true; | |
| 127 | 128 | // Every page without a card of its own shares the one brand card, so | |
| 128 | 129 | // made-up paths cannot make the service draw it again and again. | |
| 129 | 130 | const brandKey = cacheKey(new URL("/", url)); | |
| 131 | + | // A miss looks the page up and draws its card: limited per address. | |
| 132 | + | // Past the limit, the brand card, kept only briefly so the page's own | |
| 133 | + | // card is asked for again later. | |
| 134 | + | if (key !== brandKey && (await isLimited(env.OG_RENDER_LIMIT, `ip:${clientAddress(request)}`))) { | |
| 135 | + | const brand = await cache.match(brandKey); | |
| 136 | + | if (brand) return withCacheControl(brand, true); | |
| 137 | + | const drawn = await brandCard(ctx, cache, brandKey); | |
| 138 | + | return drawn.ok ? withCacheControl(drawn, true) : drawn; | |
| 139 | + | } | |
| 140 | + | ||
| 141 | + | const card = await cardFor(url, env); | |
| 142 | + | const failed = card.kind === "brand" && card.failed === true; | |
| 130 | 143 | if (card.kind === "brand" && key !== brandKey) { | |
| 131 | 144 | const brand = await cache.match(brandKey); | |
| 132 | 145 | if (brand) return withCacheControl(brand, failed); | |
| ⋯ | |||
| 180 | 193 | return brief; | |
| 181 | 194 | } | |
| 182 | 195 | ||
| 196 | + | /** The brand card when the cache has lost it: drawn once and kept again. */ | |
| 197 | + | async function brandCard(ctx: ExecutionContext, cache: Cache, brandKey: string): Promise<Response> { | |
| 198 | + | try { | |
| 199 | + | const response = new Response(await cardPng(BRAND, ASSETS), { | |
| 200 | + | headers: { | |
| 201 | + | "content-type": "image/png", | |
| 202 | + | "cache-control": CACHE_CONTROL, | |
| 203 | + | "access-control-allow-origin": "*", | |
| 204 | + | "x-content-type-options": "nosniff", | |
| 205 | + | }, | |
| 206 | + | }); | |
| 207 | + | ctx.waitUntil(cache.put(brandKey, response.clone())); | |
| 208 | + | return response; | |
| 209 | + | } catch (error) { | |
| 210 | + | console.error("og: the brand card could not be drawn", error); | |
| 211 | + | return new Response("The card could not be drawn", { status: 503, headers: { "cache-control": NO_STORE } }); | |
| 212 | + | } | |
| 213 | + | } | |
| 214 | + | ||
| 183 | 215 | /** What satori can draw: PNG and JPEG. A WebP or GIF icon is left off the card. */ | |
| 184 | 216 | const DRAWABLE = new Set(["image/png", "image/jpeg"]); | |
| 185 | 217 | ||
| 33 | 33 | { "binding": "WORK", "service": "g1t-work" }, | |
| 34 | 34 | { "binding": "PROJECTS", "service": "g1t-projects" } | |
| 35 | 35 | ], | |
| 36 | + | // Cards drawn for a cache miss, per address (src/index.ts). Past it, | |
| 37 | + | // the brand card. Id and limit: RATE_LIMITS in packages/contracts. | |
| 38 | + | "ratelimits": [{ "name": "OG_RENDER_LIMIT", "namespace_id": "4501", "simple": { "limit": 60, "period": 60 } }], | |
| 36 | 39 | "observability": { "enabled": true } | |
| 37 | 40 | } |
| 24 | 24 | mod last_commits; | |
| 25 | 25 | mod license; | |
| 26 | 26 | mod lifecycle; | |
| 27 | + | mod limits; | |
| 27 | 28 | mod listing; | |
| 28 | 29 | mod meters; | |
| 29 | 30 | mod mirror; | |
| ⋯ | |||
| 1928 | 1929 | if kept_key.is_some() { | |
| 1929 | 1930 | timing.note("refs", "miss"); | |
| 1930 | 1931 | } | |
| 1932 | + | // An anonymous fetch the store is about to answer: an operation for | |
| 1933 | + | // the repository's workspace, so limited per repository (limits.rs). | |
| 1934 | + | if limits::counts_as_anonymous_fetch(call, viewer.is_none()) | |
| 1935 | + | && g1t_kit::limits::check(env, limits::ANONYMOUS_FETCH, repo.id.clone()).await.limited() | |
| 1936 | + | { | |
| 1937 | + | let response = limits::too_many_anonymous_fetches(&format!("{}/{}", repo.namespace, repo.name))?; | |
| 1938 | + | after.ended(429, Some("Too many anonymous fetches.".to_owned())); | |
| 1939 | + | after.spawn(env, ctx); | |
| 1940 | + | return Ok(response); | |
| 1941 | + | } | |
| 1931 | 1942 | // The store's credential: one made a moment ago, here or in another | |
| 1932 | 1943 | // isolate (see store.rs), or a new one. | |
| 1933 | 1944 | let access = match kept_access { | |
| ⋯ | |||
| 2055 | 2066 | // A fresh clone the bucket did not have: counted, and its pack | |
| 2056 | 2067 | // kept as it streams to git, when it is a whole one. | |
| 2057 | 2068 | meters::record(pack_cache::MISS, &key, forwarded.sent, 0); | |
| 2058 | − | if status == 200 { | |
| 2069 | + | // Past the repository's limit on writes to the bucket, the pack | |
| 2070 | + | // goes to git without being kept (limits.rs). | |
| 2071 | + | if status == 200 && !g1t_kit::limits::check(env, limits::PACK_FILL, repo.id.clone()).await.limited() { | |
| 2059 | 2072 | let store_key = key.clone(); | |
| 2060 | 2073 | let measured = Box::new(move |bytes: u64| meters::record_bytes(pack_cache::MISS, &store_key, 0, bytes)); | |
| 2061 | 2074 | let (teed, filling) = pack_cache::tee(response, packs.clone(), pack_key, measured)?; | |
| 1 | + | //! What an anonymous clone can cost a repository's owner. | |
| 2 | + | //! | |
| 3 | + | //! Anyone may clone a public repository, and each clone or fetch the git | |
| 4 | + | //! store answers is an operation counted for the repository's workspace | |
| 5 | + | //! (git_ops.rs). The site limits git requests per address (apps/web's | |
| 6 | + | //! front-door-limits.ts); here, per repository: | |
| 7 | + | //! | |
| 8 | + | //! - ANONYMOUS_FETCH_LIMIT: anonymous fetches that reach the store. Clones | |
| 9 | + | //! answered from the pack cache or kept refs never count, so the same | |
| 10 | + | //! commit cloned by many costs nothing; a flood of different requests | |
| 11 | + | //! from many addresses is answered 429 once past it. Signed-in clones | |
| 12 | + | //! are never limited here. | |
| 13 | + | //! - PACK_FILL_LIMIT: packs written to the pack cache (pack_cache.rs). Past | |
| 14 | + | //! it the pack still goes to git, only not kept, so a flood of distinct | |
| 15 | + | //! clones cannot turn into a flood of writes to the bucket. | |
| 16 | + | //! | |
| 17 | + | //! Both are Workers Rate Limiting bindings keyed by the repository's id; | |
| 18 | + | //! their limits are in `RATE_LIMITS` (packages/contracts). Without them | |
| 19 | + | //! (self-hosted) nothing is limited, and one that fails lets the request | |
| 20 | + | //! through (g1t_kit::limits). | |
| 21 | + | ||
| 22 | + | use g1t_kit::limits::PERIOD_SECONDS; | |
| 23 | + | use worker::{Response, Result}; | |
| 24 | + | ||
| 25 | + | use crate::git_ops::GitCall; | |
| 26 | + | ||
| 27 | + | pub const ANONYMOUS_FETCH: &str = "ANONYMOUS_FETCH_LIMIT"; | |
| 28 | + | pub const PACK_FILL: &str = "PACK_FILL_LIMIT"; | |
| 29 | + | ||
| 30 | + | /// Whether a request counts against ANONYMOUS_FETCH_LIMIT: a fetch of | |
| 31 | + | /// objects, by no one, that the store is about to be asked for. | |
| 32 | + | pub fn counts_as_anonymous_fetch(call: GitCall, anonymous: bool) -> bool { | |
| 33 | + | anonymous && call == GitCall::Fetch | |
| 34 | + | } | |
| 35 | + | ||
| 36 | + | /// What git is told past the limit: a plain-text answer, which git shows. | |
| 37 | + | pub fn too_many_anonymous_fetches(path: &str) -> Result<Response> { | |
| 38 | + | let message = format!( | |
| 39 | + | "Too many anonymous clones of {path} right now. Wait a minute and try again, or clone with credentials: https://docs.g1t.sh/reference/rate-limits/\n" | |
| 40 | + | ); | |
| 41 | + | let response = Response::error(message, 429)?; | |
| 42 | + | response.headers().set("retry-after", &PERIOD_SECONDS.to_string())?; | |
| 43 | + | Ok(response) | |
| 44 | + | } | |
| 45 | + | ||
| 46 | + | #[cfg(test)] | |
| 47 | + | mod tests { | |
| 48 | + | use super::*; | |
| 49 | + | ||
| 50 | + | #[test] | |
| 51 | + | fn only_anonymous_fetches_of_objects_count() { | |
| 52 | + | assert!(counts_as_anonymous_fetch(GitCall::Fetch, true)); | |
| 53 | + | assert!(!counts_as_anonymous_fetch(GitCall::Fetch, false), "signed-in clones are not limited here"); | |
| 54 | + | assert!(!counts_as_anonymous_fetch(GitCall::RefAdvertisement, true)); | |
| 55 | + | assert!(!counts_as_anonymous_fetch(GitCall::LsRefs, true)); | |
| 56 | + | assert!(!counts_as_anonymous_fetch(GitCall::ReceivePack, true)); | |
| 57 | + | } | |
| 58 | + | } |
| 124 | 124 | "queues": { | |
| 125 | 125 | "consumers": [{ "queue": "g1t-events-repos", "max_batch_size": 20, "max_batch_timeout": 1 }] | |
| 126 | 126 | }, | |
| 127 | + | // What an anonymous clone can cost a repository's owner (src/limits.rs): | |
| 128 | + | // packs written to GIT_PACKS, and anonymous fetches the git store | |
| 129 | + | // answers, each per repository. Ids and limits: RATE_LIMITS in | |
| 130 | + | // packages/contracts. Without them, nothing is limited. | |
| 131 | + | "ratelimits": [ | |
| 132 | + | { "name": "PACK_FILL_LIMIT", "namespace_id": "4301", "simple": { "limit": 30, "period": 60 } }, | |
| 133 | + | { "name": "ANONYMOUS_FETCH_LIMIT", "namespace_id": "4302", "simple": { "limit": 120, "period": 60 } } | |
| 134 | + | ], | |
| 127 | 135 | "observability": { "enabled": true } | |
| 128 | 136 | } |