Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| The docs folder is gone, and what it held lives where people read it: how a self-hosted g1t runs and how to deploy g1t to Cloudflare are pages on docs.g1t.sh under Run g1t yourself, and speed, rate limits and operating g1t.sh are sections of CONTRIBUTING.md; code that cited a file in docs/ now points to the page or section that covers it, or says what it means itself, and applied migrations and the runner images are left as they were. | 1 | //! Datasets: the safe query layer behind dashboards (Artifacts mode). |
| 2 | //! Mirrors `packages/contracts/src/datasets.ts`; the tests here keep the | |
| 3 | //! catalog the same and run both validators over `datasets.fixtures.json`. | |
| Artifacts contracts: folios, their four kinds, dashboard datasets, folio events and the artifacts scopes are typed and validated the same in TypeScript and Rust, with nothing using them yet | 4 | //! |
| 5 | //! Not SQL: a query names a dataset from a declared catalog, one measure, | |
| 6 | //! at most one dimension, an interval, filters on declared fields and a | |
| 7 | //! range. The service that owns the data ([`DatasetSpec::service`]) answers | |
| 8 | //! `query_dataset` for the viewer, over only what the viewer can read, with | |
| 9 | //! a fixed query per measure and dimension, and caps the rows. The Rust | |
| 10 | //! owners (work, actions, billing) validate with [`DatasetQuery::validate`] | |
| 11 | //! before they run anything. | |
| 12 | ||
| 13 | use serde::{Deserialize, Serialize}; | |
| 14 | ||
| 15 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)] | |
| 16 | #[serde(rename_all = "snake_case")] | |
| 17 | pub enum DatasetId { | |
| 18 | Issues, | |
| 19 | PullRequests, | |
| 20 | WorkflowRuns, | |
| 21 | Deployments, | |
| 22 | Spend, | |
| 23 | AgentSessions, | |
| 24 | } | |
| 25 | ||
| 26 | impl DatasetId { | |
| 27 | pub const ALL: [DatasetId; 6] = [ | |
| 28 | DatasetId::Issues, | |
| 29 | DatasetId::PullRequests, | |
| 30 | DatasetId::WorkflowRuns, | |
| 31 | DatasetId::Deployments, | |
| 32 | DatasetId::Spend, | |
| 33 | DatasetId::AgentSessions, | |
| 34 | ]; | |
| 35 | ||
| 36 | pub fn as_str(self) -> &'static str { | |
| 37 | match self { | |
| 38 | DatasetId::Issues => "issues", | |
| 39 | DatasetId::PullRequests => "pull_requests", | |
| 40 | DatasetId::WorkflowRuns => "workflow_runs", | |
| 41 | DatasetId::Deployments => "deployments", | |
| 42 | DatasetId::Spend => "spend", | |
| 43 | DatasetId::AgentSessions => "agent_sessions", | |
| 44 | } | |
| 45 | } | |
| 46 | ||
| 47 | /// Its entry in [`DATASETS`]. | |
| 48 | pub fn spec(self) -> &'static DatasetSpec { | |
| 49 | DATASETS.iter().find(|spec| spec.id == self).expect("every dataset is in the catalog") | |
| 50 | } | |
| 51 | } | |
| 52 | ||
| 53 | /// How rows are summed up. `count` takes no field; `rate` a rate field; the | |
| 54 | /// rest a measure field. | |
| 55 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 56 | #[serde(rename_all = "snake_case")] | |
| 57 | pub enum MeasureOp { | |
| 58 | Count, | |
| 59 | Sum, | |
| 60 | Avg, | |
| 61 | P50, | |
| 62 | P95, | |
| 63 | Rate, | |
| 64 | } | |
| 65 | ||
| 66 | impl MeasureOp { | |
| 67 | pub fn as_str(self) -> &'static str { | |
| 68 | match self { | |
| 69 | MeasureOp::Count => "count", | |
| 70 | MeasureOp::Sum => "sum", | |
| 71 | MeasureOp::Avg => "avg", | |
| 72 | MeasureOp::P50 => "p50", | |
| 73 | MeasureOp::P95 => "p95", | |
| 74 | MeasureOp::Rate => "rate", | |
| 75 | } | |
| 76 | } | |
| 77 | } | |
| 78 | ||
| 79 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 80 | pub struct Measure { | |
| 81 | pub op: MeasureOp, | |
| 82 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 83 | pub field: Option<String>, | |
| 84 | } | |
| 85 | ||
| 86 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 87 | #[serde(rename_all = "snake_case")] | |
| 88 | pub enum Interval { | |
| 89 | Day, | |
| 90 | Week, | |
| 91 | Month, | |
| 92 | } | |
| 93 | ||
| 94 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 95 | #[serde(rename_all = "snake_case")] | |
| 96 | pub enum FilterOp { | |
| 97 | Eq, | |
| 98 | Neq, | |
| 99 | In, | |
| 100 | Gte, | |
| 101 | Lte, | |
| 102 | } | |
| 103 | ||
| 104 | impl FilterOp { | |
| 105 | pub fn as_str(self) -> &'static str { | |
| 106 | match self { | |
| 107 | FilterOp::Eq => "eq", | |
| 108 | FilterOp::Neq => "neq", | |
| 109 | FilterOp::In => "in", | |
| 110 | FilterOp::Gte => "gte", | |
| 111 | FilterOp::Lte => "lte", | |
| 112 | } | |
| 113 | } | |
| 114 | } | |
| 115 | ||
| 116 | /// Text for `eq`/`neq` on a dimension, a list for `in`, a number on a | |
| 117 | /// measure. | |
| 118 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 119 | #[serde(untagged)] | |
| 120 | pub enum FilterValue { | |
| 121 | Text(String), | |
| 122 | Number(f64), | |
| 123 | List(Vec<String>), | |
| 124 | } | |
| 125 | ||
| 126 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 127 | pub struct Filter { | |
| 128 | pub field: String, | |
| 129 | pub op: FilterOp, | |
| 130 | pub value: FilterValue, | |
| 131 | } | |
| 132 | ||
| 133 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 134 | pub enum RangePreset { | |
| 135 | #[serde(rename = "7d")] | |
| 136 | Days7, | |
| 137 | #[serde(rename = "30d")] | |
| 138 | Days30, | |
| 139 | #[serde(rename = "90d")] | |
| 140 | Days90, | |
| 141 | } | |
| 142 | ||
| 143 | impl RangePreset { | |
| 144 | pub fn days(self) -> u32 { | |
| 145 | match self { | |
| 146 | RangePreset::Days7 => 7, | |
| 147 | RangePreset::Days30 => 30, | |
| 148 | RangePreset::Days90 => 90, | |
| 149 | } | |
| 150 | } | |
| 151 | } | |
| 152 | ||
| 153 | /// A preset counted back from now, or between two times: dates | |
| 154 | /// (`2026-10-01`) or RFC 3339 UTC times; `to` is exclusive. | |
| 155 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 156 | #[serde(untagged)] | |
| 157 | pub enum DatasetRange { | |
| 158 | Preset(RangePreset), | |
| 159 | Between { from: String, to: String }, | |
| 160 | } | |
| 161 | ||
| 162 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 163 | pub struct DatasetQuery { | |
| 164 | pub dataset: DatasetId, | |
| 165 | pub measure: Measure, | |
| 166 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 167 | pub group_by: Option<String>, | |
| 168 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 169 | pub interval: Option<Interval>, | |
| 170 | /// Which of the dataset's time fields the range and interval use; its | |
| 171 | /// first when absent. | |
| 172 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 173 | pub time: Option<String>, | |
| 174 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 175 | pub filters: Option<Vec<Filter>>, | |
| 176 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 177 | pub range: Option<DatasetRange>, | |
| 178 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 179 | pub limit: Option<u32>, | |
| 180 | } | |
| 181 | ||
| 182 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 183 | #[serde(rename_all = "snake_case")] | |
| 184 | pub enum ColumnType { | |
| 185 | String, | |
| 186 | Number, | |
| 187 | Time, | |
| 188 | Money, | |
| 189 | } | |
| 190 | ||
| 191 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 192 | pub struct DatasetColumn { | |
| 193 | pub name: String, | |
| 194 | #[serde(rename = "type")] | |
| 195 | pub kind: ColumnType, | |
| 196 | } | |
| 197 | ||
| 198 | /// What `query_dataset` answers. | |
| 199 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 200 | pub struct DatasetResult { | |
| 201 | pub columns: Vec<DatasetColumn>, | |
| 202 | /// Each a string, a number or null. | |
| 203 | pub rows: Vec<Vec<serde_json::Value>>, | |
| 204 | /// More rows matched than were returned. | |
| 205 | pub truncated: bool, | |
| 206 | /// The viewer cannot read everything the query covers. | |
| 207 | pub partial: bool, | |
| 208 | /// RFC 3339. | |
| 209 | pub as_of: String, | |
| 210 | } | |
| 211 | ||
| 212 | /// The service that owns a dataset. | |
| 213 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] | |
| 214 | pub enum DatasetService { | |
| 215 | Work, | |
| 216 | Actions, | |
| 217 | Deployments, | |
| 218 | Billing, | |
| 219 | Agents, | |
| 220 | } | |
| 221 | ||
| 222 | impl DatasetService { | |
| 223 | pub fn as_str(self) -> &'static str { | |
| 224 | match self { | |
| 225 | DatasetService::Work => "work", | |
| 226 | DatasetService::Actions => "actions", | |
| 227 | DatasetService::Deployments => "deployments", | |
| 228 | DatasetService::Billing => "billing", | |
| 229 | DatasetService::Agents => "agents", | |
| 230 | } | |
| 231 | } | |
| 232 | } | |
| 233 | ||
| 234 | /// What a viewer needs to query a dataset. | |
| 235 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] | |
| 236 | pub enum DatasetNeeds { | |
| 237 | Member, | |
| 238 | /// The workspace's billing role. | |
| 239 | Billing, | |
| 240 | } | |
| 241 | ||
| 242 | impl DatasetNeeds { | |
| 243 | pub fn as_str(self) -> &'static str { | |
| 244 | match self { | |
| 245 | DatasetNeeds::Member => "member", | |
| 246 | DatasetNeeds::Billing => "billing", | |
| 247 | } | |
| 248 | } | |
| 249 | } | |
| 250 | ||
| 251 | #[derive(Debug)] | |
| 252 | pub struct DatasetSpec { | |
| 253 | pub id: DatasetId, | |
| 254 | pub label: &'static str, | |
| 255 | pub service: DatasetService, | |
| 256 | pub needs: DatasetNeeds, | |
| 257 | /// Time fields, the default first. | |
| 258 | pub times: &'static [&'static str], | |
| 259 | /// Text fields to group and filter by. | |
| 260 | pub dimensions: &'static [&'static str], | |
| 261 | /// Number fields for sum, avg, p50, p95, and number filters. | |
| 262 | pub measures: &'static [&'static str], | |
| 263 | /// Yes-or-no fields `rate` gives the share of. | |
| 264 | pub rates: &'static [&'static str], | |
| 265 | } | |
| 266 | ||
| 267 | /// The catalog. | |
| 268 | pub const DATASETS: [DatasetSpec; 6] = [ | |
| 269 | DatasetSpec { | |
| 270 | id: DatasetId::Issues, | |
| 271 | label: "Issues", | |
| 272 | service: DatasetService::Work, | |
| 273 | needs: DatasetNeeds::Member, | |
| 274 | times: &["created_at", "closed_at"], | |
| 275 | dimensions: &["repo", "label", "state", "author_kind", "assignee_kind", "milestone"], | |
| 276 | measures: &["time_to_close_hours", "comments"], | |
| 277 | rates: &["closed"], | |
| 278 | }, | |
| 279 | DatasetSpec { | |
| 280 | id: DatasetId::PullRequests, | |
| 281 | label: "Pull requests", | |
| 282 | service: DatasetService::Work, | |
| 283 | needs: DatasetNeeds::Member, | |
| 284 | times: &["created_at", "merged_at", "closed_at"], | |
| 285 | dimensions: &["repo", "label", "state", "author_kind", "base_branch"], | |
| 286 | measures: &["cycle_time_hours", "time_to_first_review_hours", "review_count", "additions", "deletions", "changed_files"], | |
| 287 | rates: &["merged"], | |
| 288 | }, | |
| 289 | DatasetSpec { | |
| 290 | id: DatasetId::WorkflowRuns, | |
| 291 | label: "Workflow runs", | |
| 292 | service: DatasetService::Actions, | |
| 293 | needs: DatasetNeeds::Member, | |
| 294 | times: &["started_at", "completed_at"], | |
| 295 | dimensions: &["repo", "workflow", "branch", "event", "conclusion", "runner_kind"], | |
| 296 | measures: &["duration_seconds", "queue_seconds"], | |
| 297 | rates: &["succeeded"], | |
| 298 | }, | |
| 299 | DatasetSpec { | |
| 300 | id: DatasetId::Deployments, | |
| 301 | label: "Deployments", | |
| 302 | service: DatasetService::Deployments, | |
| 303 | needs: DatasetNeeds::Member, | |
| 304 | times: &["created_at"], | |
| 305 | dimensions: &["repo", "project", "environment", "state"], | |
| 306 | measures: &["duration_seconds", "time_to_restore_hours"], | |
| 307 | rates: &["failed"], | |
| 308 | }, | |
| 309 | DatasetSpec { | |
| 310 | id: DatasetId::Spend, | |
| 311 | label: "Spend", | |
| 312 | service: DatasetService::Billing, | |
| 313 | needs: DatasetNeeds::Billing, | |
| 314 | times: &["day"], | |
| 315 | dimensions: &["product", "project", "person", "model"], | |
| 316 | measures: &["amount_micros"], | |
| 317 | rates: &[], | |
| 318 | }, | |
| 319 | DatasetSpec { | |
| 320 | id: DatasetId::AgentSessions, | |
| 321 | label: "Agent sessions", | |
| 322 | service: DatasetService::Agents, | |
| 323 | needs: DatasetNeeds::Member, | |
| 324 | times: &["started_at"], | |
| 325 | dimensions: &["agent", "repo", "outcome", "model", "trigger"], | |
| 326 | measures: &["duration_seconds", "cost_micros", "tokens"], | |
| 327 | rates: &["succeeded"], | |
| 328 | }, | |
| 329 | ]; | |
| 330 | ||
| 331 | /// The most rows a query returns. | |
| 332 | pub const MAX_ROWS: u32 = 100; | |
| 333 | /// The most filters on one query. | |
| 334 | pub const MAX_FILTERS: usize = 10; | |
| 335 | /// The most values in an `in` filter. | |
| 336 | pub const MAX_IN_VALUES: usize = 50; | |
| 337 | /// The longest `{ from, to }` range, in days. | |
| 338 | pub const MAX_RANGE_DAYS: u64 = 366; | |
| 339 | ||
| 340 | const DAY_MS: u64 = 86_400_000; | |
| 341 | ||
| 342 | /// A range end as milliseconds since the epoch: a real date (`2026-10-01`) | |
| 343 | /// or an RFC 3339 UTC time with at most milliseconds. `None` otherwise. | |
| 344 | pub fn range_time(text: &str) -> Option<u64> { | |
| 345 | let bytes = text.as_bytes(); | |
| 346 | let digits = |range: std::ops::Range<usize>| -> Option<u64> { | |
| 347 | let part = text.get(range)?; | |
| 348 | if part.is_empty() || !part.bytes().all(|b| b.is_ascii_digit()) { | |
| 349 | return None; | |
| 350 | } | |
| 351 | part.parse().ok() | |
| 352 | }; | |
| 353 | if bytes.len() < 10 || bytes[4] != b'-' || bytes[7] != b'-' { | |
| 354 | return None; | |
| 355 | } | |
| 356 | let (year, month, day) = (digits(0..4)?, digits(5..7)?, digits(8..10)?); | |
| 357 | if !(1..=12).contains(&month) || day < 1 || day > days_in_month(year, month) { | |
| 358 | return None; | |
| 359 | } | |
| 360 | let full = if bytes.len() == 10 { | |
| 361 | format!("{text}T00:00:00Z") | |
| 362 | } else { | |
| 363 | // T hh:mm:ss, optional .f{1,3}, Z. | |
| 364 | if bytes.len() < 20 || bytes[10] != b'T' || bytes[13] != b':' || bytes[16] != b':' || *bytes.last()? != b'Z' { | |
| 365 | return None; | |
| 366 | } | |
| 367 | let (hour, minute, second) = (digits(11..13)?, digits(14..16)?, digits(17..19)?); | |
| 368 | if hour > 23 || minute > 59 || second > 59 { | |
| 369 | return None; | |
| 370 | } | |
| 371 | match bytes.len() { | |
| 372 | 20 => {} | |
| 373 | 22..=24 if bytes[19] == b'.' => { | |
| 374 | digits(20..bytes.len() - 1)?; | |
| 375 | } | |
| 376 | _ => return None, | |
| 377 | } | |
| 378 | text.to_owned() | |
| 379 | }; | |
| 380 | crate::time::parse_rfc3339(&full) | |
| 381 | } | |
| 382 | ||
| 383 | fn days_in_month(year: u64, month: u64) -> u64 { | |
| 384 | match month { | |
| 385 | 2 if (year % 4 == 0 && year % 100 != 0) || year % 400 == 0 => 29, | |
| 386 | 2 => 28, | |
| 387 | 4 | 6 | 9 | 11 => 30, | |
| 388 | _ => 31, | |
| 389 | } | |
| 390 | } | |
| 391 | ||
| 392 | impl DatasetQuery { | |
| 393 | /// What is wrong with it, if anything: the same rules, and the same | |
| 394 | /// words, as `datasetQueryError` in TypeScript. It checks the query | |
| 395 | /// against the catalog only; who may run it is the owning service's to | |
| 396 | /// decide. | |
| 397 | pub fn validate(&self) -> Result<(), String> { | |
| 398 | let spec = self.dataset.spec(); | |
| 399 | let name = self.dataset.as_str(); | |
| 400 | let field = self.measure.field.as_deref(); | |
| 401 | match self.measure.op { | |
| 402 | MeasureOp::Count => { | |
| 403 | if field.is_some() { | |
| 404 | return Err("count takes no field.".to_owned()); | |
| 405 | } | |
| 406 | } | |
| 407 | MeasureOp::Rate => { | |
| 408 | let Some(field) = field else { return Err("rate needs a field.".to_owned()) }; | |
| 409 | if !spec.rates.contains(&field) { | |
| 410 | return Err(format!("{name} has no yes-or-no field {field} to take the rate of.")); | |
| 411 | } | |
| 412 | } | |
| 413 | op => { | |
| 414 | let Some(field) = field else { return Err(format!("{} needs a field.", op.as_str())) }; | |
| 415 | if !spec.measures.contains(&field) { | |
| 416 | return Err(format!("{name} has no number field {field}.")); | |
| 417 | } | |
| 418 | } | |
| 419 | } | |
| 420 | if let Some(group_by) = self.group_by.as_deref() | |
| 421 | && !spec.dimensions.contains(&group_by) | |
| 422 | { | |
| 423 | return Err(format!("{name} can't be grouped by {group_by}.")); | |
| 424 | } | |
| 425 | if let Some(time) = self.time.as_deref() | |
| 426 | && !spec.times.contains(&time) | |
| 427 | { | |
| 428 | return Err(format!("{name} has no time field {time}.")); | |
| 429 | } | |
| 430 | let filters = self.filters.as_deref().unwrap_or_default(); | |
| 431 | if filters.len() > MAX_FILTERS { | |
| 432 | return Err(format!("A query takes at most {MAX_FILTERS} filters.")); | |
| 433 | } | |
| 434 | for filter in filters { | |
| 435 | let (field, op) = (filter.field.as_str(), filter.op.as_str()); | |
| 436 | if spec.dimensions.contains(&field) { | |
| 437 | match (filter.op, &filter.value) { | |
| 438 | (FilterOp::In, FilterValue::List(values)) if !values.is_empty() && values.len() <= MAX_IN_VALUES => {} | |
| 439 | (FilterOp::In, _) => return Err(format!("in on {field} takes a list of 1 to {MAX_IN_VALUES} values.")), | |
| 440 | (FilterOp::Eq | FilterOp::Neq, FilterValue::Text(_)) => {} | |
| 441 | (FilterOp::Eq | FilterOp::Neq, _) => return Err(format!("{op} on {field} takes text.")), | |
| 442 | _ => return Err(format!("{field} is text: filter it with eq, neq or in.")), | |
| 443 | } | |
| 444 | } else if spec.measures.contains(&field) { | |
| 445 | match (filter.op, &filter.value) { | |
| 446 | (FilterOp::In, _) => return Err(format!("{field} is a number: filter it with eq, neq, gte or lte.")), | |
| 447 | (_, FilterValue::Number(n)) if n.is_finite() => {} | |
| 448 | _ => return Err(format!("{op} on {field} takes a number.")), | |
| 449 | } | |
| 450 | } else { | |
| 451 | return Err(format!("{name} can't be filtered by {field}.")); | |
| 452 | } | |
| 453 | } | |
| 454 | if let Some(DatasetRange::Between { from, to }) = &self.range { | |
| 455 | let (Some(from), Some(to)) = (range_time(from), range_time(to)) else { | |
| 456 | return Err("A range's from and to are dates or RFC 3339 UTC times.".to_owned()); | |
| 457 | }; | |
| 458 | if from >= to { | |
| 459 | return Err("A range's from comes before its to.".to_owned()); | |
| 460 | } | |
| 461 | if to - from > MAX_RANGE_DAYS * DAY_MS { | |
| 462 | return Err(format!("A range is at most {MAX_RANGE_DAYS} days.")); | |
| 463 | } | |
| 464 | } | |
| 465 | if let Some(limit) = self.limit | |
| 466 | && !(1..=MAX_ROWS).contains(&limit) | |
| 467 | { | |
| 468 | return Err(format!("limit is between 1 and {MAX_ROWS}.")); | |
| 469 | } | |
| 470 | Ok(()) | |
| 471 | } | |
| 472 | } | |
| 473 | ||
| 474 | #[cfg(test)] | |
| 475 | mod tests { | |
| 476 | use super::*; | |
| 477 | ||
| 478 | /// The two validators agree on every case in the shared fixtures. | |
| 479 | #[test] | |
| 480 | fn agrees_with_the_typescript_validator_on_the_fixtures() { | |
| 481 | let fixtures: serde_json::Value = serde_json::from_str(include_str!("../../../packages/contracts/src/datasets.fixtures.json")).unwrap(); | |
| 482 | let cases = fixtures["cases"].as_array().unwrap(); | |
| 483 | assert!(cases.len() > 20); | |
| 484 | for case in cases { | |
| 485 | let outcome = serde_json::from_value::<DatasetQuery>(case["query"].clone()) | |
| 486 | .map_err(|_| "*".to_owned()) | |
| 487 | .and_then(|query| query.validate()); | |
| 488 | match case["error"].as_str() { | |
| 489 | None => assert_eq!(outcome, Ok(()), "{}", case["query"]), | |
| 490 | Some("*") => assert!(outcome.is_err(), "{} should be refused", case["query"]), | |
| 491 | Some(error) => assert_eq!(outcome, Err(error.to_owned()), "{}", case["query"]), | |
| 492 | } | |
| 493 | } | |
| 494 | } | |
| 495 | ||
| 496 | /// The site's copy of the catalog names the same datasets, services and | |
| 497 | /// fields, in the same order. | |
| 498 | #[test] | |
| 499 | fn the_typescript_mirror_has_the_same_catalog() { | |
| 500 | let ts = include_str!("../../../packages/contracts/src/datasets.ts"); | |
| 501 | let catalog = ts | |
| 502 | .split_once("export const DATASETS: Record<DatasetId, DatasetSpec> = {") | |
| 503 | .and_then(|(_, rest)| rest.split_once("\n};")) | |
| 504 | .map(|(table, _)| table) | |
| 505 | .expect("DATASETS in datasets.ts"); | |
| 506 | let quoted = |line: &str, key: &str| -> Vec<String> { | |
| 507 | let start = format!("{key}: ["); | |
| 508 | let list = line.split_once(&start).and_then(|(_, rest)| rest.split_once(']')).map(|(list, _)| list).unwrap_or_else(|| panic!("{key} in {line}")); | |
| 509 | list.split('"').skip(1).step_by(2).map(str::to_owned).collect() | |
| 510 | }; | |
| 511 | let lines: Vec<&str> = catalog.lines().filter(|line| line.contains("service:")).collect(); | |
| 512 | assert_eq!(lines.len(), DATASETS.len()); | |
| 513 | for (line, spec) in lines.iter().zip(DATASETS.iter()) { | |
| 514 | assert!(line.trim_start().starts_with(&format!("{}: {{", spec.id.as_str())), "{line}"); | |
| 515 | assert!(line.contains(&format!("label: \"{}\"", spec.label)), "{line}"); | |
| 516 | assert!(line.contains(&format!("service: \"{}\"", spec.service.as_str())), "{line}"); | |
| 517 | assert!(line.contains(&format!("needs: \"{}\"", spec.needs.as_str())), "{line}"); | |
| 518 | assert_eq!(quoted(line, "times"), spec.times, "{line}"); | |
| 519 | assert_eq!(quoted(line, "dimensions"), spec.dimensions, "{line}"); | |
| 520 | assert_eq!(quoted(line, "measures"), spec.measures, "{line}"); | |
| 521 | assert_eq!(quoted(line, "rates"), spec.rates, "{line}"); | |
| 522 | } | |
| 523 | for (name, value) in [("DATASET_MAX_ROWS", MAX_ROWS as usize), ("DATASET_MAX_FILTERS", MAX_FILTERS), ("DATASET_MAX_IN_VALUES", MAX_IN_VALUES), ("DATASET_MAX_RANGE_DAYS", MAX_RANGE_DAYS as usize)] { | |
| 524 | assert!(ts.contains(&format!("export const {name} = {value};")), "{name}"); | |
| 525 | } | |
| 526 | } | |
| 527 | ||
| 528 | #[test] | |
| 529 | fn every_dataset_reads_back_and_has_a_time_field() { | |
| 530 | for id in DatasetId::ALL { | |
| 531 | assert_eq!(serde_json::to_value(id).unwrap(), id.as_str()); | |
| 532 | let spec = id.spec(); | |
| 533 | assert!(!spec.times.is_empty(), "{}", id.as_str()); | |
| 534 | // A field is one thing: never both a dimension and a number. | |
| 535 | for field in spec.dimensions { | |
| 536 | assert!(!spec.measures.contains(field) && !spec.rates.contains(field), "{field}"); | |
| 537 | } | |
| 538 | } | |
| 539 | assert_eq!(DatasetId::Spend.spec().needs, DatasetNeeds::Billing); | |
| 540 | } | |
| 541 | ||
| 542 | #[test] | |
| 543 | fn a_query_travels_snake_case() { | |
| 544 | let query = DatasetQuery { | |
| 545 | dataset: DatasetId::PullRequests, | |
| 546 | measure: Measure { op: MeasureOp::P95, field: Some("cycle_time_hours".to_owned()) }, | |
| 547 | group_by: Some("repo".to_owned()), | |
| 548 | interval: Some(Interval::Week), | |
| 549 | time: None, | |
| 550 | filters: Some(vec![Filter { field: "state".to_owned(), op: FilterOp::Eq, value: FilterValue::Text("merged".to_owned()) }]), | |
| 551 | range: Some(DatasetRange::Preset(RangePreset::Days90)), | |
| 552 | limit: None, | |
| 553 | }; | |
| 554 | let json = serde_json::to_value(&query).unwrap(); | |
| 555 | assert_eq!( | |
| 556 | json, | |
| 557 | serde_json::json!({ | |
| 558 | "dataset": "pull_requests", | |
| 559 | "measure": { "op": "p95", "field": "cycle_time_hours" }, | |
| 560 | "group_by": "repo", | |
| 561 | "interval": "week", | |
| 562 | "filters": [{ "field": "state", "op": "eq", "value": "merged" }], | |
| 563 | "range": "90d" | |
| 564 | }) | |
| 565 | ); | |
| 566 | assert_eq!(serde_json::from_value::<DatasetQuery>(json).unwrap(), query); | |
| 567 | let between: DatasetRange = serde_json::from_value(serde_json::json!({ "from": "2026-01-01", "to": "2026-02-01" })).unwrap(); | |
| 568 | assert!(matches!(between, DatasetRange::Between { .. })); | |
| 569 | } | |
| 570 | ||
| 571 | #[test] | |
| 572 | fn range_times_are_real_dates_or_utc_times() { | |
| 573 | assert_eq!(range_time("1970-01-02"), Some(DAY_MS)); | |
| 574 | assert_eq!(range_time("2026-10-02T05:16:19Z"), Some(1_790_918_179_000)); | |
| 575 | assert_eq!(range_time("2026-10-02T05:16:19.5Z"), Some(1_790_918_179_500)); | |
| 576 | assert_eq!(range_time("2024-02-29"), crate::time::parse_rfc3339("2024-02-29T00:00:00Z")); | |
| 577 | for bad in ["2026-02-29", "2026-04-31", "2026-13-01", "2026-10-02T24:00:00Z", "2026-10-02T05:16:19", "2026-10-02T05:16:19+02:00", "26-10-02", "2026-1-02", "today", ""] { | |
| 578 | assert_eq!(range_time(bad), None, "{bad}"); | |
| 579 | } | |
| 580 | } | |
| 581 | } |