Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Artifacts contracts: folios, their four kinds, dashboard datasets, folio events and the artifacts scopes are typed and validated the same in TypeScript and Rust, with nothing using them yet | 1 | //! Datasets: the safe query layer behind dashboards (Artifacts mode, |
| 2 | //! docs/ARTIFACTS_MODE.md section 3.4). Mirrors | |
| 3 | //! `packages/contracts/src/datasets.ts`; the tests here keep the catalog the | |
| 4 | //! same and run both validators over `datasets.fixtures.json`. | |
| 5 | //! | |
| 6 | //! Not SQL: a query names a dataset from a declared catalog, one measure, | |
| 7 | //! at most one dimension, an interval, filters on declared fields and a | |
| 8 | //! range. The service that owns the data ([`DatasetSpec::service`]) answers | |
| 9 | //! `query_dataset` for the viewer, over only what the viewer can read, with | |
| 10 | //! a fixed query per measure and dimension, and caps the rows. The Rust | |
| 11 | //! owners (work, actions, billing) validate with [`DatasetQuery::validate`] | |
| 12 | //! before they run anything. | |
| 13 | ||
| 14 | use serde::{Deserialize, Serialize}; | |
| 15 | ||
| 16 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)] | |
| 17 | #[serde(rename_all = "snake_case")] | |
| 18 | pub enum DatasetId { | |
| 19 | Issues, | |
| 20 | PullRequests, | |
| 21 | WorkflowRuns, | |
| 22 | Deployments, | |
| 23 | Spend, | |
| 24 | AgentSessions, | |
| 25 | } | |
| 26 | ||
| 27 | impl DatasetId { | |
| 28 | pub const ALL: [DatasetId; 6] = [ | |
| 29 | DatasetId::Issues, | |
| 30 | DatasetId::PullRequests, | |
| 31 | DatasetId::WorkflowRuns, | |
| 32 | DatasetId::Deployments, | |
| 33 | DatasetId::Spend, | |
| 34 | DatasetId::AgentSessions, | |
| 35 | ]; | |
| 36 | ||
| 37 | pub fn as_str(self) -> &'static str { | |
| 38 | match self { | |
| 39 | DatasetId::Issues => "issues", | |
| 40 | DatasetId::PullRequests => "pull_requests", | |
| 41 | DatasetId::WorkflowRuns => "workflow_runs", | |
| 42 | DatasetId::Deployments => "deployments", | |
| 43 | DatasetId::Spend => "spend", | |
| 44 | DatasetId::AgentSessions => "agent_sessions", | |
| 45 | } | |
| 46 | } | |
| 47 | ||
| 48 | /// Its entry in [`DATASETS`]. | |
| 49 | pub fn spec(self) -> &'static DatasetSpec { | |
| 50 | DATASETS.iter().find(|spec| spec.id == self).expect("every dataset is in the catalog") | |
| 51 | } | |
| 52 | } | |
| 53 | ||
| 54 | /// How rows are summed up. `count` takes no field; `rate` a rate field; the | |
| 55 | /// rest a measure field. | |
| 56 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 57 | #[serde(rename_all = "snake_case")] | |
| 58 | pub enum MeasureOp { | |
| 59 | Count, | |
| 60 | Sum, | |
| 61 | Avg, | |
| 62 | P50, | |
| 63 | P95, | |
| 64 | Rate, | |
| 65 | } | |
| 66 | ||
| 67 | impl MeasureOp { | |
| 68 | pub fn as_str(self) -> &'static str { | |
| 69 | match self { | |
| 70 | MeasureOp::Count => "count", | |
| 71 | MeasureOp::Sum => "sum", | |
| 72 | MeasureOp::Avg => "avg", | |
| 73 | MeasureOp::P50 => "p50", | |
| 74 | MeasureOp::P95 => "p95", | |
| 75 | MeasureOp::Rate => "rate", | |
| 76 | } | |
| 77 | } | |
| 78 | } | |
| 79 | ||
| 80 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 81 | pub struct Measure { | |
| 82 | pub op: MeasureOp, | |
| 83 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 84 | pub field: Option<String>, | |
| 85 | } | |
| 86 | ||
| 87 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 88 | #[serde(rename_all = "snake_case")] | |
| 89 | pub enum Interval { | |
| 90 | Day, | |
| 91 | Week, | |
| 92 | Month, | |
| 93 | } | |
| 94 | ||
| 95 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 96 | #[serde(rename_all = "snake_case")] | |
| 97 | pub enum FilterOp { | |
| 98 | Eq, | |
| 99 | Neq, | |
| 100 | In, | |
| 101 | Gte, | |
| 102 | Lte, | |
| 103 | } | |
| 104 | ||
| 105 | impl FilterOp { | |
| 106 | pub fn as_str(self) -> &'static str { | |
| 107 | match self { | |
| 108 | FilterOp::Eq => "eq", | |
| 109 | FilterOp::Neq => "neq", | |
| 110 | FilterOp::In => "in", | |
| 111 | FilterOp::Gte => "gte", | |
| 112 | FilterOp::Lte => "lte", | |
| 113 | } | |
| 114 | } | |
| 115 | } | |
| 116 | ||
| 117 | /// Text for `eq`/`neq` on a dimension, a list for `in`, a number on a | |
| 118 | /// measure. | |
| 119 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 120 | #[serde(untagged)] | |
| 121 | pub enum FilterValue { | |
| 122 | Text(String), | |
| 123 | Number(f64), | |
| 124 | List(Vec<String>), | |
| 125 | } | |
| 126 | ||
| 127 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 128 | pub struct Filter { | |
| 129 | pub field: String, | |
| 130 | pub op: FilterOp, | |
| 131 | pub value: FilterValue, | |
| 132 | } | |
| 133 | ||
| 134 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 135 | pub enum RangePreset { | |
| 136 | #[serde(rename = "7d")] | |
| 137 | Days7, | |
| 138 | #[serde(rename = "30d")] | |
| 139 | Days30, | |
| 140 | #[serde(rename = "90d")] | |
| 141 | Days90, | |
| 142 | } | |
| 143 | ||
| 144 | impl RangePreset { | |
| 145 | pub fn days(self) -> u32 { | |
| 146 | match self { | |
| 147 | RangePreset::Days7 => 7, | |
| 148 | RangePreset::Days30 => 30, | |
| 149 | RangePreset::Days90 => 90, | |
| 150 | } | |
| 151 | } | |
| 152 | } | |
| 153 | ||
| 154 | /// A preset counted back from now, or between two times: dates | |
| 155 | /// (`2026-10-01`) or RFC 3339 UTC times; `to` is exclusive. | |
| 156 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 157 | #[serde(untagged)] | |
| 158 | pub enum DatasetRange { | |
| 159 | Preset(RangePreset), | |
| 160 | Between { from: String, to: String }, | |
| 161 | } | |
| 162 | ||
| 163 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 164 | pub struct DatasetQuery { | |
| 165 | pub dataset: DatasetId, | |
| 166 | pub measure: Measure, | |
| 167 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 168 | pub group_by: Option<String>, | |
| 169 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 170 | pub interval: Option<Interval>, | |
| 171 | /// Which of the dataset's time fields the range and interval use; its | |
| 172 | /// first when absent. | |
| 173 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 174 | pub time: Option<String>, | |
| 175 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 176 | pub filters: Option<Vec<Filter>>, | |
| 177 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 178 | pub range: Option<DatasetRange>, | |
| 179 | #[serde(default, skip_serializing_if = "Option::is_none")] | |
| 180 | pub limit: Option<u32>, | |
| 181 | } | |
| 182 | ||
| 183 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 184 | #[serde(rename_all = "snake_case")] | |
| 185 | pub enum ColumnType { | |
| 186 | String, | |
| 187 | Number, | |
| 188 | Time, | |
| 189 | Money, | |
| 190 | } | |
| 191 | ||
| 192 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 193 | pub struct DatasetColumn { | |
| 194 | pub name: String, | |
| 195 | #[serde(rename = "type")] | |
| 196 | pub kind: ColumnType, | |
| 197 | } | |
| 198 | ||
| 199 | /// What `query_dataset` answers. | |
| 200 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] | |
| 201 | pub struct DatasetResult { | |
| 202 | pub columns: Vec<DatasetColumn>, | |
| 203 | /// Each a string, a number or null. | |
| 204 | pub rows: Vec<Vec<serde_json::Value>>, | |
| 205 | /// More rows matched than were returned. | |
| 206 | pub truncated: bool, | |
| 207 | /// The viewer cannot read everything the query covers. | |
| 208 | pub partial: bool, | |
| 209 | /// RFC 3339. | |
| 210 | pub as_of: String, | |
| 211 | } | |
| 212 | ||
| 213 | /// The service that owns a dataset. | |
| 214 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] | |
| 215 | pub enum DatasetService { | |
| 216 | Work, | |
| 217 | Actions, | |
| 218 | Deployments, | |
| 219 | Billing, | |
| 220 | Agents, | |
| 221 | } | |
| 222 | ||
| 223 | impl DatasetService { | |
| 224 | pub fn as_str(self) -> &'static str { | |
| 225 | match self { | |
| 226 | DatasetService::Work => "work", | |
| 227 | DatasetService::Actions => "actions", | |
| 228 | DatasetService::Deployments => "deployments", | |
| 229 | DatasetService::Billing => "billing", | |
| 230 | DatasetService::Agents => "agents", | |
| 231 | } | |
| 232 | } | |
| 233 | } | |
| 234 | ||
| 235 | /// What a viewer needs to query a dataset. | |
| 236 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] | |
| 237 | pub enum DatasetNeeds { | |
| 238 | Member, | |
| 239 | /// The workspace's billing role. | |
| 240 | Billing, | |
| 241 | } | |
| 242 | ||
| 243 | impl DatasetNeeds { | |
| 244 | pub fn as_str(self) -> &'static str { | |
| 245 | match self { | |
| 246 | DatasetNeeds::Member => "member", | |
| 247 | DatasetNeeds::Billing => "billing", | |
| 248 | } | |
| 249 | } | |
| 250 | } | |
| 251 | ||
| 252 | #[derive(Debug)] | |
| 253 | pub struct DatasetSpec { | |
| 254 | pub id: DatasetId, | |
| 255 | pub label: &'static str, | |
| 256 | pub service: DatasetService, | |
| 257 | pub needs: DatasetNeeds, | |
| 258 | /// Time fields, the default first. | |
| 259 | pub times: &'static [&'static str], | |
| 260 | /// Text fields to group and filter by. | |
| 261 | pub dimensions: &'static [&'static str], | |
| 262 | /// Number fields for sum, avg, p50, p95, and number filters. | |
| 263 | pub measures: &'static [&'static str], | |
| 264 | /// Yes-or-no fields `rate` gives the share of. | |
| 265 | pub rates: &'static [&'static str], | |
| 266 | } | |
| 267 | ||
| 268 | /// The catalog. | |
| 269 | pub const DATASETS: [DatasetSpec; 6] = [ | |
| 270 | DatasetSpec { | |
| 271 | id: DatasetId::Issues, | |
| 272 | label: "Issues", | |
| 273 | service: DatasetService::Work, | |
| 274 | needs: DatasetNeeds::Member, | |
| 275 | times: &["created_at", "closed_at"], | |
| 276 | dimensions: &["repo", "label", "state", "author_kind", "assignee_kind", "milestone"], | |
| 277 | measures: &["time_to_close_hours", "comments"], | |
| 278 | rates: &["closed"], | |
| 279 | }, | |
| 280 | DatasetSpec { | |
| 281 | id: DatasetId::PullRequests, | |
| 282 | label: "Pull requests", | |
| 283 | service: DatasetService::Work, | |
| 284 | needs: DatasetNeeds::Member, | |
| 285 | times: &["created_at", "merged_at", "closed_at"], | |
| 286 | dimensions: &["repo", "label", "state", "author_kind", "base_branch"], | |
| 287 | measures: &["cycle_time_hours", "time_to_first_review_hours", "review_count", "additions", "deletions", "changed_files"], | |
| 288 | rates: &["merged"], | |
| 289 | }, | |
| 290 | DatasetSpec { | |
| 291 | id: DatasetId::WorkflowRuns, | |
| 292 | label: "Workflow runs", | |
| 293 | service: DatasetService::Actions, | |
| 294 | needs: DatasetNeeds::Member, | |
| 295 | times: &["started_at", "completed_at"], | |
| 296 | dimensions: &["repo", "workflow", "branch", "event", "conclusion", "runner_kind"], | |
| 297 | measures: &["duration_seconds", "queue_seconds"], | |
| 298 | rates: &["succeeded"], | |
| 299 | }, | |
| 300 | DatasetSpec { | |
| 301 | id: DatasetId::Deployments, | |
| 302 | label: "Deployments", | |
| 303 | service: DatasetService::Deployments, | |
| 304 | needs: DatasetNeeds::Member, | |
| 305 | times: &["created_at"], | |
| 306 | dimensions: &["repo", "project", "environment", "state"], | |
| 307 | measures: &["duration_seconds", "time_to_restore_hours"], | |
| 308 | rates: &["failed"], | |
| 309 | }, | |
| 310 | DatasetSpec { | |
| 311 | id: DatasetId::Spend, | |
| 312 | label: "Spend", | |
| 313 | service: DatasetService::Billing, | |
| 314 | needs: DatasetNeeds::Billing, | |
| 315 | times: &["day"], | |
| 316 | dimensions: &["product", "project", "person", "model"], | |
| 317 | measures: &["amount_micros"], | |
| 318 | rates: &[], | |
| 319 | }, | |
| 320 | DatasetSpec { | |
| 321 | id: DatasetId::AgentSessions, | |
| 322 | label: "Agent sessions", | |
| 323 | service: DatasetService::Agents, | |
| 324 | needs: DatasetNeeds::Member, | |
| 325 | times: &["started_at"], | |
| 326 | dimensions: &["agent", "repo", "outcome", "model", "trigger"], | |
| 327 | measures: &["duration_seconds", "cost_micros", "tokens"], | |
| 328 | rates: &["succeeded"], | |
| 329 | }, | |
| 330 | ]; | |
| 331 | ||
| 332 | /// The most rows a query returns. | |
| 333 | pub const MAX_ROWS: u32 = 100; | |
| 334 | /// The most filters on one query. | |
| 335 | pub const MAX_FILTERS: usize = 10; | |
| 336 | /// The most values in an `in` filter. | |
| 337 | pub const MAX_IN_VALUES: usize = 50; | |
| 338 | /// The longest `{ from, to }` range, in days. | |
| 339 | pub const MAX_RANGE_DAYS: u64 = 366; | |
| 340 | ||
| 341 | const DAY_MS: u64 = 86_400_000; | |
| 342 | ||
| 343 | /// A range end as milliseconds since the epoch: a real date (`2026-10-01`) | |
| 344 | /// or an RFC 3339 UTC time with at most milliseconds. `None` otherwise. | |
| 345 | pub fn range_time(text: &str) -> Option<u64> { | |
| 346 | let bytes = text.as_bytes(); | |
| 347 | let digits = |range: std::ops::Range<usize>| -> Option<u64> { | |
| 348 | let part = text.get(range)?; | |
| 349 | if part.is_empty() || !part.bytes().all(|b| b.is_ascii_digit()) { | |
| 350 | return None; | |
| 351 | } | |
| 352 | part.parse().ok() | |
| 353 | }; | |
| 354 | if bytes.len() < 10 || bytes[4] != b'-' || bytes[7] != b'-' { | |
| 355 | return None; | |
| 356 | } | |
| 357 | let (year, month, day) = (digits(0..4)?, digits(5..7)?, digits(8..10)?); | |
| 358 | if !(1..=12).contains(&month) || day < 1 || day > days_in_month(year, month) { | |
| 359 | return None; | |
| 360 | } | |
| 361 | let full = if bytes.len() == 10 { | |
| 362 | format!("{text}T00:00:00Z") | |
| 363 | } else { | |
| 364 | // T hh:mm:ss, optional .f{1,3}, Z. | |
| 365 | if bytes.len() < 20 || bytes[10] != b'T' || bytes[13] != b':' || bytes[16] != b':' || *bytes.last()? != b'Z' { | |
| 366 | return None; | |
| 367 | } | |
| 368 | let (hour, minute, second) = (digits(11..13)?, digits(14..16)?, digits(17..19)?); | |
| 369 | if hour > 23 || minute > 59 || second > 59 { | |
| 370 | return None; | |
| 371 | } | |
| 372 | match bytes.len() { | |
| 373 | 20 => {} | |
| 374 | 22..=24 if bytes[19] == b'.' => { | |
| 375 | digits(20..bytes.len() - 1)?; | |
| 376 | } | |
| 377 | _ => return None, | |
| 378 | } | |
| 379 | text.to_owned() | |
| 380 | }; | |
| 381 | crate::time::parse_rfc3339(&full) | |
| 382 | } | |
| 383 | ||
| 384 | fn days_in_month(year: u64, month: u64) -> u64 { | |
| 385 | match month { | |
| 386 | 2 if (year % 4 == 0 && year % 100 != 0) || year % 400 == 0 => 29, | |
| 387 | 2 => 28, | |
| 388 | 4 | 6 | 9 | 11 => 30, | |
| 389 | _ => 31, | |
| 390 | } | |
| 391 | } | |
| 392 | ||
| 393 | impl DatasetQuery { | |
| 394 | /// What is wrong with it, if anything: the same rules, and the same | |
| 395 | /// words, as `datasetQueryError` in TypeScript. It checks the query | |
| 396 | /// against the catalog only; who may run it is the owning service's to | |
| 397 | /// decide. | |
| 398 | pub fn validate(&self) -> Result<(), String> { | |
| 399 | let spec = self.dataset.spec(); | |
| 400 | let name = self.dataset.as_str(); | |
| 401 | let field = self.measure.field.as_deref(); | |
| 402 | match self.measure.op { | |
| 403 | MeasureOp::Count => { | |
| 404 | if field.is_some() { | |
| 405 | return Err("count takes no field.".to_owned()); | |
| 406 | } | |
| 407 | } | |
| 408 | MeasureOp::Rate => { | |
| 409 | let Some(field) = field else { return Err("rate needs a field.".to_owned()) }; | |
| 410 | if !spec.rates.contains(&field) { | |
| 411 | return Err(format!("{name} has no yes-or-no field {field} to take the rate of.")); | |
| 412 | } | |
| 413 | } | |
| 414 | op => { | |
| 415 | let Some(field) = field else { return Err(format!("{} needs a field.", op.as_str())) }; | |
| 416 | if !spec.measures.contains(&field) { | |
| 417 | return Err(format!("{name} has no number field {field}.")); | |
| 418 | } | |
| 419 | } | |
| 420 | } | |
| 421 | if let Some(group_by) = self.group_by.as_deref() | |
| 422 | && !spec.dimensions.contains(&group_by) | |
| 423 | { | |
| 424 | return Err(format!("{name} can't be grouped by {group_by}.")); | |
| 425 | } | |
| 426 | if let Some(time) = self.time.as_deref() | |
| 427 | && !spec.times.contains(&time) | |
| 428 | { | |
| 429 | return Err(format!("{name} has no time field {time}.")); | |
| 430 | } | |
| 431 | let filters = self.filters.as_deref().unwrap_or_default(); | |
| 432 | if filters.len() > MAX_FILTERS { | |
| 433 | return Err(format!("A query takes at most {MAX_FILTERS} filters.")); | |
| 434 | } | |
| 435 | for filter in filters { | |
| 436 | let (field, op) = (filter.field.as_str(), filter.op.as_str()); | |
| 437 | if spec.dimensions.contains(&field) { | |
| 438 | match (filter.op, &filter.value) { | |
| 439 | (FilterOp::In, FilterValue::List(values)) if !values.is_empty() && values.len() <= MAX_IN_VALUES => {} | |
| 440 | (FilterOp::In, _) => return Err(format!("in on {field} takes a list of 1 to {MAX_IN_VALUES} values.")), | |
| 441 | (FilterOp::Eq | FilterOp::Neq, FilterValue::Text(_)) => {} | |
| 442 | (FilterOp::Eq | FilterOp::Neq, _) => return Err(format!("{op} on {field} takes text.")), | |
| 443 | _ => return Err(format!("{field} is text: filter it with eq, neq or in.")), | |
| 444 | } | |
| 445 | } else if spec.measures.contains(&field) { | |
| 446 | match (filter.op, &filter.value) { | |
| 447 | (FilterOp::In, _) => return Err(format!("{field} is a number: filter it with eq, neq, gte or lte.")), | |
| 448 | (_, FilterValue::Number(n)) if n.is_finite() => {} | |
| 449 | _ => return Err(format!("{op} on {field} takes a number.")), | |
| 450 | } | |
| 451 | } else { | |
| 452 | return Err(format!("{name} can't be filtered by {field}.")); | |
| 453 | } | |
| 454 | } | |
| 455 | if let Some(DatasetRange::Between { from, to }) = &self.range { | |
| 456 | let (Some(from), Some(to)) = (range_time(from), range_time(to)) else { | |
| 457 | return Err("A range's from and to are dates or RFC 3339 UTC times.".to_owned()); | |
| 458 | }; | |
| 459 | if from >= to { | |
| 460 | return Err("A range's from comes before its to.".to_owned()); | |
| 461 | } | |
| 462 | if to - from > MAX_RANGE_DAYS * DAY_MS { | |
| 463 | return Err(format!("A range is at most {MAX_RANGE_DAYS} days.")); | |
| 464 | } | |
| 465 | } | |
| 466 | if let Some(limit) = self.limit | |
| 467 | && !(1..=MAX_ROWS).contains(&limit) | |
| 468 | { | |
| 469 | return Err(format!("limit is between 1 and {MAX_ROWS}.")); | |
| 470 | } | |
| 471 | Ok(()) | |
| 472 | } | |
| 473 | } | |
| 474 | ||
| 475 | #[cfg(test)] | |
| 476 | mod tests { | |
| 477 | use super::*; | |
| 478 | ||
| 479 | /// The two validators agree on every case in the shared fixtures. | |
| 480 | #[test] | |
| 481 | fn agrees_with_the_typescript_validator_on_the_fixtures() { | |
| 482 | let fixtures: serde_json::Value = serde_json::from_str(include_str!("../../../packages/contracts/src/datasets.fixtures.json")).unwrap(); | |
| 483 | let cases = fixtures["cases"].as_array().unwrap(); | |
| 484 | assert!(cases.len() > 20); | |
| 485 | for case in cases { | |
| 486 | let outcome = serde_json::from_value::<DatasetQuery>(case["query"].clone()) | |
| 487 | .map_err(|_| "*".to_owned()) | |
| 488 | .and_then(|query| query.validate()); | |
| 489 | match case["error"].as_str() { | |
| 490 | None => assert_eq!(outcome, Ok(()), "{}", case["query"]), | |
| 491 | Some("*") => assert!(outcome.is_err(), "{} should be refused", case["query"]), | |
| 492 | Some(error) => assert_eq!(outcome, Err(error.to_owned()), "{}", case["query"]), | |
| 493 | } | |
| 494 | } | |
| 495 | } | |
| 496 | ||
| 497 | /// The site's copy of the catalog names the same datasets, services and | |
| 498 | /// fields, in the same order. | |
| 499 | #[test] | |
| 500 | fn the_typescript_mirror_has_the_same_catalog() { | |
| 501 | let ts = include_str!("../../../packages/contracts/src/datasets.ts"); | |
| 502 | let catalog = ts | |
| 503 | .split_once("export const DATASETS: Record<DatasetId, DatasetSpec> = {") | |
| 504 | .and_then(|(_, rest)| rest.split_once("\n};")) | |
| 505 | .map(|(table, _)| table) | |
| 506 | .expect("DATASETS in datasets.ts"); | |
| 507 | let quoted = |line: &str, key: &str| -> Vec<String> { | |
| 508 | let start = format!("{key}: ["); | |
| 509 | let list = line.split_once(&start).and_then(|(_, rest)| rest.split_once(']')).map(|(list, _)| list).unwrap_or_else(|| panic!("{key} in {line}")); | |
| 510 | list.split('"').skip(1).step_by(2).map(str::to_owned).collect() | |
| 511 | }; | |
| 512 | let lines: Vec<&str> = catalog.lines().filter(|line| line.contains("service:")).collect(); | |
| 513 | assert_eq!(lines.len(), DATASETS.len()); | |
| 514 | for (line, spec) in lines.iter().zip(DATASETS.iter()) { | |
| 515 | assert!(line.trim_start().starts_with(&format!("{}: {{", spec.id.as_str())), "{line}"); | |
| 516 | assert!(line.contains(&format!("label: \"{}\"", spec.label)), "{line}"); | |
| 517 | assert!(line.contains(&format!("service: \"{}\"", spec.service.as_str())), "{line}"); | |
| 518 | assert!(line.contains(&format!("needs: \"{}\"", spec.needs.as_str())), "{line}"); | |
| 519 | assert_eq!(quoted(line, "times"), spec.times, "{line}"); | |
| 520 | assert_eq!(quoted(line, "dimensions"), spec.dimensions, "{line}"); | |
| 521 | assert_eq!(quoted(line, "measures"), spec.measures, "{line}"); | |
| 522 | assert_eq!(quoted(line, "rates"), spec.rates, "{line}"); | |
| 523 | } | |
| 524 | for (name, value) in [("DATASET_MAX_ROWS", MAX_ROWS as usize), ("DATASET_MAX_FILTERS", MAX_FILTERS), ("DATASET_MAX_IN_VALUES", MAX_IN_VALUES), ("DATASET_MAX_RANGE_DAYS", MAX_RANGE_DAYS as usize)] { | |
| 525 | assert!(ts.contains(&format!("export const {name} = {value};")), "{name}"); | |
| 526 | } | |
| 527 | } | |
| 528 | ||
| 529 | #[test] | |
| 530 | fn every_dataset_reads_back_and_has_a_time_field() { | |
| 531 | for id in DatasetId::ALL { | |
| 532 | assert_eq!(serde_json::to_value(id).unwrap(), id.as_str()); | |
| 533 | let spec = id.spec(); | |
| 534 | assert!(!spec.times.is_empty(), "{}", id.as_str()); | |
| 535 | // A field is one thing: never both a dimension and a number. | |
| 536 | for field in spec.dimensions { | |
| 537 | assert!(!spec.measures.contains(field) && !spec.rates.contains(field), "{field}"); | |
| 538 | } | |
| 539 | } | |
| 540 | assert_eq!(DatasetId::Spend.spec().needs, DatasetNeeds::Billing); | |
| 541 | } | |
| 542 | ||
| 543 | #[test] | |
| 544 | fn a_query_travels_snake_case() { | |
| 545 | let query = DatasetQuery { | |
| 546 | dataset: DatasetId::PullRequests, | |
| 547 | measure: Measure { op: MeasureOp::P95, field: Some("cycle_time_hours".to_owned()) }, | |
| 548 | group_by: Some("repo".to_owned()), | |
| 549 | interval: Some(Interval::Week), | |
| 550 | time: None, | |
| 551 | filters: Some(vec![Filter { field: "state".to_owned(), op: FilterOp::Eq, value: FilterValue::Text("merged".to_owned()) }]), | |
| 552 | range: Some(DatasetRange::Preset(RangePreset::Days90)), | |
| 553 | limit: None, | |
| 554 | }; | |
| 555 | let json = serde_json::to_value(&query).unwrap(); | |
| 556 | assert_eq!( | |
| 557 | json, | |
| 558 | serde_json::json!({ | |
| 559 | "dataset": "pull_requests", | |
| 560 | "measure": { "op": "p95", "field": "cycle_time_hours" }, | |
| 561 | "group_by": "repo", | |
| 562 | "interval": "week", | |
| 563 | "filters": [{ "field": "state", "op": "eq", "value": "merged" }], | |
| 564 | "range": "90d" | |
| 565 | }) | |
| 566 | ); | |
| 567 | assert_eq!(serde_json::from_value::<DatasetQuery>(json).unwrap(), query); | |
| 568 | let between: DatasetRange = serde_json::from_value(serde_json::json!({ "from": "2026-01-01", "to": "2026-02-01" })).unwrap(); | |
| 569 | assert!(matches!(between, DatasetRange::Between { .. })); | |
| 570 | } | |
| 571 | ||
| 572 | #[test] | |
| 573 | fn range_times_are_real_dates_or_utc_times() { | |
| 574 | assert_eq!(range_time("1970-01-02"), Some(DAY_MS)); | |
| 575 | assert_eq!(range_time("2026-10-02T05:16:19Z"), Some(1_790_918_179_000)); | |
| 576 | assert_eq!(range_time("2026-10-02T05:16:19.5Z"), Some(1_790_918_179_500)); | |
| 577 | assert_eq!(range_time("2024-02-29"), crate::time::parse_rfc3339("2024-02-29T00:00:00Z")); | |
| 578 | for bad in ["2026-02-29", "2026-04-31", "2026-13-01", "2026-10-02T24:00:00Z", "2026-10-02T05:16:19", "2026-10-02T05:16:19+02:00", "26-10-02", "2026-1-02", "today", ""] { | |
| 579 | assert_eq!(range_time(bad), None, "{bad}"); | |
| 580 | } | |
| 581 | } | |
| 582 | } |