| 1 | //! Datasets: the safe query layer behind dashboards (Artifacts mode, |
| 2 | //! docs/ARTIFACTS_MODE.md section 3.4). Mirrors |
| 3 | //! `packages/contracts/src/datasets.ts`; the tests here keep the catalog the |
| 4 | //! same and run both validators over `datasets.fixtures.json`. |
| 5 | //! |
| 6 | //! Not SQL: a query names a dataset from a declared catalog, one measure, |
| 7 | //! at most one dimension, an interval, filters on declared fields and a |
| 8 | //! range. The service that owns the data ([`DatasetSpec::service`]) answers |
| 9 | //! `query_dataset` for the viewer, over only what the viewer can read, with |
| 10 | //! a fixed query per measure and dimension, and caps the rows. The Rust |
| 11 | //! owners (work, actions, billing) validate with [`DatasetQuery::validate`] |
| 12 | //! before they run anything. |
| 13 | |
| 14 | use serde::{Deserialize, Serialize}; |
| 15 | |
| 16 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Serialize, Deserialize)] |
| 17 | #[serde(rename_all = "snake_case")] |
| 18 | pub enum DatasetId { |
| 19 | Issues, |
| 20 | PullRequests, |
| 21 | WorkflowRuns, |
| 22 | Deployments, |
| 23 | Spend, |
| 24 | AgentSessions, |
| 25 | } |
| 26 | |
| 27 | impl DatasetId { |
| 28 | pub const ALL: [DatasetId; 6] = [ |
| 29 | DatasetId::Issues, |
| 30 | DatasetId::PullRequests, |
| 31 | DatasetId::WorkflowRuns, |
| 32 | DatasetId::Deployments, |
| 33 | DatasetId::Spend, |
| 34 | DatasetId::AgentSessions, |
| 35 | ]; |
| 36 | |
| 37 | pub fn as_str(self) -> &'static str { |
| 38 | match self { |
| 39 | DatasetId::Issues => "issues", |
| 40 | DatasetId::PullRequests => "pull_requests", |
| 41 | DatasetId::WorkflowRuns => "workflow_runs", |
| 42 | DatasetId::Deployments => "deployments", |
| 43 | DatasetId::Spend => "spend", |
| 44 | DatasetId::AgentSessions => "agent_sessions", |
| 45 | } |
| 46 | } |
| 47 | |
| 48 | /// Its entry in [`DATASETS`]. |
| 49 | pub fn spec(self) -> &'static DatasetSpec { |
| 50 | DATASETS.iter().find(|spec| spec.id == self).expect("every dataset is in the catalog") |
| 51 | } |
| 52 | } |
| 53 | |
| 54 | /// How rows are summed up. `count` takes no field; `rate` a rate field; the |
| 55 | /// rest a measure field. |
| 56 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] |
| 57 | #[serde(rename_all = "snake_case")] |
| 58 | pub enum MeasureOp { |
| 59 | Count, |
| 60 | Sum, |
| 61 | Avg, |
| 62 | P50, |
| 63 | P95, |
| 64 | Rate, |
| 65 | } |
| 66 | |
| 67 | impl MeasureOp { |
| 68 | pub fn as_str(self) -> &'static str { |
| 69 | match self { |
| 70 | MeasureOp::Count => "count", |
| 71 | MeasureOp::Sum => "sum", |
| 72 | MeasureOp::Avg => "avg", |
| 73 | MeasureOp::P50 => "p50", |
| 74 | MeasureOp::P95 => "p95", |
| 75 | MeasureOp::Rate => "rate", |
| 76 | } |
| 77 | } |
| 78 | } |
| 79 | |
| 80 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] |
| 81 | pub struct Measure { |
| 82 | pub op: MeasureOp, |
| 83 | #[serde(default, skip_serializing_if = "Option::is_none")] |
| 84 | pub field: Option<String>, |
| 85 | } |
| 86 | |
| 87 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] |
| 88 | #[serde(rename_all = "snake_case")] |
| 89 | pub enum Interval { |
| 90 | Day, |
| 91 | Week, |
| 92 | Month, |
| 93 | } |
| 94 | |
| 95 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] |
| 96 | #[serde(rename_all = "snake_case")] |
| 97 | pub enum FilterOp { |
| 98 | Eq, |
| 99 | Neq, |
| 100 | In, |
| 101 | Gte, |
| 102 | Lte, |
| 103 | } |
| 104 | |
| 105 | impl FilterOp { |
| 106 | pub fn as_str(self) -> &'static str { |
| 107 | match self { |
| 108 | FilterOp::Eq => "eq", |
| 109 | FilterOp::Neq => "neq", |
| 110 | FilterOp::In => "in", |
| 111 | FilterOp::Gte => "gte", |
| 112 | FilterOp::Lte => "lte", |
| 113 | } |
| 114 | } |
| 115 | } |
| 116 | |
| 117 | /// Text for `eq`/`neq` on a dimension, a list for `in`, a number on a |
| 118 | /// measure. |
| 119 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] |
| 120 | #[serde(untagged)] |
| 121 | pub enum FilterValue { |
| 122 | Text(String), |
| 123 | Number(f64), |
| 124 | List(Vec<String>), |
| 125 | } |
| 126 | |
| 127 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] |
| 128 | pub struct Filter { |
| 129 | pub field: String, |
| 130 | pub op: FilterOp, |
| 131 | pub value: FilterValue, |
| 132 | } |
| 133 | |
| 134 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] |
| 135 | pub enum RangePreset { |
| 136 | #[serde(rename = "7d")] |
| 137 | Days7, |
| 138 | #[serde(rename = "30d")] |
| 139 | Days30, |
| 140 | #[serde(rename = "90d")] |
| 141 | Days90, |
| 142 | } |
| 143 | |
| 144 | impl RangePreset { |
| 145 | pub fn days(self) -> u32 { |
| 146 | match self { |
| 147 | RangePreset::Days7 => 7, |
| 148 | RangePreset::Days30 => 30, |
| 149 | RangePreset::Days90 => 90, |
| 150 | } |
| 151 | } |
| 152 | } |
| 153 | |
| 154 | /// A preset counted back from now, or between two times: dates |
| 155 | /// (`2026-10-01`) or RFC 3339 UTC times; `to` is exclusive. |
| 156 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] |
| 157 | #[serde(untagged)] |
| 158 | pub enum DatasetRange { |
| 159 | Preset(RangePreset), |
| 160 | Between { from: String, to: String }, |
| 161 | } |
| 162 | |
| 163 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] |
| 164 | pub struct DatasetQuery { |
| 165 | pub dataset: DatasetId, |
| 166 | pub measure: Measure, |
| 167 | #[serde(default, skip_serializing_if = "Option::is_none")] |
| 168 | pub group_by: Option<String>, |
| 169 | #[serde(default, skip_serializing_if = "Option::is_none")] |
| 170 | pub interval: Option<Interval>, |
| 171 | /// Which of the dataset's time fields the range and interval use; its |
| 172 | /// first when absent. |
| 173 | #[serde(default, skip_serializing_if = "Option::is_none")] |
| 174 | pub time: Option<String>, |
| 175 | #[serde(default, skip_serializing_if = "Option::is_none")] |
| 176 | pub filters: Option<Vec<Filter>>, |
| 177 | #[serde(default, skip_serializing_if = "Option::is_none")] |
| 178 | pub range: Option<DatasetRange>, |
| 179 | #[serde(default, skip_serializing_if = "Option::is_none")] |
| 180 | pub limit: Option<u32>, |
| 181 | } |
| 182 | |
| 183 | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] |
| 184 | #[serde(rename_all = "snake_case")] |
| 185 | pub enum ColumnType { |
| 186 | String, |
| 187 | Number, |
| 188 | Time, |
| 189 | Money, |
| 190 | } |
| 191 | |
| 192 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] |
| 193 | pub struct DatasetColumn { |
| 194 | pub name: String, |
| 195 | #[serde(rename = "type")] |
| 196 | pub kind: ColumnType, |
| 197 | } |
| 198 | |
| 199 | /// What `query_dataset` answers. |
| 200 | #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)] |
| 201 | pub struct DatasetResult { |
| 202 | pub columns: Vec<DatasetColumn>, |
| 203 | /// Each a string, a number or null. |
| 204 | pub rows: Vec<Vec<serde_json::Value>>, |
| 205 | /// More rows matched than were returned. |
| 206 | pub truncated: bool, |
| 207 | /// The viewer cannot read everything the query covers. |
| 208 | pub partial: bool, |
| 209 | /// RFC 3339. |
| 210 | pub as_of: String, |
| 211 | } |
| 212 | |
| 213 | /// The service that owns a dataset. |
| 214 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] |
| 215 | pub enum DatasetService { |
| 216 | Work, |
| 217 | Actions, |
| 218 | Deployments, |
| 219 | Billing, |
| 220 | Agents, |
| 221 | } |
| 222 | |
| 223 | impl DatasetService { |
| 224 | pub fn as_str(self) -> &'static str { |
| 225 | match self { |
| 226 | DatasetService::Work => "work", |
| 227 | DatasetService::Actions => "actions", |
| 228 | DatasetService::Deployments => "deployments", |
| 229 | DatasetService::Billing => "billing", |
| 230 | DatasetService::Agents => "agents", |
| 231 | } |
| 232 | } |
| 233 | } |
| 234 | |
| 235 | /// What a viewer needs to query a dataset. |
| 236 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] |
| 237 | pub enum DatasetNeeds { |
| 238 | Member, |
| 239 | /// The workspace's billing role. |
| 240 | Billing, |
| 241 | } |
| 242 | |
| 243 | impl DatasetNeeds { |
| 244 | pub fn as_str(self) -> &'static str { |
| 245 | match self { |
| 246 | DatasetNeeds::Member => "member", |
| 247 | DatasetNeeds::Billing => "billing", |
| 248 | } |
| 249 | } |
| 250 | } |
| 251 | |
| 252 | #[derive(Debug)] |
| 253 | pub struct DatasetSpec { |
| 254 | pub id: DatasetId, |
| 255 | pub label: &'static str, |
| 256 | pub service: DatasetService, |
| 257 | pub needs: DatasetNeeds, |
| 258 | /// Time fields, the default first. |
| 259 | pub times: &'static [&'static str], |
| 260 | /// Text fields to group and filter by. |
| 261 | pub dimensions: &'static [&'static str], |
| 262 | /// Number fields for sum, avg, p50, p95, and number filters. |
| 263 | pub measures: &'static [&'static str], |
| 264 | /// Yes-or-no fields `rate` gives the share of. |
| 265 | pub rates: &'static [&'static str], |
| 266 | } |
| 267 | |
| 268 | /// The catalog. |
| 269 | pub const DATASETS: [DatasetSpec; 6] = [ |
| 270 | DatasetSpec { |
| 271 | id: DatasetId::Issues, |
| 272 | label: "Issues", |
| 273 | service: DatasetService::Work, |
| 274 | needs: DatasetNeeds::Member, |
| 275 | times: &["created_at", "closed_at"], |
| 276 | dimensions: &["repo", "label", "state", "author_kind", "assignee_kind", "milestone"], |
| 277 | measures: &["time_to_close_hours", "comments"], |
| 278 | rates: &["closed"], |
| 279 | }, |
| 280 | DatasetSpec { |
| 281 | id: DatasetId::PullRequests, |
| 282 | label: "Pull requests", |
| 283 | service: DatasetService::Work, |
| 284 | needs: DatasetNeeds::Member, |
| 285 | times: &["created_at", "merged_at", "closed_at"], |
| 286 | dimensions: &["repo", "label", "state", "author_kind", "base_branch"], |
| 287 | measures: &["cycle_time_hours", "time_to_first_review_hours", "review_count", "additions", "deletions", "changed_files"], |
| 288 | rates: &["merged"], |
| 289 | }, |
| 290 | DatasetSpec { |
| 291 | id: DatasetId::WorkflowRuns, |
| 292 | label: "Workflow runs", |
| 293 | service: DatasetService::Actions, |
| 294 | needs: DatasetNeeds::Member, |
| 295 | times: &["started_at", "completed_at"], |
| 296 | dimensions: &["repo", "workflow", "branch", "event", "conclusion", "runner_kind"], |
| 297 | measures: &["duration_seconds", "queue_seconds"], |
| 298 | rates: &["succeeded"], |
| 299 | }, |
| 300 | DatasetSpec { |
| 301 | id: DatasetId::Deployments, |
| 302 | label: "Deployments", |
| 303 | service: DatasetService::Deployments, |
| 304 | needs: DatasetNeeds::Member, |
| 305 | times: &["created_at"], |
| 306 | dimensions: &["repo", "project", "environment", "state"], |
| 307 | measures: &["duration_seconds", "time_to_restore_hours"], |
| 308 | rates: &["failed"], |
| 309 | }, |
| 310 | DatasetSpec { |
| 311 | id: DatasetId::Spend, |
| 312 | label: "Spend", |
| 313 | service: DatasetService::Billing, |
| 314 | needs: DatasetNeeds::Billing, |
| 315 | times: &["day"], |
| 316 | dimensions: &["product", "project", "person", "model"], |
| 317 | measures: &["amount_micros"], |
| 318 | rates: &[], |
| 319 | }, |
| 320 | DatasetSpec { |
| 321 | id: DatasetId::AgentSessions, |
| 322 | label: "Agent sessions", |
| 323 | service: DatasetService::Agents, |
| 324 | needs: DatasetNeeds::Member, |
| 325 | times: &["started_at"], |
| 326 | dimensions: &["agent", "repo", "outcome", "model", "trigger"], |
| 327 | measures: &["duration_seconds", "cost_micros", "tokens"], |
| 328 | rates: &["succeeded"], |
| 329 | }, |
| 330 | ]; |
| 331 | |
| 332 | /// The most rows a query returns. |
| 333 | pub const MAX_ROWS: u32 = 100; |
| 334 | /// The most filters on one query. |
| 335 | pub const MAX_FILTERS: usize = 10; |
| 336 | /// The most values in an `in` filter. |
| 337 | pub const MAX_IN_VALUES: usize = 50; |
| 338 | /// The longest `{ from, to }` range, in days. |
| 339 | pub const MAX_RANGE_DAYS: u64 = 366; |
| 340 | |
| 341 | const DAY_MS: u64 = 86_400_000; |
| 342 | |
| 343 | /// A range end as milliseconds since the epoch: a real date (`2026-10-01`) |
| 344 | /// or an RFC 3339 UTC time with at most milliseconds. `None` otherwise. |
| 345 | pub fn range_time(text: &str) -> Option<u64> { |
| 346 | let bytes = text.as_bytes(); |
| 347 | let digits = |range: std::ops::Range<usize>| -> Option<u64> { |
| 348 | let part = text.get(range)?; |
| 349 | if part.is_empty() || !part.bytes().all(|b| b.is_ascii_digit()) { |
| 350 | return None; |
| 351 | } |
| 352 | part.parse().ok() |
| 353 | }; |
| 354 | if bytes.len() < 10 || bytes[4] != b'-' || bytes[7] != b'-' { |
| 355 | return None; |
| 356 | } |
| 357 | let (year, month, day) = (digits(0..4)?, digits(5..7)?, digits(8..10)?); |
| 358 | if !(1..=12).contains(&month) || day < 1 || day > days_in_month(year, month) { |
| 359 | return None; |
| 360 | } |
| 361 | let full = if bytes.len() == 10 { |
| 362 | format!("{text}T00:00:00Z") |
| 363 | } else { |
| 364 | // T hh:mm:ss, optional .f{1,3}, Z. |
| 365 | if bytes.len() < 20 || bytes[10] != b'T' || bytes[13] != b':' || bytes[16] != b':' || *bytes.last()? != b'Z' { |
| 366 | return None; |
| 367 | } |
| 368 | let (hour, minute, second) = (digits(11..13)?, digits(14..16)?, digits(17..19)?); |
| 369 | if hour > 23 || minute > 59 || second > 59 { |
| 370 | return None; |
| 371 | } |
| 372 | match bytes.len() { |
| 373 | 20 => {} |
| 374 | 22..=24 if bytes[19] == b'.' => { |
| 375 | digits(20..bytes.len() - 1)?; |
| 376 | } |
| 377 | _ => return None, |
| 378 | } |
| 379 | text.to_owned() |
| 380 | }; |
| 381 | crate::time::parse_rfc3339(&full) |
| 382 | } |
| 383 | |
| 384 | fn days_in_month(year: u64, month: u64) -> u64 { |
| 385 | match month { |
| 386 | 2 if (year % 4 == 0 && year % 100 != 0) || year % 400 == 0 => 29, |
| 387 | 2 => 28, |
| 388 | 4 | 6 | 9 | 11 => 30, |
| 389 | _ => 31, |
| 390 | } |
| 391 | } |
| 392 | |
| 393 | impl DatasetQuery { |
| 394 | /// What is wrong with it, if anything: the same rules, and the same |
| 395 | /// words, as `datasetQueryError` in TypeScript. It checks the query |
| 396 | /// against the catalog only; who may run it is the owning service's to |
| 397 | /// decide. |
| 398 | pub fn validate(&self) -> Result<(), String> { |
| 399 | let spec = self.dataset.spec(); |
| 400 | let name = self.dataset.as_str(); |
| 401 | let field = self.measure.field.as_deref(); |
| 402 | match self.measure.op { |
| 403 | MeasureOp::Count => { |
| 404 | if field.is_some() { |
| 405 | return Err("count takes no field.".to_owned()); |
| 406 | } |
| 407 | } |
| 408 | MeasureOp::Rate => { |
| 409 | let Some(field) = field else { return Err("rate needs a field.".to_owned()) }; |
| 410 | if !spec.rates.contains(&field) { |
| 411 | return Err(format!("{name} has no yes-or-no field {field} to take the rate of.")); |
| 412 | } |
| 413 | } |
| 414 | op => { |
| 415 | let Some(field) = field else { return Err(format!("{} needs a field.", op.as_str())) }; |
| 416 | if !spec.measures.contains(&field) { |
| 417 | return Err(format!("{name} has no number field {field}.")); |
| 418 | } |
| 419 | } |
| 420 | } |
| 421 | if let Some(group_by) = self.group_by.as_deref() |
| 422 | && !spec.dimensions.contains(&group_by) |
| 423 | { |
| 424 | return Err(format!("{name} can't be grouped by {group_by}.")); |
| 425 | } |
| 426 | if let Some(time) = self.time.as_deref() |
| 427 | && !spec.times.contains(&time) |
| 428 | { |
| 429 | return Err(format!("{name} has no time field {time}.")); |
| 430 | } |
| 431 | let filters = self.filters.as_deref().unwrap_or_default(); |
| 432 | if filters.len() > MAX_FILTERS { |
| 433 | return Err(format!("A query takes at most {MAX_FILTERS} filters.")); |
| 434 | } |
| 435 | for filter in filters { |
| 436 | let (field, op) = (filter.field.as_str(), filter.op.as_str()); |
| 437 | if spec.dimensions.contains(&field) { |
| 438 | match (filter.op, &filter.value) { |
| 439 | (FilterOp::In, FilterValue::List(values)) if !values.is_empty() && values.len() <= MAX_IN_VALUES => {} |
| 440 | (FilterOp::In, _) => return Err(format!("in on {field} takes a list of 1 to {MAX_IN_VALUES} values.")), |
| 441 | (FilterOp::Eq | FilterOp::Neq, FilterValue::Text(_)) => {} |
| 442 | (FilterOp::Eq | FilterOp::Neq, _) => return Err(format!("{op} on {field} takes text.")), |
| 443 | _ => return Err(format!("{field} is text: filter it with eq, neq or in.")), |
| 444 | } |
| 445 | } else if spec.measures.contains(&field) { |
| 446 | match (filter.op, &filter.value) { |
| 447 | (FilterOp::In, _) => return Err(format!("{field} is a number: filter it with eq, neq, gte or lte.")), |
| 448 | (_, FilterValue::Number(n)) if n.is_finite() => {} |
| 449 | _ => return Err(format!("{op} on {field} takes a number.")), |
| 450 | } |
| 451 | } else { |
| 452 | return Err(format!("{name} can't be filtered by {field}.")); |
| 453 | } |
| 454 | } |
| 455 | if let Some(DatasetRange::Between { from, to }) = &self.range { |
| 456 | let (Some(from), Some(to)) = (range_time(from), range_time(to)) else { |
| 457 | return Err("A range's from and to are dates or RFC 3339 UTC times.".to_owned()); |
| 458 | }; |
| 459 | if from >= to { |
| 460 | return Err("A range's from comes before its to.".to_owned()); |
| 461 | } |
| 462 | if to - from > MAX_RANGE_DAYS * DAY_MS { |
| 463 | return Err(format!("A range is at most {MAX_RANGE_DAYS} days.")); |
| 464 | } |
| 465 | } |
| 466 | if let Some(limit) = self.limit |
| 467 | && !(1..=MAX_ROWS).contains(&limit) |
| 468 | { |
| 469 | return Err(format!("limit is between 1 and {MAX_ROWS}.")); |
| 470 | } |
| 471 | Ok(()) |
| 472 | } |
| 473 | } |
| 474 | |
| 475 | #[cfg(test)] |
| 476 | mod tests { |
| 477 | use super::*; |
| 478 | |
| 479 | /// The two validators agree on every case in the shared fixtures. |
| 480 | #[test] |
| 481 | fn agrees_with_the_typescript_validator_on_the_fixtures() { |
| 482 | let fixtures: serde_json::Value = serde_json::from_str(include_str!("../../../packages/contracts/src/datasets.fixtures.json")).unwrap(); |
| 483 | let cases = fixtures["cases"].as_array().unwrap(); |
| 484 | assert!(cases.len() > 20); |
| 485 | for case in cases { |
| 486 | let outcome = serde_json::from_value::<DatasetQuery>(case["query"].clone()) |
| 487 | .map_err(|_| "*".to_owned()) |
| 488 | .and_then(|query| query.validate()); |
| 489 | match case["error"].as_str() { |
| 490 | None => assert_eq!(outcome, Ok(()), "{}", case["query"]), |
| 491 | Some("*") => assert!(outcome.is_err(), "{} should be refused", case["query"]), |
| 492 | Some(error) => assert_eq!(outcome, Err(error.to_owned()), "{}", case["query"]), |
| 493 | } |
| 494 | } |
| 495 | } |
| 496 | |
| 497 | /// The site's copy of the catalog names the same datasets, services and |
| 498 | /// fields, in the same order. |
| 499 | #[test] |
| 500 | fn the_typescript_mirror_has_the_same_catalog() { |
| 501 | let ts = include_str!("../../../packages/contracts/src/datasets.ts"); |
| 502 | let catalog = ts |
| 503 | .split_once("export const DATASETS: Record<DatasetId, DatasetSpec> = {") |
| 504 | .and_then(|(_, rest)| rest.split_once("\n};")) |
| 505 | .map(|(table, _)| table) |
| 506 | .expect("DATASETS in datasets.ts"); |
| 507 | let quoted = |line: &str, key: &str| -> Vec<String> { |
| 508 | let start = format!("{key}: ["); |
| 509 | let list = line.split_once(&start).and_then(|(_, rest)| rest.split_once(']')).map(|(list, _)| list).unwrap_or_else(|| panic!("{key} in {line}")); |
| 510 | list.split('"').skip(1).step_by(2).map(str::to_owned).collect() |
| 511 | }; |
| 512 | let lines: Vec<&str> = catalog.lines().filter(|line| line.contains("service:")).collect(); |
| 513 | assert_eq!(lines.len(), DATASETS.len()); |
| 514 | for (line, spec) in lines.iter().zip(DATASETS.iter()) { |
| 515 | assert!(line.trim_start().starts_with(&format!("{}: {{", spec.id.as_str())), "{line}"); |
| 516 | assert!(line.contains(&format!("label: \"{}\"", spec.label)), "{line}"); |
| 517 | assert!(line.contains(&format!("service: \"{}\"", spec.service.as_str())), "{line}"); |
| 518 | assert!(line.contains(&format!("needs: \"{}\"", spec.needs.as_str())), "{line}"); |
| 519 | assert_eq!(quoted(line, "times"), spec.times, "{line}"); |
| 520 | assert_eq!(quoted(line, "dimensions"), spec.dimensions, "{line}"); |
| 521 | assert_eq!(quoted(line, "measures"), spec.measures, "{line}"); |
| 522 | assert_eq!(quoted(line, "rates"), spec.rates, "{line}"); |
| 523 | } |
| 524 | for (name, value) in [("DATASET_MAX_ROWS", MAX_ROWS as usize), ("DATASET_MAX_FILTERS", MAX_FILTERS), ("DATASET_MAX_IN_VALUES", MAX_IN_VALUES), ("DATASET_MAX_RANGE_DAYS", MAX_RANGE_DAYS as usize)] { |
| 525 | assert!(ts.contains(&format!("export const {name} = {value};")), "{name}"); |
| 526 | } |
| 527 | } |
| 528 | |
| 529 | #[test] |
| 530 | fn every_dataset_reads_back_and_has_a_time_field() { |
| 531 | for id in DatasetId::ALL { |
| 532 | assert_eq!(serde_json::to_value(id).unwrap(), id.as_str()); |
| 533 | let spec = id.spec(); |
| 534 | assert!(!spec.times.is_empty(), "{}", id.as_str()); |
| 535 | // A field is one thing: never both a dimension and a number. |
| 536 | for field in spec.dimensions { |
| 537 | assert!(!spec.measures.contains(field) && !spec.rates.contains(field), "{field}"); |
| 538 | } |
| 539 | } |
| 540 | assert_eq!(DatasetId::Spend.spec().needs, DatasetNeeds::Billing); |
| 541 | } |
| 542 | |
| 543 | #[test] |
| 544 | fn a_query_travels_snake_case() { |
| 545 | let query = DatasetQuery { |
| 546 | dataset: DatasetId::PullRequests, |
| 547 | measure: Measure { op: MeasureOp::P95, field: Some("cycle_time_hours".to_owned()) }, |
| 548 | group_by: Some("repo".to_owned()), |
| 549 | interval: Some(Interval::Week), |
| 550 | time: None, |
| 551 | filters: Some(vec![Filter { field: "state".to_owned(), op: FilterOp::Eq, value: FilterValue::Text("merged".to_owned()) }]), |
| 552 | range: Some(DatasetRange::Preset(RangePreset::Days90)), |
| 553 | limit: None, |
| 554 | }; |
| 555 | let json = serde_json::to_value(&query).unwrap(); |
| 556 | assert_eq!( |
| 557 | json, |
| 558 | serde_json::json!({ |
| 559 | "dataset": "pull_requests", |
| 560 | "measure": { "op": "p95", "field": "cycle_time_hours" }, |
| 561 | "group_by": "repo", |
| 562 | "interval": "week", |
| 563 | "filters": [{ "field": "state", "op": "eq", "value": "merged" }], |
| 564 | "range": "90d" |
| 565 | }) |
| 566 | ); |
| 567 | assert_eq!(serde_json::from_value::<DatasetQuery>(json).unwrap(), query); |
| 568 | let between: DatasetRange = serde_json::from_value(serde_json::json!({ "from": "2026-01-01", "to": "2026-02-01" })).unwrap(); |
| 569 | assert!(matches!(between, DatasetRange::Between { .. })); |
| 570 | } |
| 571 | |
| 572 | #[test] |
| 573 | fn range_times_are_real_dates_or_utc_times() { |
| 574 | assert_eq!(range_time("1970-01-02"), Some(DAY_MS)); |
| 575 | assert_eq!(range_time("2026-10-02T05:16:19Z"), Some(1_790_918_179_000)); |
| 576 | assert_eq!(range_time("2026-10-02T05:16:19.5Z"), Some(1_790_918_179_500)); |
| 577 | assert_eq!(range_time("2024-02-29"), crate::time::parse_rfc3339("2024-02-29T00:00:00Z")); |
| 578 | for bad in ["2026-02-29", "2026-04-31", "2026-13-01", "2026-10-02T24:00:00Z", "2026-10-02T05:16:19", "2026-10-02T05:16:19+02:00", "26-10-02", "2026-1-02", "today", ""] { |
| 579 | assert_eq!(range_time(bad), None, "{bad}"); |
| 580 | } |
| 581 | } |
| 582 | } |