Skip to content
251 linesCodeBlameRaw

Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.

Artifacts contracts: folios, their four kinds, dashboard datasets, folio events and the artifacts scopes are typed and validated the same in TypeScript and Rust, with nothing using them yet1/**
2 * Datasets: the safe query layer behind dashboards (Artifacts mode,
3 * docs/ARTIFACTS_MODE.md, section 3.4). Mirrors
4 * `crates/contracts/src/datasets.rs`; a Rust test keeps the catalog the
5 * same and runs both validators over `datasets.fixtures.json`.
6 *
7 * Not SQL. A query names a dataset from a declared catalog, one measure,
8 * at most one dimension to group by, a time interval, filters on declared
9 * fields and a range. The service that owns the data (`service` in the
10 * catalog) runs it for the viewer, over only what the viewer can read, with
11 * a fixed query per measure and dimension, and caps the rows.
12 *
13 * Values never live in a folio: not in its Yjs document, versions, text,
14 * preview, index or templates. Only the query does.
15 *
16 * Wire shapes are snake_case.
17 */
18
19export type DatasetId = "issues" | "pull_requests" | "workflow_runs" | "deployments" | "spend" | "agent_sessions";
20
21export const DATASET_IDS: readonly DatasetId[] = ["issues", "pull_requests", "workflow_runs", "deployments", "spend", "agent_sessions"];
22
23/** How rows are summed up. `count` takes no field; `rate` takes a rate field; the rest a measure field. */
24export type DatasetMeasureOp = "count" | "sum" | "avg" | "p50" | "p95" | "rate";
25export const DATASET_MEASURE_OPS: readonly DatasetMeasureOp[] = ["count", "sum", "avg", "p50", "p95", "rate"];
26
27export type DatasetInterval = "day" | "week" | "month";
28export const DATASET_INTERVALS: readonly DatasetInterval[] = ["day", "week", "month"];
29
30export type DatasetFilterOp = "eq" | "neq" | "in" | "gte" | "lte";
31export const DATASET_FILTER_OPS: readonly DatasetFilterOp[] = ["eq", "neq", "in", "gte", "lte"];
32
33export type DatasetRangePreset = "7d" | "30d" | "90d";
34export const DATASET_RANGE_PRESETS: readonly DatasetRangePreset[] = ["7d", "30d", "90d"];
35
36/**
37 * A time range: a preset counted back from now, or between two times.
38 * `from` and `to` are dates (`2026-10-01`) or RFC 3339 UTC times
39 * (`2026-10-01T09:00:00Z`); `to` is exclusive.
40 */
41export type DatasetRange = DatasetRangePreset | { from: string; to: string };
42
43export type DatasetFilter = {
44 field: string;
45 op: DatasetFilterOp;
46 /** Text for `eq`/`neq` on a dimension, a list for `in`, a number on a measure. */
47 value: string | number | string[];
48};
49
50export type DatasetQuery = {
51 dataset: DatasetId;
52 measure: { op: DatasetMeasureOp; field?: string | null };
53 /** A declared dimension only. */
54 group_by?: string | null;
55 /** A time series, over the dataset's time field. */
56 interval?: DatasetInterval | null;
57 /** Which of the dataset's time fields the range and interval use; its first when left out. */
58 time?: string | null;
59 filters?: DatasetFilter[] | null;
60 /** The tile's own range; else the dashboard's. */
61 range?: DatasetRange | null;
62 /** At most `DATASET_MAX_ROWS`. */
63 limit?: number | null;
64};
65
66export type DatasetColumnType = "string" | "number" | "time" | "money";
67
68export type DatasetResult = {
69 columns: { name: string; type: DatasetColumnType }[];
70 rows: (string | number | null)[][];
71 /** More rows matched than were returned. */
72 truncated: boolean;
73 /** The viewer cannot read everything the query covers: shown as "Based on what you can see". */
74 partial: boolean;
75 /** When the numbers were computed, RFC 3339. */
76 as_of: string;
77};
78
79/** The services that own datasets, each answering `query_dataset` for its own. */
80export type DatasetService = "work" | "actions" | "deployments" | "billing" | "agents";
81
82/** What a viewer needs to query a dataset: membership, or the workspace's billing role. */
83export type DatasetNeeds = "member" | "billing";
84
85export type DatasetSpec = {
86 label: string;
87 service: DatasetService;
88 needs: DatasetNeeds;
89 /** Time fields, the default first. */
90 times: string[];
91 /** Fields to group and filter by (text). */
92 dimensions: string[];
93 /** Number fields for sum, avg, p50, p95, and number filters. */
94 measures: string[];
95 /** Yes-or-no fields `rate` gives the share of. */
96 rates: string[];
97};
98
99/** The catalog. One dataset per line: the Rust mirror test reads it that way. */
100export const DATASETS: Record<DatasetId, DatasetSpec> = {
101 issues: { label: "Issues", service: "work", needs: "member", times: ["created_at", "closed_at"], dimensions: ["repo", "label", "state", "author_kind", "assignee_kind", "milestone"], measures: ["time_to_close_hours", "comments"], rates: ["closed"] },
102 pull_requests: { label: "Pull requests", service: "work", needs: "member", times: ["created_at", "merged_at", "closed_at"], dimensions: ["repo", "label", "state", "author_kind", "base_branch"], measures: ["cycle_time_hours", "time_to_first_review_hours", "review_count", "additions", "deletions", "changed_files"], rates: ["merged"] },
103 workflow_runs: { label: "Workflow runs", service: "actions", needs: "member", times: ["started_at", "completed_at"], dimensions: ["repo", "workflow", "branch", "event", "conclusion", "runner_kind"], measures: ["duration_seconds", "queue_seconds"], rates: ["succeeded"] },
104 deployments: { label: "Deployments", service: "deployments", needs: "member", times: ["created_at"], dimensions: ["repo", "project", "environment", "state"], measures: ["duration_seconds", "time_to_restore_hours"], rates: ["failed"] },
105 spend: { label: "Spend", service: "billing", needs: "billing", times: ["day"], dimensions: ["product", "project", "person", "model"], measures: ["amount_micros"], rates: [] },
106 agent_sessions: { label: "Agent sessions", service: "agents", needs: "member", times: ["started_at"], dimensions: ["agent", "repo", "outcome", "model", "trigger"], measures: ["duration_seconds", "cost_micros", "tokens"], rates: ["succeeded"] },
107};
108
109/** The most rows a query returns. */
110export const DATASET_MAX_ROWS = 100;
111/** The most filters on one query. */
112export const DATASET_MAX_FILTERS = 10;
113/** The most values in an `in` filter. */
114export const DATASET_MAX_IN_VALUES = 50;
115/** The longest `{ from, to }` range, in days. */
116export const DATASET_MAX_RANGE_DAYS = 366;
117
118const DAY_MS = 86_400_000;
119const TIME = /^(\d{4})-(\d{2})-(\d{2})(?:T(\d{2}):(\d{2}):(\d{2})(?:\.(\d{1,3}))?Z)?$/;
120
121/** A range end as milliseconds since the epoch, or null when it is not a real date or RFC 3339 UTC time. */
122export function datasetTime(text: string): number | null {
123 const match = TIME.exec(text);
124 if (!match) return null;
125 const [year, month, day] = [Number(match[1]), Number(match[2]), Number(match[3])];
126 const [hour, minute, second] = [Number(match[4] ?? 0), Number(match[5] ?? 0), Number(match[6] ?? 0)];
127 const millis = Number((match[7] ?? "").padEnd(3, "0") || 0);
128 if (month < 1 || month > 12 || day < 1 || day > daysInMonth(year, month)) return null;
129 if (hour > 23 || minute > 59 || second > 59) return null;
130 return Date.UTC(year, month - 1, day, hour, minute, second, millis);
131}
132
133function daysInMonth(year: number, month: number): number {
134 if (month === 2) return (year % 4 === 0 && year % 100 !== 0) || year % 400 === 0 ? 29 : 28;
135 return [4, 6, 9, 11].includes(month) ? 30 : 31;
136}
137
138/**
139 * What is wrong with a query, or null. The same rules, and the same
140 * words, as `DatasetQuery::validate` in Rust. It checks the query against
141 * the catalog only; who may run it is the owning service's to decide.
142 */
143export function datasetQueryError(query: DatasetQuery): string | null {
144 const spec = DATASETS[query.dataset];
145 if (!spec) return `There is no dataset called ${query.dataset}.`;
146 const name = query.dataset;
147 const { op } = query.measure;
148 const field = query.measure.field ?? null;
149 if (op === "count") {
150 if (field !== null) return "count takes no field.";
151 } else if (op === "rate") {
152 if (field === null) return "rate needs a field.";
153 if (!spec.rates.includes(field)) return `${name} has no yes-or-no field ${field} to take the rate of.`;
154 } else {
155 if (field === null) return `${op} needs a field.`;
156 if (!spec.measures.includes(field)) return `${name} has no number field ${field}.`;
157 }
158 const groupBy = query.group_by ?? null;
159 if (groupBy !== null && !spec.dimensions.includes(groupBy)) return `${name} can't be grouped by ${groupBy}.`;
160 const time = query.time ?? null;
161 if (time !== null && !spec.times.includes(time)) return `${name} has no time field ${time}.`;
162 const filters = query.filters ?? [];
163 if (filters.length > DATASET_MAX_FILTERS) return `A query takes at most ${DATASET_MAX_FILTERS} filters.`;
164 for (const filter of filters) {
165 if (spec.dimensions.includes(filter.field)) {
166 if (filter.op === "in") {
167 if (!Array.isArray(filter.value) || filter.value.length === 0 || filter.value.length > DATASET_MAX_IN_VALUES) {
168 return `in on ${filter.field} takes a list of 1 to ${DATASET_MAX_IN_VALUES} values.`;
169 }
170 } else if (filter.op === "eq" || filter.op === "neq") {
171 if (typeof filter.value !== "string") return `${filter.op} on ${filter.field} takes text.`;
172 } else {
173 return `${filter.field} is text: filter it with eq, neq or in.`;
174 }
175 } else if (spec.measures.includes(filter.field)) {
176 if (filter.op === "in") return `${filter.field} is a number: filter it with eq, neq, gte or lte.`;
177 if (typeof filter.value !== "number" || !Number.isFinite(filter.value)) return `${filter.op} on ${filter.field} takes a number.`;
178 } else {
179 return `${name} can't be filtered by ${filter.field}.`;
180 }
181 }
182 const range = query.range ?? null;
183 if (range !== null && typeof range === "object") {
184 const from = datasetTime(range.from);
185 const to = datasetTime(range.to);
186 if (from === null || to === null) return "A range's from and to are dates or RFC 3339 UTC times.";
187 if (from >= to) return "A range's from comes before its to.";
188 if (to - from > DATASET_MAX_RANGE_DAYS * DAY_MS) return `A range is at most ${DATASET_MAX_RANGE_DAYS} days.`;
189 }
190 const limit = query.limit ?? null;
191 if (limit !== null && (!Number.isInteger(limit) || limit < 1 || limit > DATASET_MAX_ROWS)) return `limit is between 1 and ${DATASET_MAX_ROWS}.`;
192 return null;
193}
194
195/** A query from untrusted JSON: well-formed and valid, or why not. Unknown keys are dropped. */
196export function parseDatasetQuery(input: unknown): { query: DatasetQuery } | { error: string } {
197 if (!isObject(input)) return { error: "A query is an object." };
198 if (typeof input.dataset !== "string" || !(DATASET_IDS as readonly string[]).includes(input.dataset)) {
199 return { error: `There is no dataset called ${String(input.dataset)}.` };
200 }
201 const measure = input.measure;
202 if (!isObject(measure) || typeof measure.op !== "string" || !(DATASET_MEASURE_OPS as readonly string[]).includes(measure.op)) {
203 return { error: "measure.op is count, sum, avg, p50, p95 or rate." };
204 }
205 if (!optional(measure.field, "string")) return { error: "measure.field is text." };
206 if (!optional(input.group_by, "string")) return { error: "group_by is text." };
207 if (!optional(input.time, "string")) return { error: "time is text." };
208 if (input.interval != null && !(DATASET_INTERVALS as readonly unknown[]).includes(input.interval)) return { error: "interval is day, week or month." };
209 if (input.limit != null && typeof input.limit !== "number") return { error: "limit is a number." };
210 let filters: DatasetFilter[] | null = null;
211 if (input.filters != null) {
212 if (!Array.isArray(input.filters)) return { error: "filters is a list." };
213 filters = [];
214 for (const filter of input.filters) {
215 if (!isObject(filter) || typeof filter.field !== "string" || !(DATASET_FILTER_OPS as readonly unknown[]).includes(filter.op)) {
216 return { error: "A filter has a field and an op: eq, neq, in, gte or lte." };
217 }
218 const value = filter.value;
219 const ok =
220 typeof value === "string" || (typeof value === "number" && Number.isFinite(value)) || (Array.isArray(value) && value.every((item) => typeof item === "string"));
221 if (!ok) return { error: "A filter's value is text, a number or a list of text." };
222 filters.push({ field: filter.field, op: filter.op as DatasetFilterOp, value: value as DatasetFilter["value"] });
223 }
224 }
225 let range: DatasetRange | null = null;
226 if (input.range != null) {
227 if (typeof input.range === "string" && (DATASET_RANGE_PRESETS as readonly string[]).includes(input.range)) range = input.range as DatasetRangePreset;
228 else if (isObject(input.range) && typeof input.range.from === "string" && typeof input.range.to === "string") range = { from: input.range.from, to: input.range.to };
229 else return { error: "range is 7d, 30d, 90d or { from, to }." };
230 }
231 const query: DatasetQuery = {
232 dataset: input.dataset as DatasetId,
233 measure: { op: measure.op as DatasetMeasureOp, field: (measure.field as string | null | undefined) ?? null },
234 group_by: (input.group_by as string | null | undefined) ?? null,
235 interval: (input.interval as DatasetInterval | null | undefined) ?? null,
236 time: (input.time as string | null | undefined) ?? null,
237 filters,
238 range,
239 limit: (input.limit as number | null | undefined) ?? null,
240 };
241 const error = datasetQueryError(query);
242 return error ? { error } : { query };
243}
244
245function isObject(value: unknown): value is Record<string, unknown> {
246 return typeof value === "object" && value !== null && !Array.isArray(value);
247}
248
249function optional(value: unknown, type: "string"): boolean {
250 return value === undefined || value === null || typeof value === type;
251}