g1t/services/actions/src/plan.rs
| 1 | //! A run's life: made, its jobs waiting on the jobs they need, each job |
| 2 | //! skipped or expanded into its matrix and queued, started in a sandbox |
| 3 | //! when its workspace has room, reporting its steps and logs as it goes, |
| 4 | //! and finished; the run finishes with its last job. |
| 5 | |
| 6 | use g1t_actions::events::{RunInfo, runner_context}; |
| 7 | use g1t_actions::expr::{self, Scope, Status}; |
| 8 | use g1t_actions::matrix; |
| 9 | use g1t_actions::workflow::{self, Workflow}; |
| 10 | use g1t_contracts::access::Capability; |
| 11 | use g1t_contracts::actions::{JobCallArgs, RunActionArgs, StartJobArgs, WorkflowRun}; |
| 12 | use g1t_contracts::identity::{CreateAccessTokenArgs, CreatedAccessToken}; |
| 13 | use g1t_contracts::repos::{Repo, RepoPath}; |
| 14 | use g1t_contracts::time::rfc3339; |
| 15 | use g1t_contracts::{FailureCode, Outcome, new_id}; |
| 16 | use g1t_kit::now_ms; |
| 17 | use g1t_secrets::{random_hex, same, sha256_hex}; |
| 18 | use serde::Deserialize; |
| 19 | use serde_json::{Map, Value, json}; |
| 20 | use worker::Result; |
| 21 | |
| 22 | use crate::sync::WorkflowRow; |
| 23 | use crate::{Actions, Count, MAX_TIMEOUT_MINUTES, RUNNING_PER_WORKSPACE, SELF_HOSTED_MAX_TIMEOUT_MINUTES, SILENT_MS, SITE, check, fail, optional, repo_path}; |
| 24 | use g1t_contracts::runners::{Wanted, waiting_reason}; |
| 25 | |
| 26 | /// The most log one job keeps, in bytes; past it, the log says so and stops. |
| 27 | const MAX_LOG_BYTES: usize = 4 * 1024 * 1024; |
| 28 | /// The most a single log report may add. |
| 29 | const MAX_CHUNK_BYTES: usize = 256 * 1024; |
| 30 | const MAX_ANNOTATIONS: usize = 50; |
| 31 | |
| 32 | /// The repository whose unfinished runs `event` stops, by id: one deleted, |
| 33 | /// or archived (not unarchived). |
| 34 | pub fn stops_runs(event: &g1t_contracts::events::Event) -> Option<String> { |
| 35 | let stops = match event.kind.as_str() { |
| 36 | "repo.deleted" => true, |
| 37 | "repo.archived" => event.data["archived"].as_bool() != Some(false), |
| 38 | _ => false, |
| 39 | }; |
| 40 | if !stops { |
| 41 | return None; |
| 42 | } |
| 43 | event.repo_id.clone().or_else(|| event.data["repoId"].as_str().map(str::to_owned)) |
| 44 | } |
| 45 | |
| 46 | pub struct NewRun { |
| 47 | pub repo: Repo, |
| 48 | pub path: String, |
| 49 | pub source: String, |
| 50 | pub workflow: Workflow, |
| 51 | pub info: RunInfo, |
| 52 | pub action: Option<String>, |
| 53 | pub pull: Option<u32>, |
| 54 | pub title: String, |
| 55 | pub inputs: Map<String, Value>, |
| 56 | pub event_key: String, |
| 57 | pub actor_id: Option<String>, |
| 58 | pub actor: Option<String>, |
| 59 | pub trusted: bool, |
| 60 | } |
| 61 | |
| 62 | #[derive(Clone, Deserialize)] |
| 63 | pub struct RunRow { |
| 64 | pub id: String, |
| 65 | pub workflow_id: String, |
| 66 | pub repo_id: String, |
| 67 | pub repo: String, |
| 68 | pub path: String, |
| 69 | pub name: String, |
| 70 | pub title: String, |
| 71 | pub number: u64, |
| 72 | pub attempt: u64, |
| 73 | pub event: String, |
| 74 | pub action: Option<String>, |
| 75 | pub git_ref: String, |
| 76 | pub sha: String, |
| 77 | pub pull: Option<u32>, |
| 78 | pub status: String, |
| 79 | pub conclusion: Option<String>, |
| 80 | pub error: Option<String>, |
| 81 | pub actor: Option<String>, |
| 82 | pub actor_id: Option<String>, |
| 83 | pub source: String, |
| 84 | pub info: String, |
| 85 | pub inputs: String, |
| 86 | pub trusted: u32, |
| 87 | pub concurrency_group: Option<String>, |
| 88 | pub created_at: String, |
| 89 | pub started_at: Option<String>, |
| 90 | pub finished_at: Option<String>, |
| 91 | } |
| 92 | |
| 93 | impl RunRow { |
| 94 | pub fn info(&self) -> RunInfo { |
| 95 | let mut info: RunInfo = serde_json::from_str(&self.info).unwrap_or_default(); |
| 96 | info.run_id = self.id.clone(); |
| 97 | info.run_number = self.number; |
| 98 | info.run_attempt = self.attempt; |
| 99 | info.workflow_path = self.path.clone(); |
| 100 | if info.workflow.is_empty() { |
| 101 | info.workflow = self.name.clone(); |
| 102 | } |
| 103 | info |
| 104 | } |
| 105 | |
| 106 | pub fn inputs(&self) -> Map<String, Value> { |
| 107 | serde_json::from_str(&self.inputs).unwrap_or_default() |
| 108 | } |
| 109 | |
| 110 | pub fn summary(&self) -> WorkflowRun { |
| 111 | WorkflowRun { |
| 112 | id: self.id.clone(), |
| 113 | workflow_id: self.workflow_id.clone(), |
| 114 | path: self.path.clone(), |
| 115 | name: self.name.clone(), |
| 116 | title: self.title.clone(), |
| 117 | number: self.number, |
| 118 | attempt: self.attempt, |
| 119 | event: self.event.clone(), |
| 120 | git_ref: self.git_ref.clone(), |
| 121 | sha: self.sha.clone(), |
| 122 | pull: self.pull, |
| 123 | status: self.status.clone(), |
| 124 | conclusion: self.conclusion.clone(), |
| 125 | error: self.error.clone(), |
| 126 | actor: self.actor.clone(), |
| 127 | created_at: self.created_at.clone(), |
| 128 | started_at: self.started_at.clone(), |
| 129 | finished_at: self.finished_at.clone(), |
| 130 | } |
| 131 | } |
| 132 | } |
| 133 | |
| 134 | #[derive(Clone, Deserialize)] |
| 135 | pub struct JobRow { |
| 136 | pub id: String, |
| 137 | pub run_id: String, |
| 138 | pub repo_id: String, |
| 139 | pub namespace: String, |
| 140 | pub key: String, |
| 141 | pub ordinal: u32, |
| 142 | pub name: String, |
| 143 | pub needs: String, |
| 144 | pub matrix: Option<String>, |
| 145 | pub status: String, |
| 146 | pub conclusion: Option<String>, |
| 147 | pub steps: String, |
| 148 | pub annotations: String, |
| 149 | pub outputs: String, |
| 150 | pub reason: Option<String>, |
| 151 | pub token_hash: Option<String>, |
| 152 | pub timeout_minutes: u32, |
| 153 | pub continue_on_error: u32, |
| 154 | pub max_parallel: Option<u32>, |
| 155 | /// Set for a job that calls a reusable workflow, and for that |
| 156 | /// workflow's jobs (see migration 0002). |
| 157 | pub call: Option<String>, |
| 158 | pub seen_at: Option<String>, |
| 159 | pub started_at: Option<String>, |
| 160 | pub finished_at: Option<String>, |
| 161 | /// For a job whose `runs-on` names self-hosted runners: what it asks |
| 162 | /// for, as a JSON array (see `g1t_contracts::runners::Wanted`), when it |
| 163 | /// started waiting, and the runner that took it (migration 0004). |
| 164 | #[serde(default)] |
| 165 | pub labels: Option<String>, |
| 166 | #[serde(default)] |
| 167 | pub queued_at: Option<String>, |
| 168 | #[serde(default)] |
| 169 | pub runner_id: Option<String>, |
| 170 | #[serde(default)] |
| 171 | pub runner_name: Option<String>, |
| 172 | } |
| 173 | |
| 174 | impl JobRow { |
| 175 | pub fn needs(&self) -> Vec<String> { |
| 176 | serde_json::from_str(&self.needs).unwrap_or_default() |
| 177 | } |
| 178 | |
| 179 | pub fn call(&self) -> Option<Value> { |
| 180 | self.call.as_deref().and_then(|call| serde_json::from_str(call).ok()) |
| 181 | } |
| 182 | |
| 183 | /// For a job of a called workflow: that workflow, the job's own id in |
| 184 | /// it, and the job. |
| 185 | pub fn callee(&self) -> Option<(Workflow, workflow::Job, Value)> { |
| 186 | let call = self.call().filter(|call| call["role"] == "callee")?; |
| 187 | let called = workflow::parse(call["source"].as_str()?).ok()?; |
| 188 | let job = called.jobs.iter().find(|job| call["job"].as_str() == Some(job.id.as_str()))?.clone(); |
| 189 | Some((called, job, call)) |
| 190 | } |
| 191 | } |
| 192 | |
| 193 | /// A job to decide on: its key, its definition, and what it needs, as |
| 194 | /// (name in `needs`, key of the jobs). |
| 195 | type Unit = (String, workflow::Job, Vec<(String, String)>); |
| 196 | |
| 197 | /// How deep reusable workflows may call one another, as on GitHub. |
| 198 | const MAX_CALL_DEPTH: u64 = 4; |
| 199 | |
| 200 | /// What the jobs of one key came to, for `needs.<key>`. |
| 201 | fn key_result(rows: &[&JobRow]) -> &'static str { |
| 202 | let failed = |row: &&&JobRow| row.conclusion.as_deref() == Some("failure") && row.continue_on_error == 0; |
| 203 | if rows.iter().any(|row| failed(&row)) { |
| 204 | "failure" |
| 205 | } else if rows.iter().any(|row| row.conclusion.as_deref() == Some("cancelled")) { |
| 206 | "cancelled" |
| 207 | } else if rows.iter().all(|row| row.conclusion.as_deref() == Some("skipped")) { |
| 208 | "skipped" |
| 209 | } else { |
| 210 | "success" |
| 211 | } |
| 212 | } |
| 213 | |
| 214 | /// Whether any job before `key` failed: one it needs, or one those need, |
| 215 | /// however far back. A job after a skipped one still sees the failure |
| 216 | /// that skipped it, as GitHub's failure() does. `needs_of`: each key's |
| 217 | /// needs, as keys; `failed`: whether a key's jobs came to a failure. |
| 218 | fn ancestor_failed(needs_of: &std::collections::HashMap<&str, Vec<&str>>, key: &str, failed: impl Fn(&str) -> bool) -> bool { |
| 219 | let mut seen = std::collections::HashSet::new(); |
| 220 | let mut stack: Vec<&str> = needs_of.get(key).cloned().unwrap_or_default(); |
| 221 | while let Some(next) = stack.pop() { |
| 222 | if !seen.insert(next) { |
| 223 | continue; |
| 224 | } |
| 225 | if failed(next) { |
| 226 | return true; |
| 227 | } |
| 228 | stack.extend(needs_of.get(next).into_iter().flatten().copied()); |
| 229 | } |
| 230 | false |
| 231 | } |
| 232 | |
| 233 | /// What the runner is told about a job it starts, from its workflow: the |
| 234 | /// environment it names plainly, and the machine its `runs-on` asks for |
| 235 | /// (`instance_for`; none for the standard one). |
| 236 | #[derive(Default)] |
| 237 | struct StartDetails { |
| 238 | environment: Option<String>, |
| 239 | instance: Option<String>, |
| 240 | } |
| 241 | |
| 242 | fn start_details(run: &RunRow, job: &JobRow) -> StartDetails { |
| 243 | let spec = match job.callee() { |
| 244 | Some((_, spec, _)) => spec, |
| 245 | None => match workflow::parse(&run.source).ok().and_then(|w| w.jobs.into_iter().find(|j| j.id == job.key)) { |
| 246 | Some(spec) => spec, |
| 247 | None => return StartDetails::default(), |
| 248 | }, |
| 249 | }; |
| 250 | let environment = match spec.raw.get("environment") { |
| 251 | Some(Value::String(name)) if !name.contains("${{") => Some(name.clone()), |
| 252 | Some(Value::Object(env)) => env.get("name").and_then(Value::as_str).filter(|n| !n.contains("${{")).map(str::to_owned), |
| 253 | _ => None, |
| 254 | }; |
| 255 | // `runs-on` as the job was queued with: its matrix and the run's |
| 256 | // inputs. A label that needs more than those is the standard machine. |
| 257 | let mut contexts = Map::new(); |
| 258 | contexts.insert("github".into(), run.info().context(&spec.id, "", run.action.as_deref())); |
| 259 | contexts.insert("inputs".into(), Value::Object(run.inputs())); |
| 260 | contexts.insert("matrix".into(), job.matrix.as_deref().and_then(|m| serde_json::from_str(m).ok()).unwrap_or_else(|| json!({}))); |
| 261 | let scope = Scope { contexts: &contexts, status: Status::Success, hash_files: None }; |
| 262 | let labels: Vec<String> = match expr::interpolate_value(&spec.runs_on, &scope).unwrap_or(Value::Null) { |
| 263 | Value::String(label) => vec![label], |
| 264 | Value::Array(labels) => labels.iter().map(expr::to_text).collect(), |
| 265 | Value::Object(given) => match given.get("labels") { |
| 266 | Some(Value::Array(labels)) => labels.iter().map(expr::to_text).collect(), |
| 267 | Some(label) => vec![expr::to_text(label)], |
| 268 | None => Vec::new(), |
| 269 | }, |
| 270 | _ => Vec::new(), |
| 271 | }; |
| 272 | let instance = g1t_contracts::actions::instance_for(&labels); |
| 273 | StartDetails { |
| 274 | environment, |
| 275 | instance: (instance != g1t_contracts::actions::STANDARD_INSTANCE).then(|| instance.label.to_owned()), |
| 276 | } |
| 277 | } |
| 278 | |
| 279 | fn now() -> String { |
| 280 | rfc3339(now_ms()) |
| 281 | } |
| 282 | |
| 283 | impl Actions { |
| 284 | pub async fn run_row(&self, id: &str) -> Result<Option<RunRow>> { |
| 285 | self.db.prepare("SELECT * FROM runs WHERE id = ?").bind(&[id.into()])?.first::<RunRow>(None).await |
| 286 | } |
| 287 | |
| 288 | pub async fn job_rows(&self, run_id: &str) -> Result<Vec<JobRow>> { |
| 289 | self.db |
| 290 | .prepare("SELECT * FROM jobs WHERE run_id = ? ORDER BY rowid") |
| 291 | .bind(&[run_id.into()])? |
| 292 | .all() |
| 293 | .await? |
| 294 | .results::<JobRow>() |
| 295 | } |
| 296 | |
| 297 | pub async fn run_summary(&self, id: &str) -> Result<Outcome<WorkflowRun>> { |
| 298 | Ok(match self.run_row(id).await? { |
| 299 | Some(row) => Outcome::Ok(row.summary()), |
| 300 | None => fail(FailureCode::NotFound, "No such run."), |
| 301 | }) |
| 302 | } |
| 303 | |
| 304 | /// The contexts every expression outside a job's steps may use. |
| 305 | fn base_contexts(run: &RunRow, vars: &Map<String, Value>, job: &str) -> Map<String, Value> { |
| 306 | let mut contexts = Map::new(); |
| 307 | contexts.insert("github".into(), run.info().context(job, "", run.action.as_deref())); |
| 308 | contexts.insert("inputs".into(), Value::Object(run.inputs())); |
| 309 | contexts.insert("vars".into(), Value::Object(vars.clone())); |
| 310 | contexts.insert("needs".into(), json!({})); |
| 311 | contexts.insert("runner".into(), runner_context()); |
| 312 | contexts |
| 313 | } |
| 314 | |
| 315 | /// Makes a run and its jobs, and starts what can start. `None` when the |
| 316 | /// event already started this workflow. |
| 317 | pub async fn create_run(&self, new: NewRun) -> Result<Option<String>> { |
| 318 | // Nothing starts on an archived repository. A deleted one is never |
| 319 | // found to start anything on. |
| 320 | if new.repo.archived() { |
| 321 | return Ok(None); |
| 322 | } |
| 323 | let workflow_row = self.workflow_row(&new.repo, &new.path, &new.workflow.display_name(&new.path), &new.source).await?; |
| 324 | let numbered = self |
| 325 | .db |
| 326 | .prepare("UPDATE workflows SET run_count = run_count + 1 WHERE id = ? RETURNING run_count AS n") |
| 327 | .bind(&[workflow_row.id.as_str().into()])? |
| 328 | .first::<Count>(None) |
| 329 | .await? |
| 330 | .map_or(1, |count| count.n); |
| 331 | let id = new_id("run", now_ms()); |
| 332 | let mut info = new.info.clone(); |
| 333 | info.workflow = new.workflow.display_name(&new.path); |
| 334 | info.workflow_path = new.path.clone(); |
| 335 | info.run_id = id.clone(); |
| 336 | info.run_number = u64::from(numbered); |
| 337 | let vars = self |
| 338 | .variables_for(&new.repo.id, &format!("{}/{}", new.repo.namespace, new.repo.name), None, new.trusted) |
| 339 | .await?; |
| 340 | |
| 341 | // run-name and the concurrency group read github, inputs and vars. |
| 342 | let mut contexts = Map::new(); |
| 343 | contexts.insert("github".into(), info.context("", "", new.action.as_deref())); |
| 344 | contexts.insert("inputs".into(), Value::Object(new.inputs.clone())); |
| 345 | contexts.insert("vars".into(), Value::Object(vars)); |
| 346 | let scope = Scope { |
| 347 | contexts: &contexts, |
| 348 | status: Status::Success, |
| 349 | hash_files: None, |
| 350 | }; |
| 351 | let title = new |
| 352 | .workflow |
| 353 | .run_name |
| 354 | .as_deref() |
| 355 | .and_then(|run_name| expr::interpolate(run_name, &scope).ok()) |
| 356 | .filter(|title| !title.trim().is_empty()) |
| 357 | .unwrap_or(new.title.clone()); |
| 358 | let group = new.workflow.concurrency.as_ref().and_then(|c| expr::interpolate(&c.group, &scope).ok()); |
| 359 | let cancel_in_progress = new |
| 360 | .workflow |
| 361 | .concurrency |
| 362 | .as_ref() |
| 363 | .and_then(|c| expr::interpolate_value(&c.cancel_in_progress, &scope).ok()) |
| 364 | .is_some_and(|value| expr::truthy(&value)); |
| 365 | |
| 366 | let inserted = self |
| 367 | .db |
| 368 | .prepare( |
| 369 | "INSERT OR IGNORE INTO runs (id, workflow_id, repo_id, repo, path, name, title, number, event, action, git_ref, sha, |
| 370 | pull, status, actor, actor_id, source, info, inputs, trusted, concurrency_group, event_key, created_at) |
| 371 | VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 'queued', ?, ?, ?, ?, ?, ?, ?, ?, ?) RETURNING id", |
| 372 | ) |
| 373 | .bind(&[ |
| 374 | id.as_str().into(), |
| 375 | workflow_row.id.as_str().into(), |
| 376 | new.repo.id.as_str().into(), |
| 377 | format!("{}/{}", new.repo.namespace, new.repo.name).into(), |
| 378 | new.path.as_str().into(), |
| 379 | info.workflow.as_str().into(), |
| 380 | title.as_str().into(), |
| 381 | numbered.into(), |
| 382 | info.event_name.as_str().into(), |
| 383 | optional(new.action.as_deref()), |
| 384 | info.git_ref.as_str().into(), |
| 385 | info.sha.as_str().into(), |
| 386 | new.pull.map_or(worker::wasm_bindgen::JsValue::NULL, Into::into), |
| 387 | optional(new.actor.as_deref()), |
| 388 | optional(new.actor_id.as_deref()), |
| 389 | new.source.as_str().into(), |
| 390 | serde_json::to_string(&info)?.into(), |
| 391 | serde_json::to_string(&new.inputs)?.into(), |
| 392 | u32::from(new.trusted).into(), |
| 393 | optional(group.as_deref()), |
| 394 | new.event_key.as_str().into(), |
| 395 | now().into(), |
| 396 | ])? |
| 397 | .first::<Value>(None) |
| 398 | .await?; |
| 399 | if inserted.is_none() { |
| 400 | return Ok(None); |
| 401 | } |
| 402 | if let Some(run) = self.run_row(&id).await? { |
| 403 | self.report_pending(&run).await?; |
| 404 | } |
| 405 | |
| 406 | // Every job, waiting; each is expanded when the jobs it needs are done. |
| 407 | let mut statements = Vec::new(); |
| 408 | for job in &new.workflow.jobs { |
| 409 | statements.push( |
| 410 | self.db |
| 411 | .prepare( |
| 412 | "INSERT INTO jobs (id, run_id, repo_id, namespace, key, name, needs, status) VALUES (?, ?, ?, ?, ?, ?, ?, 'waiting')", |
| 413 | ) |
| 414 | .bind(&[ |
| 415 | new_id("job", now_ms()).into(), |
| 416 | id.as_str().into(), |
| 417 | new.repo.id.as_str().into(), |
| 418 | new.repo.namespace.as_str().into(), |
| 419 | job.id.as_str().into(), |
| 420 | job.name.clone().filter(|n| !expr::has_expression(n)).unwrap_or(job.id.clone()).into(), |
| 421 | serde_json::to_string(&job.needs)?.into(), |
| 422 | ])?, |
| 423 | ); |
| 424 | } |
| 425 | self.db.batch(statements).await?; |
| 426 | |
| 427 | // One run at a time per concurrency group. |
| 428 | if let Some(group) = &group { |
| 429 | let others = self |
| 430 | .db |
| 431 | .prepare("SELECT * FROM runs WHERE repo_id = ? AND concurrency_group = ? AND id != ? AND status != 'completed' ORDER BY id") |
| 432 | .bind(&[new.repo.id.as_str().into(), group.as_str().into(), id.as_str().into()])? |
| 433 | .all() |
| 434 | .await? |
| 435 | .results::<RunRow>()?; |
| 436 | for other in &others { |
| 437 | if cancel_in_progress || other.status == "pending" { |
| 438 | // A newer run replaces a waiting one, as on GitHub. |
| 439 | self.cancel_run(other, "A newer run in the same concurrency group replaced it.").await?; |
| 440 | } |
| 441 | } |
| 442 | if !cancel_in_progress && others.iter().any(|other| other.status != "pending") { |
| 443 | self.db |
| 444 | .prepare("UPDATE runs SET status = 'pending' WHERE id = ?") |
| 445 | .bind(&[id.as_str().into()])? |
| 446 | .run() |
| 447 | .await?; |
| 448 | return Ok(Some(id)); |
| 449 | } |
| 450 | } |
| 451 | self.advance(&id).await?; |
| 452 | Ok(Some(id)) |
| 453 | } |
| 454 | |
| 455 | /// A run that could not start, such as for a workflow file that does not read. |
| 456 | #[allow(clippy::too_many_arguments)] |
| 457 | pub async fn record_failed_run( |
| 458 | &self, |
| 459 | row: &WorkflowRow, |
| 460 | git_ref: &str, |
| 461 | sha: &str, |
| 462 | event_key: &str, |
| 463 | actor_id: Option<&str>, |
| 464 | actor: &str, |
| 465 | problem: &str, |
| 466 | ) -> Result<()> { |
| 467 | let numbered = self |
| 468 | .db |
| 469 | .prepare("UPDATE workflows SET run_count = run_count + 1 WHERE id = ? RETURNING run_count AS n") |
| 470 | .bind(&[row.id.as_str().into()])? |
| 471 | .first::<Count>(None) |
| 472 | .await? |
| 473 | .map_or(1, |count| count.n); |
| 474 | let at = now(); |
| 475 | self.db |
| 476 | .prepare( |
| 477 | "INSERT OR IGNORE INTO runs (id, workflow_id, repo_id, repo, path, name, title, number, event, git_ref, sha, status, |
| 478 | conclusion, error, actor, actor_id, source, info, event_key, created_at, finished_at) |
| 479 | VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'push', ?, ?, 'completed', 'failure', ?, ?, ?, ?, '{}', ?, ?, ?)", |
| 480 | ) |
| 481 | .bind(&[ |
| 482 | new_id("run", now_ms()).into(), |
| 483 | row.id.as_str().into(), |
| 484 | row.repo_id.as_str().into(), |
| 485 | row.repo.as_str().into(), |
| 486 | row.path.as_str().into(), |
| 487 | row.path.as_str().into(), |
| 488 | "Invalid workflow file".into(), |
| 489 | numbered.into(), |
| 490 | git_ref.into(), |
| 491 | sha.into(), |
| 492 | problem.into(), |
| 493 | actor.into(), |
| 494 | optional(actor_id), |
| 495 | row.source.as_str().into(), |
| 496 | event_key.into(), |
| 497 | at.as_str().into(), |
| 498 | at.as_str().into(), |
| 499 | ])? |
| 500 | .run() |
| 501 | .await?; |
| 502 | Ok(()) |
| 503 | } |
| 504 | |
| 505 | /// Moves a run along: jobs whose needs are done are decided on, and |
| 506 | /// jobs that can start are started. |
| 507 | pub async fn advance(&self, run_id: &str) -> Result<()> { |
| 508 | // Each pass may finish jobs (skipped ones), which may free others. |
| 509 | for _ in 0..20 { |
| 510 | let Some(run) = self.run_row(run_id).await? else { return Ok(()) }; |
| 511 | if run.status == "completed" || run.status == "pending" { |
| 512 | return Ok(()); |
| 513 | } |
| 514 | let workflow = match workflow::parse(&run.source) { |
| 515 | Ok(workflow) => workflow, |
| 516 | Err(problem) => { |
| 517 | self.finish_run(&run, Some(&problem)).await?; |
| 518 | return Ok(()); |
| 519 | } |
| 520 | }; |
| 521 | let jobs = self.job_rows(run_id).await?; |
| 522 | let mut changed = false; |
| 523 | // What to decide on: the workflow's jobs, and the jobs of the |
| 524 | // workflows they call, each with its needs as (name, key). |
| 525 | let mut units: Vec<Unit> = workflow |
| 526 | .jobs |
| 527 | .iter() |
| 528 | .map(|job| (job.id.clone(), job.clone(), job.needs.iter().map(|n| (n.clone(), n.clone())).collect())) |
| 529 | .collect(); |
| 530 | let mut seen = std::collections::HashSet::new(); |
| 531 | for row in &jobs { |
| 532 | if !seen.insert(row.key.clone()) { |
| 533 | continue; |
| 534 | } |
| 535 | if let Some((_, job, call)) = row.callee() { |
| 536 | let parent = call["parent"].as_str().unwrap_or_default().to_owned(); |
| 537 | let needs = job.needs.iter().map(|n| (n.clone(), format!("{parent}/{n}"))).collect(); |
| 538 | units.push((row.key.clone(), job, needs)); |
| 539 | } |
| 540 | } |
| 541 | // Each key's needs, as keys, to look back through for failure(). |
| 542 | let needs_of: std::collections::HashMap<&str, Vec<&str>> = |
| 543 | units.iter().map(|(key, _, needs)| (key.as_str(), needs.iter().map(|(_, need)| need.as_str()).collect())).collect(); |
| 544 | let key_failed = |key: &str| key_result(&jobs.iter().filter(|row| row.key == key).collect::<Vec<_>>()) == "failure"; |
| 545 | for (key, job, needs) in &units { |
| 546 | let rows: Vec<&JobRow> = jobs.iter().filter(|row| &row.key == key).collect(); |
| 547 | if rows.is_empty() || !rows.iter().all(|row| row.status == "waiting") { |
| 548 | continue; |
| 549 | } |
| 550 | let needed: Vec<(&String, Vec<&JobRow>)> = |
| 551 | needs.iter().map(|(name, need)| (name, jobs.iter().filter(|row| &row.key == need).collect())).collect(); |
| 552 | if !needed.iter().all(|(_, rows)| rows.iter().all(|row| row.status == "completed")) { |
| 553 | continue; |
| 554 | } |
| 555 | let failed_before = ancestor_failed(&needs_of, key, key_failed); |
| 556 | self.decide(&run, job, rows[0], &needed, failed_before).await?; |
| 557 | changed = true; |
| 558 | } |
| 559 | // A job that called a workflow finishes with that workflow's jobs. |
| 560 | for row in jobs.iter().filter(|row| row.status == "calling") { |
| 561 | let children: Vec<&JobRow> = jobs |
| 562 | .iter() |
| 563 | .filter(|child| child.call().is_some_and(|call| call["role"] == "callee" && call["parent"].as_str() == Some(row.key.as_str()))) |
| 564 | .collect(); |
| 565 | if !children.is_empty() && children.iter().all(|child| child.status == "completed") { |
| 566 | self.finish_call(row, &children).await?; |
| 567 | changed = true; |
| 568 | } |
| 569 | } |
| 570 | if !changed { |
| 571 | break; |
| 572 | } |
| 573 | } |
| 574 | self.start_queued().await?; |
| 575 | self.finish_if_done(run_id).await |
| 576 | } |
| 577 | |
| 578 | /// Decides on one job whose needs are done: skip it, fail it, or expand |
| 579 | /// it into its matrix and queue it. `failed_before`: whether any job |
| 580 | /// before it failed, however far back (`ancestor_failed`). |
| 581 | async fn decide(&self, run: &RunRow, job: &workflow::Job, row: &JobRow, needed: &[(&String, Vec<&JobRow>)], failed_before: bool) -> Result<()> { |
| 582 | let vars = self.variables_for(&run.repo_id, &run.repo, None, run.trusted != 0).await?; |
| 583 | let mut contexts = Self::base_contexts(run, &vars, &job.id); |
| 584 | // A called workflow's jobs read the inputs they were called with. |
| 585 | let call = row.call(); |
| 586 | let parent = call.as_ref().filter(|c| c["role"] == "callee").and_then(|c| c["parent"].as_str().map(str::to_owned)); |
| 587 | if let Some(call) = call.as_ref().filter(|c| c["role"] == "callee") { |
| 588 | contexts.insert("inputs".into(), call["inputs"].clone()); |
| 589 | } |
| 590 | let mut needs = Map::new(); |
| 591 | let mut results = Vec::new(); |
| 592 | for (key, rows) in needed { |
| 593 | let result = key_result(rows); |
| 594 | let mut outputs = Map::new(); |
| 595 | for row in rows { |
| 596 | if let Ok(Value::Object(more)) = serde_json::from_str::<Value>(&row.outputs) { |
| 597 | outputs.extend(more); |
| 598 | } |
| 599 | } |
| 600 | results.push(result); |
| 601 | needs.insert((*key).clone(), json!({ "result": result, "outputs": outputs })); |
| 602 | } |
| 603 | // As on GitHub: a need that was skipped makes success() false but |
| 604 | // not failure(); failure() is a failure anywhere before the job. |
| 605 | let status = expr::job_status(results, failed_before, run.conclusion.as_deref() == Some("cancelled")); |
| 606 | contexts.insert("needs".into(), Value::Object(needs)); |
| 607 | let scope = Scope { |
| 608 | contexts: &contexts, |
| 609 | status, |
| 610 | hash_files: None, |
| 611 | }; |
| 612 | let condition = job.condition.as_deref().unwrap_or_default(); |
| 613 | match expr::condition(condition, &scope) { |
| 614 | Ok(true) => {} |
| 615 | Ok(false) => return self.skip_job(row, None).await, |
| 616 | Err(problem) => return self.fail_job(row, &format!("Its `if` does not read: {problem}")).await, |
| 617 | } |
| 618 | if let Some(uses) = &job.uses { |
| 619 | return self.call_workflow(run, job, row, uses, &scope).await; |
| 620 | } |
| 621 | |
| 622 | // Its matrix, which may come from a needed job's outputs. |
| 623 | let combinations = match &job.matrix { |
| 624 | None => vec![Map::new()], |
| 625 | Some(matrix) => { |
| 626 | let value = match expr::interpolate_value(matrix, &scope) { |
| 627 | Ok(value) => value, |
| 628 | Err(problem) => return self.fail_job(row, &format!("Its matrix does not read: {problem}")).await, |
| 629 | }; |
| 630 | match matrix::expand(&value) { |
| 631 | Ok(combinations) if !combinations.is_empty() => combinations, |
| 632 | Ok(_) => return self.fail_job(row, "Its matrix makes no jobs.").await, |
| 633 | Err(problem) => return self.fail_job(row, &problem).await, |
| 634 | } |
| 635 | } |
| 636 | }; |
| 637 | let total = combinations.len(); |
| 638 | let raw = job.raw.as_object().cloned().unwrap_or_default(); |
| 639 | // Whether pull requests from forks may use self-hosted runners here, |
| 640 | // asked once, and only for a run that is not trusted. |
| 641 | let mut forks_allowed: Option<bool> = None; |
| 642 | let mut statements = Vec::new(); |
| 643 | for (index, combination) in combinations.iter().enumerate() { |
| 644 | let mut contexts = contexts.clone(); |
| 645 | contexts.insert("matrix".into(), Value::Object(combination.clone())); |
| 646 | contexts.insert( |
| 647 | "strategy".into(), |
| 648 | json!({ "fail-fast": job.fail_fast, "job-index": index, "job-total": total, "max-parallel": job.max_parallel.unwrap_or(total as u32) }), |
| 649 | ); |
| 650 | let scope = Scope { |
| 651 | contexts: &contexts, |
| 652 | status: Status::Success, |
| 653 | hash_files: None, |
| 654 | }; |
| 655 | let base_name = job.name.clone().unwrap_or(job.id.clone()); |
| 656 | // A called workflow's job is shown under the job that called it. |
| 657 | let base_name = match &parent { |
| 658 | Some(parent) => format!("{} / {base_name}", parent.replace('/', " / ")), |
| 659 | None => base_name, |
| 660 | }; |
| 661 | let name = if expr::has_expression(&base_name) { |
| 662 | expr::interpolate(&base_name, &scope).unwrap_or(base_name) |
| 663 | } else if job.matrix.is_some() { |
| 664 | matrix::job_name(&base_name, combination) |
| 665 | } else { |
| 666 | base_name |
| 667 | }; |
| 668 | let runs_on = expr::interpolate_value(&job.runs_on, &scope).unwrap_or(Value::Null); |
| 669 | let labels = match &runs_on { |
| 670 | Value::String(label) => vec![label.clone()], |
| 671 | Value::Array(labels) => labels.iter().map(expr::to_text).collect(), |
| 672 | Value::Object(spec) => spec.get("labels").map(|l| match l { |
| 673 | Value::Array(labels) => labels.iter().map(expr::to_text).collect(), |
| 674 | other => vec![expr::to_text(other)], |
| 675 | }).unwrap_or_default(), |
| 676 | _ => Vec::new(), |
| 677 | }; |
| 678 | // `runs-on: self-hosted` (or a group): the workspace's own |
| 679 | // machines, which may run any OS. Otherwise g1t's Linux sandboxes. |
| 680 | let group = match &runs_on { |
| 681 | Value::Object(spec) => spec.get("group").map(expr::to_text), |
| 682 | _ => None, |
| 683 | }; |
| 684 | let wanted = Wanted::of(&labels, group.as_deref()); |
| 685 | let mut reason = if wanted.self_hosted { |
| 686 | None |
| 687 | } else { |
| 688 | labels |
| 689 | .iter() |
| 690 | .find(|label| { |
| 691 | let lower = label.to_ascii_lowercase(); |
| 692 | lower.contains("windows") || lower.contains("macos") |
| 693 | }) |
| 694 | .map(|label| { |
| 695 | let os = if label.to_ascii_lowercase().contains("windows") { "windows" } else { "macos" }; |
| 696 | format!("`runs-on: {label}`: g1t's own runners are Linux. A self-hosted runner can run it: `runs-on: [self-hosted, {os}]`.") |
| 697 | }) |
| 698 | }; |
| 699 | // A pull request from a fork runs code anyone could write: never |
| 700 | // on the workspace's machines unless it said they may. |
| 701 | if wanted.self_hosted && run.trusted == 0 { |
| 702 | let allowed = match forks_allowed { |
| 703 | Some(allowed) => allowed, |
| 704 | None => { |
| 705 | let allowed = self.effective_runner_settings(&row.namespace, Some(&run.repo_id)).await?.fork_pull_requests; |
| 706 | forks_allowed = Some(allowed); |
| 707 | allowed |
| 708 | } |
| 709 | }; |
| 710 | if !allowed { |
| 711 | reason = Some( |
| 712 | "Pull requests from forks do not run on self-hosted runners here. An admin can allow it under Settings, Runners.".to_owned(), |
| 713 | ); |
| 714 | } |
| 715 | } |
| 716 | let max_minutes = if wanted.self_hosted { SELF_HOSTED_MAX_TIMEOUT_MINUTES } else { MAX_TIMEOUT_MINUTES }; |
| 717 | let timeout = raw |
| 718 | .get("timeout-minutes") |
| 719 | .and_then(|value| expr::interpolate_value(value, &scope).ok()) |
| 720 | .and_then(|value| value.as_f64().or_else(|| expr::to_text(&value).parse().ok())) |
| 721 | .map_or(MAX_TIMEOUT_MINUTES, |minutes| (minutes.ceil() as u32).clamp(1, max_minutes)); |
| 722 | let continue_on_error = raw |
| 723 | .get("continue-on-error") |
| 724 | .and_then(|value| expr::interpolate_value(value, &scope).ok()) |
| 725 | .is_some_and(|value| expr::truthy(&value)); |
| 726 | let (status, conclusion, finished) = match &reason { |
| 727 | Some(_) => ("completed", Some("failure"), Some(now())), |
| 728 | None => ("queued", None, None), |
| 729 | }; |
| 730 | // A self-hosted job waits, saying for what, until a runner takes it. |
| 731 | let (labels_json, queued_at) = if wanted.self_hosted && reason.is_none() { |
| 732 | reason = Some(waiting_reason(&wanted)); |
| 733 | (Some(serde_json::to_string(&wanted.stored())?), Some(now())) |
| 734 | } else { |
| 735 | (None, None) |
| 736 | }; |
| 737 | let values: Vec<worker::wasm_bindgen::JsValue> = vec![ |
| 738 | name.into(), |
| 739 | serde_json::to_string(combination)?.into(), |
| 740 | status.into(), |
| 741 | optional(conclusion), |
| 742 | optional(reason.as_deref()), |
| 743 | timeout.into(), |
| 744 | u32::from(continue_on_error).into(), |
| 745 | job.max_parallel.map_or(worker::wasm_bindgen::JsValue::NULL, Into::into), |
| 746 | optional(finished.as_deref()), |
| 747 | optional(labels_json.as_deref()), |
| 748 | optional(queued_at.as_deref()), |
| 749 | ]; |
| 750 | if index == 0 { |
| 751 | let mut bound = values; |
| 752 | bound.push(row.id.as_str().into()); |
| 753 | statements.push( |
| 754 | self.db |
| 755 | .prepare( |
| 756 | "UPDATE jobs SET name = ?, matrix = ?, status = ?, conclusion = ?, reason = ?, timeout_minutes = ?, |
| 757 | continue_on_error = ?, max_parallel = ?, finished_at = ?, labels = ?, queued_at = ? WHERE id = ?", |
| 758 | ) |
| 759 | .bind(&bound)?, |
| 760 | ); |
| 761 | } else { |
| 762 | let mut bound: Vec<worker::wasm_bindgen::JsValue> = vec![ |
| 763 | new_id("job", now_ms()).into(), |
| 764 | row.run_id.as_str().into(), |
| 765 | row.repo_id.as_str().into(), |
| 766 | row.namespace.as_str().into(), |
| 767 | row.key.as_str().into(), |
| 768 | (index as u32).into(), |
| 769 | row.needs.as_str().into(), |
| 770 | optional(row.call.as_deref()), |
| 771 | ]; |
| 772 | bound.extend(values); |
| 773 | statements.push( |
| 774 | self.db |
| 775 | .prepare( |
| 776 | "INSERT INTO jobs (id, run_id, repo_id, namespace, key, ordinal, needs, call, name, matrix, status, conclusion, reason, |
| 777 | timeout_minutes, continue_on_error, max_parallel, finished_at, labels, queued_at) |
| 778 | VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)", |
| 779 | ) |
| 780 | .bind(&bound)?, |
| 781 | ); |
| 782 | } |
| 783 | } |
| 784 | self.db.batch(statements).await?; |
| 785 | Ok(()) |
| 786 | } |
| 787 | |
| 788 | /// A job that calls a reusable workflow in the repository: that |
| 789 | /// workflow's jobs join the run under it, with the inputs it passes. |
| 790 | async fn call_workflow(&self, run: &RunRow, job: &workflow::Job, row: &JobRow, uses: &str, scope: &Scope<'_>) -> Result<()> { |
| 791 | let Some(local) = uses.strip_prefix("./") else { |
| 792 | return self |
| 793 | .fail_job(row, "Reusable workflows from other repositories are not called on g1t yet; ones in this repository (`./.g1t/workflows/…`) are.") |
| 794 | .await; |
| 795 | }; |
| 796 | let depth = row.call().and_then(|c| c["depth"].as_u64()).unwrap_or(0) + 1; |
| 797 | if depth > MAX_CALL_DEPTH { |
| 798 | return self.fail_job(row, &format!("Reusable workflows call each other more than {MAX_CALL_DEPTH} deep.")).await; |
| 799 | } |
| 800 | let local = local.split('@').next().unwrap_or(local).to_owned(); |
| 801 | let path = repo_path(&run.repo); |
| 802 | let Some(ws) = self.workspace_actor(&path.namespace).await? else { |
| 803 | return self.fail_job(row, "The workspace is gone.").await; |
| 804 | }; |
| 805 | // A repository moved from GitHub keeps saying `.github/…`. |
| 806 | let mut found = self.read_file(&path, &ws, &run.sha, &local).await?.map(|text| (local.clone(), text)); |
| 807 | if found.is_none() |
| 808 | && let Some(rest) = local.strip_prefix(".github/") |
| 809 | { |
| 810 | let moved = format!(".g1t/{rest}"); |
| 811 | found = self.read_file(&path, &ws, &run.sha, &moved).await?.map(|text| (moved, text)); |
| 812 | } |
| 813 | let Some((file, source)) = found else { |
| 814 | return self.fail_job(row, &format!("`{uses}` is not in the repository at this commit.")).await; |
| 815 | }; |
| 816 | let called = match workflow::parse(&source) { |
| 817 | Ok(called) => called, |
| 818 | Err(problem) => return self.fail_job(row, &format!("`{file}` does not read: {problem}")).await, |
| 819 | }; |
| 820 | let Some(trigger) = called.trigger("workflow_call") else { |
| 821 | return self.fail_job(row, &format!("`{file}` cannot be called: it has no `on: workflow_call`.")).await; |
| 822 | }; |
| 823 | // Inputs: what the caller passes, else the called workflow's defaults. |
| 824 | let given = match job.raw.get("with") { |
| 825 | Some(with) => match expr::interpolate_value(with, scope) { |
| 826 | Ok(Value::Object(given)) => given, |
| 827 | Ok(_) => Map::new(), |
| 828 | Err(problem) => return self.fail_job(row, &format!("Its `with` does not read: {problem}")).await, |
| 829 | }, |
| 830 | None => Map::new(), |
| 831 | }; |
| 832 | let mut inputs = Map::new(); |
| 833 | for (name, spec) in &trigger.inputs { |
| 834 | let value = given.get(name).cloned().or_else(|| spec.get("default").cloned()).unwrap_or(Value::Null); |
| 835 | if value.is_null() && spec.get("required").and_then(Value::as_bool) == Some(true) { |
| 836 | return self.fail_job(row, &format!("`{file}` needs the input `{name}`.")).await; |
| 837 | } |
| 838 | inputs.insert(name.clone(), value); |
| 839 | } |
| 840 | for (name, value) in given { |
| 841 | inputs.entry(name).or_insert(value); |
| 842 | } |
| 843 | let mut statements = Vec::new(); |
| 844 | for called_job in &called.jobs { |
| 845 | let needs: Vec<String> = called_job.needs.iter().map(|n| format!("{}/{n}", row.key)).collect(); |
| 846 | let call = json!({ |
| 847 | "role": "callee", "parent": row.key, "job": called_job.id, "path": file, |
| 848 | "source": source, "inputs": inputs, "depth": depth, |
| 849 | }); |
| 850 | statements.push( |
| 851 | self.db |
| 852 | .prepare("INSERT INTO jobs (id, run_id, repo_id, namespace, key, name, needs, status, call) VALUES (?, ?, ?, ?, ?, ?, ?, 'waiting', ?)") |
| 853 | .bind(&[ |
| 854 | new_id("job", now_ms()).into(), |
| 855 | row.run_id.as_str().into(), |
| 856 | row.repo_id.as_str().into(), |
| 857 | row.namespace.as_str().into(), |
| 858 | format!("{}/{}", row.key, called_job.id).into(), |
| 859 | format!("{} / {}", row.name, called_job.name.clone().unwrap_or(called_job.id.clone())).into(), |
| 860 | serde_json::to_string(&needs)?.into(), |
| 861 | serde_json::to_string(&call)?.into(), |
| 862 | ])?, |
| 863 | ); |
| 864 | } |
| 865 | statements.push( |
| 866 | self.db |
| 867 | .prepare("UPDATE jobs SET status = 'calling', call = ?, reason = ?, started_at = ? WHERE id = ?") |
| 868 | .bind(&[ |
| 869 | serde_json::to_string(&json!({ "role": "caller", "path": file, "source": source }))?.into(), |
| 870 | format!("Calls `{file}`.").into(), |
| 871 | now().into(), |
| 872 | row.id.as_str().into(), |
| 873 | ])?, |
| 874 | ); |
| 875 | self.db.batch(statements).await?; |
| 876 | Ok(()) |
| 877 | } |
| 878 | |
| 879 | /// A job that called a workflow, finished with its jobs: their result, |
| 880 | /// and the outputs the workflow declares. |
| 881 | async fn finish_call(&self, row: &JobRow, children: &[&JobRow]) -> Result<()> { |
| 882 | let call = row.call().unwrap_or_default(); |
| 883 | let called = call["source"].as_str().and_then(|s| workflow::parse(s).ok()); |
| 884 | let mut jobs_context = Map::new(); |
| 885 | let mut by_key: std::collections::BTreeMap<String, Vec<&JobRow>> = std::collections::BTreeMap::new(); |
| 886 | for child in children { |
| 887 | by_key.entry(child.key.clone()).or_default().push(child); |
| 888 | } |
| 889 | for (key, rows) in &by_key { |
| 890 | let mut outputs = Map::new(); |
| 891 | for child in rows { |
| 892 | if let Ok(Value::Object(more)) = serde_json::from_str::<Value>(&child.outputs) { |
| 893 | outputs.extend(more); |
| 894 | } |
| 895 | } |
| 896 | let id = key.rsplit('/').next().unwrap_or(key); |
| 897 | jobs_context.insert(id.to_owned(), json!({ "result": key_result(rows), "outputs": outputs })); |
| 898 | } |
| 899 | let inputs = children.first().and_then(|c| c.call()).map(|c| c["inputs"].clone()).unwrap_or(json!({})); |
| 900 | let mut contexts = Map::new(); |
| 901 | contexts.insert("jobs".into(), Value::Object(jobs_context)); |
| 902 | contexts.insert("inputs".into(), inputs); |
| 903 | let scope = Scope { contexts: &contexts, status: Status::Success, hash_files: None }; |
| 904 | let mut outputs = Map::new(); |
| 905 | if let Some(Value::Object(declared)) = called.as_ref().map(|w| { |
| 906 | let on = w.raw.get("on").or_else(|| w.raw.get("true")).cloned().unwrap_or(Value::Null); |
| 907 | on.get("workflow_call").and_then(|c| c.get("outputs")).cloned().unwrap_or(Value::Null) |
| 908 | }) { |
| 909 | for (name, spec) in declared { |
| 910 | if let Some(value) = spec.get("value") { |
| 911 | let value = expr::interpolate_value(value, &scope).unwrap_or(Value::Null); |
| 912 | outputs.insert(name, Value::String(expr::to_text(&value))); |
| 913 | } |
| 914 | } |
| 915 | } |
| 916 | let conclusion = key_result(children); |
| 917 | self.db |
| 918 | .prepare("UPDATE jobs SET status = 'completed', conclusion = ?, outputs = ?, finished_at = ? WHERE id = ? AND status = 'calling'") |
| 919 | .bind(&[conclusion.into(), serde_json::to_string(&outputs)?.into(), now().into(), row.id.as_str().into()])? |
| 920 | .run() |
| 921 | .await?; |
| 922 | Ok(()) |
| 923 | } |
| 924 | |
| 925 | /// A file's text at a commit, if it is there. |
| 926 | async fn read_file(&self, path: &RepoPath, ws: &g1t_contracts::User, sha: &str, file: &str) -> Result<Option<String>> { |
| 927 | let blob: Outcome<g1t_contracts::repos::BlobView> = g1t_kit::call( |
| 928 | &self.repos, |
| 929 | "blob", |
| 930 | &g1t_contracts::repos::BlobArgs { |
| 931 | path: path.clone(), |
| 932 | viewer: Some(ws.clone()), |
| 933 | git_ref: sha.to_owned(), |
| 934 | file_path: file.to_owned(), |
| 935 | }, |
| 936 | ) |
| 937 | .await?; |
| 938 | Ok(match blob { |
| 939 | Outcome::Ok(view) => view.text, |
| 940 | Outcome::Fail(_) => None, |
| 941 | }) |
| 942 | } |
| 943 | |
| 944 | async fn skip_job(&self, row: &JobRow, reason: Option<&str>) -> Result<()> { |
| 945 | self.db |
| 946 | .prepare("UPDATE jobs SET status = 'completed', conclusion = 'skipped', reason = ?, finished_at = ? WHERE id = ?") |
| 947 | .bind(&[optional(reason), now().into(), row.id.as_str().into()])? |
| 948 | .run() |
| 949 | .await?; |
| 950 | Ok(()) |
| 951 | } |
| 952 | |
| 953 | async fn fail_job(&self, row: &JobRow, reason: &str) -> Result<()> { |
| 954 | self.db |
| 955 | .prepare("UPDATE jobs SET status = 'completed', conclusion = 'failure', reason = ?, finished_at = ? WHERE id = ? AND status != 'completed'") |
| 956 | .bind(&[reason.into(), now().into(), row.id.as_str().into()])? |
| 957 | .run() |
| 958 | .await?; |
| 959 | Ok(()) |
| 960 | } |
| 961 | |
| 962 | /// Starts queued jobs, oldest first, while their workspace has room. |
| 963 | pub async fn start_queued(&self) -> Result<()> { |
| 964 | let queued = self |
| 965 | .db |
| 966 | // Self-hosted jobs are taken by their runners (runners.rs). |
| 967 | .prepare("SELECT * FROM jobs WHERE status = 'queued' AND labels IS NULL ORDER BY rowid LIMIT 50") |
| 968 | .all() |
| 969 | .await? |
| 970 | .results::<JobRow>()?; |
| 971 | let mut running: std::collections::HashMap<String, u32> = std::collections::HashMap::new(); |
| 972 | for job in queued { |
| 973 | let in_workspace = match running.get(&job.namespace) { |
| 974 | Some(n) => *n, |
| 975 | None => { |
| 976 | let n = self |
| 977 | .db |
| 978 | // Only g1t's own sandboxes count against the workspace's room. |
| 979 | .prepare("SELECT COUNT(*) AS n FROM jobs WHERE status = 'in_progress' AND namespace = ? AND runner_id IS NULL") |
| 980 | .bind(&[job.namespace.as_str().into()])? |
| 981 | .first::<Count>(None) |
| 982 | .await? |
| 983 | .map_or(0, |count| count.n); |
| 984 | running.insert(job.namespace.clone(), n); |
| 985 | n |
| 986 | } |
| 987 | }; |
| 988 | if in_workspace >= RUNNING_PER_WORKSPACE { |
| 989 | continue; |
| 990 | } |
| 991 | if let Some(max) = job.max_parallel { |
| 992 | let siblings = self |
| 993 | .db |
| 994 | .prepare("SELECT COUNT(*) AS n FROM jobs WHERE run_id = ? AND key = ? AND status = 'in_progress'") |
| 995 | .bind(&[job.run_id.as_str().into(), job.key.as_str().into()])? |
| 996 | .first::<Count>(None) |
| 997 | .await? |
| 998 | .map_or(0, |count| count.n); |
| 999 | if siblings >= max { |
| 1000 | continue; |
| 1001 | } |
| 1002 | } |
| 1003 | let token = random_hex(24); |
| 1004 | let at = now(); |
| 1005 | let claimed = self |
| 1006 | .db |
| 1007 | .prepare( |
| 1008 | "UPDATE jobs SET status = 'in_progress', token_hash = ?, started_at = ?, seen_at = ? WHERE id = ? AND status = 'queued' RETURNING id", |
| 1009 | ) |
| 1010 | .bind(&[sha256_hex(&token).into(), at.as_str().into(), at.as_str().into(), job.id.as_str().into()])? |
| 1011 | .first::<Value>(None) |
| 1012 | .await?; |
| 1013 | if claimed.is_none() { |
| 1014 | continue; |
| 1015 | } |
| 1016 | running.insert(job.namespace.clone(), in_workspace + 1); |
| 1017 | self.db |
| 1018 | .prepare("UPDATE runs SET status = 'in_progress', started_at = COALESCE(started_at, ?) WHERE id = ? AND status = 'queued'") |
| 1019 | .bind(&[at.as_str().into(), job.run_id.as_str().into()])? |
| 1020 | .run() |
| 1021 | .await?; |
| 1022 | let run = self.run_row(&job.run_id).await?; |
| 1023 | let repo: RepoPath = run.as_ref().map(|run| repo_path(&run.repo)).unwrap_or(RepoPath { |
| 1024 | namespace: job.namespace.clone(), |
| 1025 | name: String::new(), |
| 1026 | }); |
| 1027 | // Its environment and the machine it asked for, for the runner. |
| 1028 | let details = run.as_ref().map(|run| start_details(run, &job)).unwrap_or_default(); |
| 1029 | let started: Outcome<Value> = g1t_kit::call( |
| 1030 | &self.runner, |
| 1031 | "start_actions_job", |
| 1032 | &StartJobArgs { |
| 1033 | job: job.id.clone(), |
| 1034 | token, |
| 1035 | repo, |
| 1036 | timeout_minutes: job.timeout_minutes, |
| 1037 | workflow: run.as_ref().map(|run| run.path.clone()), |
| 1038 | environment: details.environment, |
| 1039 | trusted: run.as_ref().is_some_and(|run| run.trusted != 0), |
| 1040 | instance: details.instance, |
| 1041 | }, |
| 1042 | ) |
| 1043 | .await |
| 1044 | .unwrap_or_else(|error| fail(FailureCode::Conflict, format!("The runner could not be reached: {error}"))); |
| 1045 | if let Outcome::Fail(refused) = started { |
| 1046 | Box::pin(self.finish_job(&job.id, "failure", Some(&refused.message), None)).await?; |
| 1047 | } |
| 1048 | } |
| 1049 | Ok(()) |
| 1050 | } |
| 1051 | |
| 1052 | /// Finishes a job and moves its run along. |
| 1053 | pub async fn finish_job(&self, job_id: &str, conclusion: &str, reason: Option<&str>, outputs: Option<&Map<String, Value>>) -> Result<()> { |
| 1054 | let finished = self |
| 1055 | .db |
| 1056 | .prepare( |
| 1057 | "UPDATE jobs SET status = 'completed', conclusion = ?, reason = COALESCE(?, reason), outputs = COALESCE(?, outputs), |
| 1058 | finished_at = ?, token_hash = NULL WHERE id = ? AND status != 'completed' RETURNING *", |
| 1059 | ) |
| 1060 | .bind(&[ |
| 1061 | conclusion.into(), |
| 1062 | optional(reason), |
| 1063 | outputs.map(|o| serde_json::to_string(o).unwrap_or_default()).as_deref().map_or(worker::wasm_bindgen::JsValue::NULL, Into::into), |
| 1064 | now().into(), |
| 1065 | job_id.into(), |
| 1066 | ])? |
| 1067 | .first::<JobRow>(None) |
| 1068 | .await?; |
| 1069 | let Some(job) = finished else { return Ok(()) }; |
| 1070 | // A self-hosted runner's job: the runner is free again, and its time |
| 1071 | // is recorded, at nothing. |
| 1072 | if job.runner_id.is_some() { |
| 1073 | self.released(&job).await?; |
| 1074 | } |
| 1075 | // Steps still marked as going are not going any more. |
| 1076 | let mut steps: Vec<Value> = serde_json::from_str(&job.steps).unwrap_or_default(); |
| 1077 | let mut touched = false; |
| 1078 | for step in steps.iter_mut() { |
| 1079 | if step["status"] != "completed" { |
| 1080 | let was_running = step["status"] == "in_progress"; |
| 1081 | step["status"] = json!("completed"); |
| 1082 | step["conclusion"] = json!(if was_running { conclusion } else { "skipped" }); |
| 1083 | touched = true; |
| 1084 | } |
| 1085 | } |
| 1086 | if touched { |
| 1087 | self.db |
| 1088 | .prepare("UPDATE jobs SET steps = ? WHERE id = ?") |
| 1089 | .bind(&[serde_json::to_string(&steps)?.into(), job.id.as_str().into()])? |
| 1090 | .run() |
| 1091 | .await?; |
| 1092 | } |
| 1093 | // fail-fast: one failed combination stops the rest of its matrix. |
| 1094 | if conclusion == "failure" && job.continue_on_error == 0 && job.matrix.as_deref().is_some_and(|m| m != "{}") { |
| 1095 | let run = self.run_row(&job.run_id).await?; |
| 1096 | let fail_fast = run |
| 1097 | .as_ref() |
| 1098 | .and_then(|run| workflow::parse(&run.source).ok()) |
| 1099 | .and_then(|workflow| workflow.jobs.into_iter().find(|j| j.id == job.key)) |
| 1100 | .is_none_or(|j| j.fail_fast); |
| 1101 | if fail_fast { |
| 1102 | let siblings = self |
| 1103 | .db |
| 1104 | .prepare("SELECT * FROM jobs WHERE run_id = ? AND key = ? AND status != 'completed'") |
| 1105 | .bind(&[job.run_id.as_str().into(), job.key.as_str().into()])? |
| 1106 | .all() |
| 1107 | .await? |
| 1108 | .results::<JobRow>()?; |
| 1109 | for sibling in siblings { |
| 1110 | self.stop_job(&sibling, "Another job of its matrix failed, and the matrix is fail-fast.").await?; |
| 1111 | } |
| 1112 | } |
| 1113 | } |
| 1114 | Box::pin(self.advance(&job.run_id)).await |
| 1115 | } |
| 1116 | |
| 1117 | /// Cancels a job, stopping its sandbox if it has one. |
| 1118 | async fn stop_job(&self, job: &JobRow, reason: &str) -> Result<()> { |
| 1119 | // A self-hosted runner hears it was cancelled on its next poll. |
| 1120 | if job.status == "in_progress" && job.runner_id.is_none() { |
| 1121 | let _: Result<Value> = g1t_kit::call(&self.runner, "stop_actions_job", &json!({ "job": job.id })).await; |
| 1122 | } |
| 1123 | let stopped = self |
| 1124 | .db |
| 1125 | .prepare("UPDATE jobs SET status = 'completed', conclusion = 'cancelled', reason = ?, finished_at = ?, token_hash = NULL WHERE id = ? AND status != 'completed' RETURNING *") |
| 1126 | .bind(&[reason.into(), now().into(), job.id.as_str().into()])? |
| 1127 | .first::<JobRow>(None) |
| 1128 | .await?; |
| 1129 | if let Some(stopped) = stopped.filter(|row| row.runner_id.is_some()) { |
| 1130 | self.released(&stopped).await?; |
| 1131 | } |
| 1132 | Ok(()) |
| 1133 | } |
| 1134 | |
| 1135 | /// Finishes the run when every job has. |
| 1136 | async fn finish_if_done(&self, run_id: &str) -> Result<()> { |
| 1137 | let Some(run) = self.run_row(run_id).await? else { return Ok(()) }; |
| 1138 | if run.status == "completed" || run.status == "pending" { |
| 1139 | return Ok(()); |
| 1140 | } |
| 1141 | let jobs = self.job_rows(run_id).await?; |
| 1142 | if !jobs.iter().all(|job| job.status == "completed") { |
| 1143 | return Ok(()); |
| 1144 | } |
| 1145 | self.finish_run(&run, None).await |
| 1146 | } |
| 1147 | |
| 1148 | async fn finish_run(&self, run: &RunRow, error: Option<&str>) -> Result<()> { |
| 1149 | let jobs = self.job_rows(&run.id).await?; |
| 1150 | let rows: Vec<&JobRow> = jobs.iter().collect(); |
| 1151 | let conclusion = if error.is_some() { |
| 1152 | "failure" |
| 1153 | } else if run.conclusion.as_deref() == Some("cancelled") { |
| 1154 | "cancelled" |
| 1155 | } else if rows.is_empty() { |
| 1156 | "skipped" |
| 1157 | } else { |
| 1158 | key_result(&rows) |
| 1159 | }; |
| 1160 | let done = self |
| 1161 | .db |
| 1162 | .prepare("UPDATE runs SET status = 'completed', conclusion = ?, error = COALESCE(?, error), finished_at = ? WHERE id = ? AND status != 'completed' RETURNING id") |
| 1163 | .bind(&[conclusion.into(), optional(error), now().into(), run.id.as_str().into()])? |
| 1164 | .first::<Value>(None) |
| 1165 | .await?; |
| 1166 | if done.is_none() { |
| 1167 | return Ok(()); |
| 1168 | } |
| 1169 | self.report_status(run, conclusion).await?; |
| 1170 | let published: Result<()> = g1t_kit::call( |
| 1171 | &self.events, |
| 1172 | "publish", |
| 1173 | &g1t_contracts::events::Publish { |
| 1174 | events: vec![g1t_contracts::events::NewEvent { |
| 1175 | kind: "workflow.completed", |
| 1176 | source: "actions", |
| 1177 | repo_id: Some(run.repo_id.clone()), |
| 1178 | actor: run.actor_id.clone(), |
| 1179 | data: g1t_contracts::events::WorkflowEvent { |
| 1180 | run_id: run.id.clone(), |
| 1181 | repo_id: run.repo_id.clone(), |
| 1182 | workflow: run.name.clone(), |
| 1183 | path: run.path.clone(), |
| 1184 | number: run.number, |
| 1185 | event: run.event.clone(), |
| 1186 | conclusion: conclusion.to_owned(), |
| 1187 | git_ref: run.git_ref.clone(), |
| 1188 | sha: run.sha.clone(), |
| 1189 | pull: run.pull, |
| 1190 | }, |
| 1191 | }], |
| 1192 | }, |
| 1193 | ) |
| 1194 | .await; |
| 1195 | if let Err(error) = published { |
| 1196 | worker::console_error!("actions: could not publish workflow.completed: {error}"); |
| 1197 | } |
| 1198 | // The next run waiting in its concurrency group. |
| 1199 | if let Some(group) = &run.concurrency_group { |
| 1200 | let next = self |
| 1201 | .db |
| 1202 | .prepare("SELECT * FROM runs WHERE repo_id = ? AND concurrency_group = ? AND status = 'pending' ORDER BY id LIMIT 1") |
| 1203 | .bind(&[run.repo_id.as_str().into(), group.as_str().into()])? |
| 1204 | .first::<RunRow>(None) |
| 1205 | .await?; |
| 1206 | if let Some(next) = next { |
| 1207 | self.db.prepare("UPDATE runs SET status = 'queued' WHERE id = ?").bind(&[next.id.as_str().into()])?.run().await?; |
| 1208 | Box::pin(self.advance(&next.id)).await?; |
| 1209 | } |
| 1210 | } |
| 1211 | Ok(()) |
| 1212 | } |
| 1213 | |
| 1214 | /// Tells the pull request (or commit) how the run went, as a status. |
| 1215 | async fn report_status(&self, run: &RunRow, conclusion: &str) -> Result<()> { |
| 1216 | let state = match conclusion { |
| 1217 | "success" | "skipped" => "success", |
| 1218 | "cancelled" => "error", |
| 1219 | _ => "failure", |
| 1220 | }; |
| 1221 | let _: Result<Value> = g1t_kit::call( |
| 1222 | &self.work, |
| 1223 | "set_commit_status", |
| 1224 | &json!({ |
| 1225 | "repoId": run.repo_id, |
| 1226 | "sha": run.sha, |
| 1227 | "context": format!("{} / {}", run.name, run.event), |
| 1228 | "state": state, |
| 1229 | "description": format!("{} {}", run.name, match conclusion { |
| 1230 | "success" => "passed", |
| 1231 | "skipped" => "was skipped", |
| 1232 | "cancelled" => "was cancelled", |
| 1233 | _ => "failed", |
| 1234 | }), |
| 1235 | "targetUrl": format!("{SITE}/{}/actions/runs/{}", run.repo, run.id), |
| 1236 | }), |
| 1237 | ) |
| 1238 | .await; |
| 1239 | Ok(()) |
| 1240 | } |
| 1241 | |
| 1242 | /// Tells the pull request a run has started on its head. |
| 1243 | pub async fn report_pending(&self, run: &RunRow) -> Result<()> { |
| 1244 | let _: Result<Value> = g1t_kit::call( |
| 1245 | &self.work, |
| 1246 | "set_commit_status", |
| 1247 | &json!({ |
| 1248 | "repoId": run.repo_id, |
| 1249 | "sha": run.sha, |
| 1250 | "context": format!("{} / {}", run.name, run.event), |
| 1251 | "state": "pending", |
| 1252 | "description": format!("{} is running", run.name), |
| 1253 | "targetUrl": format!("{SITE}/{}/actions/runs/{}", run.repo, run.id), |
| 1254 | }), |
| 1255 | ) |
| 1256 | .await; |
| 1257 | Ok(()) |
| 1258 | } |
| 1259 | |
| 1260 | /// Cancels a run: its waiting and queued jobs, and stops its running ones. |
| 1261 | pub async fn cancel_run(&self, run: &RunRow, reason: &str) -> Result<()> { |
| 1262 | self.db |
| 1263 | .prepare("UPDATE runs SET conclusion = 'cancelled' WHERE id = ? AND status != 'completed'") |
| 1264 | .bind(&[run.id.as_str().into()])? |
| 1265 | .run() |
| 1266 | .await?; |
| 1267 | for job in self.job_rows(&run.id).await?.iter().filter(|job| job.status != "completed") { |
| 1268 | self.stop_job(job, reason).await?; |
| 1269 | } |
| 1270 | if run.status == "pending" { |
| 1271 | self.db.prepare("UPDATE runs SET status = 'queued' WHERE id = ?").bind(&[run.id.as_str().into()])?.run().await?; |
| 1272 | } |
| 1273 | self.advance(&run.id).await |
| 1274 | } |
| 1275 | |
| 1276 | /// Cancels every run of the repository that has not finished, for |
| 1277 | /// `repo.deleted` and `repo.archived`. |
| 1278 | pub async fn stop_runs(&self, repo_id: &str) -> Result<()> { |
| 1279 | let runs = self |
| 1280 | .db |
| 1281 | .prepare("SELECT * FROM runs WHERE repo_id = ? AND status != 'completed'") |
| 1282 | .bind(&[repo_id.into()])? |
| 1283 | .all() |
| 1284 | .await? |
| 1285 | .results::<RunRow>()?; |
| 1286 | for run in runs { |
| 1287 | self.cancel_run(&run, "The repository was archived or deleted.").await?; |
| 1288 | } |
| 1289 | Ok(()) |
| 1290 | } |
| 1291 | |
| 1292 | pub async fn cancel(&self, a: RunActionArgs) -> Result<Outcome<WorkflowRun>> { |
| 1293 | if let Outcome::Fail(refused) = self.may(&a.actor, &a.repo, Capability::Run).await? { |
| 1294 | return Ok(Outcome::Fail(refused)); |
| 1295 | } |
| 1296 | let run = check!(self.run_in(&a.repo, &a.id).await?); |
| 1297 | if run.status == "completed" { |
| 1298 | return Ok(fail(FailureCode::Conflict, "The run has already finished.")); |
| 1299 | } |
| 1300 | self.cancel_run(&run, &format!("{} cancelled the run.", a.actor.username)).await?; |
| 1301 | self.run_summary(&run.id).await |
| 1302 | } |
| 1303 | |
| 1304 | /// Runs again: every job, or with `failed_only` those that did not |
| 1305 | /// succeed and the jobs that need them. |
| 1306 | pub async fn rerun(&self, a: RunActionArgs) -> Result<Outcome<WorkflowRun>> { |
| 1307 | if let Outcome::Fail(refused) = self.may(&a.actor, &a.repo, Capability::Run).await? { |
| 1308 | return Ok(Outcome::Fail(refused)); |
| 1309 | } |
| 1310 | let run = check!(self.run_in(&a.repo, &a.id).await?); |
| 1311 | if run.status != "completed" { |
| 1312 | return Ok(fail(FailureCode::Conflict, "The run is still going: cancel it first.")); |
| 1313 | } |
| 1314 | // Nothing starts again on an archived repository. |
| 1315 | match self.visible_repo(&a.repo, &Some(a.actor.clone())).await? { |
| 1316 | Some(repo) if repo.archived() => { |
| 1317 | return Ok(fail(FailureCode::Forbidden, g1t_contracts::repos::archived_message(&repo.namespace, &repo.name))); |
| 1318 | } |
| 1319 | Some(_) => {} |
| 1320 | None => return Ok(fail(FailureCode::NotFound, "There is no such repository.")), |
| 1321 | } |
| 1322 | if run.error.is_some() { |
| 1323 | return Ok(fail(FailureCode::Conflict, "This run never started: fix the workflow file and push again.")); |
| 1324 | } |
| 1325 | let jobs = self.job_rows(&run.id).await?; |
| 1326 | let workflow = workflow::parse(&run.source).ok(); |
| 1327 | // Which keys run again: failed ones and, transitively, those needing them. |
| 1328 | let mut again: Vec<String> = Vec::new(); |
| 1329 | for key in workflow.as_ref().map(|w| w.job_order()).unwrap_or_default() { |
| 1330 | let rows: Vec<&JobRow> = jobs.iter().filter(|j| j.key == key).collect(); |
| 1331 | let failed = rows.iter().any(|row| row.conclusion.as_deref() != Some("success")); |
| 1332 | let needs_again = rows.first().is_some_and(|row| row.needs().iter().any(|need| again.contains(need))); |
| 1333 | if !a.failed_only || failed || needs_again { |
| 1334 | again.push(key.to_owned()); |
| 1335 | } |
| 1336 | } |
| 1337 | if again.is_empty() { |
| 1338 | return Ok(fail(FailureCode::Conflict, "Every job succeeded: there is nothing to run again.")); |
| 1339 | } |
| 1340 | let mut statements = Vec::new(); |
| 1341 | for key in &again { |
| 1342 | statements.push(self.db.prepare("DELETE FROM logs WHERE job_id IN (SELECT id FROM jobs WHERE run_id = ? AND key = ?)").bind(&[run.id.as_str().into(), key.as_str().into()])?); |
| 1343 | statements.push(self.db.prepare("DELETE FROM jobs WHERE run_id = ? AND key = ? AND ordinal > 0").bind(&[run.id.as_str().into(), key.as_str().into()])?); |
| 1344 | // The jobs of a workflow it called are made again when it calls it again. |
| 1345 | statements.push( |
| 1346 | self.db |
| 1347 | .prepare("DELETE FROM logs WHERE job_id IN (SELECT id FROM jobs WHERE run_id = ? AND key LIKE ?)") |
| 1348 | .bind(&[run.id.as_str().into(), format!("{key}/%").into()])?, |
| 1349 | ); |
| 1350 | statements.push( |
| 1351 | self.db |
| 1352 | .prepare("DELETE FROM jobs WHERE run_id = ? AND key LIKE ?") |
| 1353 | .bind(&[run.id.as_str().into(), format!("{key}/%").into()])?, |
| 1354 | ); |
| 1355 | statements.push( |
| 1356 | self.db |
| 1357 | .prepare( |
| 1358 | "UPDATE jobs SET status = 'waiting', conclusion = NULL, steps = '[]', annotations = '[]', outputs = '{}', reason = NULL, |
| 1359 | matrix = NULL, call = NULL, token_hash = NULL, seen_at = NULL, started_at = NULL, finished_at = NULL, |
| 1360 | labels = NULL, queued_at = NULL, runner_id = NULL, runner_name = NULL WHERE run_id = ? AND key = ?", |
| 1361 | ) |
| 1362 | .bind(&[run.id.as_str().into(), key.as_str().into()])?, |
| 1363 | ); |
| 1364 | } |
| 1365 | statements.push( |
| 1366 | self.db |
| 1367 | .prepare("UPDATE runs SET status = 'queued', conclusion = NULL, attempt = attempt + 1, started_at = NULL, finished_at = NULL WHERE id = ?") |
| 1368 | .bind(&[run.id.as_str().into()])?, |
| 1369 | ); |
| 1370 | self.db.batch(statements).await?; |
| 1371 | if let Some(run) = self.run_row(&run.id).await? { |
| 1372 | self.report_pending(&run).await?; |
| 1373 | } |
| 1374 | self.advance(&run.id).await?; |
| 1375 | self.run_summary(&run.id).await |
| 1376 | } |
| 1377 | |
| 1378 | pub async fn run_in(&self, repo: &RepoPath, id: &str) -> Result<Outcome<RunRow>> { |
| 1379 | let row = self |
| 1380 | .db |
| 1381 | .prepare("SELECT * FROM runs WHERE id = ? AND lower(repo) = lower(?)") |
| 1382 | .bind(&[id.into(), format!("{}/{}", repo.namespace, repo.name).into()])? |
| 1383 | .first::<RunRow>(None) |
| 1384 | .await?; |
| 1385 | Ok(row.map_or_else(|| fail(FailureCode::NotFound, "No such run."), Outcome::Ok)) |
| 1386 | } |
| 1387 | |
| 1388 | // --- The sandbox's side ----------------------------------------------------- |
| 1389 | |
| 1390 | pub(crate) async fn job_for_token(&self, a: &JobCallArgs) -> Result<Outcome<JobRow>> { |
| 1391 | let job = self.db.prepare("SELECT * FROM jobs WHERE id = ?").bind(&[a.job.as_str().into()])?.first::<JobRow>(None).await?; |
| 1392 | Ok(match job { |
| 1393 | Some(job) if job.status == "in_progress" && job.token_hash.as_deref().is_some_and(|hash| same(hash, &sha256_hex(&a.token))) => { |
| 1394 | Outcome::Ok(job) |
| 1395 | } |
| 1396 | _ => fail(FailureCode::Unauthenticated, "That job is not running, or the token is not its."), |
| 1397 | }) |
| 1398 | } |
| 1399 | |
| 1400 | /// `job_auth`: which run and repository a running job's token is for, |
| 1401 | /// so the API can keep its artifacts and cache. |
| 1402 | pub async fn job_auth(&self, a: JobCallArgs) -> Result<Outcome<Value>> { |
| 1403 | let job = check!(self.job_for_token(&a).await?); |
| 1404 | Ok(Outcome::Ok(json!({ "run": job.run_id, "repoId": job.repo_id }))) |
| 1405 | } |
| 1406 | |
| 1407 | /// `job_spec`: everything the sandbox needs to run the job. |
| 1408 | pub async fn job_spec(&self, a: JobCallArgs) -> Result<Outcome<Value>> { |
| 1409 | let job = check!(self.job_for_token(&a).await?); |
| 1410 | let Some(run) = self.run_row(&job.run_id).await? else { |
| 1411 | return Ok(fail(FailureCode::NotFound, "No such run.")); |
| 1412 | }; |
| 1413 | let Ok(caller) = workflow::parse(&run.source) else { |
| 1414 | return Ok(fail(FailureCode::Invalid, "The workflow no longer reads.")); |
| 1415 | }; |
| 1416 | // A called workflow's job runs as that workflow defines it. |
| 1417 | let callee = job.callee(); |
| 1418 | let (workflow, spec, call_inputs) = match callee { |
| 1419 | Some((called, spec, call)) => (called, spec, Some(call["inputs"].clone())), |
| 1420 | None => match caller.jobs.iter().find(|j| j.id == job.key) { |
| 1421 | Some(spec) => (caller.clone(), spec.clone(), None), |
| 1422 | None => return Ok(fail(FailureCode::NotFound, "The job is not in the workflow.")), |
| 1423 | }, |
| 1424 | }; |
| 1425 | let spec = &spec; |
| 1426 | let repo = repo_path(&run.repo); |
| 1427 | let trusted = run.trusted != 0; |
| 1428 | // The job's `environment:`, by name: entries with a value for it give |
| 1429 | // that value instead of their default, as GitHub's environment |
| 1430 | // secrets do. |
| 1431 | let environment: Option<String> = match spec.raw.get("environment") { |
| 1432 | Some(Value::String(name)) if !name.contains("${{") => Some(name.clone()), |
| 1433 | Some(Value::Object(env)) => env.get("name").and_then(Value::as_str).filter(|n| !n.contains("${{")).map(str::to_owned), |
| 1434 | _ => None, |
| 1435 | }; |
| 1436 | // G1T_TOKEN, and GITHUB_TOKEN as its alias: the workspace's own |
| 1437 | // token, for as long as the job may run. |
| 1438 | let token = if trusted { |
| 1439 | match self.workspace_actor(&repo.namespace).await? { |
| 1440 | Some(workspace) => { |
| 1441 | let created: CreatedAccessToken = g1t_kit::call( |
| 1442 | &self.identity, |
| 1443 | "create_access_token", |
| 1444 | &CreateAccessTokenArgs { |
| 1445 | user: workspace, |
| 1446 | name: format!("G1T_TOKEN for {} run {}", run.repo, run.number), |
| 1447 | ttl_seconds: Some(u64::from(job.timeout_minutes) * 60 + 600), |
| 1448 | scopes: None, |
| 1449 | listed: false, |
| 1450 | }, |
| 1451 | ) |
| 1452 | .await?; |
| 1453 | created.token |
| 1454 | } |
| 1455 | None => String::new(), |
| 1456 | } |
| 1457 | } else { |
| 1458 | String::new() |
| 1459 | }; |
| 1460 | // A run that is not trusted (a pull request from outside the |
| 1461 | // workspace) gets no secrets and an empty token. |
| 1462 | let mut secrets = if trusted { |
| 1463 | self.secrets_for(&run.repo_id, &run.repo, environment.as_deref(), true).await? |
| 1464 | } else { |
| 1465 | Map::new() |
| 1466 | }; |
| 1467 | secrets.insert("G1T_TOKEN".into(), Value::String(token.clone())); |
| 1468 | secrets.insert("GITHUB_TOKEN".into(), Value::String(token.clone())); |
| 1469 | let masks: Vec<String> = secrets.values().filter_map(|v| v.as_str()).filter(|v| v.len() >= 4).map(str::to_owned).collect(); |
| 1470 | let vars = self.variables_for(&run.repo_id, &run.repo, environment.as_deref(), trusted).await?; |
| 1471 | |
| 1472 | let jobs = self.job_rows(&run.id).await?; |
| 1473 | let mut needs = Map::new(); |
| 1474 | // In a called workflow, its jobs' keys sit under the job that called it. |
| 1475 | let parent = job.call().filter(|c| c["role"] == "callee").and_then(|c| c["parent"].as_str().map(str::to_owned)); |
| 1476 | for need in &spec.needs { |
| 1477 | let key = match &parent { |
| 1478 | Some(parent) => format!("{parent}/{need}"), |
| 1479 | None => need.clone(), |
| 1480 | }; |
| 1481 | let rows: Vec<&JobRow> = jobs.iter().filter(|row| row.key == key).collect(); |
| 1482 | let mut outputs = Map::new(); |
| 1483 | for row in &rows { |
| 1484 | if let Ok(Value::Object(more)) = serde_json::from_str::<Value>(&row.outputs) { |
| 1485 | outputs.extend(more); |
| 1486 | } |
| 1487 | } |
| 1488 | needs.insert(need.clone(), json!({ "result": key_result(&rows), "outputs": outputs })); |
| 1489 | } |
| 1490 | let siblings = jobs.iter().filter(|row| row.key == job.key).count(); |
| 1491 | let matrix: Value = job.matrix.as_deref().and_then(|m| serde_json::from_str(m).ok()).unwrap_or(json!({})); |
| 1492 | let info = run.info(); |
| 1493 | let mut github = info.context(&job.key, &token, run.action.as_deref()); |
| 1494 | github["token"] = json!(token); |
| 1495 | // On a self-hosted runner, `runner` and `RUNNER_*` describe that |
| 1496 | // machine rather than g1t's sandbox. |
| 1497 | let mut variables = info.variables(&job.key); |
| 1498 | let runner = match &job.runner_id { |
| 1499 | Some(id) => self.runner_context_for(id, &mut variables).await?, |
| 1500 | None => runner_context(), |
| 1501 | }; |
| 1502 | |
| 1503 | // Where to check out: a pull request's fork, or the repository. |
| 1504 | let clone_url = match run.pull { |
| 1505 | Some(number) if run.event.starts_with("pull_request") && run.event != "pull_request_target" => { |
| 1506 | let located: Outcome<g1t_contracts::work::PullDetail> = g1t_kit::call( |
| 1507 | &self.work, |
| 1508 | "get_pull", |
| 1509 | &g1t_contracts::work::ViewArgs { |
| 1510 | repo: repo.clone(), |
| 1511 | number, |
| 1512 | viewer: self.workspace_actor(&repo.namespace).await?, |
| 1513 | after_seq: 0, |
| 1514 | }, |
| 1515 | ) |
| 1516 | .await?; |
| 1517 | match located { |
| 1518 | Outcome::Ok(detail) => match detail.pull.fork { |
| 1519 | Some(fork) => format!("{SITE}/{}/{}.git", fork.namespace, fork.name), |
| 1520 | None => format!("{SITE}/{}.git", run.repo), |
| 1521 | }, |
| 1522 | Outcome::Fail(_) => format!("{SITE}/{}.git", run.repo), |
| 1523 | } |
| 1524 | } |
| 1525 | _ => format!("{SITE}/{}.git", run.repo), |
| 1526 | }; |
| 1527 | |
| 1528 | Ok(Outcome::Ok(json!({ |
| 1529 | "job": job.id, |
| 1530 | "run": run.id, |
| 1531 | "key": job.key, |
| 1532 | "name": job.name, |
| 1533 | "spec": spec.raw, |
| 1534 | "workflow": { |
| 1535 | "env": workflow.env, |
| 1536 | "defaults": workflow.raw.get("defaults").cloned().unwrap_or(Value::Null), |
| 1537 | }, |
| 1538 | "github": github, |
| 1539 | "variables": variables, |
| 1540 | "event": info.event, |
| 1541 | "contexts": { |
| 1542 | "vars": vars, |
| 1543 | "secrets": secrets, |
| 1544 | "inputs": call_inputs.unwrap_or_else(|| Value::Object(run.inputs())), |
| 1545 | "matrix": matrix, |
| 1546 | "needs": needs, |
| 1547 | "strategy": { |
| 1548 | "fail-fast": spec.fail_fast, |
| 1549 | "job-index": job.ordinal, |
| 1550 | "job-total": siblings, |
| 1551 | "max-parallel": spec.max_parallel.unwrap_or(siblings as u32), |
| 1552 | }, |
| 1553 | "runner": runner, |
| 1554 | }, |
| 1555 | "checkout": { |
| 1556 | "repository": run.repo, |
| 1557 | "url": clone_url, |
| 1558 | "sha": run.sha, |
| 1559 | "ref": run.git_ref, |
| 1560 | "token": token, |
| 1561 | }, |
| 1562 | "timeoutMinutes": job.timeout_minutes, |
| 1563 | "masks": masks, |
| 1564 | }))) |
| 1565 | } |
| 1566 | |
| 1567 | /// `job_report`: the sandbox telling how the job is going. |
| 1568 | pub async fn job_report(&self, a: JobCallArgs) -> Result<Outcome<Value>> { |
| 1569 | let job = check!(self.job_for_token(&a).await?); |
| 1570 | let report = &a.report; |
| 1571 | let at = now(); |
| 1572 | match report["kind"].as_str().unwrap_or_default() { |
| 1573 | "steps" => { |
| 1574 | // The list can grow as the job goes (post steps), so steps |
| 1575 | // already reported keep where they stand. |
| 1576 | let known: Vec<Value> = serde_json::from_str(&job.steps).unwrap_or_default(); |
| 1577 | let steps: Vec<Value> = report["steps"] |
| 1578 | .as_array() |
| 1579 | .map(|names| { |
| 1580 | names |
| 1581 | .iter() |
| 1582 | .enumerate() |
| 1583 | .map(|(i, name)| match known.get(i) { |
| 1584 | Some(step) if step["status"] != "queued" => step.clone(), |
| 1585 | _ => json!({ "number": i + 1, "name": expr::to_text(name), "status": "queued", "conclusion": null, "startedAt": null, "finishedAt": null }), |
| 1586 | }) |
| 1587 | .collect() |
| 1588 | }) |
| 1589 | .unwrap_or_default(); |
| 1590 | self.db |
| 1591 | .prepare("UPDATE jobs SET steps = ?, seen_at = ? WHERE id = ?") |
| 1592 | .bind(&[serde_json::to_string(&steps)?.into(), at.as_str().into(), job.id.as_str().into()])? |
| 1593 | .run() |
| 1594 | .await?; |
| 1595 | } |
| 1596 | "step" => { |
| 1597 | let number = report["number"].as_u64().unwrap_or(0) as usize; |
| 1598 | let mut steps: Vec<Value> = serde_json::from_str(&job.steps).unwrap_or_default(); |
| 1599 | if let Some(step) = number.checked_sub(1).and_then(|i| steps.get_mut(i)) { |
| 1600 | let status = report["status"].as_str().unwrap_or("in_progress"); |
| 1601 | step["status"] = json!(status); |
| 1602 | if status == "in_progress" { |
| 1603 | step["startedAt"] = json!(at); |
| 1604 | } |
| 1605 | if status == "completed" { |
| 1606 | step["finishedAt"] = json!(at); |
| 1607 | step["conclusion"] = report["conclusion"].clone(); |
| 1608 | } |
| 1609 | if let Some(name) = report["name"].as_str() { |
| 1610 | step["name"] = json!(name); |
| 1611 | } |
| 1612 | } |
| 1613 | self.db |
| 1614 | .prepare("UPDATE jobs SET steps = ?, seen_at = ? WHERE id = ?") |
| 1615 | .bind(&[serde_json::to_string(&steps)?.into(), at.as_str().into(), job.id.as_str().into()])? |
| 1616 | .run() |
| 1617 | .await?; |
| 1618 | } |
| 1619 | "log" => { |
| 1620 | let mut text = report["text"].as_str().unwrap_or_default().to_owned(); |
| 1621 | if text.len() > MAX_CHUNK_BYTES { |
| 1622 | let mut cut = MAX_CHUNK_BYTES; |
| 1623 | while !text.is_char_boundary(cut) { |
| 1624 | cut -= 1; |
| 1625 | } |
| 1626 | text.truncate(cut); |
| 1627 | } |
| 1628 | #[derive(Deserialize)] |
| 1629 | struct Size { |
| 1630 | n: Option<f64>, |
| 1631 | seq: Option<f64>, |
| 1632 | } |
| 1633 | let size = self |
| 1634 | .db |
| 1635 | .prepare("SELECT SUM(LENGTH(text)) AS n, MAX(seq) AS seq FROM logs WHERE job_id = ?") |
| 1636 | .bind(&[job.id.as_str().into()])? |
| 1637 | .first::<Size>(None) |
| 1638 | .await?; |
| 1639 | let (used, seq) = size.map_or((0, 0.0), |s| (s.n.unwrap_or(0.0) as usize, s.seq.unwrap_or(0.0))); |
| 1640 | if used < MAX_LOG_BYTES { |
| 1641 | if used + text.len() >= MAX_LOG_BYTES { |
| 1642 | text.push_str("\n… The log reached its limit of 4 MB; the rest is not kept.\n"); |
| 1643 | } |
| 1644 | self.db |
| 1645 | .prepare("INSERT INTO logs (job_id, seq, step, text) VALUES (?, ?, ?, ?)") |
| 1646 | .bind(&[job.id.as_str().into(), // Numbers go to D1 as f64: a u64 would be a BigInt, which it refuses. |
| 1647 | (seq + 1.0).into(), (report["step"].as_u64().unwrap_or(0) as u32).into(), text.into()])? |
| 1648 | .run() |
| 1649 | .await?; |
| 1650 | } |
| 1651 | self.db.prepare("UPDATE jobs SET seen_at = ? WHERE id = ?").bind(&[at.into(), job.id.as_str().into()])?.run().await?; |
| 1652 | } |
| 1653 | "annotation" => { |
| 1654 | let mut annotations: Vec<Value> = serde_json::from_str(&job.annotations).unwrap_or_default(); |
| 1655 | if annotations.len() < MAX_ANNOTATIONS { |
| 1656 | annotations.push(json!({ |
| 1657 | "level": report["level"].as_str().unwrap_or("notice"), |
| 1658 | "message": report["message"].as_str().unwrap_or_default().chars().take(4000).collect::<String>(), |
| 1659 | "title": report["title"], |
| 1660 | "file": report["file"], |
| 1661 | "line": report["line"], |
| 1662 | })); |
| 1663 | self.db |
| 1664 | .prepare("UPDATE jobs SET annotations = ?, seen_at = ? WHERE id = ?") |
| 1665 | .bind(&[serde_json::to_string(&annotations)?.into(), at.as_str().into(), job.id.as_str().into()])? |
| 1666 | .run() |
| 1667 | .await?; |
| 1668 | } |
| 1669 | } |
| 1670 | "done" => { |
| 1671 | let conclusion = report["conclusion"] |
| 1672 | .as_str() |
| 1673 | .filter(|c| matches!(*c, "success" | "failure" | "cancelled")) |
| 1674 | .unwrap_or("failure"); |
| 1675 | let outputs = report["outputs"].as_object().cloned(); |
| 1676 | Box::pin(self.finish_job(&job.id, conclusion, report["reason"].as_str(), outputs.as_ref())).await?; |
| 1677 | } |
| 1678 | other => return Ok(fail(FailureCode::Invalid, format!("There is no report called `{other}`."))), |
| 1679 | } |
| 1680 | Ok(Outcome::Ok(json!({ "ok": true }))) |
| 1681 | } |
| 1682 | |
| 1683 | // --- Every minute --------------------------------------------------------------- |
| 1684 | |
| 1685 | pub async fn on_minute(&self, now_ms: u64) -> Result<()> { |
| 1686 | let minute = now_ms / 60_000 * 60_000; |
| 1687 | if let Err(error) = self.run_schedules(minute).await { |
| 1688 | worker::console_error!("actions: schedules failed: {error}"); |
| 1689 | } |
| 1690 | // Jobs whose sandbox went quiet or ran past their time. |
| 1691 | let running = self.db.prepare("SELECT * FROM jobs WHERE status = 'in_progress'").all().await?.results::<JobRow>()?; |
| 1692 | for job in running { |
| 1693 | // Times in g1t's format compare as text. |
| 1694 | let before = |ms: u64| rfc3339(now_ms.saturating_sub(ms)); |
| 1695 | let silent = job.seen_at.as_deref().is_some_and(|seen| seen < before(SILENT_MS).as_str()); |
| 1696 | let limit = (u64::from(job.timeout_minutes) * 60 + 120) * 1000; |
| 1697 | let over = job.started_at.as_deref().is_some_and(|started| started < before(limit).as_str()); |
| 1698 | if over { |
| 1699 | let reason = format!("It ran longer than its time limit of {} minutes.", job.timeout_minutes); |
| 1700 | // A self-hosted runner is told to stop on its next poll. |
| 1701 | if job.runner_id.is_none() { |
| 1702 | let _: Result<Value> = g1t_kit::call(&self.runner, "stop_actions_job", &json!({ "job": job.id })).await; |
| 1703 | } |
| 1704 | self.finish_job(&job.id, "failure", Some(&reason), None).await?; |
| 1705 | } else if silent { |
| 1706 | let reason = match &job.runner_name { |
| 1707 | Some(name) => format!("The self-hosted runner {name} stopped answering."), |
| 1708 | None => "The runner stopped answering.".to_owned(), |
| 1709 | }; |
| 1710 | self.finish_job(&job.id, "failure", Some(&reason), None).await?; |
| 1711 | } |
| 1712 | } |
| 1713 | if let Err(error) = self.sweep_runners(now_ms).await { |
| 1714 | worker::console_error!("actions: the runners' sweep failed: {error}"); |
| 1715 | } |
| 1716 | // Once an hour: the cache's expired entries, and its storage. |
| 1717 | if (now_ms / 60_000) % 60 == 7 |
| 1718 | && let Err(error) = self.sweep_cache(now_ms).await |
| 1719 | { |
| 1720 | worker::console_error!("actions: the cache's sweep failed: {error}"); |
| 1721 | } |
| 1722 | self.start_queued().await |
| 1723 | } |
| 1724 | } |
| 1725 | |
| 1726 | |
| 1727 | #[cfg(test)] |
| 1728 | mod stopping { |
| 1729 | use super::stops_runs; |
| 1730 | use g1t_contracts::events::Event; |
| 1731 | use serde_json::{Value, json}; |
| 1732 | |
| 1733 | fn event(kind: &str, data: Value) -> Event { |
| 1734 | Event { |
| 1735 | id: "evt_1".into(), |
| 1736 | kind: kind.into(), |
| 1737 | source: "repos".into(), |
| 1738 | time: "2026-10-05T00:00:00Z".into(), |
| 1739 | repo_id: Some("rep_1".into()), |
| 1740 | actor: None, |
| 1741 | data, |
| 1742 | } |
| 1743 | } |
| 1744 | |
| 1745 | #[test] |
| 1746 | fn deleting_or_archiving_stops_runs() { |
| 1747 | assert_eq!(stops_runs(&event("repo.deleted", json!({ "repoId": "rep_1" }))).as_deref(), Some("rep_1")); |
| 1748 | assert_eq!(stops_runs(&event("repo.archived", json!({ "archived": true }))).as_deref(), Some("rep_1")); |
| 1749 | assert_eq!(stops_runs(&event("repo.unarchived", json!({ "archived": false }))), None); |
| 1750 | assert_eq!(stops_runs(&event("repo.restored", json!({}))), None); |
| 1751 | assert_eq!(stops_runs(&event("git.push", json!({}))), None); |
| 1752 | } |
| 1753 | } |
| 1754 | |
| 1755 | #[cfg(test)] |
| 1756 | mod status_of_needs { |
| 1757 | use std::collections::HashMap; |
| 1758 | |
| 1759 | use super::ancestor_failed; |
| 1760 | |
| 1761 | /// check -> plan -> (migrate) -> core -> edge, as deploy.yml has them, |
| 1762 | /// and a job that needs only the last. |
| 1763 | fn graph() -> HashMap<&'static str, Vec<&'static str>> { |
| 1764 | HashMap::from([ |
| 1765 | ("check", vec![]), |
| 1766 | ("plan", vec!["check"]), |
| 1767 | ("migrate", vec!["plan"]), |
| 1768 | ("core", vec!["plan", "migrate"]), |
| 1769 | ("edge", vec!["plan", "migrate", "core"]), |
| 1770 | ("notify", vec!["edge"]), |
| 1771 | ]) |
| 1772 | } |
| 1773 | |
| 1774 | #[test] |
| 1775 | fn a_failure_is_seen_however_far_back() { |
| 1776 | let needs = graph(); |
| 1777 | let failed = |which: &'static str| move |key: &str| key == which; |
| 1778 | // check failed; plan, and everything after, was skipped for it. |
| 1779 | assert!(ancestor_failed(&needs, "notify", failed("check"))); |
| 1780 | assert!(ancestor_failed(&needs, "core", failed("check"))); |
| 1781 | assert!(ancestor_failed(&needs, "edge", failed("core"))); |
| 1782 | // Nothing before a job failed: a skipped migrate is not a failure. |
| 1783 | assert!(!ancestor_failed(&needs, "edge", |_| false)); |
| 1784 | assert!(!ancestor_failed(&needs, "core", failed("edge"))); |
| 1785 | assert!(!ancestor_failed(&needs, "check", failed("check"))); |
| 1786 | } |
| 1787 | |
| 1788 | #[test] |
| 1789 | fn cycles_and_unknown_keys_end() { |
| 1790 | let needs = HashMap::from([("a", vec!["b"]), ("b", vec!["a"])]); |
| 1791 | assert!(!ancestor_failed(&needs, "a", |_| false)); |
| 1792 | assert!(!ancestor_failed(&needs, "missing", |_| true)); |
| 1793 | } |
| 1794 | } |