| 1 | //! A run's life: made, its jobs waiting on the jobs they need, each job |
| 2 | //! skipped or expanded into its matrix and queued, started in a sandbox |
| 3 | //! when its workspace has room, reporting its steps and logs as it goes, |
| 4 | //! and finished; the run finishes with its last job. |
| 5 | |
| 6 | use g1t_actions::events::{RunInfo, runner_context}; |
| 7 | use g1t_actions::expr::{self, Scope, Status}; |
| 8 | use g1t_actions::matrix; |
| 9 | use g1t_actions::workflow::{self, Workflow}; |
| 10 | use g1t_contracts::access::Capability; |
| 11 | use g1t_contracts::actions::{JobCallArgs, RunActionArgs, RunApproval, StartJobArgs, WorkflowRun}; |
| 12 | use g1t_contracts::identity::{CreateJobTokenArgs, CreatedAccessToken, RevokeJobTokensArgs}; |
| 13 | use g1t_contracts::repos::{Repo, RepoPath}; |
| 14 | use g1t_contracts::time::rfc3339; |
| 15 | use g1t_contracts::{FailureCode, Outcome, new_id}; |
| 16 | use g1t_kit::now_ms; |
| 17 | use g1t_secrets::{random_hex, same, sha256_hex}; |
| 18 | use serde::Deserialize; |
| 19 | use serde_json::{Map, Value, json}; |
| 20 | use worker::Result; |
| 21 | |
| 22 | use crate::protection::Gate; |
| 23 | use crate::sync::WorkflowRow; |
| 24 | use crate::{Actions, Count, MAX_TIMEOUT_MINUTES, RUNNING_PER_WORKSPACE, SELF_HOSTED_MAX_TIMEOUT_MINUTES, SILENT_MS, SITE, check, fail, optional, repo_path}; |
| 25 | use g1t_contracts::runners::{Wanted, waiting_reason}; |
| 26 | |
| 27 | /// The most log one job keeps, in bytes; past it, the log says so and stops. |
| 28 | const MAX_LOG_BYTES: usize = 4 * 1024 * 1024; |
| 29 | /// The most a single log report may add. |
| 30 | const MAX_CHUNK_BYTES: usize = 256 * 1024; |
| 31 | const MAX_ANNOTATIONS: usize = 50; |
| 32 | /// The most one step's job summary keeps, in bytes, and how many steps of a |
| 33 | /// job may have one, as on GitHub. |
| 34 | pub(crate) const MAX_SUMMARY_BYTES: usize = 1024 * 1024; |
| 35 | pub(crate) const MAX_SUMMARIES: u32 = 20; |
| 36 | |
| 37 | /// Whether a step's summary of `adding` bytes is kept: a job keeps up to |
| 38 | /// 20 steps' summaries (`steps` it has, `mine` this step's bytes so far) |
| 39 | /// of up to 1 MiB each. |
| 40 | pub(crate) fn summary_fits(steps: u32, mine: Option<usize>, adding: usize) -> bool { |
| 41 | if adding == 0 { |
| 42 | return false; |
| 43 | } |
| 44 | match mine { |
| 45 | Some(used) => used + adding <= MAX_SUMMARY_BYTES, |
| 46 | None => steps < MAX_SUMMARIES && adding <= MAX_SUMMARY_BYTES, |
| 47 | } |
| 48 | } |
| 49 | |
| 50 | /// The repository whose unfinished runs `event` stops, by id: one deleted, |
| 51 | /// or archived (not unarchived). |
| 52 | pub fn stops_runs(event: &g1t_contracts::events::Event) -> Option<String> { |
| 53 | let stops = match event.kind.as_str() { |
| 54 | "repo.deleted" => true, |
| 55 | "repo.archived" => event.data["archived"].as_bool() != Some(false), |
| 56 | _ => false, |
| 57 | }; |
| 58 | if !stops { |
| 59 | return None; |
| 60 | } |
| 61 | event.repo_id.clone().or_else(|| event.data["repoId"].as_str().map(str::to_owned)) |
| 62 | } |
| 63 | |
| 64 | pub struct NewRun { |
| 65 | pub repo: Repo, |
| 66 | pub path: String, |
| 67 | pub source: String, |
| 68 | pub workflow: Workflow, |
| 69 | pub info: RunInfo, |
| 70 | pub action: Option<String>, |
| 71 | pub pull: Option<u32>, |
| 72 | pub title: String, |
| 73 | pub inputs: Map<String, Value>, |
| 74 | pub event_key: String, |
| 75 | pub actor_id: Option<String>, |
| 76 | pub actor: Option<String>, |
| 77 | pub trusted: bool, |
| 78 | /// Why the run waits for someone with the Write role to approve it |
| 79 | /// before anything starts: a pull request from outside (protection.rs). |
| 80 | pub approval: Option<String>, |
| 81 | } |
| 82 | |
| 83 | #[derive(Clone, Deserialize)] |
| 84 | pub struct RunRow { |
| 85 | pub id: String, |
| 86 | pub workflow_id: String, |
| 87 | pub repo_id: String, |
| 88 | pub repo: String, |
| 89 | pub path: String, |
| 90 | pub name: String, |
| 91 | pub title: String, |
| 92 | pub number: u64, |
| 93 | pub attempt: u64, |
| 94 | pub event: String, |
| 95 | pub action: Option<String>, |
| 96 | pub git_ref: String, |
| 97 | pub sha: String, |
| 98 | pub pull: Option<u32>, |
| 99 | pub status: String, |
| 100 | pub conclusion: Option<String>, |
| 101 | pub error: Option<String>, |
| 102 | pub actor: Option<String>, |
| 103 | pub actor_id: Option<String>, |
| 104 | pub source: String, |
| 105 | pub info: String, |
| 106 | pub inputs: String, |
| 107 | pub trusted: u32, |
| 108 | pub concurrency_group: Option<String>, |
| 109 | pub created_at: String, |
| 110 | pub started_at: Option<String>, |
| 111 | pub finished_at: Option<String>, |
| 112 | /// JSON `RunApproval`, for a run that needed approval (migration 0006). |
| 113 | #[serde(default)] |
| 114 | pub approval: Option<String>, |
| 115 | /// Whether its concurrency group cancels what it replaces. |
| 116 | #[serde(default)] |
| 117 | pub cancel_in_progress: u32, |
| 118 | /// Who started the current attempt, once it is re-run (migration 0009). |
| 119 | #[serde(default)] |
| 120 | pub triggering_actor: Option<String>, |
| 121 | /// The current attempt runs with debug logging. |
| 122 | #[serde(default)] |
| 123 | pub debug: u32, |
| 124 | } |
| 125 | |
| 126 | impl RunRow { |
| 127 | pub fn approval(&self) -> Option<RunApproval> { |
| 128 | self.approval.as_deref().and_then(|text| serde_json::from_str(text).ok()) |
| 129 | } |
| 130 | |
| 131 | pub fn info(&self) -> RunInfo { |
| 132 | let mut info: RunInfo = serde_json::from_str(&self.info).unwrap_or_default(); |
| 133 | info.run_id = self.id.clone(); |
| 134 | info.run_number = self.number; |
| 135 | info.run_attempt = self.attempt; |
| 136 | info.workflow_path = self.path.clone(); |
| 137 | if info.workflow.is_empty() { |
| 138 | info.workflow = self.name.clone(); |
| 139 | } |
| 140 | info |
| 141 | } |
| 142 | |
| 143 | pub fn inputs(&self) -> Map<String, Value> { |
| 144 | serde_json::from_str(&self.inputs).unwrap_or_default() |
| 145 | } |
| 146 | |
| 147 | pub fn summary(&self) -> WorkflowRun { |
| 148 | WorkflowRun { |
| 149 | id: self.id.clone(), |
| 150 | workflow_id: self.workflow_id.clone(), |
| 151 | path: self.path.clone(), |
| 152 | name: self.name.clone(), |
| 153 | title: self.title.clone(), |
| 154 | number: self.number, |
| 155 | attempt: self.attempt, |
| 156 | event: self.event.clone(), |
| 157 | git_ref: self.git_ref.clone(), |
| 158 | sha: self.sha.clone(), |
| 159 | pull: self.pull, |
| 160 | status: self.status.clone(), |
| 161 | conclusion: self.conclusion.clone(), |
| 162 | error: self.error.clone(), |
| 163 | actor: self.actor.clone(), |
| 164 | created_at: self.created_at.clone(), |
| 165 | started_at: self.started_at.clone(), |
| 166 | finished_at: self.finished_at.clone(), |
| 167 | } |
| 168 | } |
| 169 | } |
| 170 | |
| 171 | #[derive(Clone, Deserialize)] |
| 172 | pub struct JobRow { |
| 173 | pub id: String, |
| 174 | pub run_id: String, |
| 175 | pub repo_id: String, |
| 176 | pub namespace: String, |
| 177 | pub key: String, |
| 178 | pub ordinal: u32, |
| 179 | pub name: String, |
| 180 | pub needs: String, |
| 181 | pub matrix: Option<String>, |
| 182 | pub status: String, |
| 183 | pub conclusion: Option<String>, |
| 184 | pub steps: String, |
| 185 | pub annotations: String, |
| 186 | pub outputs: String, |
| 187 | pub reason: Option<String>, |
| 188 | pub token_hash: Option<String>, |
| 189 | pub timeout_minutes: u32, |
| 190 | pub continue_on_error: u32, |
| 191 | pub max_parallel: Option<u32>, |
| 192 | /// Set for a job that calls a reusable workflow, and for that |
| 193 | /// workflow's jobs (see migration 0002). |
| 194 | pub call: Option<String>, |
| 195 | pub seen_at: Option<String>, |
| 196 | pub started_at: Option<String>, |
| 197 | pub finished_at: Option<String>, |
| 198 | /// For a job whose `runs-on` names self-hosted runners: what it asks |
| 199 | /// for, as a JSON array (see `g1t_contracts::runners::Wanted`), when it |
| 200 | /// started waiting, and the runner that took it (migration 0004). |
| 201 | #[serde(default)] |
| 202 | pub labels: Option<String>, |
| 203 | #[serde(default)] |
| 204 | pub queued_at: Option<String>, |
| 205 | #[serde(default)] |
| 206 | pub runner_id: Option<String>, |
| 207 | #[serde(default)] |
| 208 | pub runner_name: Option<String>, |
| 209 | /// The environment it names, read when its needs were done; a job its |
| 210 | /// rules hold is `pending` (migration 0006). |
| 211 | #[serde(default)] |
| 212 | pub environment: Option<String>, |
| 213 | /// Its own `concurrency` group, and whether that cancels what it replaces. |
| 214 | #[serde(default)] |
| 215 | pub concurrency_group: Option<String>, |
| 216 | #[serde(default)] |
| 217 | pub cancel_in_progress: u32, |
| 218 | /// When it was told to stop, while it runs its cleanup steps |
| 219 | /// (migration 0009). |
| 220 | #[serde(default)] |
| 221 | pub cancel_requested_at: Option<String>, |
| 222 | } |
| 223 | |
| 224 | /// How long a cancelled job has to run its `if: always()` and `cancelled()` |
| 225 | /// steps and its post steps before it is stopped outright, as on GitHub. |
| 226 | pub const CANCEL_GRACE_MS: u64 = 5 * 60 * 1000; |
| 227 | |
| 228 | /// Which jobs a re-run runs again. |
| 229 | #[derive(Clone, Copy, Debug, PartialEq)] |
| 230 | pub enum Rerun<'a> { |
| 231 | All, |
| 232 | /// Those that did not succeed. |
| 233 | Failed, |
| 234 | /// One job's key. |
| 235 | Job(&'a str), |
| 236 | } |
| 237 | |
| 238 | /// The keys a re-run runs again, in `order`: those `which` names and, |
| 239 | /// transitively, every key that needs one of them. `needs` gives a key's |
| 240 | /// needs; `succeeded` whether every job of a key succeeded. |
| 241 | pub fn rerun_keys(order: &[&str], needs: impl Fn(&str) -> Vec<String>, succeeded: impl Fn(&str) -> bool, which: Rerun) -> Vec<String> { |
| 242 | let mut again: Vec<String> = Vec::new(); |
| 243 | for key in order { |
| 244 | let named = match which { |
| 245 | Rerun::All => true, |
| 246 | Rerun::Failed => !succeeded(key), |
| 247 | Rerun::Job(job) => *key == job, |
| 248 | }; |
| 249 | if named || needs(key).iter().any(|need| again.contains(need)) { |
| 250 | again.push((*key).to_owned()); |
| 251 | } |
| 252 | } |
| 253 | again |
| 254 | } |
| 255 | |
| 256 | /// The key under `jobs:` a job belongs to: a called workflow's jobs |
| 257 | /// (`build/test`) run again with the job that calls it (`build`). |
| 258 | pub fn top_key(key: &str) -> &str { |
| 259 | key.split('/').next().unwrap_or(key) |
| 260 | } |
| 261 | |
| 262 | /// Debug logging for a job, as a re-run with it turned on gives GitHub's: |
| 263 | /// `RUNNER_DEBUG=1` and `runner.debug`, and `ACTIONS_STEP_DEBUG` and |
| 264 | /// `ACTIONS_RUNNER_DEBUG` set, which the runner reads as the secrets of |
| 265 | /// those names (`::debug::` lines shown, and how each step's `if` read). |
| 266 | pub fn debug_logging(variables: &mut Map<String, Value>, runner: &mut Value) { |
| 267 | variables.insert("RUNNER_DEBUG".into(), json!("1")); |
| 268 | variables.insert("ACTIONS_STEP_DEBUG".into(), json!("true")); |
| 269 | variables.insert("ACTIONS_RUNNER_DEBUG".into(), json!("true")); |
| 270 | runner["debug"] = json!("1"); |
| 271 | } |
| 272 | |
| 273 | impl JobRow { |
| 274 | pub fn needs(&self) -> Vec<String> { |
| 275 | serde_json::from_str(&self.needs).unwrap_or_default() |
| 276 | } |
| 277 | |
| 278 | pub fn call(&self) -> Option<Value> { |
| 279 | self.call.as_deref().and_then(|call| serde_json::from_str(call).ok()) |
| 280 | } |
| 281 | |
| 282 | /// For a job of a called workflow: that workflow, the job's own id in |
| 283 | /// it, and the job. |
| 284 | pub fn callee(&self) -> Option<(Workflow, workflow::Job, Value)> { |
| 285 | let call = self.call().filter(|call| call["role"] == "callee")?; |
| 286 | let called = workflow::parse(call["source"].as_str()?).ok()?; |
| 287 | let job = called.jobs.iter().find(|job| call["job"].as_str() == Some(job.id.as_str()))?.clone(); |
| 288 | Some((called, job, call)) |
| 289 | } |
| 290 | } |
| 291 | |
| 292 | /// A job to decide on: its key, its definition, and what it needs, as |
| 293 | /// (name in `needs`, key of the jobs). |
| 294 | type Unit = (String, workflow::Job, Vec<(String, String)>); |
| 295 | |
| 296 | /// How deep reusable workflows may call one another, as on GitHub. |
| 297 | const MAX_CALL_DEPTH: u64 = 4; |
| 298 | |
| 299 | /// What the jobs of one key came to, for `needs.<key>`. |
| 300 | fn key_result(rows: &[&JobRow]) -> &'static str { |
| 301 | let failed = |row: &&&JobRow| row.conclusion.as_deref() == Some("failure") && row.continue_on_error == 0; |
| 302 | if rows.iter().any(|row| failed(&row)) { |
| 303 | "failure" |
| 304 | } else if rows.iter().any(|row| row.conclusion.as_deref() == Some("cancelled")) { |
| 305 | "cancelled" |
| 306 | } else if rows.iter().all(|row| row.conclusion.as_deref() == Some("skipped")) { |
| 307 | "skipped" |
| 308 | } else { |
| 309 | "success" |
| 310 | } |
| 311 | } |
| 312 | |
| 313 | /// Whether any job before `key` failed: one it needs, or one those need, |
| 314 | /// however far back. A job after a skipped one still sees the failure |
| 315 | /// that skipped it, as GitHub's failure() does. `needs_of`: each key's |
| 316 | /// needs, as keys; `failed`: whether a key's jobs came to a failure. |
| 317 | fn ancestor_failed(needs_of: &std::collections::HashMap<&str, Vec<&str>>, key: &str, failed: impl Fn(&str) -> bool) -> bool { |
| 318 | let mut seen = std::collections::HashSet::new(); |
| 319 | let mut stack: Vec<&str> = needs_of.get(key).cloned().unwrap_or_default(); |
| 320 | while let Some(next) = stack.pop() { |
| 321 | if !seen.insert(next) { |
| 322 | continue; |
| 323 | } |
| 324 | if failed(next) { |
| 325 | return true; |
| 326 | } |
| 327 | stack.extend(needs_of.get(next).into_iter().flatten().copied()); |
| 328 | } |
| 329 | false |
| 330 | } |
| 331 | |
| 332 | /// A job's `environment:`, as deployments read it: its name, the address in |
| 333 | /// `environment.url` once its expressions are filled in from the run, and |
| 334 | /// whether the job deploys to it. A job with `deployment: false` only reads |
| 335 | /// the environment's secrets and variables, and makes no deployment. |
| 336 | #[derive(Clone, Debug, Default, PartialEq)] |
| 337 | pub(crate) struct JobEnvironment { |
| 338 | pub(crate) name: String, |
| 339 | pub(crate) url: Option<String>, |
| 340 | pub(crate) deploys: bool, |
| 341 | } |
| 342 | |
| 343 | /// The environment `environment:` names, as written in `raw` (a job), with |
| 344 | /// `contexts` to fill in expressions: null when it names none, or only |
| 345 | /// through an expression this cannot read before the job runs. |
| 346 | pub(crate) fn environment_of(raw: &Value, contexts: &Map<String, Value>) -> Option<JobEnvironment> { |
| 347 | let scope = Scope { contexts, status: Status::Success, hash_files: None }; |
| 348 | let plain = |value: &Value| -> Option<String> { |
| 349 | let text = match value { |
| 350 | Value::String(text) if text.contains("${{") => expr::interpolate_value(value, &scope).ok().map(|v| expr::to_text(&v))?, |
| 351 | Value::String(text) => text.clone(), |
| 352 | _ => return None, |
| 353 | }; |
| 354 | let text = text.trim().to_owned(); |
| 355 | (!text.is_empty()).then_some(text) |
| 356 | }; |
| 357 | match raw.get("environment")? { |
| 358 | name @ Value::String(_) => Some(JobEnvironment { name: plain(name)?, url: None, deploys: true }), |
| 359 | Value::Object(env) => { |
| 360 | let name = plain(env.get("name")?)?; |
| 361 | let url = env |
| 362 | .get("url") |
| 363 | .and_then(plain) |
| 364 | .filter(|url| url.starts_with("https://") || url.starts_with("http://")); |
| 365 | let deploys = !matches!(env.get("deployment"), Some(Value::Bool(false))) |
| 366 | && !matches!(env.get("deployment"), Some(Value::String(text)) if text.trim() == "false"); |
| 367 | Some(JobEnvironment { name, url, deploys }) |
| 368 | } |
| 369 | _ => None, |
| 370 | } |
| 371 | } |
| 372 | |
| 373 | /// The contexts a job's `runs-on` and `environment` are read with before it |
| 374 | /// runs: the run's `github`, its inputs, and the job's matrix. |
| 375 | fn start_contexts(run: &RunRow, job: &JobRow, key: &str) -> Map<String, Value> { |
| 376 | let mut contexts = Map::new(); |
| 377 | contexts.insert("github".into(), run.info().context(key, "", run.action.as_deref())); |
| 378 | contexts.insert("inputs".into(), Value::Object(run.inputs())); |
| 379 | contexts.insert("matrix".into(), job.matrix.as_deref().and_then(|m| serde_json::from_str(m).ok()).unwrap_or_else(|| json!({}))); |
| 380 | contexts |
| 381 | } |
| 382 | |
| 383 | /// The job as its workflow (or the workflow it calls) defines it. |
| 384 | fn job_spec(run: &RunRow, job: &JobRow) -> Option<workflow::Job> { |
| 385 | match job.callee() { |
| 386 | Some((_, spec, _)) => Some(spec), |
| 387 | None => workflow::parse(&run.source).ok().and_then(|w| w.jobs.into_iter().find(|j| j.id == job.key)), |
| 388 | } |
| 389 | } |
| 390 | |
| 391 | /// The environment a job deploys to, if it deploys: by the name read when |
| 392 | /// its needs were done (which may have needed their outputs), else as |
| 393 | /// its `environment:` reads now. |
| 394 | pub(crate) fn deploys_to(run: &RunRow, job: &JobRow) -> Option<JobEnvironment> { |
| 395 | let spec = job_spec(run, job)?; |
| 396 | let read = environment_of(&spec.raw, &start_contexts(run, job, &spec.id)); |
| 397 | let env = match (read, &job.environment) { |
| 398 | (Some(env), Some(name)) => Some(JobEnvironment { name: name.clone(), ..env }), |
| 399 | (Some(env), None) => Some(env), |
| 400 | (None, Some(name)) => { |
| 401 | let deployment = spec.raw.get("environment").and_then(|env| env.get("deployment")); |
| 402 | let deploys = !matches!(deployment, Some(Value::Bool(false))) |
| 403 | && !matches!(deployment, Some(Value::String(text)) if text.trim() == "false"); |
| 404 | Some(JobEnvironment { name: name.clone(), url: None, deploys }) |
| 405 | } |
| 406 | (None, None) => None, |
| 407 | }; |
| 408 | env.filter(|env| env.deploys) |
| 409 | } |
| 410 | |
| 411 | /// What the runner is told about a job it starts, from its workflow: the |
| 412 | /// environment it names plainly, and the machine its `runs-on` asks for |
| 413 | /// (`instance_for`; none for the standard one). |
| 414 | #[derive(Default)] |
| 415 | struct StartDetails { |
| 416 | environment: Option<String>, |
| 417 | instance: Option<String>, |
| 418 | } |
| 419 | |
| 420 | fn start_details(run: &RunRow, job: &JobRow) -> StartDetails { |
| 421 | let spec = match job.callee() { |
| 422 | Some((_, spec, _)) => spec, |
| 423 | None => match workflow::parse(&run.source).ok().and_then(|w| w.jobs.into_iter().find(|j| j.id == job.key)) { |
| 424 | Some(spec) => spec, |
| 425 | None => return StartDetails::default(), |
| 426 | }, |
| 427 | }; |
| 428 | // As read when its needs were done, an expression's included. |
| 429 | let environment = job.environment.clone(); |
| 430 | // `runs-on` as the job was queued with: its matrix and the run's |
| 431 | // inputs. A label that needs more than those is the standard machine. |
| 432 | let contexts = start_contexts(run, job, &spec.id); |
| 433 | let scope = Scope { contexts: &contexts, status: Status::Success, hash_files: None }; |
| 434 | let labels: Vec<String> = match expr::interpolate_value(&spec.runs_on, &scope).unwrap_or(Value::Null) { |
| 435 | Value::String(label) => vec![label], |
| 436 | Value::Array(labels) => labels.iter().map(expr::to_text).collect(), |
| 437 | Value::Object(given) => match given.get("labels") { |
| 438 | Some(Value::Array(labels)) => labels.iter().map(expr::to_text).collect(), |
| 439 | Some(label) => vec![expr::to_text(label)], |
| 440 | None => Vec::new(), |
| 441 | }, |
| 442 | _ => Vec::new(), |
| 443 | }; |
| 444 | let instance = g1t_contracts::actions::instance_for(&labels); |
| 445 | StartDetails { |
| 446 | environment, |
| 447 | instance: (instance != g1t_contracts::actions::STANDARD_INSTANCE).then(|| instance.label.to_owned()), |
| 448 | } |
| 449 | } |
| 450 | |
| 451 | fn now() -> String { |
| 452 | rfc3339(now_ms()) |
| 453 | } |
| 454 | |
| 455 | /// How a finished run went for one environment, from the conclusions of |
| 456 | /// the jobs that deploy to it: failed if any failed, an error if any was |
| 457 | /// cancelled, else a success; nothing when none ran (all skipped). |
| 458 | pub(crate) fn deployment_outcome(conclusions: &[Option<String>]) -> Option<&'static str> { |
| 459 | let ran: Vec<&str> = conclusions.iter().flatten().map(String::as_str).filter(|c| *c != "skipped").collect(); |
| 460 | if ran.is_empty() { |
| 461 | return None; |
| 462 | } |
| 463 | if ran.iter().any(|c| *c == "failure" || *c == "timed_out") { |
| 464 | return Some("failure"); |
| 465 | } |
| 466 | if ran.contains(&"cancelled") { |
| 467 | return Some("error"); |
| 468 | } |
| 469 | Some("success") |
| 470 | } |
| 471 | |
| 472 | impl Actions { |
| 473 | pub async fn run_row(&self, id: &str) -> Result<Option<RunRow>> { |
| 474 | self.db.prepare("SELECT * FROM runs WHERE id = ?").bind(&[id.into()])?.first::<RunRow>(None).await |
| 475 | } |
| 476 | |
| 477 | pub async fn job_rows(&self, run_id: &str) -> Result<Vec<JobRow>> { |
| 478 | self.db |
| 479 | .prepare("SELECT * FROM jobs WHERE run_id = ? ORDER BY rowid") |
| 480 | .bind(&[run_id.into()])? |
| 481 | .all() |
| 482 | .await? |
| 483 | .results::<JobRow>() |
| 484 | } |
| 485 | |
| 486 | pub async fn run_summary(&self, id: &str) -> Result<Outcome<WorkflowRun>> { |
| 487 | Ok(match self.run_row(id).await? { |
| 488 | Some(row) => Outcome::Ok(row.summary()), |
| 489 | None => fail(FailureCode::NotFound, "No such run."), |
| 490 | }) |
| 491 | } |
| 492 | |
| 493 | /// The contexts every expression outside a job's steps may use. |
| 494 | fn base_contexts(run: &RunRow, vars: &Map<String, Value>, job: &str) -> Map<String, Value> { |
| 495 | let mut contexts = Map::new(); |
| 496 | contexts.insert("github".into(), run.info().context(job, "", run.action.as_deref())); |
| 497 | contexts.insert("inputs".into(), Value::Object(run.inputs())); |
| 498 | contexts.insert("vars".into(), Value::Object(vars.clone())); |
| 499 | contexts.insert("needs".into(), json!({})); |
| 500 | contexts.insert("runner".into(), runner_context()); |
| 501 | contexts |
| 502 | } |
| 503 | |
| 504 | /// Makes a run and its jobs, and starts what can start. `None` when the |
| 505 | /// event already started this workflow. |
| 506 | pub async fn create_run(&self, new: NewRun) -> Result<Option<String>> { |
| 507 | // Nothing starts on an archived repository. A deleted one is never |
| 508 | // found to start anything on. |
| 509 | if new.repo.archived() { |
| 510 | return Ok(None); |
| 511 | } |
| 512 | let workflow_row = self.workflow_row(&new.repo, &new.path, &new.workflow.display_name(&new.path), &new.source).await?; |
| 513 | let numbered = self |
| 514 | .db |
| 515 | .prepare("UPDATE workflows SET run_count = run_count + 1 WHERE id = ? RETURNING run_count AS n") |
| 516 | .bind(&[workflow_row.id.as_str().into()])? |
| 517 | .first::<Count>(None) |
| 518 | .await? |
| 519 | .map_or(1, |count| count.n); |
| 520 | let id = new_id("run", now_ms()); |
| 521 | let mut info = new.info.clone(); |
| 522 | info.workflow = new.workflow.display_name(&new.path); |
| 523 | info.workflow_path = new.path.clone(); |
| 524 | info.run_id = id.clone(); |
| 525 | info.run_number = u64::from(numbered); |
| 526 | let vars = self |
| 527 | .variables_for(&new.repo.id, &format!("{}/{}", new.repo.namespace, new.repo.name), None, new.trusted) |
| 528 | .await?; |
| 529 | |
| 530 | // run-name and the concurrency group read github, inputs and vars. |
| 531 | let mut contexts = Map::new(); |
| 532 | contexts.insert("github".into(), info.context("", "", new.action.as_deref())); |
| 533 | contexts.insert("inputs".into(), Value::Object(new.inputs.clone())); |
| 534 | contexts.insert("vars".into(), Value::Object(vars)); |
| 535 | let scope = Scope { |
| 536 | contexts: &contexts, |
| 537 | status: Status::Success, |
| 538 | hash_files: None, |
| 539 | }; |
| 540 | let title = new |
| 541 | .workflow |
| 542 | .run_name |
| 543 | .as_deref() |
| 544 | .and_then(|run_name| expr::interpolate(run_name, &scope).ok()) |
| 545 | .filter(|title| !title.trim().is_empty()) |
| 546 | .unwrap_or(new.title.clone()); |
| 547 | let group = new.workflow.concurrency.as_ref().and_then(|c| expr::interpolate(&c.group, &scope).ok()); |
| 548 | let cancel_in_progress = new |
| 549 | .workflow |
| 550 | .concurrency |
| 551 | .as_ref() |
| 552 | .and_then(|c| expr::interpolate_value(&c.cancel_in_progress, &scope).ok()) |
| 553 | .is_some_and(|value| expr::truthy(&value)); |
| 554 | |
| 555 | let approval = new.approval.as_ref().map(|reason| RunApproval { state: "required".into(), reason: reason.clone(), approved_by: None }); |
| 556 | let inserted = self |
| 557 | .db |
| 558 | .prepare( |
| 559 | "INSERT OR IGNORE INTO runs (id, workflow_id, repo_id, repo, path, name, title, number, event, action, git_ref, sha, |
| 560 | pull, status, actor, actor_id, source, info, inputs, trusted, concurrency_group, event_key, created_at, |
| 561 | approval, cancel_in_progress) |
| 562 | VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) RETURNING id", |
| 563 | ) |
| 564 | .bind(&[ |
| 565 | id.as_str().into(), |
| 566 | workflow_row.id.as_str().into(), |
| 567 | new.repo.id.as_str().into(), |
| 568 | format!("{}/{}", new.repo.namespace, new.repo.name).into(), |
| 569 | new.path.as_str().into(), |
| 570 | info.workflow.as_str().into(), |
| 571 | title.as_str().into(), |
| 572 | numbered.into(), |
| 573 | info.event_name.as_str().into(), |
| 574 | optional(new.action.as_deref()), |
| 575 | info.git_ref.as_str().into(), |
| 576 | info.sha.as_str().into(), |
| 577 | new.pull.map_or(worker::wasm_bindgen::JsValue::NULL, Into::into), |
| 578 | if approval.is_some() { "action_required" } else { "queued" }.into(), |
| 579 | optional(new.actor.as_deref()), |
| 580 | optional(new.actor_id.as_deref()), |
| 581 | new.source.as_str().into(), |
| 582 | serde_json::to_string(&info)?.into(), |
| 583 | serde_json::to_string(&new.inputs)?.into(), |
| 584 | u32::from(new.trusted).into(), |
| 585 | optional(group.as_deref()), |
| 586 | new.event_key.as_str().into(), |
| 587 | now().into(), |
| 588 | approval.as_ref().map(serde_json::to_string).transpose()?.as_deref().map_or(worker::wasm_bindgen::JsValue::NULL, Into::into), |
| 589 | u32::from(cancel_in_progress).into(), |
| 590 | ])? |
| 591 | .first::<Value>(None) |
| 592 | .await?; |
| 593 | if inserted.is_none() { |
| 594 | return Ok(None); |
| 595 | } |
| 596 | if let Some(run) = self.run_row(&id).await? { |
| 597 | self.report_pending(&run).await?; |
| 598 | } |
| 599 | |
| 600 | // Every job, waiting; each is expanded when the jobs it needs are done. |
| 601 | let mut statements = Vec::new(); |
| 602 | for job in &new.workflow.jobs { |
| 603 | statements.push( |
| 604 | self.db |
| 605 | .prepare( |
| 606 | "INSERT INTO jobs (id, run_id, repo_id, namespace, key, name, needs, status) VALUES (?, ?, ?, ?, ?, ?, ?, 'waiting')", |
| 607 | ) |
| 608 | .bind(&[ |
| 609 | new_id("job", now_ms()).into(), |
| 610 | id.as_str().into(), |
| 611 | new.repo.id.as_str().into(), |
| 612 | new.repo.namespace.as_str().into(), |
| 613 | job.id.as_str().into(), |
| 614 | job.name.clone().filter(|n| !expr::has_expression(n)).unwrap_or(job.id.clone()).into(), |
| 615 | serde_json::to_string(&job.needs)?.into(), |
| 616 | ])?, |
| 617 | ); |
| 618 | } |
| 619 | self.db.batch(statements).await?; |
| 620 | |
| 621 | // A pull request's run from outside waits to be approved; it joins |
| 622 | // its concurrency group once it is (protection.rs, approve_run). |
| 623 | if approval.is_some() { |
| 624 | return Ok(Some(id)); |
| 625 | } |
| 626 | if let Some(run) = self.run_row(&id).await? { |
| 627 | self.enter_group(&run).await?; |
| 628 | } |
| 629 | Ok(Some(id)) |
| 630 | } |
| 631 | |
| 632 | /// A run about to start: one run at a time per concurrency group, as on |
| 633 | /// GitHub. A newer run replaces one that waits in the group, and with |
| 634 | /// `cancel-in-progress` the one that runs; otherwise it waits as |
| 635 | /// `pending` for the one that runs. Then it moves along. |
| 636 | pub(crate) async fn enter_group(&self, run: &RunRow) -> Result<()> { |
| 637 | if let Some(group) = &run.concurrency_group { |
| 638 | let cancel_in_progress = run.cancel_in_progress != 0; |
| 639 | let others = self |
| 640 | .db |
| 641 | .prepare( |
| 642 | "SELECT * FROM runs WHERE repo_id = ? AND concurrency_group = ? AND id != ? AND status NOT IN ('completed', 'action_required') ORDER BY id", |
| 643 | ) |
| 644 | .bind(&[run.repo_id.as_str().into(), group.as_str().into(), run.id.as_str().into()])? |
| 645 | .all() |
| 646 | .await? |
| 647 | .results::<RunRow>()?; |
| 648 | for other in &others { |
| 649 | if cancel_in_progress || other.status == "pending" { |
| 650 | // A newer run replaces a waiting one, as on GitHub. |
| 651 | self.cancel_run(other, "A newer run in the same concurrency group replaced it.", false).await?; |
| 652 | } |
| 653 | } |
| 654 | if !cancel_in_progress && others.iter().any(|other| other.status != "pending") { |
| 655 | self.db |
| 656 | .prepare("UPDATE runs SET status = 'pending' WHERE id = ?") |
| 657 | .bind(&[run.id.as_str().into()])? |
| 658 | .run() |
| 659 | .await?; |
| 660 | return Ok(()); |
| 661 | } |
| 662 | } |
| 663 | self.advance(&run.id).await |
| 664 | } |
| 665 | |
| 666 | /// A run that could not start, such as for a workflow file that does not read. |
| 667 | #[allow(clippy::too_many_arguments)] |
| 668 | pub async fn record_failed_run( |
| 669 | &self, |
| 670 | row: &WorkflowRow, |
| 671 | git_ref: &str, |
| 672 | sha: &str, |
| 673 | event_key: &str, |
| 674 | actor_id: Option<&str>, |
| 675 | actor: &str, |
| 676 | problem: &str, |
| 677 | ) -> Result<()> { |
| 678 | let numbered = self |
| 679 | .db |
| 680 | .prepare("UPDATE workflows SET run_count = run_count + 1 WHERE id = ? RETURNING run_count AS n") |
| 681 | .bind(&[row.id.as_str().into()])? |
| 682 | .first::<Count>(None) |
| 683 | .await? |
| 684 | .map_or(1, |count| count.n); |
| 685 | let at = now(); |
| 686 | self.db |
| 687 | .prepare( |
| 688 | "INSERT OR IGNORE INTO runs (id, workflow_id, repo_id, repo, path, name, title, number, event, git_ref, sha, status, |
| 689 | conclusion, error, actor, actor_id, source, info, event_key, created_at, finished_at) |
| 690 | VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'push', ?, ?, 'completed', 'failure', ?, ?, ?, ?, '{}', ?, ?, ?)", |
| 691 | ) |
| 692 | .bind(&[ |
| 693 | new_id("run", now_ms()).into(), |
| 694 | row.id.as_str().into(), |
| 695 | row.repo_id.as_str().into(), |
| 696 | row.repo.as_str().into(), |
| 697 | row.path.as_str().into(), |
| 698 | row.path.as_str().into(), |
| 699 | "Invalid workflow file".into(), |
| 700 | numbered.into(), |
| 701 | git_ref.into(), |
| 702 | sha.into(), |
| 703 | problem.into(), |
| 704 | actor.into(), |
| 705 | optional(actor_id), |
| 706 | row.source.as_str().into(), |
| 707 | event_key.into(), |
| 708 | at.as_str().into(), |
| 709 | at.as_str().into(), |
| 710 | ])? |
| 711 | .run() |
| 712 | .await?; |
| 713 | Ok(()) |
| 714 | } |
| 715 | |
| 716 | /// Moves a run along: jobs whose needs are done are decided on, and |
| 717 | /// jobs that can start are started. |
| 718 | pub async fn advance(&self, run_id: &str) -> Result<()> { |
| 719 | // Each pass may finish jobs (skipped ones), which may free others. |
| 720 | for _ in 0..20 { |
| 721 | let Some(run) = self.run_row(run_id).await? else { return Ok(()) }; |
| 722 | if matches!(run.status.as_str(), "completed" | "pending" | "action_required") { |
| 723 | return Ok(()); |
| 724 | } |
| 725 | let workflow = match workflow::parse(&run.source) { |
| 726 | Ok(workflow) => workflow, |
| 727 | Err(problem) => { |
| 728 | self.finish_run(&run, Some(&problem)).await?; |
| 729 | return Ok(()); |
| 730 | } |
| 731 | }; |
| 732 | let jobs = self.job_rows(run_id).await?; |
| 733 | let mut changed = false; |
| 734 | // What to decide on: the workflow's jobs, and the jobs of the |
| 735 | // workflows they call, each with its needs as (name, key). |
| 736 | let mut units: Vec<Unit> = workflow |
| 737 | .jobs |
| 738 | .iter() |
| 739 | .map(|job| (job.id.clone(), job.clone(), job.needs.iter().map(|n| (n.clone(), n.clone())).collect())) |
| 740 | .collect(); |
| 741 | let mut seen = std::collections::HashSet::new(); |
| 742 | for row in &jobs { |
| 743 | if !seen.insert(row.key.clone()) { |
| 744 | continue; |
| 745 | } |
| 746 | if let Some((_, job, call)) = row.callee() { |
| 747 | let parent = call["parent"].as_str().unwrap_or_default().to_owned(); |
| 748 | let needs = job.needs.iter().map(|n| (n.clone(), format!("{parent}/{n}"))).collect(); |
| 749 | units.push((row.key.clone(), job, needs)); |
| 750 | } |
| 751 | } |
| 752 | // Each key's needs, as keys, to look back through for failure(). |
| 753 | let needs_of: std::collections::HashMap<&str, Vec<&str>> = |
| 754 | units.iter().map(|(key, _, needs)| (key.as_str(), needs.iter().map(|(_, need)| need.as_str()).collect())).collect(); |
| 755 | let key_failed = |key: &str| key_result(&jobs.iter().filter(|row| row.key == key).collect::<Vec<_>>()) == "failure"; |
| 756 | for (key, job, needs) in &units { |
| 757 | let rows: Vec<&JobRow> = jobs.iter().filter(|row| &row.key == key).collect(); |
| 758 | if rows.is_empty() || !rows.iter().all(|row| row.status == "waiting") { |
| 759 | continue; |
| 760 | } |
| 761 | let needed: Vec<(&String, Vec<&JobRow>)> = |
| 762 | needs.iter().map(|(name, need)| (name, jobs.iter().filter(|row| &row.key == need).collect())).collect(); |
| 763 | if !needed.iter().all(|(_, rows)| rows.iter().all(|row| row.status == "completed")) { |
| 764 | continue; |
| 765 | } |
| 766 | let failed_before = ancestor_failed(&needs_of, key, key_failed); |
| 767 | self.decide(&run, job, rows[0], &needed, failed_before).await?; |
| 768 | changed = true; |
| 769 | } |
| 770 | // A job that called a workflow finishes with that workflow's jobs. |
| 771 | for row in jobs.iter().filter(|row| row.status == "calling") { |
| 772 | let children: Vec<&JobRow> = jobs |
| 773 | .iter() |
| 774 | .filter(|child| child.call().is_some_and(|call| call["role"] == "callee" && call["parent"].as_str() == Some(row.key.as_str()))) |
| 775 | .collect(); |
| 776 | if !children.is_empty() && children.iter().all(|child| child.status == "completed") { |
| 777 | self.finish_call(row, &children).await?; |
| 778 | changed = true; |
| 779 | } |
| 780 | } |
| 781 | if !changed { |
| 782 | break; |
| 783 | } |
| 784 | } |
| 785 | self.start_queued().await?; |
| 786 | self.finish_if_done(run_id).await |
| 787 | } |
| 788 | |
| 789 | /// Decides on one job whose needs are done: skip it, fail it, or expand |
| 790 | /// it into its matrix and queue it. `failed_before`: whether any job |
| 791 | /// before it failed, however far back (`ancestor_failed`). |
| 792 | async fn decide(&self, run: &RunRow, job: &workflow::Job, row: &JobRow, needed: &[(&String, Vec<&JobRow>)], failed_before: bool) -> Result<()> { |
| 793 | let vars = self.variables_for(&run.repo_id, &run.repo, None, run.trusted != 0).await?; |
| 794 | let mut contexts = Self::base_contexts(run, &vars, &job.id); |
| 795 | // A called workflow's jobs read the inputs they were called with. |
| 796 | let call = row.call(); |
| 797 | let parent = call.as_ref().filter(|c| c["role"] == "callee").and_then(|c| c["parent"].as_str().map(str::to_owned)); |
| 798 | if let Some(call) = call.as_ref().filter(|c| c["role"] == "callee") { |
| 799 | contexts.insert("inputs".into(), call["inputs"].clone()); |
| 800 | } |
| 801 | let mut needs = Map::new(); |
| 802 | let mut results = Vec::new(); |
| 803 | for (key, rows) in needed { |
| 804 | let result = key_result(rows); |
| 805 | let mut outputs = Map::new(); |
| 806 | for row in rows { |
| 807 | if let Ok(Value::Object(more)) = serde_json::from_str::<Value>(&row.outputs) { |
| 808 | outputs.extend(more); |
| 809 | } |
| 810 | } |
| 811 | results.push(result); |
| 812 | needs.insert((*key).clone(), json!({ "result": result, "outputs": outputs })); |
| 813 | } |
| 814 | // As on GitHub: a need that was skipped makes success() false but |
| 815 | // not failure(); failure() is a failure anywhere before the job. |
| 816 | let status = expr::job_status(results, failed_before, run.conclusion.as_deref() == Some("cancelled")); |
| 817 | contexts.insert("needs".into(), Value::Object(needs)); |
| 818 | let scope = Scope { |
| 819 | contexts: &contexts, |
| 820 | status, |
| 821 | hash_files: None, |
| 822 | }; |
| 823 | let condition = job.condition.as_deref().unwrap_or_default(); |
| 824 | match expr::condition(condition, &scope) { |
| 825 | Ok(true) => {} |
| 826 | Ok(false) => return self.skip_job(row, None).await, |
| 827 | Err(problem) => return self.fail_job(row, &format!("Its `if` does not read: {problem}")).await, |
| 828 | } |
| 829 | if let Some(uses) = &job.uses { |
| 830 | return self.call_workflow(run, job, row, uses, &scope).await; |
| 831 | } |
| 832 | |
| 833 | // Its matrix, which may come from a needed job's outputs. |
| 834 | let combinations = match &job.matrix { |
| 835 | None => vec![Map::new()], |
| 836 | Some(matrix) => { |
| 837 | let value = match expr::interpolate_value(matrix, &scope) { |
| 838 | Ok(value) => value, |
| 839 | Err(problem) => return self.fail_job(row, &format!("Its matrix does not read: {problem}")).await, |
| 840 | }; |
| 841 | match matrix::expand(&value) { |
| 842 | Ok(combinations) if !combinations.is_empty() => combinations, |
| 843 | Ok(_) => return self.fail_job(row, "Its matrix makes no jobs.").await, |
| 844 | Err(problem) => return self.fail_job(row, &problem).await, |
| 845 | } |
| 846 | } |
| 847 | }; |
| 848 | let total = combinations.len(); |
| 849 | let raw = job.raw.as_object().cloned().unwrap_or_default(); |
| 850 | // Whether pull requests from forks may use self-hosted runners here, |
| 851 | // asked once, and only for a run that is not trusted. |
| 852 | let mut forks_allowed: Option<bool> = None; |
| 853 | // Each concurrency group its jobs join, and whether it cancels. |
| 854 | let mut groups: Vec<(String, bool)> = Vec::new(); |
| 855 | let mut statements = Vec::new(); |
| 856 | for (index, combination) in combinations.iter().enumerate() { |
| 857 | let mut contexts = contexts.clone(); |
| 858 | contexts.insert("matrix".into(), Value::Object(combination.clone())); |
| 859 | contexts.insert( |
| 860 | "strategy".into(), |
| 861 | json!({ "fail-fast": job.fail_fast, "job-index": index, "job-total": total, "max-parallel": job.max_parallel.unwrap_or(total as u32) }), |
| 862 | ); |
| 863 | let scope = Scope { |
| 864 | contexts: &contexts, |
| 865 | status: Status::Success, |
| 866 | hash_files: None, |
| 867 | }; |
| 868 | let base_name = job.name.clone().unwrap_or(job.id.clone()); |
| 869 | // A called workflow's job is shown under the job that called it. |
| 870 | let base_name = match &parent { |
| 871 | Some(parent) => format!("{} / {base_name}", parent.replace('/', " / ")), |
| 872 | None => base_name, |
| 873 | }; |
| 874 | let name = if expr::has_expression(&base_name) { |
| 875 | expr::interpolate(&base_name, &scope).unwrap_or(base_name) |
| 876 | } else if job.matrix.is_some() { |
| 877 | matrix::job_name(&base_name, combination) |
| 878 | } else { |
| 879 | base_name |
| 880 | }; |
| 881 | let runs_on = expr::interpolate_value(&job.runs_on, &scope).unwrap_or(Value::Null); |
| 882 | let labels = match &runs_on { |
| 883 | Value::String(label) => vec![label.clone()], |
| 884 | Value::Array(labels) => labels.iter().map(expr::to_text).collect(), |
| 885 | Value::Object(spec) => spec.get("labels").map(|l| match l { |
| 886 | Value::Array(labels) => labels.iter().map(expr::to_text).collect(), |
| 887 | other => vec![expr::to_text(other)], |
| 888 | }).unwrap_or_default(), |
| 889 | _ => Vec::new(), |
| 890 | }; |
| 891 | // `runs-on: self-hosted` (or a group): the workspace's own |
| 892 | // machines, which may run any OS. Otherwise g1t's Linux sandboxes. |
| 893 | let group = match &runs_on { |
| 894 | Value::Object(spec) => spec.get("group").map(expr::to_text), |
| 895 | _ => None, |
| 896 | }; |
| 897 | let wanted = Wanted::of(&labels, group.as_deref()); |
| 898 | let mut reason = if wanted.self_hosted { |
| 899 | None |
| 900 | } else { |
| 901 | labels |
| 902 | .iter() |
| 903 | .find(|label| { |
| 904 | let lower = label.to_ascii_lowercase(); |
| 905 | lower.contains("windows") || lower.contains("macos") |
| 906 | }) |
| 907 | .map(|label| { |
| 908 | let os = if label.to_ascii_lowercase().contains("windows") { "windows" } else { "macos" }; |
| 909 | format!("`runs-on: {label}`: g1t's own runners are Linux. A self-hosted runner can run it: `runs-on: [self-hosted, {os}]`.") |
| 910 | }) |
| 911 | }; |
| 912 | // A pull request from a fork runs code anyone could write: never |
| 913 | // on the workspace's machines unless it said they may. |
| 914 | if wanted.self_hosted && run.trusted == 0 { |
| 915 | let allowed = match forks_allowed { |
| 916 | Some(allowed) => allowed, |
| 917 | None => { |
| 918 | let allowed = self.effective_runner_settings(&row.namespace, Some(&run.repo_id)).await?.fork_pull_requests; |
| 919 | forks_allowed = Some(allowed); |
| 920 | allowed |
| 921 | } |
| 922 | }; |
| 923 | if !allowed { |
| 924 | reason = Some( |
| 925 | "Pull requests from forks do not run on self-hosted runners here. An admin can allow it under Settings, Runners.".to_owned(), |
| 926 | ); |
| 927 | } |
| 928 | } |
| 929 | let max_minutes = if wanted.self_hosted { SELF_HOSTED_MAX_TIMEOUT_MINUTES } else { MAX_TIMEOUT_MINUTES }; |
| 930 | let timeout = raw |
| 931 | .get("timeout-minutes") |
| 932 | .and_then(|value| expr::interpolate_value(value, &scope).ok()) |
| 933 | .and_then(|value| value.as_f64().or_else(|| expr::to_text(&value).parse().ok())) |
| 934 | .map_or(MAX_TIMEOUT_MINUTES, |minutes| (minutes.ceil() as u32).clamp(1, max_minutes)); |
| 935 | let continue_on_error = raw |
| 936 | .get("continue-on-error") |
| 937 | .and_then(|value| expr::interpolate_value(value, &scope).ok()) |
| 938 | .is_some_and(|value| expr::truthy(&value)); |
| 939 | let (mut status, mut conclusion, mut finished) = match &reason { |
| 940 | Some(_) => ("completed", Some("failure"), Some(now())), |
| 941 | None => ("queued", None, None), |
| 942 | }; |
| 943 | // A self-hosted job waits, saying for what, until a runner takes it. |
| 944 | let (labels_json, queued_at) = if wanted.self_hosted && reason.is_none() { |
| 945 | reason = Some(waiting_reason(&wanted)); |
| 946 | (Some(serde_json::to_string(&wanted.stored())?), Some(now())) |
| 947 | } else { |
| 948 | (None, None) |
| 949 | }; |
| 950 | // The environment it names, an expression read now, and what |
| 951 | // that environment's protection rules say: it starts, waits as |
| 952 | // `pending` (protection.rs), or may not deploy there. |
| 953 | let environment = environment_of(&job.raw, &contexts).map(|env| env.name); |
| 954 | if status == "queued" |
| 955 | && let Some(name) = &environment |
| 956 | { |
| 957 | match self.gate(run, name).await? { |
| 958 | Gate::Open => {} |
| 959 | Gate::Held(why) => { |
| 960 | status = "pending"; |
| 961 | reason = Some(why); |
| 962 | } |
| 963 | Gate::Refused(why) => { |
| 964 | (status, conclusion, finished) = ("completed", Some("failure"), Some(now())); |
| 965 | reason = Some(why); |
| 966 | } |
| 967 | } |
| 968 | } |
| 969 | // Its own concurrency group. |
| 970 | let job_group = job |
| 971 | .concurrency |
| 972 | .as_ref() |
| 973 | .and_then(|c| expr::interpolate(&c.group, &scope).ok()) |
| 974 | .map(|group| group.trim().to_owned()) |
| 975 | .filter(|group| !group.is_empty()); |
| 976 | let job_cancel = job |
| 977 | .concurrency |
| 978 | .as_ref() |
| 979 | .and_then(|c| expr::interpolate_value(&c.cancel_in_progress, &scope).ok()) |
| 980 | .is_some_and(|value| expr::truthy(&value)); |
| 981 | if status != "completed" |
| 982 | && let Some(group) = &job_group |
| 983 | && !groups.iter().any(|(known, _)| known == group) |
| 984 | { |
| 985 | groups.push((group.clone(), job_cancel)); |
| 986 | } |
| 987 | let values: Vec<worker::wasm_bindgen::JsValue> = vec![ |
| 988 | name.into(), |
| 989 | serde_json::to_string(combination)?.into(), |
| 990 | status.into(), |
| 991 | optional(conclusion), |
| 992 | optional(reason.as_deref()), |
| 993 | timeout.into(), |
| 994 | u32::from(continue_on_error).into(), |
| 995 | job.max_parallel.map_or(worker::wasm_bindgen::JsValue::NULL, Into::into), |
| 996 | optional(finished.as_deref()), |
| 997 | optional(labels_json.as_deref()), |
| 998 | optional(queued_at.as_deref()), |
| 999 | optional(environment.as_deref()), |
| 1000 | optional(job_group.as_deref()), |
| 1001 | u32::from(job_cancel).into(), |
| 1002 | ]; |
| 1003 | if index == 0 { |
| 1004 | let mut bound = values; |
| 1005 | bound.push(row.id.as_str().into()); |
| 1006 | statements.push( |
| 1007 | self.db |
| 1008 | .prepare( |
| 1009 | "UPDATE jobs SET name = ?, matrix = ?, status = ?, conclusion = ?, reason = ?, timeout_minutes = ?, |
| 1010 | continue_on_error = ?, max_parallel = ?, finished_at = ?, labels = ?, queued_at = ?, environment = ?, |
| 1011 | concurrency_group = ?, cancel_in_progress = ? WHERE id = ?", |
| 1012 | ) |
| 1013 | .bind(&bound)?, |
| 1014 | ); |
| 1015 | } else { |
| 1016 | let mut bound: Vec<worker::wasm_bindgen::JsValue> = vec![ |
| 1017 | new_id("job", now_ms()).into(), |
| 1018 | row.run_id.as_str().into(), |
| 1019 | row.repo_id.as_str().into(), |
| 1020 | row.namespace.as_str().into(), |
| 1021 | row.key.as_str().into(), |
| 1022 | (index as u32).into(), |
| 1023 | row.needs.as_str().into(), |
| 1024 | optional(row.call.as_deref()), |
| 1025 | ]; |
| 1026 | bound.extend(values); |
| 1027 | statements.push( |
| 1028 | self.db |
| 1029 | .prepare( |
| 1030 | "INSERT INTO jobs (id, run_id, repo_id, namespace, key, ordinal, needs, call, name, matrix, status, conclusion, reason, |
| 1031 | timeout_minutes, continue_on_error, max_parallel, finished_at, labels, queued_at, environment, |
| 1032 | concurrency_group, cancel_in_progress) |
| 1033 | VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)", |
| 1034 | ) |
| 1035 | .bind(&bound)?, |
| 1036 | ); |
| 1037 | } |
| 1038 | } |
| 1039 | self.db.batch(statements).await?; |
| 1040 | self.replace_in_groups(run, &row.key, &groups).await |
| 1041 | } |
| 1042 | |
| 1043 | /// A job queued in its own concurrency group: with `cancel-in-progress` |
| 1044 | /// it cancels the group's other jobs; otherwise it replaces any that |
| 1045 | /// are queued and not started, and waits for the one running |
| 1046 | /// (`start_queued` starts one of a group at a time). As on GitHub. |
| 1047 | async fn replace_in_groups(&self, run: &RunRow, key: &str, groups: &[(String, bool)]) -> Result<()> { |
| 1048 | if groups.is_empty() { |
| 1049 | return Ok(()); |
| 1050 | } |
| 1051 | #[derive(Deserialize)] |
| 1052 | struct Id { |
| 1053 | id: String, |
| 1054 | } |
| 1055 | let mine: Vec<String> = self |
| 1056 | .db |
| 1057 | .prepare("SELECT id FROM jobs WHERE run_id = ? AND key = ?") |
| 1058 | .bind(&[run.id.as_str().into(), key.into()])? |
| 1059 | .all() |
| 1060 | .await? |
| 1061 | .results::<Id>()? |
| 1062 | .into_iter() |
| 1063 | .map(|row| row.id) |
| 1064 | .collect(); |
| 1065 | let mut moved: Vec<String> = Vec::new(); |
| 1066 | for (group, cancel) in groups { |
| 1067 | let others = self |
| 1068 | .db |
| 1069 | .prepare("SELECT * FROM jobs WHERE repo_id = ? AND concurrency_group = ? AND status IN ('queued', 'pending', 'in_progress')") |
| 1070 | .bind(&[run.repo_id.as_str().into(), group.as_str().into()])? |
| 1071 | .all() |
| 1072 | .await? |
| 1073 | .results::<JobRow>()?; |
| 1074 | for other in others.iter().filter(|other| !mine.contains(&other.id)) { |
| 1075 | if *cancel || other.status == "queued" { |
| 1076 | self.stop_job(other, "A newer job in the same concurrency group replaced it.").await?; |
| 1077 | if !moved.contains(&other.run_id) { |
| 1078 | moved.push(other.run_id.clone()); |
| 1079 | } |
| 1080 | } |
| 1081 | } |
| 1082 | } |
| 1083 | for other_run in moved.iter().filter(|id| **id != run.id) { |
| 1084 | Box::pin(self.advance(other_run)).await?; |
| 1085 | } |
| 1086 | Ok(()) |
| 1087 | } |
| 1088 | |
| 1089 | /// A job that calls a reusable workflow, in the repository or another |
| 1090 | /// (reach.rs): that workflow's jobs join the run under it, with the |
| 1091 | /// inputs and secrets it passes. |
| 1092 | async fn call_workflow(&self, run: &RunRow, job: &workflow::Job, row: &JobRow, uses: &str, scope: &Scope<'_>) -> Result<()> { |
| 1093 | let depth = row.call().and_then(|c| c["depth"].as_u64()).unwrap_or(0) + 1; |
| 1094 | if depth > MAX_CALL_DEPTH { |
| 1095 | return self.fail_job(row, &format!("Reusable workflows call each other more than {MAX_CALL_DEPTH} deep.")).await; |
| 1096 | } |
| 1097 | let (file, source, origin) = match self.called_workflow(run, row, uses).await? { |
| 1098 | Ok(found) => found, |
| 1099 | Err(why) => return self.fail_job(row, &why).await, |
| 1100 | }; |
| 1101 | let called = match workflow::parse(&source) { |
| 1102 | Ok(called) => called, |
| 1103 | Err(problem) => return self.fail_job(row, &format!("`{file}` does not read: {problem}")).await, |
| 1104 | }; |
| 1105 | let Some(trigger) = called.trigger("workflow_call") else { |
| 1106 | return self.fail_job(row, &format!("`{file}` cannot be called: it has no `on: workflow_call`.")).await; |
| 1107 | }; |
| 1108 | // Inputs: what the caller passes, else the called workflow's defaults. |
| 1109 | let given = match job.raw.get("with") { |
| 1110 | Some(with) => match expr::interpolate_value(with, scope) { |
| 1111 | Ok(Value::Object(given)) => given, |
| 1112 | Ok(_) => Map::new(), |
| 1113 | Err(problem) => return self.fail_job(row, &format!("Its `with` does not read: {problem}")).await, |
| 1114 | }, |
| 1115 | None => Map::new(), |
| 1116 | }; |
| 1117 | let mut inputs = Map::new(); |
| 1118 | for (name, spec) in &trigger.inputs { |
| 1119 | let value = given.get(name).cloned().or_else(|| spec.get("default").cloned()).unwrap_or(Value::Null); |
| 1120 | if value.is_null() && spec.get("required").and_then(Value::as_bool) == Some(true) { |
| 1121 | return self.fail_job(row, &format!("`{file}` needs the input `{name}`.")).await; |
| 1122 | } |
| 1123 | inputs.insert(name.clone(), value); |
| 1124 | } |
| 1125 | for (name, value) in given { |
| 1126 | inputs.entry(name).or_insert(value); |
| 1127 | } |
| 1128 | // Secrets: none but the job's token unless `secrets:` passes them, |
| 1129 | // by name or with `inherit`; read when each job starts (job_spec). |
| 1130 | let outer = row.call().filter(|c| c["role"] == "callee").and_then(|c| c.get("secrets").cloned()); |
| 1131 | let secrets = crate::reach::secrets_plan(&job.raw, scope.contexts, outer); |
| 1132 | if let Some(name) = crate::reach::missing_secrets(&called.raw, &secrets).first() { |
| 1133 | return self.fail_job(row, &format!("`{file}` needs the secret `{name}`: pass it under `secrets:`, or use `secrets: inherit`.")).await; |
| 1134 | } |
| 1135 | let mut statements = Vec::new(); |
| 1136 | for called_job in &called.jobs { |
| 1137 | let needs: Vec<String> = called_job.needs.iter().map(|n| format!("{}/{n}", row.key)).collect(); |
| 1138 | let call = json!({ |
| 1139 | "role": "callee", "parent": row.key, "job": called_job.id, "path": file, |
| 1140 | "source": source, "inputs": inputs, "depth": depth, "origin": origin, "secrets": secrets, |
| 1141 | }); |
| 1142 | statements.push( |
| 1143 | self.db |
| 1144 | .prepare("INSERT INTO jobs (id, run_id, repo_id, namespace, key, name, needs, status, call) VALUES (?, ?, ?, ?, ?, ?, ?, 'waiting', ?)") |
| 1145 | .bind(&[ |
| 1146 | new_id("job", now_ms()).into(), |
| 1147 | row.run_id.as_str().into(), |
| 1148 | row.repo_id.as_str().into(), |
| 1149 | row.namespace.as_str().into(), |
| 1150 | format!("{}/{}", row.key, called_job.id).into(), |
| 1151 | format!("{} / {}", row.name, called_job.name.clone().unwrap_or(called_job.id.clone())).into(), |
| 1152 | serde_json::to_string(&needs)?.into(), |
| 1153 | serde_json::to_string(&call)?.into(), |
| 1154 | ])?, |
| 1155 | ); |
| 1156 | } |
| 1157 | statements.push( |
| 1158 | self.db |
| 1159 | .prepare("UPDATE jobs SET status = 'calling', call = ?, reason = ?, started_at = ? WHERE id = ?") |
| 1160 | .bind(&[ |
| 1161 | serde_json::to_string(&json!({ "role": "caller", "path": file, "source": source }))?.into(), |
| 1162 | format!("Calls `{file}`.").into(), |
| 1163 | now().into(), |
| 1164 | row.id.as_str().into(), |
| 1165 | ])?, |
| 1166 | ); |
| 1167 | self.db.batch(statements).await?; |
| 1168 | Ok(()) |
| 1169 | } |
| 1170 | |
| 1171 | /// A job that called a workflow, finished with its jobs: their result, |
| 1172 | /// and the outputs the workflow declares. |
| 1173 | async fn finish_call(&self, row: &JobRow, children: &[&JobRow]) -> Result<()> { |
| 1174 | let call = row.call().unwrap_or_default(); |
| 1175 | let called = call["source"].as_str().and_then(|s| workflow::parse(s).ok()); |
| 1176 | let mut jobs_context = Map::new(); |
| 1177 | let mut by_key: std::collections::BTreeMap<String, Vec<&JobRow>> = std::collections::BTreeMap::new(); |
| 1178 | for child in children { |
| 1179 | by_key.entry(child.key.clone()).or_default().push(child); |
| 1180 | } |
| 1181 | for (key, rows) in &by_key { |
| 1182 | let mut outputs = Map::new(); |
| 1183 | for child in rows { |
| 1184 | if let Ok(Value::Object(more)) = serde_json::from_str::<Value>(&child.outputs) { |
| 1185 | outputs.extend(more); |
| 1186 | } |
| 1187 | } |
| 1188 | let id = key.rsplit('/').next().unwrap_or(key); |
| 1189 | jobs_context.insert(id.to_owned(), json!({ "result": key_result(rows), "outputs": outputs })); |
| 1190 | } |
| 1191 | let inputs = children.first().and_then(|c| c.call()).map(|c| c["inputs"].clone()).unwrap_or(json!({})); |
| 1192 | let mut contexts = Map::new(); |
| 1193 | contexts.insert("jobs".into(), Value::Object(jobs_context)); |
| 1194 | contexts.insert("inputs".into(), inputs); |
| 1195 | let scope = Scope { contexts: &contexts, status: Status::Success, hash_files: None }; |
| 1196 | let mut outputs = Map::new(); |
| 1197 | if let Some(Value::Object(declared)) = called.as_ref().map(|w| { |
| 1198 | let on = w.raw.get("on").or_else(|| w.raw.get("true")).cloned().unwrap_or(Value::Null); |
| 1199 | on.get("workflow_call").and_then(|c| c.get("outputs")).cloned().unwrap_or(Value::Null) |
| 1200 | }) { |
| 1201 | for (name, spec) in declared { |
| 1202 | if let Some(value) = spec.get("value") { |
| 1203 | let value = expr::interpolate_value(value, &scope).unwrap_or(Value::Null); |
| 1204 | outputs.insert(name, Value::String(expr::to_text(&value))); |
| 1205 | } |
| 1206 | } |
| 1207 | } |
| 1208 | let conclusion = key_result(children); |
| 1209 | self.db |
| 1210 | .prepare("UPDATE jobs SET status = 'completed', conclusion = ?, outputs = ?, finished_at = ? WHERE id = ? AND status = 'calling'") |
| 1211 | .bind(&[conclusion.into(), serde_json::to_string(&outputs)?.into(), now().into(), row.id.as_str().into()])? |
| 1212 | .run() |
| 1213 | .await?; |
| 1214 | Ok(()) |
| 1215 | } |
| 1216 | |
| 1217 | /// A file's text at a commit, if it is there. |
| 1218 | pub(crate) async fn read_file(&self, path: &RepoPath, ws: &g1t_contracts::User, sha: &str, file: &str) -> Result<Option<String>> { |
| 1219 | let blob: Outcome<g1t_contracts::repos::BlobView> = g1t_kit::call( |
| 1220 | &self.repos, |
| 1221 | "blob", |
| 1222 | &g1t_contracts::repos::BlobArgs { |
| 1223 | path: path.clone(), |
| 1224 | viewer: Some(ws.clone()), |
| 1225 | git_ref: sha.to_owned(), |
| 1226 | file_path: file.to_owned(), |
| 1227 | }, |
| 1228 | ) |
| 1229 | .await?; |
| 1230 | Ok(match blob { |
| 1231 | Outcome::Ok(view) => view.text, |
| 1232 | Outcome::Fail(_) => None, |
| 1233 | }) |
| 1234 | } |
| 1235 | |
| 1236 | async fn skip_job(&self, row: &JobRow, reason: Option<&str>) -> Result<()> { |
| 1237 | self.db |
| 1238 | .prepare("UPDATE jobs SET status = 'completed', conclusion = 'skipped', reason = ?, finished_at = ? WHERE id = ?") |
| 1239 | .bind(&[optional(reason), now().into(), row.id.as_str().into()])? |
| 1240 | .run() |
| 1241 | .await?; |
| 1242 | Ok(()) |
| 1243 | } |
| 1244 | |
| 1245 | async fn fail_job(&self, row: &JobRow, reason: &str) -> Result<()> { |
| 1246 | self.db |
| 1247 | .prepare("UPDATE jobs SET status = 'completed', conclusion = 'failure', reason = ?, finished_at = ? WHERE id = ? AND status != 'completed'") |
| 1248 | .bind(&[reason.into(), now().into(), row.id.as_str().into()])? |
| 1249 | .run() |
| 1250 | .await?; |
| 1251 | Ok(()) |
| 1252 | } |
| 1253 | |
| 1254 | /// Starts queued jobs, oldest first, while their workspace has room. |
| 1255 | pub async fn start_queued(&self) -> Result<()> { |
| 1256 | let queued = self |
| 1257 | .db |
| 1258 | // Self-hosted jobs are taken by their runners (runners.rs). |
| 1259 | .prepare("SELECT * FROM jobs WHERE status = 'queued' AND labels IS NULL ORDER BY rowid LIMIT 50") |
| 1260 | .all() |
| 1261 | .await? |
| 1262 | .results::<JobRow>()?; |
| 1263 | let mut running: std::collections::HashMap<String, u32> = std::collections::HashMap::new(); |
| 1264 | for job in queued { |
| 1265 | let in_workspace = match running.get(&job.namespace) { |
| 1266 | Some(n) => *n, |
| 1267 | None => { |
| 1268 | let n = self |
| 1269 | .db |
| 1270 | // Only g1t's own sandboxes count against the workspace's room. |
| 1271 | .prepare("SELECT COUNT(*) AS n FROM jobs WHERE status = 'in_progress' AND namespace = ? AND runner_id IS NULL") |
| 1272 | .bind(&[job.namespace.as_str().into()])? |
| 1273 | .first::<Count>(None) |
| 1274 | .await? |
| 1275 | .map_or(0, |count| count.n); |
| 1276 | running.insert(job.namespace.clone(), n); |
| 1277 | n |
| 1278 | } |
| 1279 | }; |
| 1280 | if in_workspace >= RUNNING_PER_WORKSPACE { |
| 1281 | continue; |
| 1282 | } |
| 1283 | if let Some(max) = job.max_parallel { |
| 1284 | let siblings = self |
| 1285 | .db |
| 1286 | .prepare("SELECT COUNT(*) AS n FROM jobs WHERE run_id = ? AND key = ? AND status = 'in_progress'") |
| 1287 | .bind(&[job.run_id.as_str().into(), job.key.as_str().into()])? |
| 1288 | .first::<Count>(None) |
| 1289 | .await? |
| 1290 | .map_or(0, |count| count.n); |
| 1291 | if siblings >= max { |
| 1292 | continue; |
| 1293 | } |
| 1294 | } |
| 1295 | // One job of a concurrency group runs at a time. |
| 1296 | if let Some(group) = &job.concurrency_group { |
| 1297 | let running = self |
| 1298 | .db |
| 1299 | .prepare("SELECT COUNT(*) AS n FROM jobs WHERE repo_id = ? AND concurrency_group = ? AND status = 'in_progress' AND id != ?") |
| 1300 | .bind(&[job.repo_id.as_str().into(), group.as_str().into(), job.id.as_str().into()])? |
| 1301 | .first::<Count>(None) |
| 1302 | .await? |
| 1303 | .map_or(0, |count| count.n); |
| 1304 | if running > 0 { |
| 1305 | continue; |
| 1306 | } |
| 1307 | } |
| 1308 | let token = random_hex(24); |
| 1309 | let at = now(); |
| 1310 | let claimed = self |
| 1311 | .db |
| 1312 | .prepare( |
| 1313 | "UPDATE jobs SET status = 'in_progress', token_hash = ?, started_at = ?, seen_at = ? WHERE id = ? AND status = 'queued' RETURNING id", |
| 1314 | ) |
| 1315 | .bind(&[sha256_hex(&token).into(), at.as_str().into(), at.as_str().into(), job.id.as_str().into()])? |
| 1316 | .first::<Value>(None) |
| 1317 | .await?; |
| 1318 | if claimed.is_none() { |
| 1319 | continue; |
| 1320 | } |
| 1321 | running.insert(job.namespace.clone(), in_workspace + 1); |
| 1322 | self.db |
| 1323 | .prepare("UPDATE runs SET status = 'in_progress', started_at = COALESCE(started_at, ?) WHERE id = ? AND status = 'queued'") |
| 1324 | .bind(&[at.as_str().into(), job.run_id.as_str().into()])? |
| 1325 | .run() |
| 1326 | .await?; |
| 1327 | let run = self.run_row(&job.run_id).await?; |
| 1328 | let repo: RepoPath = run.as_ref().map(|run| repo_path(&run.repo)).unwrap_or(RepoPath { |
| 1329 | namespace: job.namespace.clone(), |
| 1330 | name: String::new(), |
| 1331 | }); |
| 1332 | // Its environment and the machine it asked for, for the runner. |
| 1333 | let details = run.as_ref().map(|run| start_details(run, &job)).unwrap_or_default(); |
| 1334 | let started: Outcome<Value> = g1t_kit::call( |
| 1335 | &self.runner, |
| 1336 | "start_actions_job", |
| 1337 | &StartJobArgs { |
| 1338 | job: job.id.clone(), |
| 1339 | token, |
| 1340 | repo, |
| 1341 | timeout_minutes: job.timeout_minutes, |
| 1342 | workflow: run.as_ref().map(|run| run.path.clone()), |
| 1343 | environment: details.environment, |
| 1344 | trusted: run.as_ref().is_some_and(|run| run.trusted != 0), |
| 1345 | instance: details.instance, |
| 1346 | }, |
| 1347 | ) |
| 1348 | .await |
| 1349 | .unwrap_or_else(|error| fail(FailureCode::Conflict, format!("The runner could not be reached: {error}"))); |
| 1350 | if let Outcome::Fail(refused) = started { |
| 1351 | Box::pin(self.finish_job(&job.id, "failure", Some(&refused.message), None)).await?; |
| 1352 | } else { |
| 1353 | self.job_started(&job.id).await?; |
| 1354 | } |
| 1355 | } |
| 1356 | Ok(()) |
| 1357 | } |
| 1358 | |
| 1359 | /// Finishes a job and moves its run along. |
| 1360 | pub async fn finish_job(&self, job_id: &str, conclusion: &str, reason: Option<&str>, outputs: Option<&Map<String, Value>>) -> Result<()> { |
| 1361 | let finished = self |
| 1362 | .db |
| 1363 | .prepare( |
| 1364 | "UPDATE jobs SET status = 'completed', conclusion = ?, reason = COALESCE(?, reason), outputs = COALESCE(?, outputs), |
| 1365 | finished_at = ?, token_hash = NULL WHERE id = ? AND status != 'completed' RETURNING *", |
| 1366 | ) |
| 1367 | .bind(&[ |
| 1368 | conclusion.into(), |
| 1369 | optional(reason), |
| 1370 | outputs.map(|o| serde_json::to_string(o).unwrap_or_default()).as_deref().map_or(worker::wasm_bindgen::JsValue::NULL, Into::into), |
| 1371 | now().into(), |
| 1372 | job_id.into(), |
| 1373 | ])? |
| 1374 | .first::<JobRow>(None) |
| 1375 | .await?; |
| 1376 | let Some(job) = finished else { return Ok(()) }; |
| 1377 | // Its G1T_TOKEN stops working with it. |
| 1378 | self.revoke_job_tokens(&job.id).await; |
| 1379 | // A job that deploys failed: so did its run's deployment, now. |
| 1380 | if conclusion == "failure" && job.continue_on_error == 0 && job.started_at.is_some() |
| 1381 | && let Some(run) = self.run_row(&job.run_id).await? |
| 1382 | && let Some(env) = deploys_to(&run, &job) |
| 1383 | { |
| 1384 | self.report_deployment(&run, &env, "failure", false).await; |
| 1385 | } |
| 1386 | // A self-hosted runner's job: the runner is free again, and its time |
| 1387 | // is recorded, at nothing. |
| 1388 | if job.runner_id.is_some() { |
| 1389 | self.released(&job).await?; |
| 1390 | } |
| 1391 | // Steps still marked as going are not going any more. |
| 1392 | let mut steps: Vec<Value> = serde_json::from_str(&job.steps).unwrap_or_default(); |
| 1393 | let mut touched = false; |
| 1394 | for step in steps.iter_mut() { |
| 1395 | if step["status"] != "completed" { |
| 1396 | let was_running = step["status"] == "in_progress"; |
| 1397 | step["status"] = json!("completed"); |
| 1398 | step["conclusion"] = json!(if was_running { conclusion } else { "skipped" }); |
| 1399 | touched = true; |
| 1400 | } |
| 1401 | } |
| 1402 | if touched { |
| 1403 | self.db |
| 1404 | .prepare("UPDATE jobs SET steps = ? WHERE id = ?") |
| 1405 | .bind(&[serde_json::to_string(&steps)?.into(), job.id.as_str().into()])? |
| 1406 | .run() |
| 1407 | .await?; |
| 1408 | } |
| 1409 | // fail-fast: one failed combination stops the rest of its matrix. |
| 1410 | if conclusion == "failure" && job.continue_on_error == 0 && job.matrix.as_deref().is_some_and(|m| m != "{}") { |
| 1411 | let run = self.run_row(&job.run_id).await?; |
| 1412 | let fail_fast = run |
| 1413 | .as_ref() |
| 1414 | .and_then(|run| workflow::parse(&run.source).ok()) |
| 1415 | .and_then(|workflow| workflow.jobs.into_iter().find(|j| j.id == job.key)) |
| 1416 | .is_none_or(|j| j.fail_fast); |
| 1417 | if fail_fast { |
| 1418 | let siblings = self |
| 1419 | .db |
| 1420 | .prepare("SELECT * FROM jobs WHERE run_id = ? AND key = ? AND status != 'completed'") |
| 1421 | .bind(&[job.run_id.as_str().into(), job.key.as_str().into()])? |
| 1422 | .all() |
| 1423 | .await? |
| 1424 | .results::<JobRow>()?; |
| 1425 | for sibling in siblings { |
| 1426 | self.stop_job(&sibling, "Another job of its matrix failed, and the matrix is fail-fast.").await?; |
| 1427 | } |
| 1428 | } |
| 1429 | } |
| 1430 | Box::pin(self.advance(&job.run_id)).await |
| 1431 | } |
| 1432 | |
| 1433 | /// Cancels a job. One that is running is told to stop (the answer to |
| 1434 | /// its next report), and runs its `if: always()` and `cancelled()` |
| 1435 | /// steps and its post steps before it ends `cancelled`; one that has |
| 1436 | /// not started is cancelled at once. |
| 1437 | async fn stop_job(&self, job: &JobRow, reason: &str) -> Result<()> { |
| 1438 | if job.status == "in_progress" && job.started_at.is_some() { |
| 1439 | self.db |
| 1440 | .prepare("UPDATE jobs SET cancel_requested_at = COALESCE(cancel_requested_at, ?), reason = ? WHERE id = ? AND status = 'in_progress'") |
| 1441 | .bind(&[now().into(), reason.into(), job.id.as_str().into()])? |
| 1442 | .run() |
| 1443 | .await?; |
| 1444 | return Ok(()); |
| 1445 | } |
| 1446 | self.hard_stop(job, reason).await |
| 1447 | } |
| 1448 | |
| 1449 | /// Cancels a job outright, stopping its sandbox if it has one. |
| 1450 | async fn hard_stop(&self, job: &JobRow, reason: &str) -> Result<()> { |
| 1451 | // A self-hosted runner hears it was cancelled on its next poll. |
| 1452 | if job.status == "in_progress" && job.runner_id.is_none() { |
| 1453 | let _: Result<Value> = g1t_kit::call(&self.runner, "stop_actions_job", &json!({ "job": job.id })).await; |
| 1454 | } |
| 1455 | let stopped = self |
| 1456 | .db |
| 1457 | .prepare("UPDATE jobs SET status = 'completed', conclusion = 'cancelled', reason = ?, finished_at = ?, token_hash = NULL WHERE id = ? AND status != 'completed' RETURNING *") |
| 1458 | .bind(&[reason.into(), now().into(), job.id.as_str().into()])? |
| 1459 | .first::<JobRow>(None) |
| 1460 | .await?; |
| 1461 | if stopped.as_ref().is_some_and(|row| row.started_at.is_some()) { |
| 1462 | self.revoke_job_tokens(&job.id).await; |
| 1463 | } |
| 1464 | if let Some(stopped) = stopped.filter(|row| row.runner_id.is_some()) { |
| 1465 | self.released(&stopped).await?; |
| 1466 | } |
| 1467 | Ok(()) |
| 1468 | } |
| 1469 | |
| 1470 | /// Ends a job's tokens at once. A failure is logged: the token expires |
| 1471 | /// on its own soon after the job's time limit. |
| 1472 | async fn revoke_job_tokens(&self, job_id: &str) { |
| 1473 | let revoked: Result<bool> = |
| 1474 | g1t_kit::call(&self.identity, "revoke_job_tokens", &RevokeJobTokensArgs { job_id: job_id.to_owned() }).await; |
| 1475 | if let Err(error) = revoked { |
| 1476 | worker::console_error!("actions: the tokens of job {job_id} were not revoked: {error}"); |
| 1477 | } |
| 1478 | } |
| 1479 | |
| 1480 | /// Finishes the run when every job has. |
| 1481 | async fn finish_if_done(&self, run_id: &str) -> Result<()> { |
| 1482 | let Some(run) = self.run_row(run_id).await? else { return Ok(()) }; |
| 1483 | if matches!(run.status.as_str(), "completed" | "pending" | "action_required") { |
| 1484 | return Ok(()); |
| 1485 | } |
| 1486 | let jobs = self.job_rows(run_id).await?; |
| 1487 | if !jobs.iter().all(|job| job.status == "completed") { |
| 1488 | return Ok(()); |
| 1489 | } |
| 1490 | self.finish_run(&run, None).await |
| 1491 | } |
| 1492 | |
| 1493 | async fn finish_run(&self, run: &RunRow, error: Option<&str>) -> Result<()> { |
| 1494 | let jobs = self.job_rows(&run.id).await?; |
| 1495 | let rows: Vec<&JobRow> = jobs.iter().collect(); |
| 1496 | let conclusion = if error.is_some() { |
| 1497 | "failure" |
| 1498 | } else if run.conclusion.as_deref() == Some("cancelled") { |
| 1499 | "cancelled" |
| 1500 | } else if rows.is_empty() { |
| 1501 | "skipped" |
| 1502 | } else { |
| 1503 | key_result(&rows) |
| 1504 | }; |
| 1505 | let done = self |
| 1506 | .db |
| 1507 | .prepare("UPDATE runs SET status = 'completed', conclusion = ?, error = COALESCE(?, error), finished_at = ? WHERE id = ? AND status != 'completed' RETURNING id") |
| 1508 | .bind(&[conclusion.into(), optional(error), now().into(), run.id.as_str().into()])? |
| 1509 | .first::<Value>(None) |
| 1510 | .await?; |
| 1511 | if done.is_none() { |
| 1512 | return Ok(()); |
| 1513 | } |
| 1514 | self.report_status(run, conclusion).await?; |
| 1515 | self.settle_deployments(run, &jobs, conclusion == "cancelled").await; |
| 1516 | let published: Result<()> = g1t_kit::call( |
| 1517 | &self.events, |
| 1518 | "publish", |
| 1519 | &g1t_contracts::events::Publish { |
| 1520 | events: vec![g1t_contracts::events::NewEvent { |
| 1521 | kind: "workflow.completed", |
| 1522 | source: "actions", |
| 1523 | repo_id: Some(run.repo_id.clone()), |
| 1524 | actor: run.actor_id.clone(), |
| 1525 | data: g1t_contracts::events::WorkflowEvent { |
| 1526 | run_id: run.id.clone(), |
| 1527 | repo_id: run.repo_id.clone(), |
| 1528 | workflow: run.name.clone(), |
| 1529 | path: run.path.clone(), |
| 1530 | number: run.number, |
| 1531 | event: run.event.clone(), |
| 1532 | conclusion: conclusion.to_owned(), |
| 1533 | git_ref: run.git_ref.clone(), |
| 1534 | sha: run.sha.clone(), |
| 1535 | pull: run.pull, |
| 1536 | }, |
| 1537 | }], |
| 1538 | }, |
| 1539 | ) |
| 1540 | .await; |
| 1541 | if let Err(error) = published { |
| 1542 | worker::console_error!("actions: could not publish workflow.completed: {error}"); |
| 1543 | } |
| 1544 | // The next run waiting in its concurrency group. |
| 1545 | if let Some(group) = &run.concurrency_group { |
| 1546 | let next = self |
| 1547 | .db |
| 1548 | .prepare("SELECT * FROM runs WHERE repo_id = ? AND concurrency_group = ? AND status = 'pending' ORDER BY id LIMIT 1") |
| 1549 | .bind(&[run.repo_id.as_str().into(), group.as_str().into()])? |
| 1550 | .first::<RunRow>(None) |
| 1551 | .await?; |
| 1552 | if let Some(next) = next { |
| 1553 | self.db.prepare("UPDATE runs SET status = 'queued' WHERE id = ?").bind(&[next.id.as_str().into()])?.run().await?; |
| 1554 | Box::pin(self.advance(&next.id)).await?; |
| 1555 | } |
| 1556 | } |
| 1557 | Ok(()) |
| 1558 | } |
| 1559 | |
| 1560 | /// Tells the deployments service how a run's deployment to `env` stands: |
| 1561 | /// made when its first job naming the environment starts, failed when one |
| 1562 | /// of them fails, and settled (`last`) when the run finishes. Never fails |
| 1563 | /// the run: a deployment that could not be recorded is logged. |
| 1564 | async fn report_deployment(&self, run: &RunRow, env: &JobEnvironment, state: &str, last: bool) { |
| 1565 | let path = repo_path(&run.repo); |
| 1566 | let reported: Result<Value> = g1t_kit::call( |
| 1567 | &self.deployments, |
| 1568 | "actions_deployment", |
| 1569 | &json!({ |
| 1570 | "repoId": run.repo_id, |
| 1571 | "repo": { "namespace": path.namespace, "name": path.name }, |
| 1572 | "runId": run.id, |
| 1573 | "attempt": run.attempt, |
| 1574 | "runUrl": format!("{SITE}/{}/actions/runs/{}", run.repo, run.id), |
| 1575 | "environment": env.name, |
| 1576 | "url": env.url, |
| 1577 | "ref": run.git_ref, |
| 1578 | "sha": run.sha, |
| 1579 | "state": state, |
| 1580 | "final": last, |
| 1581 | "creator": run.actor, |
| 1582 | "workflow": run.name, |
| 1583 | }), |
| 1584 | ) |
| 1585 | .await; |
| 1586 | if let Err(error) = reported { |
| 1587 | worker::console_error!("actions: deployment to {} not recorded for run {}: {error}", env.name, run.id); |
| 1588 | } |
| 1589 | } |
| 1590 | |
| 1591 | /// A job that deploys has started: its run's deployment to the |
| 1592 | /// environment is under way. |
| 1593 | pub(crate) async fn job_started(&self, job_id: &str) -> Result<()> { |
| 1594 | let Some(job) = self.db.prepare("SELECT * FROM jobs WHERE id = ?").bind(&[job_id.into()])?.first::<JobRow>(None).await? else { |
| 1595 | return Ok(()); |
| 1596 | }; |
| 1597 | let Some(run) = self.run_row(&job.run_id).await? else { return Ok(()) }; |
| 1598 | if let Some(env) = deploys_to(&run, &job) { |
| 1599 | self.report_deployment(&run, &env, "in_progress", false).await; |
| 1600 | } |
| 1601 | Ok(()) |
| 1602 | } |
| 1603 | |
| 1604 | /// The run is over: each environment its jobs deployed to takes the |
| 1605 | /// outcome of those jobs (`deployment_outcome`). |
| 1606 | async fn settle_deployments(&self, run: &RunRow, jobs: &[JobRow], cancelled: bool) { |
| 1607 | let mut seen: Vec<(JobEnvironment, Vec<Option<String>>)> = Vec::new(); |
| 1608 | for job in jobs { |
| 1609 | let Some(env) = deploys_to(run, job) else { continue }; |
| 1610 | let conclusion = if cancelled && job.conclusion.as_deref() != Some("skipped") && job.started_at.is_some() { |
| 1611 | Some("cancelled".to_owned()) |
| 1612 | } else if job.started_at.is_none() { |
| 1613 | Some("skipped".to_owned()) |
| 1614 | } else { |
| 1615 | job.conclusion.clone() |
| 1616 | }; |
| 1617 | match seen.iter_mut().find(|(known, _)| known.name.eq_ignore_ascii_case(&env.name)) { |
| 1618 | Some((known, conclusions)) => { |
| 1619 | if known.url.is_none() { |
| 1620 | known.url = env.url.clone(); |
| 1621 | } |
| 1622 | conclusions.push(conclusion); |
| 1623 | } |
| 1624 | None => seen.push((env, vec![conclusion])), |
| 1625 | } |
| 1626 | } |
| 1627 | for (env, conclusions) in seen { |
| 1628 | if let Some(state) = deployment_outcome(&conclusions) { |
| 1629 | self.report_deployment(run, &env, state, true).await; |
| 1630 | } |
| 1631 | } |
| 1632 | } |
| 1633 | |
| 1634 | /// Tells the pull request (or commit) how the run went, as a status. |
| 1635 | async fn report_status(&self, run: &RunRow, conclusion: &str) -> Result<()> { |
| 1636 | let state = match conclusion { |
| 1637 | "success" | "skipped" => "success", |
| 1638 | "cancelled" => "error", |
| 1639 | _ => "failure", |
| 1640 | }; |
| 1641 | let _: Result<Value> = g1t_kit::call( |
| 1642 | &self.work, |
| 1643 | "set_commit_status", |
| 1644 | &json!({ |
| 1645 | "repoId": run.repo_id, |
| 1646 | "sha": run.sha, |
| 1647 | "context": format!("{} / {}", run.name, run.event), |
| 1648 | "state": state, |
| 1649 | "description": format!("{} {}", run.name, match conclusion { |
| 1650 | "success" => "passed", |
| 1651 | "skipped" => "was skipped", |
| 1652 | "cancelled" => "was cancelled", |
| 1653 | _ => "failed", |
| 1654 | }), |
| 1655 | "targetUrl": format!("{SITE}/{}/actions/runs/{}", run.repo, run.id), |
| 1656 | "source": "actions", |
| 1657 | }), |
| 1658 | ) |
| 1659 | .await; |
| 1660 | Ok(()) |
| 1661 | } |
| 1662 | |
| 1663 | /// Tells the pull request a run has started on its head. |
| 1664 | pub async fn report_pending(&self, run: &RunRow) -> Result<()> { |
| 1665 | let _: Result<Value> = g1t_kit::call( |
| 1666 | &self.work, |
| 1667 | "set_commit_status", |
| 1668 | &json!({ |
| 1669 | "repoId": run.repo_id, |
| 1670 | "sha": run.sha, |
| 1671 | "context": format!("{} / {}", run.name, run.event), |
| 1672 | "state": "pending", |
| 1673 | "description": if run.status == "action_required" { |
| 1674 | format!("{} is waiting for approval", run.name) |
| 1675 | } else { |
| 1676 | format!("{} is running", run.name) |
| 1677 | }, |
| 1678 | "targetUrl": format!("{SITE}/{}/actions/runs/{}", run.repo, run.id), |
| 1679 | "source": "actions", |
| 1680 | }), |
| 1681 | ) |
| 1682 | .await; |
| 1683 | Ok(()) |
| 1684 | } |
| 1685 | |
| 1686 | /// Cancels a run: its waiting and queued jobs at once, and its running |
| 1687 | /// ones gracefully (`stop_job`), or outright when `force`. |
| 1688 | pub async fn cancel_run(&self, run: &RunRow, reason: &str, force: bool) -> Result<()> { |
| 1689 | self.db |
| 1690 | .prepare("UPDATE runs SET conclusion = 'cancelled' WHERE id = ? AND status != 'completed'") |
| 1691 | .bind(&[run.id.as_str().into()])? |
| 1692 | .run() |
| 1693 | .await?; |
| 1694 | for job in self.job_rows(&run.id).await?.iter().filter(|job| job.status != "completed") { |
| 1695 | if force { |
| 1696 | self.hard_stop(job, reason).await?; |
| 1697 | } else { |
| 1698 | self.stop_job(job, reason).await?; |
| 1699 | } |
| 1700 | } |
| 1701 | if run.status == "pending" || run.status == "action_required" { |
| 1702 | self.db.prepare("UPDATE runs SET status = 'queued' WHERE id = ?").bind(&[run.id.as_str().into()])?.run().await?; |
| 1703 | } |
| 1704 | self.advance(&run.id).await |
| 1705 | } |
| 1706 | |
| 1707 | /// Cancels every run of the repository that has not finished, for |
| 1708 | /// `repo.deleted` and `repo.archived`. |
| 1709 | pub async fn stop_runs(&self, repo_id: &str) -> Result<()> { |
| 1710 | let runs = self |
| 1711 | .db |
| 1712 | .prepare("SELECT * FROM runs WHERE repo_id = ? AND status != 'completed'") |
| 1713 | .bind(&[repo_id.into()])? |
| 1714 | .all() |
| 1715 | .await? |
| 1716 | .results::<RunRow>()?; |
| 1717 | for run in runs { |
| 1718 | self.cancel_run(&run, "The repository was archived or deleted.", true).await?; |
| 1719 | } |
| 1720 | Ok(()) |
| 1721 | } |
| 1722 | |
| 1723 | pub async fn cancel(&self, a: RunActionArgs) -> Result<Outcome<WorkflowRun>> { |
| 1724 | if let Outcome::Fail(refused) = self.may(&a.actor, &a.repo, Capability::Run).await? { |
| 1725 | return Ok(Outcome::Fail(refused)); |
| 1726 | } |
| 1727 | let run = check!(self.run_in(&a.repo, &a.id).await?); |
| 1728 | if run.status == "completed" { |
| 1729 | return Ok(fail(FailureCode::Conflict, "The run has already finished.")); |
| 1730 | } |
| 1731 | // Cancelling a run that is already cancelling stops its jobs |
| 1732 | // outright, without waiting for their cleanup steps. |
| 1733 | let force = a.force || run.conclusion.as_deref() == Some("cancelled"); |
| 1734 | let reason = if force { |
| 1735 | format!("{} stopped the run without waiting for its cleanup steps.", a.actor.username) |
| 1736 | } else { |
| 1737 | format!("{} cancelled the run.", a.actor.username) |
| 1738 | }; |
| 1739 | self.cancel_run(&run, &reason, force).await?; |
| 1740 | self.run_summary(&run.id).await |
| 1741 | } |
| 1742 | |
| 1743 | /// Runs again, as a new attempt: every job, with `failed_only` those |
| 1744 | /// that did not succeed, or with `job` that one; each with the jobs that |
| 1745 | /// need them. The attempt that ends is kept, its jobs and their logs and |
| 1746 | /// summaries with it (`job_attempts`, `run_attempts`). |
| 1747 | pub async fn rerun(&self, a: RunActionArgs) -> Result<Outcome<WorkflowRun>> { |
| 1748 | if let Outcome::Fail(refused) = self.may(&a.actor, &a.repo, Capability::Run).await? { |
| 1749 | return Ok(Outcome::Fail(refused)); |
| 1750 | } |
| 1751 | // A job alone (`POST …/jobs/{job}/rerun`) names its run. |
| 1752 | let run_id = match (&a.job, a.id.is_empty()) { |
| 1753 | (Some(job), true) => { |
| 1754 | #[derive(Deserialize)] |
| 1755 | struct Of { |
| 1756 | run_id: String, |
| 1757 | } |
| 1758 | let of = self.db.prepare("SELECT run_id FROM jobs WHERE id = ?").bind(&[job.as_str().into()])?.first::<Of>(None).await?; |
| 1759 | match of { |
| 1760 | Some(of) => of.run_id, |
| 1761 | None => return Ok(fail(FailureCode::NotFound, "No such job.")), |
| 1762 | } |
| 1763 | } |
| 1764 | _ => a.id.clone(), |
| 1765 | }; |
| 1766 | let run = check!(self.run_in(&a.repo, &run_id).await?); |
| 1767 | if run.status != "completed" { |
| 1768 | return Ok(fail(FailureCode::Conflict, "The run is still going: cancel it first.")); |
| 1769 | } |
| 1770 | // Nothing starts again on an archived repository. |
| 1771 | match self.visible_repo(&a.repo, &Some(a.actor.clone())).await? { |
| 1772 | Some(repo) if repo.archived() => { |
| 1773 | return Ok(fail(FailureCode::Forbidden, g1t_contracts::repos::archived_message(&repo.namespace, &repo.name))); |
| 1774 | } |
| 1775 | Some(_) => {} |
| 1776 | None => return Ok(fail(FailureCode::NotFound, "There is no such repository.")), |
| 1777 | } |
| 1778 | if run.error.is_some() { |
| 1779 | return Ok(fail(FailureCode::Conflict, "This run never started: fix the workflow file and push again.")); |
| 1780 | } |
| 1781 | let jobs = self.job_rows(&run.id).await?; |
| 1782 | let workflow = workflow::parse(&run.source).ok(); |
| 1783 | let target = match &a.job { |
| 1784 | Some(id) => match jobs.iter().find(|job| &job.id == id) { |
| 1785 | Some(job) => Some(top_key(&job.key).to_owned()), |
| 1786 | None => return Ok(fail(FailureCode::NotFound, "That job is not in the run's latest attempt.")), |
| 1787 | }, |
| 1788 | None => None, |
| 1789 | }; |
| 1790 | let which = match (&target, a.failed_only) { |
| 1791 | (Some(key), _) => Rerun::Job(key), |
| 1792 | (None, true) => Rerun::Failed, |
| 1793 | (None, false) => Rerun::All, |
| 1794 | }; |
| 1795 | let order = workflow.as_ref().map(|w| w.job_order()).unwrap_or_default(); |
| 1796 | let again = rerun_keys( |
| 1797 | &order, |
| 1798 | |key| jobs.iter().find(|j| j.key == key).map(JobRow::needs).unwrap_or_default(), |
| 1799 | |key| jobs.iter().filter(|j| j.key == key).all(|j| j.conclusion.as_deref() == Some("success")), |
| 1800 | which, |
| 1801 | ); |
| 1802 | if again.is_empty() { |
| 1803 | return Ok(fail(FailureCode::Conflict, "Every job succeeded: there is nothing to run again.")); |
| 1804 | } |
| 1805 | // The jobs that run again, a called workflow's with the job calling it. |
| 1806 | let rerun_ids: Vec<&str> = jobs.iter().filter(|job| again.iter().any(|key| key == top_key(&job.key))).map(|job| job.id.as_str()).collect(); |
| 1807 | let ids = serde_json::to_string(&rerun_ids)?; |
| 1808 | let ended = run.attempt.to_string(); |
| 1809 | let ended = ended.as_str(); |
| 1810 | let mut statements = vec![ |
| 1811 | // The attempt that ends, as it ended. |
| 1812 | self.db |
| 1813 | .prepare( |
| 1814 | "INSERT OR REPLACE INTO run_attempts (run_id, attempt, repo_id, conclusion, actor, debug, started_at, finished_at) |
| 1815 | SELECT id, attempt, repo_id, conclusion, COALESCE(triggering_actor, actor), debug, started_at, finished_at FROM runs WHERE id = ?", |
| 1816 | ) |
| 1817 | .bind(&[run.id.as_str().into()])?, |
| 1818 | // Earlier attempts that showed a job's logs from where it ran |
| 1819 | // then now find them where they move to. |
| 1820 | self.db |
| 1821 | .prepare("UPDATE job_attempts SET log_id = log_id || '.' || ?1 WHERE run_id = ?2 AND log_id IN (SELECT value FROM json_each(?3))") |
| 1822 | .bind(&[ended.into(), run.id.as_str().into(), ids.as_str().into()])?, |
| 1823 | // Each of its jobs: one that runs again keeps its logs under |
| 1824 | // `{id}.{attempt}`; one left alone is still the live job's. |
| 1825 | self.db |
| 1826 | .prepare( |
| 1827 | "INSERT OR REPLACE INTO job_attempts (id, run_id, repo_id, attempt, job_id, log_id, key, ordinal, name, needs, status, conclusion, |
| 1828 | steps, annotations, reason, environment, labels, runner_name, started_at, finished_at) |
| 1829 | SELECT id || '.' || ?1, run_id, repo_id, ?1, id, |
| 1830 | CASE WHEN id IN (SELECT value FROM json_each(?3)) THEN id || '.' || ?1 ELSE id END, |
| 1831 | key, ordinal, name, needs, status, conclusion, steps, annotations, reason, environment, labels, runner_name, started_at, finished_at |
| 1832 | FROM jobs WHERE run_id = ?2 ORDER BY rowid", |
| 1833 | ) |
| 1834 | .bind(&[ended.into(), run.id.as_str().into(), ids.as_str().into()])?, |
| 1835 | self.db |
| 1836 | .prepare("UPDATE logs SET job_id = job_id || '.' || ?1 WHERE job_id IN (SELECT value FROM json_each(?2))") |
| 1837 | .bind(&[ended.into(), ids.as_str().into()])?, |
| 1838 | self.db |
| 1839 | .prepare("UPDATE job_summaries SET job_id = job_id || '.' || ?1 WHERE job_id IN (SELECT value FROM json_each(?2))") |
| 1840 | .bind(&[ended.into(), ids.as_str().into()])?, |
| 1841 | ]; |
| 1842 | for key in &again { |
| 1843 | statements.push(self.db.prepare("DELETE FROM jobs WHERE run_id = ? AND key = ? AND ordinal > 0").bind(&[run.id.as_str().into(), key.as_str().into()])?); |
| 1844 | // The jobs of a workflow it called are made again when it calls it again. |
| 1845 | statements.push( |
| 1846 | self.db |
| 1847 | .prepare("DELETE FROM jobs WHERE run_id = ? AND key LIKE ?") |
| 1848 | .bind(&[run.id.as_str().into(), format!("{key}/%").into()])?, |
| 1849 | ); |
| 1850 | statements.push( |
| 1851 | self.db |
| 1852 | .prepare( |
| 1853 | "UPDATE jobs SET status = 'waiting', conclusion = NULL, steps = '[]', annotations = '[]', outputs = '{}', reason = NULL, |
| 1854 | matrix = NULL, call = NULL, token_hash = NULL, seen_at = NULL, started_at = NULL, finished_at = NULL, |
| 1855 | labels = NULL, queued_at = NULL, runner_id = NULL, runner_name = NULL, environment = NULL, |
| 1856 | concurrency_group = NULL, cancel_in_progress = 0, cancel_requested_at = NULL WHERE run_id = ? AND key = ?", |
| 1857 | ) |
| 1858 | .bind(&[run.id.as_str().into(), key.as_str().into()])?, |
| 1859 | ); |
| 1860 | } |
| 1861 | statements.push( |
| 1862 | self.db |
| 1863 | .prepare( |
| 1864 | "UPDATE runs SET status = 'queued', conclusion = NULL, attempt = attempt + 1, started_at = NULL, finished_at = NULL, |
| 1865 | triggering_actor = ?, debug = ? WHERE id = ?", |
| 1866 | ) |
| 1867 | .bind(&[a.actor.username.as_str().into(), u32::from(a.debug).into(), run.id.as_str().into()])?, |
| 1868 | ); |
| 1869 | self.db.batch(statements).await?; |
| 1870 | if let Some(run) = self.run_row(&run.id).await? { |
| 1871 | self.report_pending(&run).await?; |
| 1872 | } |
| 1873 | self.advance(&run.id).await?; |
| 1874 | self.run_summary(&run.id).await |
| 1875 | } |
| 1876 | |
| 1877 | pub async fn run_in(&self, repo: &RepoPath, id: &str) -> Result<Outcome<RunRow>> { |
| 1878 | let row = self |
| 1879 | .db |
| 1880 | .prepare("SELECT * FROM runs WHERE id = ? AND lower(repo) = lower(?)") |
| 1881 | .bind(&[id.into(), format!("{}/{}", repo.namespace, repo.name).into()])? |
| 1882 | .first::<RunRow>(None) |
| 1883 | .await?; |
| 1884 | Ok(row.map_or_else(|| fail(FailureCode::NotFound, "No such run."), Outcome::Ok)) |
| 1885 | } |
| 1886 | |
| 1887 | // --- The sandbox's side ----------------------------------------------------- |
| 1888 | |
| 1889 | pub(crate) async fn job_for_token(&self, a: &JobCallArgs) -> Result<Outcome<JobRow>> { |
| 1890 | let job = self.db.prepare("SELECT * FROM jobs WHERE id = ?").bind(&[a.job.as_str().into()])?.first::<JobRow>(None).await?; |
| 1891 | Ok(match job { |
| 1892 | Some(job) if job.status == "in_progress" && job.token_hash.as_deref().is_some_and(|hash| same(hash, &sha256_hex(&a.token))) => { |
| 1893 | Outcome::Ok(job) |
| 1894 | } |
| 1895 | _ => fail(FailureCode::Unauthenticated, "That job is not running, or the token is not its."), |
| 1896 | }) |
| 1897 | } |
| 1898 | |
| 1899 | /// `job_auth`: which run and repository a running job's token is for, |
| 1900 | /// so the API can keep its artifacts and cache. |
| 1901 | pub async fn job_auth(&self, a: JobCallArgs) -> Result<Outcome<Value>> { |
| 1902 | let job = check!(self.job_for_token(&a).await?); |
| 1903 | Ok(Outcome::Ok(json!({ "run": job.run_id, "repoId": job.repo_id }))) |
| 1904 | } |
| 1905 | |
| 1906 | /// `job_spec`: everything the sandbox needs to run the job. |
| 1907 | pub async fn job_spec(&self, a: JobCallArgs) -> Result<Outcome<Value>> { |
| 1908 | let job = check!(self.job_for_token(&a).await?); |
| 1909 | let Some(run) = self.run_row(&job.run_id).await? else { |
| 1910 | return Ok(fail(FailureCode::NotFound, "No such run.")); |
| 1911 | }; |
| 1912 | let Ok(caller) = workflow::parse(&run.source) else { |
| 1913 | return Ok(fail(FailureCode::Invalid, "The workflow no longer reads.")); |
| 1914 | }; |
| 1915 | // A called workflow's job runs as that workflow defines it. |
| 1916 | let callee = job.callee(); |
| 1917 | let (workflow, spec, call_inputs) = match callee { |
| 1918 | Some((called, spec, call)) => (called, spec, Some(call["inputs"].clone())), |
| 1919 | None => match caller.jobs.iter().find(|j| j.id == job.key) { |
| 1920 | Some(spec) => (caller.clone(), spec.clone(), None), |
| 1921 | None => return Ok(fail(FailureCode::NotFound, "The job is not in the workflow.")), |
| 1922 | }, |
| 1923 | }; |
| 1924 | let spec = &spec; |
| 1925 | let repo = repo_path(&run.repo); |
| 1926 | let trusted = run.trusted != 0; |
| 1927 | // The job's `environment:`, by name, as read when its needs were done |
| 1928 | // (an expression included), once the environment's protection rules |
| 1929 | // let it start: entries with a value for it give that value instead |
| 1930 | // of their default, as GitHub's environment secrets do. |
| 1931 | let environment: Option<String> = job.environment.clone(); |
| 1932 | // What its token may do: its `permissions:` (a called workflow's |
| 1933 | // jobs no more than the job that calls it), else the repository's |
| 1934 | // default; read-only for a pull request from outside. |
| 1935 | // The repository's default, its workspace's taken in, and whether |
| 1936 | // its jobs may open and approve pull requests. |
| 1937 | let (default, pull_requests) = self.token_policy(&run.repo_id).await?; |
| 1938 | let mut permissions = spec.permissions(&workflow, default); |
| 1939 | if let Some(parent) = &job.call().filter(|c| c["role"] == "callee").and_then(|c| c["parent"].as_str().map(str::to_owned)) { |
| 1940 | let top = parent.split('/').next().unwrap_or(parent); |
| 1941 | if let Some(caller_job) = caller.jobs.iter().find(|j| j.id == top) { |
| 1942 | permissions = permissions.capped_by(&caller_job.permissions(&caller, default)); |
| 1943 | } |
| 1944 | } |
| 1945 | if !trusted { |
| 1946 | permissions = permissions.read_only(); |
| 1947 | } |
| 1948 | // G1T_TOKEN, and GITHUB_TOKEN as its alias: a token of the |
| 1949 | // workspace's that reaches this repository only, with the scopes |
| 1950 | // its permissions give, until the job ends. |
| 1951 | let token = match self.workspace_actor(&repo.namespace).await? { |
| 1952 | Some(workspace) => { |
| 1953 | let created: CreatedAccessToken = g1t_kit::call( |
| 1954 | &self.identity, |
| 1955 | "create_job_token", |
| 1956 | &CreateJobTokenArgs { |
| 1957 | workspace, |
| 1958 | repo: repo.clone(), |
| 1959 | run_id: run.id.clone(), |
| 1960 | job_id: job.id.clone(), |
| 1961 | name: format!("G1T_TOKEN for {} run {}", run.repo, run.number), |
| 1962 | ttl_seconds: u64::from(job.timeout_minutes) * 60 + 600, |
| 1963 | scopes: permissions.scopes().into_iter().map(str::to_owned).collect(), |
| 1964 | pull_requests: pull_requests && trusted, |
| 1965 | }, |
| 1966 | ) |
| 1967 | .await?; |
| 1968 | created.token |
| 1969 | } |
| 1970 | None => String::new(), |
| 1971 | }; |
| 1972 | // A run that is not trusted (a pull request from outside the |
| 1973 | // workspace) gets no secrets and an empty token. |
| 1974 | let passed = job.call().filter(|c| c["role"] == "callee").and_then(|c| c.get("secrets").cloned()); |
| 1975 | let mut secrets = match (trusted, passed) { |
| 1976 | (false, _) => Map::new(), |
| 1977 | (true, None) => self.secrets_for(&run.repo_id, &run.repo, environment.as_deref(), true).await?, |
| 1978 | // A called workflow's job: what its callers passed it (reach.rs), |
| 1979 | // and its own environment's secrets over them. |
| 1980 | (true, Some(plan)) => { |
| 1981 | let base = self.secrets_for(&run.repo_id, &run.repo, None, true).await?; |
| 1982 | let vars = self.variables_for(&run.repo_id, &run.repo, None, true).await?; |
| 1983 | let github = run.info().context(&job.key, "", run.action.as_deref()); |
| 1984 | let mut passed = crate::reach::resolve_secrets(&plan, &base, &github, &vars); |
| 1985 | if let Some(name) = environment.as_deref() { |
| 1986 | let own = self.secrets_for(&run.repo_id, &run.repo, Some(name), true).await?; |
| 1987 | for (key, value) in own { |
| 1988 | if base.get(&key) != Some(&value) { |
| 1989 | passed.insert(key, value); |
| 1990 | } |
| 1991 | } |
| 1992 | } |
| 1993 | passed |
| 1994 | } |
| 1995 | }; |
| 1996 | secrets.insert("G1T_TOKEN".into(), Value::String(token.clone())); |
| 1997 | secrets.insert("GITHUB_TOKEN".into(), Value::String(token.clone())); |
| 1998 | // Each secret as it is, a line at a time, base64 and JSON-escaped. |
| 1999 | let mut masks: Vec<String> = g1t_actions::mask::all_variants(secrets.values().filter_map(|v| v.as_str())); |
| 2000 | let vars = self.variables_for(&run.repo_id, &run.repo, environment.as_deref(), trusted).await?; |
| 2001 | // The toolkit's runtime token (runtime.rs), for as long as the job |
| 2002 | // may run; the API puts it and the toolkit's addresses in the |
| 2003 | // job's variables. |
| 2004 | let runtime_token = crate::runtime::runtime_token( |
| 2005 | &job.id, |
| 2006 | &job.run_id, |
| 2007 | job.token_hash.as_deref().unwrap_or_default(), |
| 2008 | now_ms() / 1000, |
| 2009 | u64::from(job.timeout_minutes) * 60 + 600, |
| 2010 | ); |
| 2011 | masks.push(runtime_token.clone()); |
| 2012 | let retention_days = self.retention_setting(&run.repo_id).await?; |
| 2013 | |
| 2014 | let jobs = self.job_rows(&run.id).await?; |
| 2015 | let mut needs = Map::new(); |
| 2016 | // In a called workflow, its jobs' keys sit under the job that called it. |
| 2017 | let parent = job.call().filter(|c| c["role"] == "callee").and_then(|c| c["parent"].as_str().map(str::to_owned)); |
| 2018 | for need in &spec.needs { |
| 2019 | let key = match &parent { |
| 2020 | Some(parent) => format!("{parent}/{need}"), |
| 2021 | None => need.clone(), |
| 2022 | }; |
| 2023 | let rows: Vec<&JobRow> = jobs.iter().filter(|row| row.key == key).collect(); |
| 2024 | let mut outputs = Map::new(); |
| 2025 | for row in &rows { |
| 2026 | if let Ok(Value::Object(more)) = serde_json::from_str::<Value>(&row.outputs) { |
| 2027 | outputs.extend(more); |
| 2028 | } |
| 2029 | } |
| 2030 | needs.insert(need.clone(), json!({ "result": key_result(&rows), "outputs": outputs })); |
| 2031 | } |
| 2032 | let siblings = jobs.iter().filter(|row| row.key == job.key).count(); |
| 2033 | let matrix: Value = job.matrix.as_deref().and_then(|m| serde_json::from_str(m).ok()).unwrap_or(json!({})); |
| 2034 | let info = run.info(); |
| 2035 | let mut github = info.context(&job.key, &token, run.action.as_deref()); |
| 2036 | github["token"] = json!(token); |
| 2037 | github["retention_days"] = json!(retention_days); |
| 2038 | // On a self-hosted runner, `runner` and `RUNNER_*` describe that |
| 2039 | // machine rather than g1t's sandbox. |
| 2040 | let mut variables = info.variables(&job.key); |
| 2041 | variables.insert("GITHUB_RETENTION_DAYS".into(), json!(retention_days.to_string())); |
| 2042 | let mut runner = match &job.runner_id { |
| 2043 | Some(id) => self.runner_context_for(id, &mut variables).await?, |
| 2044 | None => runner_context(), |
| 2045 | }; |
| 2046 | // A re-run with debug logging: what GitHub sets for one. |
| 2047 | if run.debug != 0 { |
| 2048 | debug_logging(&mut variables, &mut runner); |
| 2049 | } |
| 2050 | |
| 2051 | // Where to check out: a pull request's fork, or the repository. |
| 2052 | let clone_url = match run.pull { |
| 2053 | Some(number) if run.event.starts_with("pull_request") && run.event != "pull_request_target" => { |
| 2054 | let located: Outcome<g1t_contracts::work::PullDetail> = g1t_kit::call( |
| 2055 | &self.work, |
| 2056 | "get_pull", |
| 2057 | &g1t_contracts::work::ViewArgs { |
| 2058 | repo: repo.clone(), |
| 2059 | number, |
| 2060 | viewer: self.workspace_actor(&repo.namespace).await?, |
| 2061 | after_seq: 0, |
| 2062 | }, |
| 2063 | ) |
| 2064 | .await?; |
| 2065 | match located { |
| 2066 | Outcome::Ok(detail) => match detail.pull.fork { |
| 2067 | Some(fork) => format!("{SITE}/{}/{}.git", fork.namespace, fork.name), |
| 2068 | None => format!("{SITE}/{}.git", run.repo), |
| 2069 | }, |
| 2070 | Outcome::Fail(_) => format!("{SITE}/{}.git", run.repo), |
| 2071 | } |
| 2072 | } |
| 2073 | _ => format!("{SITE}/{}.git", run.repo), |
| 2074 | }; |
| 2075 | |
| 2076 | Ok(Outcome::Ok(json!({ |
| 2077 | "job": job.id, |
| 2078 | "run": run.id, |
| 2079 | "key": job.key, |
| 2080 | "name": job.name, |
| 2081 | "spec": spec.raw, |
| 2082 | "workflow": { |
| 2083 | "env": workflow.env, |
| 2084 | "defaults": workflow.raw.get("defaults").cloned().unwrap_or(Value::Null), |
| 2085 | }, |
| 2086 | "github": github, |
| 2087 | "variables": variables, |
| 2088 | "event": info.event, |
| 2089 | "contexts": { |
| 2090 | "vars": vars, |
| 2091 | "secrets": secrets, |
| 2092 | "inputs": call_inputs.unwrap_or_else(|| Value::Object(run.inputs())), |
| 2093 | "matrix": matrix, |
| 2094 | "needs": needs, |
| 2095 | "strategy": { |
| 2096 | "fail-fast": spec.fail_fast, |
| 2097 | "job-index": job.ordinal, |
| 2098 | "job-total": siblings, |
| 2099 | "max-parallel": spec.max_parallel.unwrap_or(siblings as u32), |
| 2100 | }, |
| 2101 | "runner": runner, |
| 2102 | }, |
| 2103 | "checkout": { |
| 2104 | "repository": run.repo, |
| 2105 | "url": clone_url, |
| 2106 | "sha": run.sha, |
| 2107 | "ref": run.git_ref, |
| 2108 | "token": token, |
| 2109 | }, |
| 2110 | "timeoutMinutes": job.timeout_minutes, |
| 2111 | "masks": masks, |
| 2112 | // As the job's log lists them at its start. |
| 2113 | "permissions": permissions.listed().into_iter().map(|(name, access)| (name.to_owned(), json!(access.as_str()))).collect::<Map<String, Value>>(), |
| 2114 | // Whether the job may ask for an OIDC token decides whether it |
| 2115 | // is told where to. |
| 2116 | "runtime": { |
| 2117 | "token": runtime_token, |
| 2118 | "idToken": self.oidc_allowed(&run, &job), |
| 2119 | }, |
| 2120 | }))) |
| 2121 | } |
| 2122 | |
| 2123 | /// `job_report`: the sandbox telling how the job is going. |
| 2124 | pub async fn job_report(&self, a: JobCallArgs) -> Result<Outcome<Value>> { |
| 2125 | let job = check!(self.job_for_token(&a).await?); |
| 2126 | let report = &a.report; |
| 2127 | let at = now(); |
| 2128 | match report["kind"].as_str().unwrap_or_default() { |
| 2129 | "steps" => { |
| 2130 | // The list can grow as the job goes (post steps), so steps |
| 2131 | // already reported keep where they stand. |
| 2132 | let known: Vec<Value> = serde_json::from_str(&job.steps).unwrap_or_default(); |
| 2133 | let steps: Vec<Value> = report["steps"] |
| 2134 | .as_array() |
| 2135 | .map(|names| { |
| 2136 | names |
| 2137 | .iter() |
| 2138 | .enumerate() |
| 2139 | .map(|(i, name)| match known.get(i) { |
| 2140 | Some(step) if step["status"] != "queued" => step.clone(), |
| 2141 | _ => json!({ "number": i + 1, "name": expr::to_text(name), "status": "queued", "conclusion": null, "startedAt": null, "finishedAt": null }), |
| 2142 | }) |
| 2143 | .collect() |
| 2144 | }) |
| 2145 | .unwrap_or_default(); |
| 2146 | self.db |
| 2147 | .prepare("UPDATE jobs SET steps = ?, seen_at = ? WHERE id = ?") |
| 2148 | .bind(&[serde_json::to_string(&steps)?.into(), at.as_str().into(), job.id.as_str().into()])? |
| 2149 | .run() |
| 2150 | .await?; |
| 2151 | } |
| 2152 | "step" => { |
| 2153 | let number = report["number"].as_u64().unwrap_or(0) as usize; |
| 2154 | let mut steps: Vec<Value> = serde_json::from_str(&job.steps).unwrap_or_default(); |
| 2155 | if let Some(step) = number.checked_sub(1).and_then(|i| steps.get_mut(i)) { |
| 2156 | let status = report["status"].as_str().unwrap_or("in_progress"); |
| 2157 | step["status"] = json!(status); |
| 2158 | if status == "in_progress" { |
| 2159 | step["startedAt"] = json!(at); |
| 2160 | } |
| 2161 | if status == "completed" { |
| 2162 | step["finishedAt"] = json!(at); |
| 2163 | step["conclusion"] = report["conclusion"].clone(); |
| 2164 | } |
| 2165 | if let Some(name) = report["name"].as_str() { |
| 2166 | step["name"] = json!(name); |
| 2167 | } |
| 2168 | } |
| 2169 | self.db |
| 2170 | .prepare("UPDATE jobs SET steps = ?, seen_at = ? WHERE id = ?") |
| 2171 | .bind(&[serde_json::to_string(&steps)?.into(), at.as_str().into(), job.id.as_str().into()])? |
| 2172 | .run() |
| 2173 | .await?; |
| 2174 | } |
| 2175 | "log" => { |
| 2176 | let mut text = report["text"].as_str().unwrap_or_default().to_owned(); |
| 2177 | if text.len() > MAX_CHUNK_BYTES { |
| 2178 | let mut cut = MAX_CHUNK_BYTES; |
| 2179 | while !text.is_char_boundary(cut) { |
| 2180 | cut -= 1; |
| 2181 | } |
| 2182 | text.truncate(cut); |
| 2183 | } |
| 2184 | #[derive(Deserialize)] |
| 2185 | struct Size { |
| 2186 | n: Option<f64>, |
| 2187 | seq: Option<f64>, |
| 2188 | } |
| 2189 | let size = self |
| 2190 | .db |
| 2191 | .prepare("SELECT SUM(LENGTH(text)) AS n, MAX(seq) AS seq FROM logs WHERE job_id = ?") |
| 2192 | .bind(&[job.id.as_str().into()])? |
| 2193 | .first::<Size>(None) |
| 2194 | .await?; |
| 2195 | let (used, seq) = size.map_or((0, 0.0), |s| (s.n.unwrap_or(0.0) as usize, s.seq.unwrap_or(0.0))); |
| 2196 | if used < MAX_LOG_BYTES { |
| 2197 | if used + text.len() >= MAX_LOG_BYTES { |
| 2198 | text.push_str("\n… The log reached its limit of 4 MB; the rest is not kept.\n"); |
| 2199 | } |
| 2200 | self.db |
| 2201 | .prepare("INSERT INTO logs (job_id, seq, step, text) VALUES (?, ?, ?, ?)") |
| 2202 | .bind(&[job.id.as_str().into(), // Numbers go to D1 as f64: a u64 would be a BigInt, which it refuses. |
| 2203 | (seq + 1.0).into(), (report["step"].as_u64().unwrap_or(0) as u32).into(), text.into()])? |
| 2204 | .run() |
| 2205 | .await?; |
| 2206 | } |
| 2207 | self.db.prepare("UPDATE jobs SET seen_at = ? WHERE id = ?").bind(&[at.into(), job.id.as_str().into()])?.run().await?; |
| 2208 | } |
| 2209 | "summary" => { |
| 2210 | // `$GITHUB_STEP_SUMMARY`, masked by the runner: added to the |
| 2211 | // step's summary, up to 1 MiB a step and 20 steps a job. |
| 2212 | let step = report["step"].as_u64().unwrap_or(0) as u32; |
| 2213 | let markdown = report["markdown"].as_str().unwrap_or_default(); |
| 2214 | #[derive(Deserialize)] |
| 2215 | struct Held { |
| 2216 | steps: u32, |
| 2217 | mine: Option<f64>, |
| 2218 | } |
| 2219 | let held = self |
| 2220 | .db |
| 2221 | .prepare("SELECT COUNT(*) AS steps, MAX(CASE WHEN step = ? THEN LENGTH(markdown) END) AS mine FROM job_summaries WHERE job_id = ?") |
| 2222 | .bind(&[step.into(), job.id.as_str().into()])? |
| 2223 | .first::<Held>(None) |
| 2224 | .await?; |
| 2225 | let (steps, mine) = held.map_or((0, None), |held| (held.steps, held.mine.map(|n| n as usize))); |
| 2226 | if summary_fits(steps, mine, markdown.len()) { |
| 2227 | self.db |
| 2228 | .prepare( |
| 2229 | "INSERT INTO job_summaries (job_id, step, markdown) VALUES (?, ?, ?) |
| 2230 | ON CONFLICT (job_id, step) DO UPDATE SET markdown = job_summaries.markdown || excluded.markdown", |
| 2231 | ) |
| 2232 | .bind(&[job.id.as_str().into(), step.into(), markdown.into()])? |
| 2233 | .run() |
| 2234 | .await?; |
| 2235 | } |
| 2236 | self.db.prepare("UPDATE jobs SET seen_at = ? WHERE id = ?").bind(&[at.into(), job.id.as_str().into()])?.run().await?; |
| 2237 | } |
| 2238 | "annotation" => { |
| 2239 | let mut annotations: Vec<Value> = serde_json::from_str(&job.annotations).unwrap_or_default(); |
| 2240 | if annotations.len() < MAX_ANNOTATIONS { |
| 2241 | annotations.push(json!({ |
| 2242 | "level": report["level"].as_str().unwrap_or("notice"), |
| 2243 | "message": report["message"].as_str().unwrap_or_default().chars().take(4000).collect::<String>(), |
| 2244 | "title": report["title"], |
| 2245 | "file": report["file"], |
| 2246 | "line": report["line"], |
| 2247 | })); |
| 2248 | self.db |
| 2249 | .prepare("UPDATE jobs SET annotations = ?, seen_at = ? WHERE id = ?") |
| 2250 | .bind(&[serde_json::to_string(&annotations)?.into(), at.as_str().into(), job.id.as_str().into()])? |
| 2251 | .run() |
| 2252 | .await?; |
| 2253 | } |
| 2254 | } |
| 2255 | // Nothing to tell; the answer says whether to stop. |
| 2256 | "ping" => { |
| 2257 | self.db.prepare("UPDATE jobs SET seen_at = ? WHERE id = ?").bind(&[at.into(), job.id.as_str().into()])?.run().await?; |
| 2258 | } |
| 2259 | "done" => { |
| 2260 | // A job told to stop ends cancelled, however its cleanup went. |
| 2261 | let conclusion = if job.cancel_requested_at.is_some() { |
| 2262 | "cancelled" |
| 2263 | } else { |
| 2264 | report["conclusion"] |
| 2265 | .as_str() |
| 2266 | .filter(|c| matches!(*c, "success" | "failure" | "cancelled")) |
| 2267 | .unwrap_or("failure") |
| 2268 | }; |
| 2269 | let outputs = report["outputs"].as_object().cloned(); |
| 2270 | Box::pin(self.finish_job(&job.id, conclusion, report["reason"].as_str(), outputs.as_ref())).await?; |
| 2271 | } |
| 2272 | other => return Ok(fail(FailureCode::Invalid, format!("There is no report called `{other}`."))), |
| 2273 | } |
| 2274 | // `cancelled`: the run was cancelled, and the runner should stop |
| 2275 | // the step it is on and run only its cleanup steps. |
| 2276 | Ok(Outcome::Ok(json!({ "ok": true, "cancelled": job.cancel_requested_at.is_some() }))) |
| 2277 | } |
| 2278 | |
| 2279 | // --- Every minute --------------------------------------------------------------- |
| 2280 | |
| 2281 | pub async fn on_minute(&self, now_ms: u64) -> Result<()> { |
| 2282 | let minute = now_ms / 60_000 * 60_000; |
| 2283 | if let Err(error) = self.run_schedules(minute).await { |
| 2284 | worker::console_error!("actions: schedules failed: {error}"); |
| 2285 | } |
| 2286 | // Jobs whose sandbox went quiet or ran past their time. |
| 2287 | let running = self.db.prepare("SELECT * FROM jobs WHERE status = 'in_progress'").all().await?.results::<JobRow>()?; |
| 2288 | for job in running { |
| 2289 | // Times in g1t's format compare as text. |
| 2290 | let before = |ms: u64| rfc3339(now_ms.saturating_sub(ms)); |
| 2291 | let silent = job.seen_at.as_deref().is_some_and(|seen| seen < before(SILENT_MS).as_str()); |
| 2292 | let limit = (u64::from(job.timeout_minutes) * 60 + 120) * 1000; |
| 2293 | let over = job.started_at.as_deref().is_some_and(|started| started < before(limit).as_str()); |
| 2294 | let cancel_overdue = job.cancel_requested_at.as_deref().is_some_and(|at| at < before(CANCEL_GRACE_MS).as_str()); |
| 2295 | if cancel_overdue { |
| 2296 | // Cancelled, and still going after its grace period. |
| 2297 | self.hard_stop(&job, "It was cancelled, and did not finish its cleanup steps within 5 minutes.").await?; |
| 2298 | self.advance(&job.run_id).await?; |
| 2299 | } else if over { |
| 2300 | let reason = format!("It ran longer than its time limit of {} minutes.", job.timeout_minutes); |
| 2301 | // A self-hosted runner is told to stop on its next poll. |
| 2302 | if job.runner_id.is_none() { |
| 2303 | let _: Result<Value> = g1t_kit::call(&self.runner, "stop_actions_job", &json!({ "job": job.id })).await; |
| 2304 | } |
| 2305 | self.finish_job(&job.id, "failure", Some(&reason), None).await?; |
| 2306 | } else if silent { |
| 2307 | let reason = match &job.runner_name { |
| 2308 | Some(name) => format!("The self-hosted runner {name} stopped answering."), |
| 2309 | None => "The runner stopped answering.".to_owned(), |
| 2310 | }; |
| 2311 | self.finish_job(&job.id, "failure", Some(&reason), None).await?; |
| 2312 | } |
| 2313 | } |
| 2314 | if let Err(error) = self.sweep_runners(now_ms).await { |
| 2315 | worker::console_error!("actions: the runners' sweep failed: {error}"); |
| 2316 | } |
| 2317 | // Jobs held at an environment whose wait timer has run out. |
| 2318 | if let Err(error) = self.release_gates().await { |
| 2319 | worker::console_error!("actions: environments' gates failed: {error}"); |
| 2320 | } |
| 2321 | // Once an hour: the cache's expired entries, and its storage. |
| 2322 | if (now_ms / 60_000) % 60 == 7 |
| 2323 | && let Err(error) = self.sweep_cache(now_ms).await |
| 2324 | { |
| 2325 | worker::console_error!("actions: the cache's sweep failed: {error}"); |
| 2326 | } |
| 2327 | // And artifacts past their time, and the toolkit's abandoned parts. |
| 2328 | if (now_ms / 60_000) % 60 == 37 { |
| 2329 | if let Err(error) = self.sweep_artifacts(now_ms).await { |
| 2330 | worker::console_error!("actions: the artifacts' sweep failed: {error}"); |
| 2331 | } |
| 2332 | if let Err(error) = self.sweep_blob_parts(now_ms).await { |
| 2333 | worker::console_error!("actions: the blob parts' sweep failed: {error}"); |
| 2334 | } |
| 2335 | } |
| 2336 | self.start_queued().await |
| 2337 | } |
| 2338 | } |
| 2339 | |
| 2340 | |
| 2341 | #[cfg(test)] |
| 2342 | mod stopping { |
| 2343 | use super::stops_runs; |
| 2344 | use g1t_contracts::events::Event; |
| 2345 | use serde_json::{Value, json}; |
| 2346 | |
| 2347 | fn event(kind: &str, data: Value) -> Event { |
| 2348 | Event { |
| 2349 | id: "evt_1".into(), |
| 2350 | kind: kind.into(), |
| 2351 | source: "repos".into(), |
| 2352 | time: "2026-10-05T00:00:00Z".into(), |
| 2353 | repo_id: Some("rep_1".into()), |
| 2354 | actor: None, |
| 2355 | data, |
| 2356 | } |
| 2357 | } |
| 2358 | |
| 2359 | #[test] |
| 2360 | fn deleting_or_archiving_stops_runs() { |
| 2361 | assert_eq!(stops_runs(&event("repo.deleted", json!({ "repoId": "rep_1" }))).as_deref(), Some("rep_1")); |
| 2362 | assert_eq!(stops_runs(&event("repo.archived", json!({ "archived": true }))).as_deref(), Some("rep_1")); |
| 2363 | assert_eq!(stops_runs(&event("repo.unarchived", json!({ "archived": false }))), None); |
| 2364 | assert_eq!(stops_runs(&event("repo.restored", json!({}))), None); |
| 2365 | assert_eq!(stops_runs(&event("git.push", json!({}))), None); |
| 2366 | } |
| 2367 | } |
| 2368 | |
| 2369 | #[cfg(test)] |
| 2370 | mod status_of_needs { |
| 2371 | use std::collections::HashMap; |
| 2372 | |
| 2373 | use super::ancestor_failed; |
| 2374 | |
| 2375 | /// check -> plan -> (migrate) -> core -> edge, as deploy.yml has them, |
| 2376 | /// and a job that needs only the last. |
| 2377 | fn graph() -> HashMap<&'static str, Vec<&'static str>> { |
| 2378 | HashMap::from([ |
| 2379 | ("check", vec![]), |
| 2380 | ("plan", vec!["check"]), |
| 2381 | ("migrate", vec!["plan"]), |
| 2382 | ("core", vec!["plan", "migrate"]), |
| 2383 | ("edge", vec!["plan", "migrate", "core"]), |
| 2384 | ("notify", vec!["edge"]), |
| 2385 | ]) |
| 2386 | } |
| 2387 | |
| 2388 | #[test] |
| 2389 | fn a_failure_is_seen_however_far_back() { |
| 2390 | let needs = graph(); |
| 2391 | let failed = |which: &'static str| move |key: &str| key == which; |
| 2392 | // check failed; plan, and everything after, was skipped for it. |
| 2393 | assert!(ancestor_failed(&needs, "notify", failed("check"))); |
| 2394 | assert!(ancestor_failed(&needs, "core", failed("check"))); |
| 2395 | assert!(ancestor_failed(&needs, "edge", failed("core"))); |
| 2396 | // Nothing before a job failed: a skipped migrate is not a failure. |
| 2397 | assert!(!ancestor_failed(&needs, "edge", |_| false)); |
| 2398 | assert!(!ancestor_failed(&needs, "core", failed("edge"))); |
| 2399 | assert!(!ancestor_failed(&needs, "check", failed("check"))); |
| 2400 | } |
| 2401 | |
| 2402 | #[test] |
| 2403 | fn cycles_and_unknown_keys_end() { |
| 2404 | let needs = HashMap::from([("a", vec!["b"]), ("b", vec!["a"])]); |
| 2405 | assert!(!ancestor_failed(&needs, "a", |_| false)); |
| 2406 | assert!(!ancestor_failed(&needs, "missing", |_| true)); |
| 2407 | } |
| 2408 | } |
| 2409 | |
| 2410 | #[cfg(test)] |
| 2411 | mod deployments { |
| 2412 | use serde_json::{Map, Value, json}; |
| 2413 | |
| 2414 | use super::{JobEnvironment, deployment_outcome, environment_of}; |
| 2415 | |
| 2416 | #[test] |
| 2417 | fn a_jobs_environment_is_read_for_deployments() { |
| 2418 | let contexts: Map<String, Value> = serde_json::from_value(json!({ |
| 2419 | "github": { "ref_name": "main", "repository": "acme/web" }, |
| 2420 | "inputs": { "target": "staging" }, |
| 2421 | "matrix": {}, |
| 2422 | })) |
| 2423 | .unwrap(); |
| 2424 | let read = |raw: Value| environment_of(&raw, &contexts); |
| 2425 | assert_eq!(read(json!({})), None); |
| 2426 | assert_eq!( |
| 2427 | read(json!({ "environment": "production" })), |
| 2428 | Some(JobEnvironment { name: "production".into(), url: None, deploys: true }) |
| 2429 | ); |
| 2430 | assert_eq!( |
| 2431 | read(json!({ "environment": { "name": "production", "url": "https://g1t.sh" } })), |
| 2432 | Some(JobEnvironment { name: "production".into(), url: Some("https://g1t.sh".into()), deploys: true }) |
| 2433 | ); |
| 2434 | // Expressions are filled in from the run. |
| 2435 | assert_eq!( |
| 2436 | read(json!({ "environment": { "name": "${{ inputs.target }}", "url": "https://${{ github.ref_name }}.example.com" } })), |
| 2437 | Some(JobEnvironment { name: "staging".into(), url: Some("https://main.example.com".into()), deploys: true }) |
| 2438 | ); |
| 2439 | // Secrets only: no deployment. |
| 2440 | assert!(!read(json!({ "environment": { "name": "production", "deployment": false } })).unwrap().deploys); |
| 2441 | // Only http(s) addresses. |
| 2442 | assert_eq!(read(json!({ "environment": { "name": "production", "url": "javascript:alert(1)" } })).unwrap().url, None); |
| 2443 | } |
| 2444 | |
| 2445 | #[test] |
| 2446 | fn a_runs_outcome_for_an_environment() { |
| 2447 | let of = |list: &[&str]| deployment_outcome(&list.iter().map(|c| Some((*c).to_owned())).collect::<Vec<_>>()); |
| 2448 | assert_eq!(of(&["success", "skipped"]), Some("success")); |
| 2449 | assert_eq!(of(&["success", "failure"]), Some("failure")); |
| 2450 | assert_eq!(of(&["success", "cancelled"]), Some("error")); |
| 2451 | assert_eq!(of(&["skipped"]), None); |
| 2452 | assert_eq!(deployment_outcome(&[None]), None); |
| 2453 | } |
| 2454 | } |
| 2455 | |
| 2456 | #[cfg(test)] |
| 2457 | mod reruns { |
| 2458 | use serde_json::{Map, Value, json}; |
| 2459 | |
| 2460 | use super::{MAX_SUMMARIES, MAX_SUMMARY_BYTES, Rerun, debug_logging, rerun_keys, summary_fits, top_key}; |
| 2461 | |
| 2462 | /// build ← test ← deploy, and lint on its own. |
| 2463 | fn keys(which: Rerun, failed: &[&str]) -> Vec<String> { |
| 2464 | let order = ["build", "lint", "test", "deploy"]; |
| 2465 | let needs = |key: &str| -> Vec<String> { |
| 2466 | match key { |
| 2467 | "test" => vec!["build".into()], |
| 2468 | "deploy" => vec!["test".into()], |
| 2469 | _ => Vec::new(), |
| 2470 | } |
| 2471 | }; |
| 2472 | rerun_keys(&order, needs, |key| !failed.contains(&key), which) |
| 2473 | } |
| 2474 | |
| 2475 | #[test] |
| 2476 | fn a_rerun_takes_the_jobs_it_names_and_those_that_need_them() { |
| 2477 | assert_eq!(keys(Rerun::All, &[]), ["build", "lint", "test", "deploy"]); |
| 2478 | assert_eq!(keys(Rerun::Failed, &["test"]), ["test", "deploy"]); |
| 2479 | assert_eq!(keys(Rerun::Failed, &["lint"]), ["lint"]); |
| 2480 | assert!(keys(Rerun::Failed, &[]).is_empty()); |
| 2481 | // One job, whatever it came to, and what depends on it. |
| 2482 | assert_eq!(keys(Rerun::Job("build"), &[]), ["build", "test", "deploy"]); |
| 2483 | assert_eq!(keys(Rerun::Job("lint"), &["test"]), ["lint"]); |
| 2484 | assert_eq!(keys(Rerun::Job("deploy"), &[]), ["deploy"]); |
| 2485 | assert!(keys(Rerun::Job("missing"), &[]).is_empty()); |
| 2486 | } |
| 2487 | |
| 2488 | #[test] |
| 2489 | fn a_called_workflows_jobs_run_again_with_the_job_that_calls_it() { |
| 2490 | assert_eq!(top_key("build/test"), "build"); |
| 2491 | assert_eq!(top_key("build/inner/test"), "build"); |
| 2492 | assert_eq!(top_key("lint"), "lint"); |
| 2493 | } |
| 2494 | |
| 2495 | #[test] |
| 2496 | fn a_job_keeps_twenty_steps_summaries_of_a_mebibyte_each() { |
| 2497 | assert!(summary_fits(0, None, 10)); |
| 2498 | assert!(!summary_fits(0, None, 0), "nothing to keep"); |
| 2499 | assert!(!summary_fits(MAX_SUMMARIES, None, 10), "a twenty-first step's summary is dropped"); |
| 2500 | assert!(summary_fits(MAX_SUMMARIES, Some(10), 10), "a step that has one may add to it"); |
| 2501 | assert!(summary_fits(1, Some(MAX_SUMMARY_BYTES - 10), 10)); |
| 2502 | assert!(!summary_fits(1, Some(MAX_SUMMARY_BYTES - 10), 11)); |
| 2503 | assert!(!summary_fits(0, None, MAX_SUMMARY_BYTES + 1)); |
| 2504 | } |
| 2505 | |
| 2506 | #[test] |
| 2507 | fn a_debug_rerun_sets_what_github_sets() { |
| 2508 | let mut variables = Map::new(); |
| 2509 | let mut runner = json!({ "name": "g1t", "debug": "" }); |
| 2510 | debug_logging(&mut variables, &mut runner); |
| 2511 | assert_eq!(variables["RUNNER_DEBUG"], "1"); |
| 2512 | assert_eq!(variables["ACTIONS_STEP_DEBUG"], "true"); |
| 2513 | assert_eq!(variables["ACTIONS_RUNNER_DEBUG"], "true"); |
| 2514 | assert_eq!(runner["debug"], Value::String("1".into())); |
| 2515 | assert_eq!(runner["name"], "g1t"); |
| 2516 | } |
| 2517 | } |