Skip to content

g1t/crates/runner/src/docker/engine.rs

414 lines19,240 bytesCodeBlame

Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.

Merge branch 'main' into actions-toolkit-oidc-artifacts1//! The job's own Docker Engine: set up when the job starts (a socket, and
2//! nothing running), and started the first time the socket is used, or a
3//! job's `services:` or `container:` need it.
4//!
5//! What Cloudflare Containers allow, and so how it is started: `dockerd`
6//! as root (a rootless Engine does not start there), with
7//! `--iptables=false --ip6tables=false --ip-forward=false`, since a sandbox
8//! may not change its packet filter or forward packets. Containers on a
9//! bridge network therefore have no way out, which is why the API proxy
10//! puts them on the job's network (api.rs). The Engine's data goes on the
11//! sandbox's disk with the overlay filesystem when the disk takes it, and
12//! with plain copies (containerd's `native` snapshotter) when it does not. Docker Hub's images come
13//! through Google's public mirror of it first, so jobs from many machines
14//! sharing addresses do not run into Docker Hub's anonymous limits.
15
16use std::path::Path;
17
18use serde_json::{Value, json};
19
20/// Everything of the Engine's that is not its data.
21pub(crate) const DIR: &str = "/run/g1t-docker";
22/// What the job's steps reach: the API proxy.
23pub(crate) const PROXY_SOCKET: &str = "/run/g1t-docker/docker.sock";
24/// The Engine itself.
25pub(crate) const ENGINE_SOCKET: &str = "/run/g1t-docker/engine.sock";
26/// Where the job's steps look for Docker.
27pub(crate) const DOCKER_SOCKET: &str = "/var/run/docker.sock";
28pub(crate) const MIRROR: &str = "https://mirror.gcr.io";
29
30/// What the job tells the Engine when it sets it up.
31#[derive(Clone, Default)]
32pub(crate) struct Options {
33 /// g1t's container registry (`g1t.sh`) and the run's token, to be
34 /// signed in to from the start. None for a run without secrets.
35 pub(crate) registry: Option<(String, String)>,
36}
37
38/// The Engine's `daemon.json`. `group`: the group whose members may use
39/// its socket (the job's user). `copies`: the disk does not take overlays.
40pub(crate) fn daemon_config(group: &str, copies: bool) -> Value {
41 let mut config = json!({
42 "hosts": [format!("unix://{ENGINE_SOCKET}")],
43 "group": group,
44 "pidfile": format!("{DIR}/dockerd.pid"),
45 "iptables": false,
46 "ip6tables": false,
47 "ip-forward": false,
48 "registry-mirrors": [MIRROR],
49 "log-driver": "json-file",
50 "log-opts": { "max-size": "20m", "max-file": "2" },
51 });
52 if copies {
53 // The containerd image store's name for plain copies (vfs).
54 config["storage-driver"] = json!("native");
55 }
56 config
57}
58
59/// `~/.docker/config.json`, signed in to `registry` with `token` unless it
60/// is already signed in there.
61pub(crate) fn with_login(mut config: Value, registry: &str, token: &str) -> Option<Value> {
62 use base64::Engine;
63 if !config.is_object() {
64 config = json!({});
65 }
66 let auths = config.as_object_mut()?.entry("auths").or_insert_with(|| json!({}));
67 let auths = auths.as_object_mut()?;
68 if auths.contains_key(registry) {
69 return None;
70 }
71 let auth = base64::engine::general_purpose::STANDARD.encode(format!("g1t:{token}"));
72 auths.insert(registry.to_owned(), json!({ "auth": auth }));
73 Some(config)
74}
75
76/// The host of a server URL: `https://g1t.sh` is `g1t.sh`.
77pub(crate) fn registry_host(server_url: &str) -> Option<String> {
78 let rest = server_url.split_once("://").map_or(server_url, |(_, rest)| rest);
79 let host = rest.split('/').next().unwrap_or_default();
80 (!host.is_empty()).then(|| host.to_ascii_lowercase())
81}
82
83/// The last lines of a file, for a message.
84fn tail(path: &Path, lines: usize) -> String {
85 let text = std::fs::read_to_string(path).unwrap_or_default();
86 let all: Vec<&str> = text.lines().filter(|l| !l.trim().is_empty()).collect();
87 all[all.len().saturating_sub(lines)..].join("\n")
88}
89
90#[cfg(target_os = "linux")]
91pub(crate) use linux::{enable, ensure_started};
92
93#[cfg(not(target_os = "linux"))]
94pub(crate) fn enable(_options: Options) -> Result<(), String> {
95 Err("Docker runs in jobs on g1t's own Linux machines.".into())
96}
97
98#[cfg(not(target_os = "linux"))]
99pub(crate) fn ensure_started() -> Result<String, String> {
100 Err("Docker runs in jobs on g1t's own Linux machines.".into())
101}
102
103#[cfg(target_os = "linux")]
104mod linux {
105 use std::collections::BTreeSet;
106 use std::io::{self, Write};
107 use std::net::{TcpListener, TcpStream};
108 use std::os::unix::fs::{MetadataExt, PermissionsExt};
109 use std::os::unix::net::{UnixListener, UnixStream};
110 use std::path::Path;
111 use std::process::{Child, Command, Stdio};
112 use std::sync::{Arc, Mutex, OnceLock};
113 use std::time::{Duration, Instant};
114
115 use serde_json::Value;
116
117 use super::super::api::{self, Duplex, Host, State};
118 use super::{DIR, DOCKER_SOCKET, ENGINE_SOCKET, Options, PROXY_SOCKET, daemon_config, tail, with_login};
119
120 const LOG: &str = "/run/g1t-docker/dockerd.log";
121 /// From moby's `hack/dind`: the sandbox's processes into a group of
122 /// their own, then every controller enabled for the groups below.
123 const CGROUP_NESTING: &str = "if [ -f /sys/fs/cgroup/cgroup.controllers ] && [ -z \"$(cat /sys/fs/cgroup/cgroup.subtree_control)\" ]; then \
124 mkdir -p /sys/fs/cgroup/init && xargs -rn1 < /sys/fs/cgroup/cgroup.procs > /sys/fs/cgroup/init/cgroup.procs 2>/dev/null; \
125 sed -e 's/ / +/g' -e 's/^/+/' < /sys/fs/cgroup/cgroup.controllers > /sys/fs/cgroup/cgroup.subtree_control; fi";
126 const CERTS: &str = "/run/g1t-docker/certs";
127 const BIN: &str = "/run/g1t-docker/bin";
128
129 /// The options the job set up with, once set up.
130 static OPTIONS: OnceLock<Options> = OnceLock::new();
131 /// How starting went: the Engine's version, or why not. Held while
132 /// starting, so everything that needs the Engine waits for it.
133 static STARTED: Mutex<Option<Result<String, String>>> = Mutex::new(None);
134 /// Host ports forwarded so far.
135 static FORWARDED: Mutex<BTreeSet<u16>> = Mutex::new(BTreeSet::new());
136
137 fn sudo(args: &[&str]) -> bool {
138 Command::new("sudo").arg("-n").args(args).stdin(Stdio::null()).stdout(Stdio::null()).stderr(Stdio::null()).status().is_ok_and(|s| s.success())
139 }
140
141 fn sudo_with_input(args: &[&str], input: &str) -> bool {
142 let Ok(mut child) = Command::new("sudo").arg("-n").args(args).stdin(Stdio::piped()).stdout(Stdio::null()).stderr(Stdio::null()).spawn() else {
143 return false;
144 };
145 if let Some(mut stdin) = child.stdin.take() {
146 let _ = stdin.write_all(input.as_bytes());
147 }
148 child.wait().is_ok_and(|s| s.success())
149 }
150
151 fn ids() -> (u32, u32) {
152 std::fs::metadata("/proc/self").map(|m| (m.uid(), m.gid())).unwrap_or((1000, 1000))
153 }
154
155 fn group_name() -> String {
156 Command::new("id").arg("-gn").output().ok().map(|o| String::from_utf8_lossy(&o.stdout).trim().to_owned()).filter(|g| !g.is_empty()).unwrap_or_else(|| "node".into())
157 }
158
159 /// Sets the socket up, with no Engine behind it yet.
160 pub(crate) fn enable(options: Options) -> Result<(), String> {
161 if !Path::new("/usr/bin/dockerd").exists() {
162 return Err("this sandbox's image has no Docker Engine".into());
163 }
164 let (uid, gid) = ids();
165 if !sudo(&["install", "-d", "-m", "0755", "-o", &uid.to_string(), "-g", &gid.to_string(), DIR]) {
166 return Err(format!("could not make {DIR}"));
167 }
168 let _ = std::fs::remove_file(PROXY_SOCKET);
169 let listener = UnixListener::bind(PROXY_SOCKET).map_err(|e| format!("could not listen on {PROXY_SOCKET}: {e}"))?;
170 // Every process in the sandbox is the job's, root or not.
171 let _ = std::fs::set_permissions(PROXY_SOCKET, std::fs::Permissions::from_mode(0o666));
172 if !sudo(&["ln", "-sfn", PROXY_SOCKET, DOCKER_SOCKET]) {
173 return Err(format!("could not link {DOCKER_SOCKET}"));
174 }
175 let _ = OPTIONS.set(options);
176 let host: Arc<dyn Host> = Arc::new(Sandbox);
177 let state = Arc::new(Mutex::new(State::default()));
178 std::thread::spawn(move || {
179 for client in listener.incoming().flatten() {
180 let (host, state) = (host.clone(), state.clone());
181 std::thread::spawn(move || api::serve(Box::new(client), host, state));
182 }
183 });
184 Ok(())
185 }
186
187 /// Starts the Engine, once; its version, or why it could not start.
188 pub(crate) fn ensure_started() -> Result<String, String> {
189 let mut started = STARTED.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
190 if let Some(result) = &*started {
191 return result.clone();
192 }
193 let begun = Instant::now();
194 let result = start();
195 match &result {
196 Ok(version) => super::super::note(format!(
197 "Docker: started this job's own Docker Engine {version} in {:.1}s. Its containers run in this job's sandbox, on the job's network and under its guardrails, and end with the job.",
198 begun.elapsed().as_secs_f64()
199 )),
200 Err(problem) => super::super::note(format!("##[error]Docker could not start: {problem}")),
201 }
202 *started = Some(result.clone());
203 result
204 }
205
206 /// Whether the overlay filesystem works where the Engine keeps its data.
207 fn overlay_works() -> bool {
208 let script = "d=/var/lib/docker/.g1t-probe; rm -rf $d; mkdir -p $d/l $d/u $d/w $d/m && echo x > $d/l/f && mount -t overlay overlay -o lowerdir=$d/l,upperdir=$d/u,workdir=$d/w $d/m && umount $d/m; s=$?; rm -rf $d; exit $s";
209 sudo(&["sh", "-c", script])
210 }
211
212 fn start() -> Result<String, String> {
213 // This program, as the runc the Engine finds first (oci.rs).
214 let exe = std::env::current_exe().map_err(|e| e.to_string())?;
215 std::fs::create_dir_all(BIN).map_err(|e| format!("could not make {BIN}: {e}"))?;
216 let shim = Path::new(BIN).join("runc");
217 let _ = std::fs::remove_file(&shim);
218 std::os::unix::fs::symlink(&exe, &shim).map_err(|e| format!("could not link runc: {e}"))?;
219
220 // A guarded job's certificate, for its containers.
221 let mut ca_dir = None;
222 if let Ok(ca) = std::env::var("G1T_EGRESS_CA")
223 && Path::new(&ca).exists()
224 {
225 let _ = std::fs::create_dir_all(CERTS);
226 let copied = std::fs::copy(&ca, Path::new(CERTS).join(super::super::oci::EGRESS)).is_ok()
227 && std::fs::copy("/etc/ssl/certs/ca-certificates.crt", Path::new(CERTS).join(super::super::oci::BUNDLE)).is_ok();
228 if copied {
229 ca_dir = Some(CERTS.to_owned());
230 }
231 }
232 // `-p 80:8080` is forwarded by this process, which is not root.
233 let _ = sudo(&["sysctl", "-q", "-w", "net.ipv4.ip_unprivileged_port_start=0"]);
234 // Containers' resource limits (`--cpus`, `--memory`) need the
235 // cgroup controllers handed down, which cgroup v2 allows only from
236 // a group with no processes of its own: move ours aside first, as
237 // the Engine's own Docker-in-Docker image does.
238 let _ = sudo(&["sh", "-c", CGROUP_NESTING]);
239
240 let mut vfs = !overlay_works();
241 let group = group_name();
242 let mut last = String::new();
243 for _ in 0..2 {
244 let config = daemon_config(&group, vfs);
245 std::fs::write(format!("{DIR}/daemon.json"), serde_json::to_vec_pretty(&config).unwrap_or_default())
246 .map_err(|e| format!("could not write daemon.json: {e}"))?;
247 let mut child = spawn(ca_dir.as_deref())?;
248 match wait_ready(&mut child, Duration::from_secs(90)) {
249 Ok(()) => {
250 let version = engine_version().unwrap_or_else(|| "?".into());
251 sign_in();
252 return Ok(if vfs { format!("{version} (plain-copy storage: this disk takes no overlays, so images take more room)") } else { version });
253 }
254 Err(problem) => {
255 last = problem;
256 let _ = child.kill();
257 let _ = sudo(&["pkill", "-x", "dockerd"]);
258 let _ = sudo(&["pkill", "-x", "containerd"]);
259 if vfs {
260 break;
261 }
262 // Overlays that mount but do not work for the Engine.
263 vfs = true;
264 }
265 }
266 }
267 Err(last)
268 }
269
270 fn spawn(ca_dir: Option<&str>) -> Result<Child, String> {
271 let log = std::fs::File::create(LOG).map_err(|e| format!("could not write {LOG}: {e}"))?;
272 let err = log.try_clone().map_err(|e| e.to_string())?;
273 let path = format!("{BIN}:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin");
274 let mut command = Command::new("sudo");
275 command.args(["-n", "env", &format!("PATH={path}"), "G1T_REAL_RUNC=/usr/bin/runc"]);
276 if let Some(dir) = ca_dir {
277 command.arg(format!("G1T_DOCKER_CA_DIR={dir}"));
278 }
279 command.args(["dockerd", "--config-file", &format!("{DIR}/daemon.json")]);
280 command.stdin(Stdio::null()).stdout(log).stderr(err);
281 command.spawn().map_err(|e| format!("could not run dockerd: {e}"))
282 }
283
284 /// A request to the Engine; its status and body.
285 fn ask(method: &str, path: &str) -> io::Result<(u16, Vec<u8>)> {
286 let mut stream = UnixStream::connect(ENGINE_SOCKET)?;
287 stream.set_read_timeout(Some(Duration::from_secs(10)))?;
288 stream.write_all(format!("{method} {path} HTTP/1.1\r\nHost: docker\r\nConnection: close\r\n\r\n").as_bytes())?;
289 let mut reader = io::BufReader::new(stream);
290 let head = super::super::http::read_head(&mut reader)?.ok_or_else(|| io::Error::other("no answer"))?;
291 let body = super::super::http::read_body(&mut reader, super::super::http::response_body(&head, method))?;
292 Ok((head.status(), body))
293 }
294
295 fn wait_ready(child: &mut Child, limit: Duration) -> Result<(), String> {
296 let until = Instant::now() + limit;
297 loop {
298 if matches!(ask("GET", "/_ping"), Ok((200, _))) {
299 return Ok(());
300 }
301 if let Ok(Some(status)) = child.try_wait() {
302 return Err(format!("dockerd stopped ({status}):\n{}", tail(Path::new(LOG), 15)));
303 }
304 if Instant::now() >= until {
305 return Err(format!("dockerd did not answer in {} s:\n{}", limit.as_secs(), tail(Path::new(LOG), 15)));
306 }
307 std::thread::sleep(Duration::from_millis(100));
308 }
309 }
310
311 fn engine_version() -> Option<String> {
312 let (_, body) = ask("GET", "/version").ok()?;
313 let value: Value = serde_json::from_slice(&body).ok()?;
314 value.get("Version").and_then(Value::as_str).map(str::to_owned)
315 }
316
317 /// Signs the job in to g1t's registry with the run's own token.
318 fn sign_in() {
319 let Some((registry, token)) = OPTIONS.get().and_then(|o| o.registry.clone()) else { return };
320 let home = std::env::var("HOME").unwrap_or_else(|_| "/home/node".into());
321 let path = Path::new(&home).join(".docker").join("config.json");
322 let current: Value = std::fs::read(&path).ok().and_then(|t| serde_json::from_slice(&t).ok()).unwrap_or(Value::Null);
323 if let Some(config) = with_login(current, &registry, &token) {
324 let _ = std::fs::create_dir_all(path.parent().expect("a folder"));
325 if std::fs::write(&path, serde_json::to_vec_pretty(&config).unwrap_or_default()).is_ok() {
326 let _ = std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o600));
327 super::super::note(format!("Docker: signed in to {registry} with this run's token."));
328 }
329 }
330 }
331
332 /// The sandbox, as the API proxy sees it.
333 struct Sandbox;
334
335 impl Host for Sandbox {
336 fn add_hosts(&self, names: &[String]) {
337 let lines: String = names.iter().map(|name| format!("127.0.0.1\t{name}\n")).collect();
338 if !sudo_with_input(&["sh", "-c", "cat >> /etc/hosts"], &lines) {
339 super::super::note(format!("Docker: could not add {} to /etc/hosts.", names.join(", ")));
340 }
341 }
342
343 fn forward(&self, host_port: u16, container_port: u16) {
344 if !FORWARDED.lock().is_ok_and(|mut set| set.insert(host_port)) {
345 return;
346 }
347 let listener = match TcpListener::bind(("0.0.0.0", host_port)) {
348 Ok(listener) => listener,
349 Err(error) => {
350 super::super::note(format!("##[warning]Docker: port {host_port} could not be published for port {container_port}: {error}"));
351 return;
352 }
353 };
354 std::thread::spawn(move || {
355 for client in listener.incoming().flatten() {
356 std::thread::spawn(move || {
357 let Ok(target) = TcpStream::connect(("127.0.0.1", container_port)) else { return };
358 pipe(client, target);
359 });
360 }
361 });
362 }
363
364 fn ensure_engine(&self) -> Result<(), String> {
365 ensure_started().map(|_| ())
366 }
367
368 fn connect(&self) -> io::Result<Box<dyn Duplex>> {
369 Ok(Box::new(UnixStream::connect(ENGINE_SOCKET)?))
370 }
371 }
372
373 /// Copies two connections into each other until both are done.
374 fn pipe(a: TcpStream, b: TcpStream) {
375 let (Ok(mut a_read), Ok(mut b_read)) = (a.try_clone(), b.try_clone()) else { return };
376 let (mut a_write, mut b_write) = (a, b);
377 let back = std::thread::spawn(move || {
378 let _ = io::copy(&mut b_read, &mut a_write);
379 let _ = a_write.shutdown(std::net::Shutdown::Write);
380 });
381 let _ = io::copy(&mut a_read, &mut b_write);
382 let _ = b_write.shutdown(std::net::Shutdown::Write);
383 let _ = back.join();
384 }
385}
386
387#[cfg(test)]
388mod tests {
389 use super::*;
390
391 #[test]
392 fn the_engine_runs_as_a_sandbox_allows() {
393 let config = daemon_config("node", false);
394 assert_eq!(config["iptables"], false);
395 assert_eq!(config["ip6tables"], false);
396 assert_eq!(config["ip-forward"], false);
397 assert_eq!(config["group"], "node");
398 assert_eq!(config["hosts"][0], "unix:///run/g1t-docker/engine.sock");
399 assert_eq!(config["registry-mirrors"][0], MIRROR);
400 assert!(config.get("storage-driver").is_none());
401 assert_eq!(daemon_config("node", true)["storage-driver"], "native");
402 }
403
404 #[test]
405 fn the_run_signs_in_to_g1t_unless_already_signed_in() {
406 let config = with_login(json!({ "auths": { "ghcr.io": { "auth": "x" } } }), "g1t.sh", "tok").unwrap();
407 assert_eq!(config["auths"]["g1t.sh"]["auth"], "ZzF0OnRvaw==");
408 assert_eq!(config["auths"]["ghcr.io"]["auth"], "x");
409 assert!(with_login(config, "g1t.sh", "other").is_none());
410 assert!(with_login(Value::Null, "g1t.sh", "tok").is_some());
411 assert_eq!(registry_host("https://g1t.sh").as_deref(), Some("g1t.sh"));
412 assert_eq!(registry_host("http://localhost:8787/").as_deref(), Some("localhost:8787"));
413 }
414}

This file's history is long; its oldest lines are credited to the oldest commit read.