zuka
zuka/src/ci/runner.rs

worklyn / zukapublic

Agent-first git hosting. One Rust binary: git over HTTP and SSH, a REST API, MCP, CI, and multi-tenant isolation.

Get a copy: git clone https://zuka.worklyn.com/worklyn/zuka.git
zuka/src/ci/runner.rs
RSrunner.rs16.1 KBDownload
1// Executing a run.
2//
3// What this actually provides, stated plainly because the gap between claimed and
4// real isolation is where people get hurt:
5//
6// * a scrubbed environment — the child inherits nothing but an allowlist, so the
7// service's credentials are not readable from `/proc/self/environ`;
8// * rlimits on CPU, address space, file size and open files (process count is
9// opt-in: `RLIMIT_NPROC` is per-UID and means nothing without a runner uid);
10// * a wall-clock timeout that kills the whole process GROUP, because a step's
11// children are grandchildren of ours and killing the shell orphans them;
12// * a bounded log, so output cannot fill the disk;
13// * a workspace that is deleted afterwards.
14//
15// What it does NOT provide, in standalone mode: a container, a network namespace, or
16// a separate uid unless the operator configured one. **Standalone CI is not a
17// security boundary** — anyone who can push can run code as the service user. It is
18// off by default and the README says this in the same words.
19//
20// A step allowlist was considered and rejected: `sh -c` defeats it in one character.
21// Enforcement is at the OS or it is not enforcement.
22
23use crate::brand;
24use crate::ci::run::{Run, RunStore, Status};
25use crate::ci::spec::Spec;
26use crate::config::Config;
27use crate::git::validate::Name;
28use std::path::Path;
29use std::process::Stdio;
30use std::time::Duration;
31use tokio::io::AsyncReadExt;
32use tokio::process::Command;
33
34/// Environment variables the child is allowed to see, beyond the ones we set.
35const INHERITED: &[&str] = &["PATH", "HOME", "LANG", "TZ"];
36
37/// Everything a step needs that does not change between steps.
38struct Context<'a> {
39 config: &'a Config,
40 runs: &'a RunStore,
41 account: &'a Name,
42 repo: &'a Name,
43 run: &'a Run,
44 workspace: &'a Path,
45}
46
47impl Context<'_> {
48 fn log(&self, text: &str) {
49 let _ = self.runs.append_log(
50 self.account,
51 self.repo,
52 &self.run.id,
53 text.as_bytes(),
54 self.config.ci_log_bytes,
55 );
56 }
57}
58
59pub struct Outcome {
60 pub status: Status,
61 pub exit_code: Option<i32>,
62 pub detail: Option<String>,
63}
64
65/// Check out a revision and run its steps.
66pub async fn execute(
67 config: &Config,
68 runs: &RunStore,
69 account: &Name,
70 repo: &Name,
71 run: &Run,
72 git_dir: &Path,
73) -> Outcome {
74 let workspace = config
75 .data_dir
76 .join("tmp")
77 .join("work")
78 .join(format!("{}-{}", run.id, run.created_at));
79
80 let outcome = execute_in(config, runs, account, repo, run, git_dir, &workspace).await;
81
82 // The workspace is scratch and may contain anything the run wrote.
83 if let Err(e) = std::fs::remove_dir_all(&workspace) {
84 if e.kind() != std::io::ErrorKind::NotFound {
85 eprintln!("[ci] could not clean {}: {e}", workspace.display());
86 }
87 }
88 outcome
89}
90
91async fn execute_in(
92 config: &Config,
93 runs: &RunStore,
94 account: &Name,
95 repo: &Name,
96 run: &Run,
97 git_dir: &Path,
98 workspace: &Path,
99) -> Outcome {
100 let ctx = Context {
101 config,
102 runs,
103 account,
104 repo,
105 run,
106 workspace,
107 };
108
109 if let Err(e) = std::fs::create_dir_all(workspace) {
110 return internal(format!("could not create a workspace: {e}"));
111 }
112
113 // Check out into the workspace without a clone: the object database is right
114 // there, and `--no-checkout` plus `checkout-index` avoids a second copy of it.
115 ctx.log(&format!("$ checkout {}\n", run.sha));
116 if let Err(e) = checkout(git_dir, &run.sha, workspace).await {
117 ctx.log(&format!("{e}\n"));
118 return Outcome {
119 status: Status::Failed,
120 exit_code: None,
121 detail: Some("checkout failed".into()),
122 };
123 }
124
125 let spec_path = workspace.join(brand::ci_filename());
126 let source = match std::fs::read_to_string(&spec_path) {
127 Ok(source) => source,
128 Err(_) => {
129 ctx.log(&format!(
130 "no {} in this revision; nothing to run\n",
131 brand::ci_filename()
132 ));
133 return Outcome {
134 status: Status::Succeeded,
135 exit_code: Some(0),
136 detail: Some("no spec".into()),
137 };
138 }
139 };
140
141 let spec = match Spec::parse(&source) {
142 Ok(spec) => spec,
143 Err(e) => {
144 ctx.log(&format!("{}\n", e.detail_for_user()));
145 return Outcome {
146 status: Status::Failed,
147 exit_code: None,
148 detail: Some("invalid spec".into()),
149 };
150 }
151 };
152
153 if !spec.runs_for(&run.git_ref) {
154 ctx.log(&format!(
155 "{} is not in run.branches; skipped\n",
156 run.git_ref
157 ));
158 return Outcome {
159 status: Status::Succeeded,
160 exit_code: Some(0),
161 detail: Some("skipped".into()),
162 };
163 }
164
165 let deadline = spec.timeout(config.ci_timeout);
166
167 // Say so when the host ceiling overrides what the repository asked for. Silently
168 // clamping is how a repository ends up reporting "timed out after 600s" when its
169 // spec plainly says 2400, and the only way to find out is to read the source.
170 let requested = Duration::from_secs(spec.run.timeout_secs);
171 if requested > deadline {
172 ctx.log(&format!(
173 "note: {} asks for a {}s timeout; this host allows at most {}s\n",
174 brand::ci_filename(),
175 requested.as_secs(),
176 deadline.as_secs(),
177 ));
178 }
179
180 let started = std::time::Instant::now();
181
182 for step in &spec.run.steps {
183 let remaining = deadline.saturating_sub(started.elapsed());
184 if remaining.is_zero() {
185 return timed_out(&ctx, deadline);
186 }
187
188 ctx.log(&format!("$ {step}\n"));
189 match run_step(&ctx, step, remaining).await {
190 StepResult::Ok => {}
191 StepResult::Failed(code) => {
192 ctx.log(&format!("step failed with exit code {code}\n"));
193 return Outcome {
194 status: Status::Failed,
195 exit_code: Some(code),
196 detail: Some(format!("step failed: {step}")),
197 };
198 }
199 StepResult::TimedOut => return timed_out(&ctx, deadline),
200 StepResult::Internal(detail) => {
201 ctx.log(&format!("{detail}\n"));
202 return internal(detail);
203 }
204 }
205 }
206
207 ctx.log("all steps succeeded\n");
208 Outcome {
209 status: Status::Succeeded,
210 exit_code: Some(0),
211 detail: None,
212 }
213}
214
215enum StepResult {
216 Ok,
217 Failed(i32),
218 TimedOut,
219 Internal(String),
220}
221
222async fn run_step(ctx: &Context<'_>, step: &str, remaining: Duration) -> StepResult {
223 let (config, runs, run) = (ctx.config, ctx.runs, ctx.run);
224 let (account, repo) = (ctx.account, ctx.repo);
225
226 let mut command = Command::new("/bin/sh");
227 command
228 .arg("-c")
229 .arg(step)
230 .current_dir(ctx.workspace)
231 .env_clear()
232 .env("CI", "1")
233 .env(brand::env_name("CI"), "1")
234 .env(brand::env_name("RUN_ID"), &run.id)
235 .env(brand::env_name("RUN_NUMBER"), run.number.to_string())
236 .env(brand::env_name("SHA"), &run.sha)
237 .env(brand::env_name("REF"), &run.git_ref)
238 // Non-secret topology is useful to a repository that lives in more than
239 // one instance: this source runs its real build on the standalone build
240 // host and is mirrored to a tenant that intentionally has no Rust toolchain.
241 .env(brand::env_name("MODE"), config.mode.as_str())
242 .stdin(Stdio::null())
243 .stdout(Stdio::piped())
244 .stderr(Stdio::piped())
245 .kill_on_drop(true);
246
247 for key in INHERITED {
248 if let Ok(value) = std::env::var(key) {
249 command.env(key, value);
250 }
251 }
252
253 #[cfg(unix)]
254 {
255 command.process_group(0);
256 apply_limits(&mut command, config);
257 }
258
259 let mut child = match command.spawn() {
260 Ok(child) => child,
261 Err(e) => return StepResult::Internal(format!("could not start the step: {e}")),
262 };
263 let pid = child.id();
264
265 let mut stdout = child.stdout.take().expect("stdout is piped");
266 let mut stderr = child.stderr.take().expect("stderr is piped");
267 let cap = config.ci_log_bytes;
268
269 // Both pipes drain concurrently. A step that fills stderr while we read only
270 // stdout blocks forever on a 64 KiB pipe.
271 let pump = async {
272 let mut out = [0u8; 8192];
273 let mut err = [0u8; 8192];
274 loop {
275 tokio::select! {
276 read = stdout.read(&mut out) => match read {
277 Ok(0) | Err(_) => break,
278 Ok(n) => { let _ = runs.append_log(account, repo, &run.id, &out[..n], cap); }
279 },
280 read = stderr.read(&mut err) => match read {
281 Ok(0) | Err(_) => break,
282 Ok(n) => { let _ = runs.append_log(account, repo, &run.id, &err[..n], cap); }
283 },
284 }
285 }
286 // Whichever pipe closed first, drain the other.
287 let mut rest = Vec::new();
288 let _ = stdout.read_to_end(&mut rest).await;
289 let _ = runs.append_log(account, repo, &run.id, &rest, cap);
290 rest.clear();
291 let _ = stderr.read_to_end(&mut rest).await;
292 let _ = runs.append_log(account, repo, &run.id, &rest, cap);
293 };
294
295 let waited = tokio::time::timeout(remaining, async {
296 pump.await;
297 child.wait().await
298 })
299 .await;
300
301 match waited {
302 Err(_) => {
303 kill_group(pid);
304 StepResult::TimedOut
305 }
306 Ok(Err(e)) => StepResult::Internal(format!("could not wait for the step: {e}")),
307 Ok(Ok(status)) if status.success() => StepResult::Ok,
308 Ok(Ok(status)) => StepResult::Failed(status.code().unwrap_or(-1)),
309 }
310}
311
312/// Apply resource limits in the child, between fork and exec.
313#[cfg(unix)]
314fn apply_limits(command: &mut Command, config: &Config) {
315 let cpu = config.ci_timeout.as_secs().max(1);
316 let memory = config.ci_memory_bytes;
317 let file_size = config.ci_file_bytes;
318 let processes = config.ci_max_processes;
319
320 // SAFETY: only async-signal-safe calls between fork and exec. `setrlimit` is.
321 unsafe {
322 command.pre_exec(move || {
323 set_limit(libc::RLIMIT_CPU, cpu);
324 // Independent of the memory limit. This was once derived from it, and
325 // turning the memory limit off therefore shrank the file limit to a few
326 // megabytes — which surfaced as the linker being killed by SIGXFSZ while
327 // writing a test binary, an error that names neither limit.
328 if file_size > 0 {
329 set_limit(libc::RLIMIT_FSIZE, file_size);
330 }
331 set_limit(libc::RLIMIT_NOFILE, 1024);
332 // RLIMIT_AS caps the address space a process may map, which is not the
333 // same thing as the memory it uses. V8, the JVM and the Go runtime all
334 // reserve multi-gigabyte regions up front and touch almost none of it,
335 // so a limit set to a sensible working-set size kills them on startup
336 // with an out-of-memory message that has nothing to do with memory.
337 // Anything running a JavaScript or Java toolchain in CI needs this set
338 // far above the real requirement, or set to 0 and bounded by a cgroup
339 // instead — which is the control that actually measures usage.
340 if memory > 0 {
341 set_limit(libc::RLIMIT_AS, memory);
342 }
343 // Off unless the operator asked for it. RLIMIT_NPROC is per-UID: with
344 // the runner sharing a uid with the service it counts processes we do
345 // not control, so a useful-looking value makes `fork` fail on the first
346 // step. It becomes meaningful — and the fork-bomb protection it looks
347 // like — only alongside a dedicated runner uid.
348 if processes > 0 {
349 set_limit(libc::RLIMIT_NPROC, processes);
350 }
351 Ok(())
352 });
353 }
354}
355
356/// Type of the `resource` argument to `setrlimit`.
357///
358/// glibc declares it `__rlimit_resource_t` (u32); musl and the BSDs use `c_int`.
359/// The discriminator is the C library, not the OS — keying on `target_os` compiles
360/// on linux-gnu and fails on linux-musl, which is the target that produces the
361/// static binary this service ships as.
362#[cfg(all(unix, target_os = "linux", target_env = "gnu"))]
363type RlimitResource = u32;
364#[cfg(all(unix, not(all(target_os = "linux", target_env = "gnu"))))]
365type RlimitResource = libc::c_int;
366
367#[cfg(unix)]
368fn set_limit(resource: RlimitResource, value: u64) {
369 let limit = libc::rlimit {
370 rlim_cur: value as libc::rlim_t,
371 rlim_max: value as libc::rlim_t,
372 };
373 // SAFETY: `limit` is a valid, fully initialised rlimit for `resource`.
374 unsafe {
375 libc::setrlimit(resource, &limit);
376 }
377}
378
379fn kill_group(pid: Option<u32>) {
380 #[cfg(unix)]
381 if let Some(pid) = pid {
382 // SAFETY: spawned with `process_group(0)`, so the pgid is the child's own
383 // pid and no unrelated process shares it.
384 unsafe {
385 libc::killpg(pid as libc::pid_t, libc::SIGKILL);
386 }
387 }
388 let _ = pid;
389}
390
391/// Materialise a revision into a workspace.
392///
393/// A temporary index plus `checkout-index`, not a clone and not `archive | tar`:
394/// the object database is already on this disk, so cloning would copy all of it to
395/// read one tree, and piping through `tar` would add an external dependency and a
396/// second process to supervise. This is pure git and writes only under
397/// `GIT_WORK_TREE`.
398async fn checkout(git_dir: &Path, sha: &str, workspace: &Path) -> std::result::Result<(), String> {
399 let index = workspace.with_extension("index");
400
401 let run = |args: Vec<String>| {
402 let index = index.clone();
403 async move {
404 let output = Command::new("git")
405 .env_clear()
406 .env(
407 "PATH",
408 std::env::var("PATH").unwrap_or_else(|_| "/usr/bin:/bin".into()),
409 )
410 .env("GIT_CONFIG_NOSYSTEM", "1")
411 .env("GIT_DIR", git_dir)
412 .env("GIT_WORK_TREE", workspace)
413 .env("GIT_INDEX_FILE", &index)
414 .args(&args)
415 .output()
416 .await
417 .map_err(|e| format!("spawn git: {e}"))?;
418
419 if output.status.success() {
420 Ok(())
421 } else {
422 Err(format!(
423 "git {} failed: {}",
424 args.first().cloned().unwrap_or_default(),
425 String::from_utf8_lossy(&output.stderr).trim()
426 ))
427 }
428 }
429 };
430
431 let outcome = async {
432 run(vec!["read-tree".into(), sha.to_string()]).await?;
433 run(vec![
434 "checkout-index".into(),
435 "--all".into(),
436 "--force".into(),
437 ])
438 .await
439 }
440 .await;
441
442 // The index is scratch and must not outlive the checkout.
443 let _ = std::fs::remove_file(&index);
444 outcome
445}
446
447fn timed_out(ctx: &Context<'_>, deadline: Duration) -> Outcome {
448 ctx.log(&format!("run exceeded its {deadline:?} timeout\n"));
449 Outcome {
450 status: Status::TimedOut,
451 exit_code: None,
452 detail: Some(format!("timed out after {deadline:?}")),
453 }
454}
455
456fn internal(detail: String) -> Outcome {
457 Outcome {
458 status: Status::Failed,
459 exit_code: None,
460 detail: Some(detail),
461 }
462}
463
464#[cfg(test)]
465mod tests {
466 use super::*;
467
468 #[test]
469 fn the_inherited_environment_is_an_allowlist_not_a_passthrough() {
470 // The service's own secrets must not be readable from a step.
471 for dangerous in [
472 "KV_URL",
473 "DENO_KV_ACCESS_TOKEN",
474 "AWS_SECRET_ACCESS_KEY",
475 "LD_PRELOAD",
476 "GIT_CONFIG_GLOBAL",
477 ] {
478 assert!(
479 !INHERITED.contains(&dangerous),
480 "{dangerous} must not be inherited by a CI step"
481 );
482 }
483 assert!(
484 INHERITED.contains(&"PATH"),
485 "a step needs to find its tools"
486 );
487 }
488}