collin/anvil · 2ba8b0fa
feat: sandbox CI jobs; document the untrusted-mode threat model
Collin Richards · 2026-06-09 22:50 UTC · 2ba8b0fa7c1f52dc47b7ef83f5844faefd4b6fb4 · parent b6833865 · browse files
modifiedDEPLOY.md+9 −0
| ⋯ 173 unchanged lines | |||
| 174 | 174 | Runs show up at `/{owner}/{repo}/ci`, with a per-commit status badge on the | |
| 175 | 175 | commit list and a full log on each run's page. | |
| 176 | 176 | ||
| 177 | + | **Job sandbox.** anvil is the only Docker client; the job container gets no | |
| 178 | + | socket, no mounts (the checkout is uploaded as a tar), all capabilities | |
| 179 | + | dropped, and `no-new-privileges`. Resource bounds are configurable under | |
| 180 | + | `[ci]`: `memory_mb` (default 2048), `cpus` (2), `pids_limit` (512), | |
| 181 | + | `timeout_secs` (1800, then the container is killed), `network` (true), | |
| 182 | + | `run_as` (empty = image default), and `allowed_images` (empty = any; a tagless | |
| 183 | + | entry like `"rust"` allows every tag). See `docs/untrusted-mode.md` for the | |
| 184 | + | threat model and what this does/doesn't protect against. | |
| 185 | + | ||
| 177 | 186 | **Continuous deployment** is deliberately scoped to **one** repository. On a | |
| 178 | 187 | successful run of `deploy_branch` (default `main`) in the repo named by | |
| 179 | 188 | `[ci] deploy_repo`, anvil POSTs JSON to `[ci] deploy_webhook`: | |
| ⋯ 32 unchanged lines | |||
modifiedanvil.example.toml+25 −0
| ⋯ 16 unchanged lines | |||
| 17 | 17 | clone_host = "localhost" | |
| 18 | 18 | clone_port = 2222 | |
| 19 | 19 | clone_user = "git" | |
| 20 | + | ||
| 21 | + | [ci] | |
| 22 | + | # Job sandbox. Containers always run with no Docker socket, no mounts, all | |
| 23 | + | # capabilities dropped, and no-new-privileges; these knobs bound resources | |
| 24 | + | # (0 = unlimited). See docs/untrusted-mode.md for the threat model. | |
| 25 | + | # Images a pipeline may use: empty allows any; a tagless entry ("rust") allows | |
| 26 | + | # every tag of that image; a tagged one ("alpine:3.20") exactly itself. | |
| 27 | + | allowed_images = [] | |
| 28 | + | memory_mb = 2048 | |
| 29 | + | cpus = 2.0 | |
| 30 | + | pids_limit = 512 | |
| 31 | + | # Wall-clock limit per job, after which the container is killed. | |
| 32 | + | timeout_secs = 1800 | |
| 33 | + | # Whether jobs get network access (most builds need it to fetch dependencies). | |
| 34 | + | network = true | |
| 35 | + | # User inside the job container, e.g. "1000:1000". Empty keeps the image default. | |
| 36 | + | run_as = "" | |
| 37 | + | ||
| 38 | + | # Continuous deployment: on a green run of deploy_branch in the ONE repo named | |
| 39 | + | # by deploy_repo, POST to deploy_webhook with the X-Anvil-Deploy-Secret header. | |
| 40 | + | # Empty deploy_repo/deploy_webhook disables deploys entirely. | |
| 41 | + | deploy_repo = "" | |
| 42 | + | deploy_branch = "main" | |
| 43 | + | deploy_webhook = "" | |
| 44 | + | deploy_secret = "" | |
modifiedcrates/anvil-ci/src/lib.rs+79 −41
| ⋯ 6 unchanged lines | |||
| 7 | 7 | //! pipeline's image. The checkout is uploaded into the container as a tar (via | |
| 8 | 8 | //! the Docker API), so it works regardless of where anvil's own filesystem | |
| 9 | 9 | //! lives and never exposes anvil's data volume to CI. | |
| 10 | + | //! | |
| 11 | + | //! anvil is the *broker*: it is the only Docker client, and the job container | |
| 12 | + | //! gets no socket, no bind mounts, and no volumes. On top of that the job runs | |
| 13 | + | //! with all capabilities dropped, `no-new-privileges`, and configurable | |
| 14 | + | //! pids/memory/cpu caps plus a wall-clock timeout and an optional image | |
| 15 | + | //! allowlist ([`anvil_core::config::CiConfig`]). | |
| 10 | 16 | ||
| 11 | 17 | use anvil_core::ci::{self, Pipeline}; | |
| 18 | + | use anvil_core::config::CiConfig; | |
| 12 | 19 | use anvil_core::{App, repos, storage, users}; | |
| 13 | 20 | use anvil_git::browse::{self, TreeFile}; | |
| 14 | 21 | use bollard::Docker; | |
| ⋯ 2 unchanged lines | |||
| 17 | 24 | UploadToContainerOptions, WaitContainerOptions, | |
| 18 | 25 | }; | |
| 19 | 26 | use bollard::image::CreateImageOptions; | |
| 27 | + | use bollard::models::HostConfig; | |
| 20 | 28 | use futures_util::StreamExt; | |
| 21 | 29 | use tokio::sync::mpsc::UnboundedReceiver; | |
| 22 | 30 | ||
| ⋯ 64 unchanged lines | |||
| 87 | 95 | owner.username, repo.name, run.ref_name, pipeline.image | |
| 88 | 96 | ); | |
| 89 | 97 | ||
| 90 | - | let status = match execute(&pipeline, tar, &mut log).await { | |
| 98 | + | let status = match execute(&pipeline, tar, &mut log, &app.config.ci).await { | |
| 91 | 99 | Ok(0) => ci::status::SUCCESS, | |
| 92 | 100 | Ok(code) => { | |
| 93 | 101 | log.push_str(&format!("\n[exited with status {code}]\n")); | |
| ⋯ 52 unchanged lines | |||
| 146 | 154 | } | |
| 147 | 155 | } | |
| 148 | 156 | ||
| 149 | - | /// Execute the pipeline in a container, streaming output into `log`. Returns the | |
| 150 | - | /// container's exit code. | |
| 151 | - | async fn execute(pipeline: &Pipeline, tar: Vec<u8>, log: &mut String) -> Result<i64, String> { | |
| 157 | + | /// Execute the pipeline in a sandboxed container, streaming output into `log`. | |
| 158 | + | /// Returns the container's exit code. | |
| 159 | + | /// | |
| 160 | + | /// The job container never sees the Docker socket and gets no mounts of any | |
| 161 | + | /// kind (the checkout is *uploaded*, not bind-mounted). All capabilities are | |
| 162 | + | /// dropped and `no-new-privileges` is set unconditionally; pids/memory/cpu | |
| 163 | + | /// caps, the wall-clock timeout, network access, the container user, and the | |
| 164 | + | /// image allowlist come from `cfg`. | |
| 165 | + | async fn execute( | |
| 166 | + | pipeline: &Pipeline, | |
| 167 | + | tar: Vec<u8>, | |
| 168 | + | log: &mut String, | |
| 169 | + | cfg: &CiConfig, | |
| 170 | + | ) -> Result<i64, String> { | |
| 171 | + | if !cfg.image_allowed(&pipeline.image) { | |
| 172 | + | return Err(format!( | |
| 173 | + | "image {} is not permitted by ci.allowed_images", | |
| 174 | + | pipeline.image | |
| 175 | + | )); | |
| 176 | + | } | |
| 152 | 177 | let docker = Docker::connect_with_socket_defaults() | |
| 153 | 178 | .map_err(|e| format!("docker unavailable (is the socket mounted?): {e}"))?; | |
| 154 | 179 | ||
| ⋯ 25 unchanged lines | |||
| 180 | 205 | script.push('\n'); | |
| 181 | 206 | } | |
| 182 | 207 | ||
| 208 | + | // The sandbox. Limits of 0 mean "unlimited" and omit the corresponding cap. | |
| 209 | + | let host_config = HostConfig { | |
| 210 | + | cap_drop: Some(vec!["ALL".to_string()]), | |
| 211 | + | security_opt: Some(vec!["no-new-privileges:true".to_string()]), | |
| 212 | + | pids_limit: (cfg.pids_limit > 0).then_some(cfg.pids_limit), | |
| 213 | + | memory: (cfg.memory_mb > 0).then(|| cfg.memory_mb * 1024 * 1024), | |
| 214 | + | memory_swap: (cfg.memory_mb > 0).then(|| cfg.memory_mb * 1024 * 1024), | |
| 215 | + | nano_cpus: (cfg.cpus > 0.0).then_some((cfg.cpus * 1e9) as i64), | |
| 216 | + | network_mode: (!cfg.network).then(|| "none".to_string()), | |
| 217 | + | ..Default::default() | |
| 218 | + | }; | |
| 183 | 219 | let config = Config { | |
| 184 | 220 | image: Some(pipeline.image.clone()), | |
| 185 | 221 | cmd: Some(vec!["sh".to_string(), "-c".to_string(), script]), | |
| 186 | 222 | working_dir: Some(WORKDIR.to_string()), | |
| 223 | + | user: (!cfg.run_as.is_empty()).then(|| cfg.run_as.clone()), | |
| 224 | + | host_config: Some(host_config), | |
| 187 | 225 | ..Default::default() | |
| 188 | 226 | }; | |
| 189 | 227 | let created = docker | |
| ⋯ 20 unchanged lines | |||
| 210 | 248 | .await | |
| 211 | 249 | .map_err(|e| format!("start container: {e}"))?; | |
| 212 | 250 | ||
| 213 | - | // Stream logs until the container stops. | |
| 214 | - | let mut logs = docker.logs( | |
| 215 | - | &id, | |
| 216 | - | Some(LogsOptions::<String> { | |
| 217 | - | follow: true, | |
| 218 | - | stdout: true, | |
| 219 | - | stderr: true, | |
| 220 | - | ..Default::default() | |
| 221 | - | }), | |
| 222 | - | ); | |
| 223 | - | while let Some(item) = logs.next().await { | |
| 224 | - | match item { | |
| 225 | - | Ok(output) => log.push_str(&String::from_utf8_lossy(&output.into_bytes())), | |
| 226 | - | Err(e) => { | |
| 227 | - | log.push_str(&format!("\n[log stream error] {e}\n")); | |
| 228 | - | break; | |
| 251 | + | // Stream logs and wait for the exit code, bounded by the wall-clock | |
| 252 | + | // timeout. The container is force-removed on every path (which also kills | |
| 253 | + | // a still-running job after a timeout). | |
| 254 | + | let run = async { | |
| 255 | + | let mut logs = docker.logs( | |
| 256 | + | &id, | |
| 257 | + | Some(LogsOptions::<String> { | |
| 258 | + | follow: true, | |
| 259 | + | stdout: true, | |
| 260 | + | stderr: true, | |
| 261 | + | ..Default::default() | |
| 262 | + | }), | |
| 263 | + | ); | |
| 264 | + | while let Some(item) = logs.next().await { | |
| 265 | + | match item { | |
| 266 | + | Ok(output) => log.push_str(&String::from_utf8_lossy(&output.into_bytes())), | |
| 267 | + | Err(e) => { | |
| 268 | + | log.push_str(&format!("\n[log stream error] {e}\n")); | |
| 269 | + | break; | |
| 270 | + | } | |
| 229 | 271 | } | |
| 230 | 272 | } | |
| 231 | - | } | |
| 232 | 273 | ||
| 233 | - | // Wait for the exit code (non-zero surfaces as a wait error in bollard). | |
| 234 | - | let mut code = 0i64; | |
| 235 | - | let mut wait = docker.wait_container(&id, None::<WaitContainerOptions<String>>); | |
| 236 | - | while let Some(item) = wait.next().await { | |
| 237 | - | match item { | |
| 238 | - | Ok(resp) => code = resp.status_code, | |
| 239 | - | Err(bollard::errors::Error::DockerContainerWaitError { code: c, .. }) => code = c, | |
| 240 | - | Err(e) => { | |
| 241 | - | let _ = docker | |
| 242 | - | .remove_container( | |
| 243 | - | &id, | |
| 244 | - | Some(RemoveContainerOptions { | |
| 245 | - | force: true, | |
| 246 | - | ..Default::default() | |
| 247 | - | }), | |
| 248 | - | ) | |
| 249 | - | .await; | |
| 250 | - | return Err(format!("wait: {e}")); | |
| 274 | + | // Non-zero exit codes surface as a wait error in bollard. | |
| 275 | + | let mut code = 0i64; | |
| 276 | + | let mut wait = docker.wait_container(&id, None::<WaitContainerOptions<String>>); | |
| 277 | + | while let Some(item) = wait.next().await { | |
| 278 | + | match item { | |
| 279 | + | Ok(resp) => code = resp.status_code, | |
| 280 | + | Err(bollard::errors::Error::DockerContainerWaitError { code: c, .. }) => code = c, | |
| 281 | + | Err(e) => return Err(format!("wait: {e}")), | |
| 251 | 282 | } | |
| 252 | 283 | } | |
| 253 | - | } | |
| 284 | + | Ok(code) | |
| 285 | + | }; | |
| 286 | + | let result = match cfg.timeout_secs { | |
| 287 | + | 0 => run.await, | |
| 288 | + | secs => tokio::time::timeout(std::time::Duration::from_secs(secs), run) | |
| 289 | + | .await | |
| 290 | + | .unwrap_or_else(|_| Err(format!("job exceeded ci.timeout_secs ({secs}s); killed"))), | |
| 291 | + | }; | |
| 254 | 292 | ||
| 255 | 293 | let _ = docker | |
| 256 | 294 | .remove_container( | |
| ⋯ 5 unchanged lines | |||
| 262 | 300 | ) | |
| 263 | 301 | .await; | |
| 264 | 302 | ||
| 265 | - | Ok(code) | |
| 303 | + | result | |
| 266 | 304 | } | |
| 267 | 305 | ||
| 268 | 306 | /// Build an uncompressed tar of the checkout, rooted at `workspace/` so it | |
| ⋯ 21 unchanged lines | |||
modifiedcrates/anvil-core/src/config.rs+73 −1
| ⋯ 21 unchanged lines | |||
| 22 | 22 | pub ci: CiConfig, | |
| 23 | 23 | } | |
| 24 | 24 | ||
| 25 | - | /// Continuous-deployment configuration. | |
| 25 | + | /// CI configuration: job sandbox limits and the single-repo redeploy webhook. | |
| 26 | 26 | /// | |
| 27 | 27 | /// On a successful CI run of [`deploy_branch`](CiConfig::deploy_branch) in the | |
| 28 | 28 | /// single repository named by [`deploy_repo`](CiConfig::deploy_repo), anvil | |
| ⋯ 15 unchanged lines | |||
| 44 | 44 | pub deploy_secret: String, | |
| 45 | 45 | /// Branch whose successful run triggers a deploy. Defaults to `main`. | |
| 46 | 46 | pub deploy_branch: String, | |
| 47 | + | /// Images a pipeline may run in. Empty allows any image. An entry without a | |
| 48 | + | /// tag (e.g. `rust`) allows every tag of that image; an entry with a tag | |
| 49 | + | /// (e.g. `rust:1.95-bookworm`) allows exactly that image. | |
| 50 | + | pub allowed_images: Vec<String>, | |
| 51 | + | /// Memory cap for a job container, in MiB (swap is capped to the same | |
| 52 | + | /// value). `0` means unlimited. Defaults to 2048. | |
| 53 | + | pub memory_mb: i64, | |
| 54 | + | /// CPU cap for a job container, in (possibly fractional) CPUs. `0` means | |
| 55 | + | /// unlimited. Defaults to 2. | |
| 56 | + | pub cpus: f64, | |
| 57 | + | /// Process-count cap inside a job container. `0` means unlimited. | |
| 58 | + | /// Defaults to 512. | |
| 59 | + | pub pids_limit: i64, | |
| 60 | + | /// Wall-clock timeout for a job, in seconds; on expiry the container is | |
| 61 | + | /// force-removed and the run errors. `0` disables the timeout. Defaults to | |
| 62 | + | /// 1800 (30 minutes). | |
| 63 | + | pub timeout_secs: u64, | |
| 64 | + | /// Whether job containers get network access (the default Docker network). | |
| 65 | + | /// Most builds need it to fetch dependencies; disable for stricter | |
| 66 | + | /// isolation. Defaults to `true`. | |
| 67 | + | pub network: bool, | |
| 68 | + | /// User to run the job as inside the container (`uid[:gid]` or a name known | |
| 69 | + | /// to the image). Empty keeps the image's default user. Note many base | |
| 70 | + | /// images assume root for e.g. `apt-get`. | |
| 71 | + | pub run_as: String, | |
| 47 | 72 | } | |
| 48 | 73 | ||
| 49 | 74 | #[derive(Debug, Clone, Serialize, Deserialize)] | |
| ⋯ 41 unchanged lines | |||
| 91 | 116 | deploy_webhook: String::new(), | |
| 92 | 117 | deploy_secret: String::new(), | |
| 93 | 118 | deploy_branch: "main".to_string(), | |
| 119 | + | allowed_images: Vec::new(), | |
| 120 | + | memory_mb: 2048, | |
| 121 | + | cpus: 2.0, | |
| 122 | + | pids_limit: 512, | |
| 123 | + | timeout_secs: 1800, | |
| 124 | + | network: true, | |
| 125 | + | run_as: String::new(), | |
| 94 | 126 | } | |
| 95 | 127 | } | |
| 96 | 128 | } | |
| ⋯ 6 unchanged lines | |||
| 103 | 135 | && self.deploy_repo == format!("{owner}/{name}") | |
| 104 | 136 | && self.deploy_branch == branch | |
| 105 | 137 | } | |
| 138 | + | ||
| 139 | + | /// Whether `image` passes [`allowed_images`](CiConfig::allowed_images). | |
| 140 | + | /// An empty allowlist permits any image; a tagless entry permits every tag | |
| 141 | + | /// of that image; a tagged entry permits exactly itself. | |
| 142 | + | pub fn image_allowed(&self, image: &str) -> bool { | |
| 143 | + | self.allowed_images.is_empty() | |
| 144 | + | || self.allowed_images.iter().any(|allowed| { | |
| 145 | + | image == allowed | |
| 146 | + | || (!allowed.contains(':') | |
| 147 | + | && image | |
| 148 | + | .strip_prefix(allowed.as_str()) | |
| 149 | + | .is_some_and(|rest| rest.starts_with(':'))) | |
| 150 | + | }) | |
| 151 | + | } | |
| 106 | 152 | } | |
| 107 | 153 | ||
| 108 | 154 | impl Default for HttpConfig { | |
| ⋯ 80 unchanged lines | |||
| 189 | 235 | } | |
| 190 | 236 | } | |
| 191 | 237 | } | |
| 238 | + | ||
| 239 | + | #[cfg(test)] | |
| 240 | + | mod tests { | |
| 241 | + | use super::*; | |
| 242 | + | ||
| 243 | + | #[test] | |
| 244 | + | fn image_allowlist_semantics() { | |
| 245 | + | let mut ci = CiConfig::default(); | |
| 246 | + | assert!(ci.image_allowed("anything:latest"), "empty list allows all"); | |
| 247 | + | ||
| 248 | + | ci.allowed_images = vec!["rust".to_string(), "alpine:3.20".to_string()]; | |
| 249 | + | assert!(ci.image_allowed("rust"), "tagless entry, tagless image"); | |
| 250 | + | assert!( | |
| 251 | + | ci.image_allowed("rust:1.95-bookworm"), | |
| 252 | + | "tagless entry allows any tag" | |
| 253 | + | ); | |
| 254 | + | assert!(ci.image_allowed("alpine:3.20"), "tagged entry, exact match"); | |
| 255 | + | assert!(!ci.image_allowed("alpine:3.21"), "tagged entry, other tag"); | |
| 256 | + | assert!(!ci.image_allowed("alpine"), "tagged entry, tagless image"); | |
| 257 | + | assert!( | |
| 258 | + | !ci.image_allowed("rustlang/rust:nightly"), | |
| 259 | + | "no prefix bleed" | |
| 260 | + | ); | |
| 261 | + | assert!(!ci.image_allowed("rusty:latest"), "no name-prefix bleed"); | |
| 262 | + | } | |
| 263 | + | } | |
addeddocs/untrusted-mode.md+109 −0
| 1 | + | # Threat model: running anvil with untrusted users | |
| 2 | + | ||
| 3 | + | anvil is built as a **single-tenant, owner-operated forge**: the operator and | |
| 4 | + | the people with accounts are assumed to trust each other (a person, a family, a | |
| 5 | + | small team). This document records what would have to be true before opening | |
| 6 | + | registration (or repo write access) to people you *don't* trust, ranked by | |
| 7 | + | severity. It is the output of the security-audit pass; keep it updated as | |
| 8 | + | items land. | |
| 9 | + | ||
| 10 | + | **Supported stance:** single-tenant / owner-operated. Untrusted multi-tenancy | |
| 11 | + | is *not* supported until at least items 1–4 below are closed. | |
| 12 | + | ||
| 13 | + | --- | |
| 14 | + | ||
| 15 | + | ## 1. CI: arbitrary code execution by design | |
| 16 | + | ||
| 17 | + | Anyone who can push to a repo with a `.anvil/ci.yml` runs arbitrary code on | |
| 18 | + | your hardware. That is the *point* of CI, so the question is only how well the | |
| 19 | + | blast radius is contained. | |
| 20 | + | ||
| 21 | + | **Broker model (implemented).** anvil itself is the only Docker client. The | |
| 22 | + | job container gets: | |
| 23 | + | ||
| 24 | + | - **no Docker socket, no bind mounts, no volumes** — the checkout is uploaded | |
| 25 | + | into the container as a tar via the Docker API, so the job can never reach | |
| 26 | + | anvil's data directory or the host filesystem; | |
| 27 | + | - **`--cap-drop=ALL` and `no-new-privileges`** unconditionally; | |
| 28 | + | - **pids / memory(+swap) / cpu caps** (`ci.pids_limit`, `ci.memory_mb`, | |
| 29 | + | `ci.cpus`; defaults 512 / 2048 MiB / 2); | |
| 30 | + | - **a wall-clock timeout** (`ci.timeout_secs`, default 30 minutes) after which | |
| 31 | + | the container is force-removed; | |
| 32 | + | - optionally **no network** (`ci.network = false`) and a **non-root user** | |
| 33 | + | (`ci.run_as = "1000:1000"`) — most real builds need network and many base | |
| 34 | + | images assume root, so these default to permissive; | |
| 35 | + | - an **image allowlist** (`ci.allowed_images`) — empty allows any image, which | |
| 36 | + | is fine single-tenant; set it before letting strangers push. | |
| 37 | + | ||
| 38 | + | **Deliberately not done:** read-only rootfs (the workspace lives in the | |
| 39 | + | container filesystem precisely so no volume is ever attached; builds also | |
| 40 | + | write `$HOME` caches), and egress *filtering* (network is all-or-nothing). | |
| 41 | + | ||
| 42 | + | **Residual risk / stronger tier.** Containers share the host kernel; a kernel | |
| 43 | + | or runc escape defeats all of the above. For genuinely hostile tenants run the | |
| 44 | + | jobs under gVisor/Kata/Firecracker (a runtime flag on the broker — the | |
| 45 | + | "isolated workers" idea), and add per-user CI-minute and disk quotas (image | |
| 46 | + | pulls consume host disk). Until then, CI for untrusted users should stay off. | |
| 47 | + | ||
| 48 | + | ## 2. Stored XSS via served content | |
| 49 | + | ||
| 50 | + | Repo browsing renders escaped text (Maud auto-escapes; highlighting emits | |
| 51 | + | sanitized HTML), so hostile file *content* does not execute in the forge | |
| 52 | + | origin today. | |
| 53 | + | ||
| 54 | + | **Pages hosting is the exception by design**: it serves attacker-authored | |
| 55 | + | HTML/JS. On a single-origin deployment, a pages site runs in the same origin | |
| 56 | + | as the forge UI — its JS could read forge pages and drive authenticated | |
| 57 | + | requests in a visitor's session. Mitigations in place: session cookies are | |
| 58 | + | `HttpOnly` (no token theft) and all mutating routes require the CSRF token. | |
| 59 | + | But same-origin JS can still *read* the token off a fetched page, so for | |
| 60 | + | untrusted users pages must move to a **separate origin** (e.g. | |
| 61 | + | `*.pages.example.com`), as GitHub does. The same applies to any future "raw | |
| 62 | + | blob" endpoint: serve `text/plain` + `nosniff` + a restrictive CSP, or a | |
| 63 | + | separate origin. | |
| 64 | + | ||
| 65 | + | ## 3. Git resource exhaustion | |
| 66 | + | ||
| 67 | + | A hostile pusher can send decompression bombs (tiny pack, enormous objects), | |
| 68 | + | deep delta chains, or millions of refs; a hostile cloner can request expensive | |
| 69 | + | packs repeatedly. Needed before untrusted use: per-repo and per-user storage | |
| 70 | + | quotas, an upload size cap on `receive-pack`, timeouts/memory bounds on pack | |
| 71 | + | ingestion and pack generation, and a cap on advertised refs. (The CI tar | |
| 72 | + | materializer also loads a full checkout into memory — bounded today only by | |
| 73 | + | push quotas not existing.) | |
| 74 | + | ||
| 75 | + | ## 4. Open registration anti-abuse | |
| 76 | + | ||
| 77 | + | Registration is currently operator-controlled (CLI), which is the real | |
| 78 | + | mitigation. Opening it requires: email verification, rate limiting on signup / | |
| 79 | + | login / repo creation, a CAPTCHA or proof-of-work, per-user quotas (repos, | |
| 80 | + | storage, CI minutes), and an admin ban/cleanup path. Reserved usernames and | |
| 81 | + | the `/-/` route namespace already prevent route-shadowing squats. | |
| 82 | + | ||
| 83 | + | ## 5. Authorization granularity | |
| 84 | + | ||
| 85 | + | Access today is owner-or-admin, repo public-or-private. Fine single-tenant; | |
| 86 | + | multi-user collaboration needs collaborator roles (read/write/admin per repo), | |
| 87 | + | per-repo deploy keys, and scoped tokens instead of full-account SSH keys. The | |
| 88 | + | deploy webhook is already scoped to exactly one configured repo. | |
| 89 | + | ||
| 90 | + | ## 6. Webhook SSRF (future) | |
| 91 | + | ||
| 92 | + | User-configurable webhooks don't exist yet (the deploy webhook URL is | |
| 93 | + | operator-set in the config file, not user data). When they land: resolve and | |
| 94 | + | block private/link-local/metadata ranges (and re-check on redirect), pin DNS, | |
| 95 | + | cap response sizes and time, and never reflect response bodies to users. | |
| 96 | + | ||
| 97 | + | --- | |
| 98 | + | ||
| 99 | + | ## Already right (keep it that way) | |
| 100 | + | ||
| 101 | + | - Private repos 404 for non-readers — no existence leak (`resolve_repo`). | |
| 102 | + | - Reserved usernames + `/-/` namespace for app routes. | |
| 103 | + | - CI checkout is tar-uploaded, never bind-mounted; job containers get no | |
| 104 | + | socket; sandbox defaults are on (see §1). | |
| 105 | + | - CD webhook gated to a single configured repo + shared-secret header. | |
| 106 | + | - Passwords: argon2. Sessions: `HttpOnly` + `SameSite=Lax` + `Secure` (auto | |
| 107 | + | when `base_url` is https). CSRF: HMAC synchronizer token, constant-time | |
| 108 | + | compare, on all mutating forms; htmx requests carry it via `hx-headers`. | |
| 109 | + | - SSH auth by exact public-key match; unknown keys rejected. |