diff --git a/crates/fakecloud-codebuild/src/runtime.rs b/crates/fakecloud-codebuild/src/runtime.rs index c6e2ad6f0..7dfd4a3d9 100644 --- a/crates/fakecloud-codebuild/src/runtime.rs +++ b/crates/fakecloud-codebuild/src/runtime.rs @@ -1049,13 +1049,23 @@ async fn finish_stopped(job: &BuildJob, guard: &BuildGuard, container: &str) { /// Create + start a detached container that stays alive (`sleep`) long enough to /// run the whole build, so we can `exec` the build script into it. `docker cp` -/// / bind mounts are avoided; the image is pulled implicitly by `run`. +/// / bind mounts are avoided; a missing image is pulled first, with retries. async fn start_container( cli: &str, image: &str, job: &BuildJob, keepalive_secs: u64, ) -> Result { + // `run` pulls a missing image itself but gives up on the first throttled + // pull; make it available first, retrying transient registry failures. + tokio::time::timeout( + Duration::from_secs(600), + fakecloud_core::container_image::ensure_image(cli, None, image), + ) + .await + .map_err(|_| "timed out pulling build image".to_string())? + .map_err(|e| format!("failed to pull build image: {e}"))?; + let mut cmd = Command::new(cli); cmd.arg("run") .arg("-d") diff --git a/crates/fakecloud-core/src/container_image.rs b/crates/fakecloud-core/src/container_image.rs index 20c00fc3f..21a24cd6d 100644 --- a/crates/fakecloud-core/src/container_image.rs +++ b/crates/fakecloud-core/src/container_image.rs @@ -1,5 +1,8 @@ -//! Image pulls for the runtimes that launch user-supplied images (ECS, and -//! Batch through it; Lambda `PackageType=Image` functions). +//! Image pulls for the runtimes that launch containers: ECS (and Batch +//! through it) and Lambda `PackageType=Image` functions pull on every launch +//! with [`pull_image`]; EC2 instances and CodeBuild builds, whose images +//! stand in for an AMI or a curated build image, pull only when the image is +//! missing with [`ensure_image`]. //! //! A bare `docker pull` always contacts the registry, even when the image is //! already in the local cache, so a momentary registry failure fails the @@ -37,6 +40,8 @@ pub enum PulledImage { /// The pull failed transiently but the image was already cached locally, /// so the cached copy is used. Carries the pull's error for logging. Cached { pull_error: String }, + /// [`ensure_image`] found the image already cached and did not pull. + Present, } /// Pull `reference` with the container `cli`. A transient registry failure @@ -54,6 +59,31 @@ pub async fn pull_image( pull_image_with(cli, docker_config, reference, BASE_RETRY_DELAY).await } +/// Make `reference` available locally, pulling it only when it is not +/// already cached -- what `docker run` does with its implicit pull, but with +/// [`pull_image`]'s retry when the registry fails transiently. `docker run` +/// gives up on the first `429 Too Many Requests`, which on a host with an +/// empty cache fails every launch that races the first pull of an image. +pub async fn ensure_image( + cli: &str, + docker_config: Option<&Path>, + reference: &str, +) -> Result { + ensure_image_with(cli, docker_config, reference, BASE_RETRY_DELAY).await +} + +async fn ensure_image_with( + cli: &str, + docker_config: Option<&Path>, + reference: &str, + base_delay: Duration, +) -> Result { + if image_cached(cli, docker_config, reference).await { + return Ok(PulledImage::Present); + } + pull_image_with(cli, docker_config, reference, base_delay).await +} + async fn pull_image_with( cli: &str, docker_config: Option<&Path>, @@ -299,6 +329,10 @@ exit 2 async fn pull_ref(&self, reference: &str) -> Result { pull_image_with(&self.cli(), None, reference, Duration::from_millis(1)).await } + + async fn ensure(&self) -> Result { + ensure_image_with(&self.cli(), None, "alpine:3.20", Duration::from_millis(1)).await + } } fn is_transient_for_test(stderr: &str) -> bool { @@ -307,6 +341,29 @@ exit 2 const THROTTLED: &str = "Error response from daemon: unexpected status from HEAD request to https://public.ecr.aws/v2/docker/library/alpine/manifests/3.20: 429 Too Many Requests"; + #[tokio::test] + async fn ensure_image_uses_a_cached_image_without_contacting_the_registry() { + let cli = FakeCli::new(u32::MAX, THROTTLED, true); + assert_eq!(cli.ensure().await, Ok(PulledImage::Present)); + assert_eq!(cli.calls(), ["image inspect alpine:3.20"]); + } + + #[tokio::test] + async fn ensure_image_retries_a_throttled_first_pull() { + let cli = FakeCli::new(2, THROTTLED, false); + assert_eq!(cli.ensure().await, Ok(PulledImage::Pulled)); + let pulls = cli.calls().iter().filter(|c| c.starts_with("pull")).count(); + assert_eq!(pulls, 3); + } + + #[tokio::test] + async fn ensure_image_fails_on_a_refused_pull() { + let missing = + "Error response from daemon: manifest for alpine:3.20 not found: manifest unknown"; + let cli = FakeCli::new(u32::MAX, missing, false); + assert_eq!(cli.ensure().await, Err(missing.to_string())); + } + #[tokio::test] async fn a_successful_pull_needs_no_cache_check() { let cli = FakeCli::new(0, "", false); diff --git a/crates/fakecloud-ec2/src/runtime/mod.rs b/crates/fakecloud-ec2/src/runtime/mod.rs index 35660f899..0ea401329 100644 --- a/crates/fakecloud-ec2/src/runtime/mod.rs +++ b/crates/fakecloud-ec2/src/runtime/mod.rs @@ -739,6 +739,13 @@ impl DockerInstances { args.push(image.to_string()); args.extend(boot_command(user_data)); + // `run` pulls a missing image itself but gives up on the first + // throttled pull; make it available first, retrying transient + // registry failures. + fakecloud_core::container_image::ensure_image(&self.cli, None, image) + .await + .map_err(RuntimeError::ContainerStartFailed)?; + let output = tokio::process::Command::new(&self.cli) .args(&args) .output() diff --git a/website/content/docs/services/codebuild.md b/website/content/docs/services/codebuild.md index 29ab2a56b..1a07ef5a1 100644 --- a/website/content/docs/services/codebuild.md +++ b/website/content/docs/services/codebuild.md @@ -54,7 +54,10 @@ When a container runtime is available, the background build task: - **Resolves the image** from `environment.image`. A user-supplied image is used verbatim; an AWS-curated `aws/codebuild/*` image (not publicly pullable) maps - to a small runnable Ubuntu so the buildspec `commands` execute unchanged. + to a small runnable Ubuntu so the buildspec `commands` execute unchanged. An + image already in the local cache is used as is; a missing one is pulled first, + retrying a transient registry failure (rate limiting such as + `429 Too Many Requests`, a 5xx, a network timeout) with backoff. - **Parses the buildspec** — the inline `source.buildspec` or a `StartBuild.buildspecOverride` — reading `env.variables` and the `install` / `pre_build` / `build` / `post_build` phase `commands` and the diff --git a/website/content/docs/services/ec2.md b/website/content/docs/services/ec2.md index b26ce6dc0..6fa3c78e7 100644 --- a/website/content/docs/services/ec2.md +++ b/website/content/docs/services/ec2.md @@ -58,7 +58,7 @@ VPC/subnet/security-group/NACL metadata isn't just stored — fakecloud gives in ## Known limitations -- **Instances run as containers, not VMs** — `RunInstances` boots a real container per instance (Amazon Linux by default, overridable via `FAKECLOUD_EC2_DEFAULT_IMAGE`) and runs user-data at boot, but it is a container, not a full virtual machine: no kernel modules, no nested virtualization, and the instance type / EBS sizing is metadata only. The container runs as a local Docker/Podman container by default, or as a **native Kubernetes Pod** when `FAKECLOUD_EC2_BACKEND=k8s` (or the global `FAKECLOUD_CONTAINER_BACKEND=k8s`) is set — mirroring the Lambda/ECS/RDS/ElastiCache backends, so fakecloud running inside Kubernetes needs no Docker daemon. On the k8s backend a stopped instance's Pod is deleted and recreated on start (instances are not persistent disks). When no container runtime is available at all, the control plane degrades to a metadata-only instance so every API call still succeeds. +- **Instances run as containers, not VMs** — `RunInstances` boots a real container per instance (Amazon Linux by default, overridable via `FAKECLOUD_EC2_DEFAULT_IMAGE`; a cached image is used as is, and a missing one is pulled first with transient registry failures such as `429 Too Many Requests` retried) and runs user-data at boot, but it is a container, not a full virtual machine: no kernel modules, no nested virtualization, and the instance type / EBS sizing is metadata only. The container runs as a local Docker/Podman container by default, or as a **native Kubernetes Pod** when `FAKECLOUD_EC2_BACKEND=k8s` (or the global `FAKECLOUD_CONTAINER_BACKEND=k8s`) is set — mirroring the Lambda/ECS/RDS/ElastiCache backends, so fakecloud running inside Kubernetes needs no Docker daemon. On the k8s backend a stopped instance's Pod is deleted and recreated on start (instances are not persistent disks). When no container runtime is available at all, the control plane degrades to a metadata-only instance so every API call still succeeds. - **A handful of model operations are absent from the vendored AWS SDK** (`DescribeIpamPoolAllocations`, `ModifyIpamPoolAllocation`, the capacity-reservation cancellation-quote pair, and the `AttachImageWatermark`/`DetachImageWatermark` AMI watermark pair). They are implemented and conformance-probed via raw `ec2Query`, and graduate to typed SDK calls on the next SDK refresh. - **Security-group / NACL packet filtering needs host privileges.** Real enforcement (above) requires `CAP_NET_ADMIN` + `nft` on Docker/Podman, or a NetworkPolicy-enforcing CNI on Kubernetes. Where neither is available the rules are stored and returned faithfully but tracked-only; L3 (per-subnet/VPC) isolation still applies. ICMP and IPv6 rules are modeled best-effort. On Docker/Podman, fakecloud enables bridge netfilter (`net.bridge.bridge-nf-call-iptables`) so same-subnet traffic is actually filtered. - **Enforced security-group source matching uses real container IPs, not the AWS address space.** A backing container has a Docker/Podman bridge IP, not an address in the VPC's CIDR. So an enforced ingress rule keyed on a specific AWS CIDR (e.g. `10.0.0.0/8`) won't match a peer instance's real source IP — only `0.0.0.0/0` (anywhere) and **referenced security groups** (which resolve to member instances' real IPs, like the `default` group's allow-from-self) match as intended. Metadata (`DescribeSecurityGroups`) is unaffected; this only concerns the optional packet-filtering path.