From c6b45060f8cdcdedf51fb9b96cde08dd8c0cd6aa Mon Sep 17 00:00:00 2001 From: mecattaf Date: Sun, 20 Sep 2026 23:16:18 +0200 Subject: [PATCH 01/37] nas: Substrate's state services, gate OFF (sandbox spike, Track S) Tom, 2026-09-20: "having local kubernetes s3 or redis or postgres or whatever it needs on the NAS". This is that sentence for the three Substrate actually reads. State on the appliance, machines on the Strix boxes -- the NAS has 8 threads and 22 GiB and runs the house's DNS, so it holds the control plane's state and schedules nothing (Appendix J section 5). A SECOND DATABASE, not a second PostgreSQL. hosts/nas/media.nix already brought the instance up and pinned its dataDir to the NVMe; this rides it, so there is one instance and one backup story. ensureDatabases/ensureUsers with ensureDBOwnership, because ateapi runs goose migrations at startup and creates its own tables. enableTCPIP is forced by the deployment and not by taste: ateapi is a pod on a Strix box, the k3s server here sets disableAgent, so there is no unix socket to peer-authenticate over. scram-sha-256 from the LAN and from the pod CIDR, never trust. THE OBJECT STORE IS A HAND-WRITTEN UNIT ON PURPOSE. This host rides nixpkgs-stable (nixos-26.05) and stable has no rustfs at all -- no module, no package. The main pin has both; the package comes across the hosts/nas/unstable-pkgs.nix seam that attic-server and Immich already use, and the unit is modelled line for line on the unstable module's own serviceConfig. Delete it for services.rustfs when the NAS next rides a stable that ships it. Every unit carries RequiresMountsFor on /mnt/fast. hosts/nas/attic.nix learned the second edition of the signing-key trap: /mnt/fast is nofail, so a failed mount otherwise yields an empty state tree and a service that starts happily against it. For an object store that is silently lost snapshots. Refusing to start is a Tuesday. Secrets are runbook-placed root-owned files, not agenix, per attic.nix's no-agenix doctrine -- the rustfs key pair and the Substrate role's password are consumed by this host alone and a runbook can place them once. The password oneshot reads its file through LoadCredential and binds it as a psql value, so it lands in no argv and no journal line. Firewall: one block, three ports, iifname "enp1s0" only. Not openFirewall, which would publish all three on the tailnet as well. All three are on hosts/nas/cloudflared.nix's never-routed list. Co-Authored-By: Claude Opus 5 (1M context) --- hosts/nas/default.nix | 13 ++ hosts/nas/state-services.nix | 362 +++++++++++++++++++++++++++++++++++ 2 files changed, 375 insertions(+) create mode 100644 hosts/nas/state-services.nix diff --git a/hosts/nas/default.nix b/hosts/nas/default.nix index 7b6de7252..e1f147bd2 100644 --- a/hosts/nas/default.nix +++ b/hosts/nas/default.nix @@ -45,6 +45,11 @@ ./headscale.nix # 2026-09-01: the fleet's OWN tailnet control plane (supersedes #233) ./tailscale-personal.nix # Additional isolated SaaS ingress; never enroll the lent laptops here ./personal-https.nix # Gated, NAS-scoped DNS-01 certificates for private media + # 2026-09-20 sandbox spike: the three state services the sandbox lane + # reads -- a second database on the PostgreSQL ./media.nix already runs, + # an S3-compatible object store on the NVMe, and a container registry. + # State here, machines on the Strix boxes (Appendix J section 5). Gate OFF. + ./state-services.nix ./headscale-backup.nix # consistent identity backup before overseas handover ../../modules/adguardhome.nix inputs.nixos-hardware.nixosModules.common-cpu-amd @@ -121,6 +126,14 @@ myNas.tailscalePersonal.funnel.policyApproved = true; myNas.headscale.serverUrl = "https://nas-saas.tail8dd1.ts.net:8443"; myNas.headscale.backup.enable = true; + + # ── State services for the sandbox lane: GATE OFF ─────────────────────── + # Three runbook-placed root-owned files stand between this and a flip (the + # rustfs key pair, the registry and object-store directories on /mnt/fast, + # and the Substrate role's password) -- ./state-services.nix's header has + # the commands. The appliance's no-agenix doctrine (./attic.nix) is why + # they are files placed by hand and not ciphertexts. + myNas.stateServices.enable = false; # Retired 2026-09-16: the Dell belongs to its owner; Tom no longer # publishes or manages Omarchy updates. Keep historical receipts only. myNas.omarchyUpdateCenter.enable = false; diff --git a/hosts/nas/state-services.nix b/hosts/nas/state-services.nix new file mode 100644 index 000000000..44c195460 --- /dev/null +++ b/hosts/nas/state-services.nix @@ -0,0 +1,362 @@ +{ + config, + lib, + pkgs, + unstablePkgs, + ... +}: +# ─── State services for the sandbox lane: PostgreSQL, object store, registry ─ +# +# Tom, 2026-09-20: "having local kubernetes s3 or redis or postgres or +# whatever it needs on the NAS". This is that sentence, item for item, for the +# three that Agent Substrate actually reads. It is state, not compute: the +# NAS has 8 threads and 22 GiB and runs the house's DNS, so control plane and +# state live here and the machines live on the Strix boxes. Appendix J +# section 5 is the long form of that split. +# +# ── WHY THIS IS ONE FILE AND NOT THREE ──────────────────────────────────── +# All three are LAN-only, all three are reached by the same two consumers +# (the k3s agents on the twins), all three land and flip together, and all +# three share one firewall block. Splitting them would mean three gates that +# are only ever flipped at once. +# +# ── THE APPLIANCE'S NO-AGENIX DOCTRINE APPLIES TO TWO OF THE THREE ──────── +# ./attic.nix, #130's ruling: "a root-owned env file placed by hand (runbook +# below) keeps the appliance's no-agenix doctrine." The doctrine was never +# "no agenix here" (hosts/nas/default.nix corrects that) -- it is "no STANDING +# decryption authority over ciphertext this box never reads". The test is +# whether the secret is consumed by this host alone and whether a runbook can +# place it once. +# +# rustfs access keys -> runbook-placed env file. Consumed here only. +# the atepg password -> runbook-placed file. Consumed here only. +# the tunnel creds -> agenix (./cloudflared.nix), because the ciphertext +# has to survive a reflash and be re-minted from +# Tom's key, not re-typed. +# +# ── /mnt/fast IS `nofail`, AND THAT IS LETHAL FOR STATE ─────────────────── +# ./disko.nix marks the 256G M.2 `nofail`, correct for a budget NVMe holding +# regenerable state. ./attic.nix learned the second edition of the +# signing-key trap the hard way: if that disk fails to mount, a StateDirectory +# cheerfully creates a fresh EMPTY tree and the service starts against it. For +# an object store that means Substrate's snapshots silently vanish; for a +# registry it means every image digest 404s mid-run. So every unit below +# carries RequiresMountsFor and refuses to start rather than inventing state. +# That is a Tuesday instead of an outage. Do not remove those lines. +# +# ── RUNBOOK — walk this before flipping the gate ────────────────────────── +# 1. Directories on the NVMe (the units will not create them; see above): +# install -d -m 0700 -o postgres -g postgres /mnt/fast/rustfs # no: see 2 +# 2. rustfs state and its key file: +# install -d -m 0750 -o rustfs -g rustfs /mnt/fast/rustfs +# install -d -m 0700 root:root /var/lib/rustfs-secrets +# printf 'RUSTFS_ACCESS_KEY=%s\nRUSTFS_SECRET_KEY=%s\n' \ +# > /var/lib/rustfs-secrets/env +# chmod 0400 /var/lib/rustfs-secrets/env +# Generate the pair with `openssl rand -hex 24` twice. They are NOT +# Cloudflare credentials and have nothing to do with R2. +# 3. registry storage: +# install -d -m 0750 -o docker-registry -g docker-registry \ +# /mnt/fast/registry +# 4. The Substrate database password (the role and database themselves are +# declarative below; only the password is by hand, because +# `services.postgresql.ensureUsers` deliberately cannot set one): +# install -d -m 0700 root:root /var/lib/postgresql-secrets +# openssl rand -hex 24 > /var/lib/postgresql-secrets/atepg-password +# chmod 0400 /var/lib/postgresql-secrets/atepg-password +# The oneshot below ALTERs the role from that file on every start, so +# rotating the password is "write the file, restart the unit". +# 5. Flip myNas.stateServices.enable, deploy the NAS. +# 6. Prove each one from the coordinator: +# psql "postgresql://atepg:$(cat …)@nas:5432/atepg?sslmode=disable" -c '\conninfo' +# curl -sS -o /dev/null -w '%{http_code}\n' http://nas:9000/ # rustfs +# curl -sS http://nas:5000/v2/_catalog # registry +# +# ── THE DSN SUBSTRATE WANTS ─────────────────────────────────────────────── +# MEASURED, ~/Downloads/substrate: `cmd/ateapi/main.go:347` reads +# ATE_API_POSTGRES_CONNECTION_STRING and `:348` reads ATE_API_POSTGRES_SCHEMA +# (default "public", hack/install-ate.sh:674). The installer skips its own +# bundled PostgreSQL StatefulSet entirely when the connection string is set +# (hack/install-ate.sh:274-286). The store is pgx v5 +# (cmd/ateapi/internal/store/atepg/atepg.go:36-38) and its own header comment +# says it passes "standard libpq sslmode/sslrootcert/sslcert/sslkey +# parameters" through to Connect. The upstream default DSN uses +# client-certificate auth against a projected pod certificate, which is a +# property of running PostgreSQL INSIDE the mesh; an external instance uses +# its own: +# +# ATE_API_POSTGRES_CONNECTION_STRING=postgresql://atepg:@nas:5432/atepg?sslmode=disable +# ATE_API_POSTGRES_SCHEMA=public +# +# `sslmode=disable` is honest rather than lazy: this is a LAN segment behind +# the house router, the traffic never leaves enp1s0, and a self-signed TLS +# layer here would add a certificate to rotate and no attacker it excludes. +# Revisit if the k3s agents ever stop being on the same wire. +# +# PEER AUTH OVER A UNIX SOCKET IS NOT AVAILABLE and was checked: ateapi runs +# as a pod on a Strix box, not on this host (the k3s server here sets +# disableAgent = true and schedules nothing), so the connection is necessarily +# TCP. That is what forces enableTCPIP and the pg_hba lines below. +# +# ── GATE OFF ────────────────────────────────────────────────────────────── +# Lands with `enable = false`. Nothing about this host changes until Tom walks +# the runbook and flips it. +let + cfg = config.myNas.stateServices; + + fastRoot = "/mnt/fast"; + lanInterface = "enp1s0"; + lanCidr = "10.42.0.0/24"; + + # Kept in lockstep with modules/k3s-fleet.nix. If those move, these move. + podCidr = "10.200.0.0/16"; + + atepgPasswordFile = "/var/lib/postgresql-secrets/atepg-password"; + rustfsEnvironmentFile = "/var/lib/rustfs-secrets/env"; +in +{ + options.myNas.stateServices = { + enable = lib.mkEnableOption "PostgreSQL/object-store/registry state services for the sandbox lane (2026-09-20 spike; Appendix J section 5)"; + + databaseName = lib.mkOption { + type = lib.types.str; + default = "atepg"; + description = "Substrate's database on the instance this box already runs. Upstream's own name; Paperless and Immich keep theirs, untouched."; + }; + + rustfsPort = lib.mkOption { + type = lib.types.port; + default = 9000; + description = "S3 API port for the object store. LAN address only, never 0.0.0.0."; + }; + + registryPort = lib.mkOption { + type = lib.types.port; + default = 5000; + description = "Container registry port. LAN address only, plain HTTP; see the k3s mirror note in modules/k3s-fleet.nix."; + }; + }; + + config = lib.mkIf cfg.enable { + assertions = [ + { + # Immich brings PostgreSQL up on this box (./media.nix). If that ever + # stops being true, this module is silently adding a database to + # nothing, and the failure would show up as a Substrate install that + # cannot reach its store rather than as a NixOS error. + assertion = config.services.postgresql.enable; + message = "myNas.stateServices expects the NAS's existing PostgreSQL (brought up by hosts/nas/media.nix). Enable it, or drop the Substrate database from this module."; + } + { + assertion = cfg.databaseName != "paperless" && cfg.databaseName != "immich"; + message = "myNas.stateServices.databaseName must not collide with an existing database on this instance."; + } + ]; + + # ── 1. A SECOND DATABASE ON THE INSTANCE THAT ALREADY RUNS ──────────── + # Not a second PostgreSQL. ./media.nix already put the data directory on + # the NVMe and pinned RequiresMountsFor; this rides that, so there is one + # instance, one dataDir, one backup story. + services.postgresql = { + ensureDatabases = [ cfg.databaseName ]; + ensureUsers = [ + { + name = cfg.databaseName; + # Substrate runs goose migrations at startup + # (cmd/ateapi/internal/store/atepg/schema.go) and creates its own + # tables, so it needs ownership of the database rather than grants + # on a schema someone else owns. + ensureDBOwnership = true; + } + ]; + + # Forced by the shape of the deployment, not by taste: ateapi is a pod + # on a Strix box, so there is no unix socket to peer-authenticate over. + # This flips listen_addresses to "*" -- the nftables block at the bottom + # of this file is the actual access control, exactly as ./paperless.nix + # says of its own port ("the firewall rule below is the actual access + # control"). + enableTCPIP = true; + + # mkBefore so these land ABOVE the module's own generated rules and the + # first match wins. scram-sha-256 and never trust: the LAN is not a + # trusted segment just because it is a LAN, and this database holds the + # control plane's record of every sandbox. + # + # Both source ranges are deliberate. Cilium masquerades pod traffic + # leaving the cluster to the node's own address, so in practice the + # connection arrives from 10.42.0.2 or 10.42.0.5; the pod CIDR line is + # there for the day masquerading is turned off for this destination, so + # that change is a Cilium edit and not also a pg_hba mystery. + authentication = lib.mkBefore '' + host ${cfg.databaseName} ${cfg.databaseName} ${lanCidr} scram-sha-256 + host ${cfg.databaseName} ${cfg.databaseName} ${podCidr} scram-sha-256 + ''; + }; + + # `ensureUsers` deliberately cannot set a password (a password in the Nix + # store is a password in git). This is the runbook's half: read the + # root-owned file placed by hand and ALTER the role from it. Idempotent, + # so rotation is "write the file, restart this unit". + systemd.services.substrate-postgres-password = { + description = "Set the Substrate database role's password from the runbook-placed file"; + after = [ "postgresql.service" ]; + requires = [ "postgresql.service" ]; + wantedBy = [ "multi-user.target" ]; + # No file, no unit: a fresh box before step 4 of the runbook stays quiet + # rather than failing every boot. + unitConfig.ConditionPathExists = atepgPasswordFile; + serviceConfig = { + Type = "oneshot"; + User = "postgres"; + Group = "postgres"; + RemainAfterExit = true; + # The password never reaches the command line (ps is world-readable) + # nor the journal: it goes in through psql's stdin as a bound value. + LoadCredential = "atepg-password:${atepgPasswordFile}"; + }; + script = '' + set -euo pipefail + # The password reaches psql as a bound value read from the credential + # file, so it appears in no argv (ps is world-readable) and in no + # journal line. + ${config.services.postgresql.package}/bin/psql \ + --no-psqlrc --quiet --set=ON_ERROR_STOP=1 --dbname=${cfg.databaseName} <<'SQL' + \set pw `cat "$CREDENTIALS_DIRECTORY/atepg-password"` + ALTER ROLE ${cfg.databaseName} WITH LOGIN PASSWORD :'pw'; + SQL + ''; + }; + + # ── 2. THE OBJECT STORE ─────────────────────────────────────────────── + # Substrate selects its backend by environment, not at compile time + # (MEASURED, cmd/atelet/main.go:226-246): ATE_STORAGE_BACKEND=s3 plus the + # standard AWS_* variables, with AWS_S3_USE_PATH_STYLE for a non-AWS + # endpoint. rustfs is what Substrate's own kind path uses, which keeps + # its manifests unchanged. Note that the base atelet.yaml hardcodes "gcs" + # and the kind overlay patches it, so a non-kind install has to carry + # that patch. + # + # WHY A HAND-WRITTEN UNIT AND NOT services.rustfs: this host rides + # nixpkgs-stable (nixos-26.05, flake.nix:37) and stable has NO rustfs at + # all, neither module nor package. The main pin has both -- + # nixos/modules/services/web-servers/rustfs.nix and rustfs 1.0.0-beta.9 + # -- but importing one nixpkgs's module tree into another's evaluation is + # how you get a module that references options stable does not have. So: + # the package comes across the ./unstable-pkgs.nix seam that attic-server + # and Immich already use, and the unit below is modelled line for line on + # the unstable module's own serviceConfig. When the NAS next rides a + # stable that ships the module, delete this block and use it. + # + # rustfs takes no flags worth the name; everything is environment. + users.users.rustfs = { + isSystemUser = true; + group = "rustfs"; + }; + users.groups.rustfs = { }; + + systemd.services.rustfs = { + description = "RustFS object store (Substrate snapshot backend)"; + documentation = [ "https://rustfs.com/docs/" ]; + after = [ "network-online.target" ]; + wants = [ "network-online.target" ]; + wantedBy = [ "multi-user.target" ]; + + environment = { + RUSTFS_VOLUMES = "${fastRoot}/rustfs"; + # The LAN address and nothing else. Binding 0.0.0.0 here would put an + # unauthenticated-by-default object store on every interface this box + # has, tailnet included. + RUSTFS_ADDRESS = "10.42.0.1:${toString cfg.rustfsPort}"; + # The bundled web console is a second attack surface for a store whose + # only client is a Go program. + RUSTFS_CONSOLE_ENABLE = "false"; + }; + + unitConfig = { + # The /mnt/fast lesson from ./attic.nix, second edition. Without this + # a failed NVMe mount yields an empty store and silently lost + # snapshots instead of a service that refuses to start. + RequiresMountsFor = [ "${fastRoot}/rustfs" ]; + # No keys, no start. The upstream module prints a warning and exits; + # this says the same thing before the process is spawned. + ConditionPathExists = rustfsEnvironmentFile; + }; + + serviceConfig = { + Type = "notify"; + NotifyAccess = "main"; + User = "rustfs"; + Group = "rustfs"; + EnvironmentFile = rustfsEnvironmentFile; + ExecStart = lib.getExe unstablePkgs.rustfs; + LimitNOFILE = 1048576; + LimitNPROC = 32768; + TasksMax = "infinity"; + Restart = "always"; + RestartSec = "10s"; + TimeoutStartSec = "30s"; + TimeoutStopSec = "30s"; + NoNewPrivileges = true; + ProtectHome = true; + PrivateTmp = true; + PrivateDevices = true; + ProtectClock = true; + ProtectKernelTunables = true; + ProtectKernelModules = true; + ProtectControlGroups = true; + RestrictSUIDSGID = true; + RestrictRealtime = true; + }; + }; + + systemd.tmpfiles.rules = [ + "d /var/lib/rustfs-secrets 0700 root root -" + "d /var/lib/postgresql-secrets 0700 root root -" + # NB deliberately no rule for ${fastRoot}/rustfs or ${fastRoot}/registry, + # for ./attic.nix's reason: a tmpfiles rule would race the mount and + # create an empty directory for a failed NVMe to find, which is the + # silent-data-loss path this module exists to avoid. Runbook steps 2 + # and 3 create them on the real disk, once. + ]; + + # ── 3. THE REGISTRY ─────────────────────────────────────────────────── + # The Nix-built /process image and the Substrate cmd/* images have to be + # pullable by both agents. Plain HTTP on the LAN address: see the k3s + # mirror note below, and note that k3s's containerd needs to be told this + # endpoint is not TLS -- that registries.yaml lives in + # modules/k3s-fleet.nix, not here, because it is the client's problem. + services.dockerRegistry = { + enable = true; + listenAddress = "10.42.0.1"; + port = cfg.registryPort; + storagePath = "${fastRoot}/registry"; + # A spike pushes the same tag many times. Without delete plus a + # collection pass, /mnt/fast accumulates every superseded layer forever, + # on the 118 GiB that the object store is also growing into. + enableDelete = true; + enableGarbageCollect = true; + garbageCollectDates = "weekly"; + # openFirewall is deliberately NOT used: it opens the port on every + # interface. The interface-scoped rule at the bottom of this file is the + # access control, same shape as ./attic.nix and ./paperless.nix. + }; + systemd.services.docker-registry.unitConfig.RequiresMountsFor = [ "${fastRoot}/registry" ]; + + # ── THE ONE FIREWALL BLOCK ──────────────────────────────────────────── + # Interface-scoped, not subnet-scoped: `iifname "enp1s0"` is the LAN leg + # and nothing else, so none of these three is reachable over the tailnet, + # over headscale, or through the tunnel in ./cloudflared.nix. All three + # are unauthenticated or weakly authenticated by design and all three are + # on the never-routed list in that file. + networking.firewall.extraInputRules = '' + iifname "${lanInterface}" tcp dport ${toString config.services.postgresql.settings.port} accept comment "substrate postgres, LAN leg only" + iifname "${lanInterface}" tcp dport ${toString cfg.rustfsPort} accept comment "rustfs S3 API, LAN leg only" + iifname "${lanInterface}" tcp dport ${toString cfg.registryPort} accept comment "container registry, LAN leg only" + ''; + + # The unstable seam this module takes. Named here so `grep unstablePkgs` + # finds every consumer; ./unstable-pkgs.nix's header lists the others. + warnings = lib.optional (unstablePkgs.rustfs.version or "" == "") "hosts/nas/state-services.nix: unstablePkgs.rustfs has no version attribute; the pin may have moved."; + }; +} From a0749a3e4f66b0fedd035443a7443364beeceac1 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Sun, 20 Sep 2026 23:24:59 +0200 Subject: [PATCH 02/37] k3s: server on the NAS, agents on the twins, Cilium and a gvisor class Tom, 2026-09-20: "kubernetes is world-class for that ... they each stay in their lane. Effects ts handles the ultracode-level json dag specification and kubernetes schedules it on the right machines." One shared modules/k3s-fleet.nix imported by all three hosts, because the numbers have to agree and a second copy of them is how they stop agreeing. Server on the NAS with disableAgent, so the appliance holds the control plane and schedules nothing; agents on the Strix boxes, which have the cores, the memory and /dev/kvm. THE POD CIDR MOVES, AND THE ASSERTION IS OUTSIDE THE GATE. k3s defaults to 10.42.0.0/16 for pods and the house LAN is 10.42.0.0/24, which contains the NAS, the coordinator, the worker, the printer and the router. Tom's 2026-09-09 research already picked the replacement: pods 10.200.0.0/16, services 10.201.0.0/16. Those are kept exactly, and the overlap check is real arithmetic in a real assertion placed OUTSIDE the enable gate, so it evaluates on every host whether k3s is on or off. Verified against six cases, including that k3s's own default IS caught. TRACK K AND TRACK S ARE THE SAME FILE, and that is the point. Design K is plain k3s plus Cilium plus the gvisor RuntimeClass; Design S adds Substrate on top of exactly this. The only thing S needs from the bottom layer that K does not is four feature-gate settings, and a gate nothing asks for is inert: no scheduling change, no new controller, no memory. So the gates are in unconditionally and there is one PR instead of two that drift. The gate names are upstream's own, MEASURED from ~/Downloads/substrate, hack/create-kind-cluster.sh:104-112: ClusterTrustBundle, ClusterTrustBundleProjection, PodCertificateRequest, plus runtimeConfig certificates.k8s.io/v1beta1. The kubelet is handed only the two that are its own, because an unrecognised gate name is fatal to kubelet and ClusterTrustBundle is apiserver-side. Whether k3s 1.35.6+k3s1 accepts them at all is U9 and is not claimed here. Cilium 1.18.14 through autoDeployCharts, so the CNI lives in the generation rather than in a `cilium install` somebody has to remember. The chart hash is measured, not fakeHash: built with fakeHash, took the reported hash, rebuilt clean. The gvisor RuntimeClass is a server manifest and its containerd runtime is an agent containerdConfigTemplate that keeps `{{ template "base" . }}` -- one line, load-bearing, and dropping it costs the node its CNI, its snapshotter and its registry mirrors at once. kata is written out and commented, pointing at tonight's U14 report. The kube API is LAN-only: 6443 on the NAS's enp1s0 and nothing else, plus the never-routed doctrine in the cloudflared PR, because a tunnel ingress bypasses nftables entirely and "firewalled" is therefore a separate guarantee from "not in the tunnel". All three hosts land with the gate OFF. secrets/k3s-token.age does not exist; minting it needs Tom's admin age key. Co-Authored-By: Claude Opus 5 (1M context) --- hosts/coordinator/default.nix | 27 +++ hosts/nas/default.nix | 12 + hosts/worker/default.nix | 21 ++ modules/k3s-fleet.nix | 414 ++++++++++++++++++++++++++++++++++ secrets.nix | 11 + 5 files changed, 485 insertions(+) create mode 100644 modules/k3s-fleet.nix diff --git a/hosts/coordinator/default.nix b/hosts/coordinator/default.nix index 9bc36001e..7f285ae7f 100644 --- a/hosts/coordinator/default.nix +++ b/hosts/coordinator/default.nix @@ -74,6 +74,11 @@ # this: it carries its own pins in hosts/nas/network.nix and keeps the # stock loopback mapping. ../../modules/fleet-hosts.nix + # 2026-09-20 sandbox spike: k3s AGENT. The desk box is where an + # interactive session's sandbox wants to be, because herdr and the seat + # are here; ../../modules/k3s-fleet.nix carries the numbers and the + # RuntimeClass wiring for both twins. + ../../modules/k3s-fleet.nix # The REWRITE kernel (github.com/mecattaf/tally, U-B1…U-B13) as one system # service against ~/.local/state/tally-rewrite/, coexisting with the live # user-bus tally-daemon.service (U-D13). Declared here, installed by U-D19's @@ -83,6 +88,28 @@ networking.hostName = "coordinator"; + # ── k3s agent: GATE OFF (2026-09-20 sandbox spike) ───────────────────── + # Flip this, the worker's and the NAS's in the SAME commit: an agent whose + # server is not up retries forever and logs nothing useful. + # + # The labels are what the ultracode DAG schedules against, so they describe + # capability and not hardware. `fleet/desk` is the one that matters and the + # one only this box can have: herdr, the seat and the human are here, so an + # item that needs to be watched, teleported into, or answered belongs on + # this node and nowhere else. `fleet/kvm` is true on both twins (nested KVM + # MEASURED = 1 on both, 2026-09-20) and is the micro-VM sandbox class's + # precondition. There is no `fleet/gpu-proximity` here on purpose: Halogen + # is declared on this box with autoStart = false ("a resident model there + # would starve the desktop, TTS and diarization", modules/halogen.nix), so + # a GPU-adjacent item belongs on the worker. + myK3sFleet.enable = false; + myK3sFleet.role = "agent"; + myK3sFleet.nodeLabels = { + "fleet/role" = "desk"; + "fleet/desk" = "true"; + "fleet/kvm" = "true"; + }; + # Primary physical seat again (2026-09-16); Zenbook remains a second seat. # Agent services stay independent of either compositor. myDisplay.enable = true; diff --git a/hosts/nas/default.nix b/hosts/nas/default.nix index 7b6de7252..e602137be 100644 --- a/hosts/nas/default.nix +++ b/hosts/nas/default.nix @@ -46,6 +46,11 @@ ./tailscale-personal.nix # Additional isolated SaaS ingress; never enroll the lent laptops here ./personal-https.nix # Gated, NAS-scoped DNS-01 certificates for private media ./headscale-backup.nix # consistent identity backup before overseas handover + # 2026-09-20 sandbox spike: the fleet k3s cluster. This box is the SERVER + # and schedules nothing (disableAgent); the machines live on the Strix + # boxes. The CIDR-overlap assertions in that file evaluate whether or not + # the gate is on, which is the point of them. + ../../modules/k3s-fleet.nix ../../modules/adguardhome.nix inputs.nixos-hardware.nixosModules.common-cpu-amd inputs.nixos-hardware.nixosModules.common-pc @@ -121,6 +126,13 @@ myNas.tailscalePersonal.funnel.policyApproved = true; myNas.headscale.serverUrl = "https://nas-saas.tail8dd1.ts.net:8443"; myNas.headscale.backup.enable = true; + + # ── The k3s control plane: GATE OFF ──────────────────────────────────── + # secrets/k3s-token.age does not exist in the tree; minting it needs Tom's + # admin age key. Flip all three hosts in the SAME commit -- an agent whose + # server is not up yet retries forever and logs nothing useful. + myK3sFleet.enable = false; + myK3sFleet.role = "server"; # Retired 2026-09-16: the Dell belongs to its owner; Tom no longer # publishes or manages Omarchy updates. Keep historical receipts only. myNas.omarchyUpdateCenter.enable = false; diff --git a/hosts/worker/default.nix b/hosts/worker/default.nix index c18052618..77c164ce6 100644 --- a/hosts/worker/default.nix +++ b/hosts/worker/default.nix @@ -79,10 +79,31 @@ # resolves to loopback, which every distributed library happily binds — the # rank-1-hangs-forever failure. The NAS must NOT import this. ../../modules/fleet-hosts.nix + # 2026-09-20 sandbox spike: k3s AGENT. Idle CPU and /dev/kvm while + # Halogen holds only the GPU, so this is where a long batch belongs. + ../../modules/k3s-fleet.nix ]; networking.hostName = "worker"; + # ── k3s agent: GATE OFF (2026-09-20 sandbox spike) ───────────────────── + # Flip this, the coordinator's and the NAS's in the SAME commit. + # + # `fleet/gpu-proximity=halogen` is the label that earns this box its work: + # Halogen Flash is RESIDENT here (modules/halogen.nix, http://worker:8731) + # and holds the GPU and ~68 GiB of weights for the life of the process. A + # pod scheduled here reaches it over the LAN with no hop, and -- the part + # that actually matters -- Halogen holds the GPU but NOT the CPU, which is + # idle. So this is where a long CPU batch belongs even though the box looks + # busy. No `fleet/desk`: there is no display, no seat and no herdr here. + myK3sFleet.enable = false; + myK3sFleet.role = "agent"; + myK3sFleet.nodeLabels = { + "fleet/role" = "compute"; + "fleet/kvm" = "true"; + "fleet/gpu-proximity" = "halogen"; + }; + # ── no display, no compositor ────────────────────────────────────────────── # One line, not two forces: myDisplay.enable (modules/display.nix) is the # fleet's "is there a seat here" option, and modules/common.nix derives diff --git a/modules/k3s-fleet.nix b/modules/k3s-fleet.nix new file mode 100644 index 000000000..d6aef8239 --- /dev/null +++ b/modules/k3s-fleet.nix @@ -0,0 +1,414 @@ +{ + config, + lib, + pkgs, + ... +}: +# ─── k3s on this fleet: one module, three roles, one set of numbers ───────── +# +# Tom, 2026-09-20: "kubernetes is world-class for that ... they each stay in +# their lane. Effects ts handles the ultracode-level json dag specification +# and kubernetes schedules it on the right machines." +# +# The shape (Appendix J section 9): server on the NAS with disableAgent, so +# the appliance holds the control plane and schedules nothing; agents on the +# two Strix boxes, which have the cores, the memory and /dev/kvm. State on the +# NAS, machines on the twins. That split is the whole design and it is the +# same split hosts/nas/state-services.nix makes for PostgreSQL and the object +# store. +# +# ── THE NUMBERS, AND WHY THE ASSERTION IS NOT INSIDE THE GATE ───────────── +# k3s defaults to 10.42.0.0/16 for pods and 10.43.0.0/16 for services. THE +# HOUSE LAN IS 10.42.0.0/24 (hosts/nas/network.nix, hosts/nas/router.nix:23's +# pool, modules/fleet-hosts.nix). Taking k3s's default would put every pod on +# an address range that contains the NAS, the coordinator, the worker, the +# printer and the router, and the failure would be a same-day whole-house +# outage on the box that is also the DNS server. +# +# Tom's own 2026-09-09 research already chose the replacement: pods +# 10.200.0.0/16, services 10.201.0.0/16. Those numbers are kept exactly, and +# the overlap check below is a REAL EVALUATED ASSERTION rather than a comment, +# placed OUTSIDE the `enable` gate on purpose. It costs nothing when k3s is +# off, and it means that the day somebody edits a CIDR the flake refuses to +# evaluate rather than the house losing DNS. An assertion in the flake is the +# only thing that makes this non-forgettable. +# +# ── WHAT IS NEVER REACHABLE FROM OUTSIDE THIS HOUSE ────────────────────── +# The kube API on :6443 is the cluster's root credential surface. It is opened +# on the NAS's LAN leg (`iifname "enp1s0"`) and on nothing else, and it is on +# hosts/nas/cloudflared.nix's never-routed list, which that file enforces with +# its own assertion. A tunnel ingress bypasses every nftables rule on the +# appliance, so "not in the tunnel" is a separate guarantee from "firewalled", +# and both are needed. Same for ateapi and ax-server when they land: neither +# implements authorization at all, so reachability IS full control. +# +# ── TRACK K AND TRACK S ARE THE SAME FILE ──────────────────────────────── +# Design K is plain k3s with Cilium and a gVisor RuntimeClass, no Substrate +# and no ax. Design S adds Substrate on top of exactly this. The only thing +# Design S needs from the bottom layer that K does not is four feature-gate +# settings, and a feature gate that nothing asks for is inert: it changes no +# scheduling, admits no new controller, and costs no memory. So the gates are +# included unconditionally and this one module is both tracks. If Substrate is +# never installed, nothing here was wasted; if it is, nothing here has to +# change. +# +# ── THE FEATURE GATES, AND THE HONEST PART ─────────────────────────────── +# MEASURED from ~/Downloads/substrate, hack/create-kind-cluster.sh:104-112, +# which is upstream's own comment on why they are not optional: +# +# # cmd/podcertcontroller depends on ClusterTrustBundle & PodCertificateRequest. +# # They are not enabled by default as of Kubernetes v1.36 +# featureGates: +# ClusterTrustBundle: true +# ClusterTrustBundleProjection: true +# PodCertificateRequest: true +# runtimeConfig: +# "certificates.k8s.io/v1beta1": "true" +# +# atelet's own pod mounts a projected podCertificate volume and a +# clusterTrustBundle volume (manifests/ate-install/atelet.yaml:271-289), and +# every Substrate component's mTLS identity comes from that signer. +# +# WHAT IS NOT MEASURED, and must not be claimed until the first switch: which +# of these three gates each component actually accepts on k3s 1.35.6+k3s1. +# ClusterTrustBundle is an apiserver-side gate and a kubelet that is handed an +# unrecognised gate name refuses to start, so the kubelet below is given only +# the two that are its own. If the first switch produces a kubelet that will +# not start, the fix is in the kubeletGates list below and nowhere else. This +# is U9 and it is the single measurement that gates all of Track S. +# +# ── GATE OFF ───────────────────────────────────────────────────────────── +# Every host lands with `enable = false`. secrets/k3s-token.age does not exist +# in the tree: minting it needs Tom's admin age key, and the overnight spike +# that opened this PR has none. Runbook in the header of each host's gate. +let + cfg = config.myK3sFleet; + + # ── The numbers. One definition, three hosts. ── + podCidr = "10.200.0.0/16"; + serviceCidr = "10.201.0.0/16"; + lanCidr = "10.42.0.0/24"; + + nasLanInterface = "enp1s0"; + nasLanAddress = "10.42.0.1"; + apiPort = 6443; + serverAddr = "https://nas:${toString apiPort}"; + + # hosts/nas/state-services.nix's registry, same box, plain HTTP on the LAN. + registryEndpoint = "${nasLanAddress}:5000"; + + # ── CIDR arithmetic, so the overlap check is arithmetic and not a wish ── + ipToInt = + s: + let + o = map lib.toInt (lib.splitString "." s); + at = builtins.elemAt o; + in + (at 0) * 16777216 + (at 1) * 65536 + (at 2) * 256 + (at 3); + + cidrRange = + c: + let + parts = lib.splitString "/" c; + base = ipToInt (builtins.head parts); + bits = lib.toInt (builtins.elemAt parts 1); + # 2 ^ (32 - bits), without a pow in lib. + size = builtins.foldl' (a: _: a * 2) 1 (lib.range 1 (32 - bits)); + in + { + lo = base; + hi = base + size - 1; + }; + + overlaps = + a: b: + let + x = cidrRange a; + y = cidrRange b; + in + x.lo <= y.hi && y.lo <= x.hi; + + # ── The feature gates Substrate needs (see the header) ── + apiserverGates = [ + "ClusterTrustBundle=true" + "ClusterTrustBundleProjection=true" + "PodCertificateRequest=true" + ]; + controllerManagerGates = apiserverGates; + # Deliberately a SUBSET: see the honest part in the header. ClusterTrustBundle + # itself is apiserver-side, and an unrecognised gate name is fatal to kubelet. + kubeletGates = [ + "ClusterTrustBundleProjection=true" + "PodCertificateRequest=true" + ]; + + serverFlags = [ + "--cluster-cidr=${podCidr}" + "--service-cidr=${serviceCidr}" + # Cilium is the CNI (autoDeployCharts below). flannel off and k3s's own + # network policy controller off, because Cilium owns both. + "--flannel-backend=none" + "--disable-network-policy" + # The API certificate has to be valid for the name the agents dial. They + # dial `nas`, resolved by the static pins in modules/common.nix:130 and + # modules/fleet-hosts.nix, not by DNS. + "--tls-san=nas" + "--kube-apiserver-arg=--feature-gates=${lib.concatStringsSep "," apiserverGates}" + "--kube-apiserver-arg=--runtime-config=certificates.k8s.io/v1beta1=true" + "--kube-controller-manager-arg=--feature-gates=${lib.concatStringsSep "," controllerManagerGates}" + "--kubelet-arg=--feature-gates=${lib.concatStringsSep "," kubeletGates}" + ]; + + agentFlags = [ + "--kubelet-arg=--feature-gates=${lib.concatStringsSep "," kubeletGates}" + ]; + + # ── containerd: add runsc WITHOUT losing the stock config ────────────── + # `{{ template "base" . }}` is the module's own documented way to keep + # k3s's generated containerd configuration and append to it (the option's + # example in nixos/modules/services/cluster/rancher/default.nix:628-646 is + # literally "Add a custom runtime"). Dropping that line replaces the whole + # config and the node loses its CNI, its snapshotter and its registry + # mirrors at once. It is one line and it is load-bearing. + # + # runc.v2 with BinaryName pointing at runsc, rather than the runsc shim: + # runsc is an OCI runtime, the runc.v2 shim is already in k3s, and this is + # the shape the module documents. The alternative (runtime_type + # "io.containerd.runsc.v1", which needs containerd-shim-runsc-v1 on k3s's + # PATH -- the nixpkgs gvisor package does build it) is the upstream-preferred + # path and is the first thing to try if a sandbox refuses to start. Whether + # Claude Code itself survives inside gVisor at all is U7 and is measured + # separately tonight; this module only makes the class available. + containerdTemplate = '' + {{ template "base" . }} + + [plugins."io.containerd.grpc.v1.cri".containerd.runtimes."runsc"] + runtime_type = "io.containerd.runc.v2" + [plugins."io.containerd.grpc.v1.cri".containerd.runtimes."runsc".options] + BinaryName = "${pkgs.gvisor}/bin/runsc" + SystemdCgroup = true + ''; + + # The NAS registry is plain HTTP on the LAN (hosts/nas/state-services.nix + # explains why: this segment never leaves enp1s0 and a self-signed layer + # here adds a certificate to rotate and excludes no attacker). containerd + # will not talk to an HTTP registry unless told, and this is how it is told. + registriesYaml = '' + mirrors: + "${registryEndpoint}": + endpoint: + - "http://${registryEndpoint}" + configs: + "${registryEndpoint}": + tls: + insecure_skip_verify: true + ''; +in +{ + options.myK3sFleet = { + enable = lib.mkEnableOption "this host's membership in the fleet k3s cluster (2026-09-20 sandbox spike; Appendix J section 9)"; + + role = lib.mkOption { + type = lib.types.enum [ + "server" + "agent" + ]; + default = "agent"; + description = '' + server on the NAS (control plane only, schedules nothing); agent on + the Strix boxes (where the machines are). The default is the safe one: + an agent that dials a server it is not. The assertion below refuses a + host whose role and hostname disagree, so a forgotten `role` on the + appliance is an evaluation error and not a second control plane. + ''; + }; + + nodeLabels = lib.mkOption { + type = lib.types.attrsOf lib.types.str; + default = { }; + example = { + "fleet/role" = "desk"; + "fleet/kvm" = "true"; + }; + description = '' + Declarative node labels, applied by kubelet at registration so a node + that reboots comes back schedulable AND labelled. This is the whole + reason the labels are here and not in a `kubectl label` someone has to + remember. + ''; + }; + + ciliumVersion = lib.mkOption { + type = lib.types.str; + default = "1.18.14"; + description = "Cilium chart version. 1.18.14 is the latest of the 1.18 line; 1.19.8 exists and is the next step up."; + }; + + ciliumHash = lib.mkOption { + type = lib.types.str; + # MEASURED 2026-09-20 on the coordinator: built the chart's + # fixed-output derivation with lib.fakeHash and took the hash the + # mismatch reported, then rebuilt clean. Not guessed, and not left as + # fakeHash -- so the first switch does not fail on it. + default = "sha256-js/NLsDWeV+xlcrBc3giFaXltFTE5gey8Zaet5cqTWk="; + description = '' + Hash of the packaged Cilium chart. The module fetches the chart at + build time as a fixed-output derivation, so a wrong hash fails the + build with the right one in the message. + ''; + }; + }; + + config = lib.mkMerge [ + { + # ── OUTSIDE THE GATE, ON PURPOSE ──────────────────────────────────── + # These evaluate on every host whether or not k3s is enabled, so an edit + # to the CIDRs is caught at `nix eval` time rather than at outage time. + assertions = [ + { + assertion = !(overlaps podCidr lanCidr); + message = "modules/k3s-fleet.nix: the pod CIDR ${podCidr} overlaps the house LAN ${lanCidr}. k3s's own default (10.42.0.0/16) does exactly this and would take the NAS, the coordinator, the worker, the printer and the router with it. Pick a range outside the LAN."; + } + { + assertion = !(overlaps serviceCidr lanCidr); + message = "modules/k3s-fleet.nix: the service CIDR ${serviceCidr} overlaps the house LAN ${lanCidr}. Pick a range outside the LAN."; + } + { + assertion = !(overlaps podCidr serviceCidr); + message = "modules/k3s-fleet.nix: the pod CIDR ${podCidr} and the service CIDR ${serviceCidr} overlap each other."; + } + ]; + } + + (lib.mkIf cfg.enable { + assertions = [ + { + assertion = config.mySecrets.enable; + message = "myK3sFleet needs agenix delivery for secrets/k3s-token.age on this host."; + } + { + # The appliance is the only server, and the only server is the + # appliance. Mechanical, because `role` has a default and a + # forgotten one would otherwise be silent. + assertion = (config.networking.hostName == "nas") == (cfg.role == "server"); + message = "modules/k3s-fleet.nix: ${config.networking.hostName} has role \"${cfg.role}\". The NAS is the server and nothing else is; the Strix boxes are agents and nothing else is."; + } + ]; + + age.secrets.k3s-token = { + file = ../secrets/k3s-token.age; + mode = "0400"; + }; + + services.k3s = { + enable = true; + inherit (cfg) role; + tokenFile = config.age.secrets.k3s-token.path; + nodeLabel = lib.mapAttrsToList (k: v: "${k}=${v}") cfg.nodeLabels; + extraFlags = if cfg.role == "server" then serverFlags else agentFlags; + }; + + # Both sides of the cluster need to pull from the NAS registry. + environment.etc."rancher/k3s/registries.yaml".text = registriesYaml; + }) + + # ── THE SERVER: the NAS ─────────────────────────────────────────────── + (lib.mkIf (cfg.enable && cfg.role == "server") { + services.k3s = { + # Embedded etcd rather than the stock sqlite. One server today and one + # server for the foreseeable future, so this buys nothing operational; + # it buys the ability to add a second server later without a datastore + # migration on the box that is also the house router. + clusterInit = true; + # THE APPLIANCE SCHEDULES NOTHING. 8 threads and 22 GiB, running DNS, + # DHCP, the binary cache, Paperless, Immich and headscale. Control + # plane only; the machines are on the twins. + disableAgent = true; + disable = [ + "traefik" + "servicelb" + ]; + + # Cilium as CNI, in the generation rather than in a `cilium install` + # somebody has to remember after every reprovision. cilium-cli 0.19.6 + # is in the pin and stays in the toolbox for `cilium status` and + # `cilium connectivity test`, which are read verbs. + # + # kubeProxyReplacement with k8sServiceHost/k8sServicePort is what lets + # Cilium reach the API server before there is a CNI to reach it + # through: the chicken-and-egg that makes a half-configured cluster + # sit at "0/1 nodes ready" with no useful log line. + autoDeployCharts.cilium = { + name = "cilium"; + repo = "https://helm.cilium.io"; + version = cfg.ciliumVersion; + hash = cfg.ciliumHash; + values = { + ipam.mode = "kubernetes"; + kubeProxyReplacement = true; + k8sServiceHost = "nas"; + k8sServicePort = apiPort; + }; + }; + + manifests = { + # The gVisor class. Handler "runsc" matches the containerd runtime + # name added by containerdTemplate above on each AGENT -- a + # RuntimeClass is a cluster-scoped name for a per-node containerd + # runtime, so both halves have to agree and they are written in the + # same file for that reason. + gvisor-runtimeclass.content = { + apiVersion = "node.k8s.io/v1"; + kind = "RuntimeClass"; + metadata.name = "gvisor"; + handler = "runsc"; + }; + + # ── kata: DELIBERATELY NOT ENABLED ───────────────────────────── + # The pin has kata-runtime 3.32.0, which builds + # containerd-shim-kata-v2 with DEFAULT_HYPERVISOR=qemu and + # HYPERVISORS=qemu, so Kata on QEMU is available from the pin today + # and Kata on cloud-hypervisor is a makeFlags override rather than a + # new package (Appendix J section 9). What is NOT known is whether a + # Kata class under k3s's containerd actually runs the /process image + # as a micro-VM on Strix silicon. That is U14, measured tonight; its + # report is at + # ~/today/review/2026-09-20/sandbox-spike/ (the U14 agent's file). + # Read that before uncommenting. gVisor is the fallback class and is + # the one enabled above. + # + # kata-runtimeclass.content = { + # apiVersion = "node.k8s.io/v1"; + # kind = "RuntimeClass"; + # metadata.name = "kata"; + # handler = "kata-qemu"; + # }; + }; + }; + + # ── THE KUBE API IS LAN-ONLY, AND IS NEVER IN THE TUNNEL ─────────── + # Interface-scoped, same shape as every other door on this box. :6443 is + # the cluster's root credential surface; anything that can reach it can + # schedule a privileged pod on either Strix box. hosts/nas/cloudflared.nix + # carries the matching doctrine block and an assertion that refuses to + # route it, because a tunnel ingress bypasses this rule entirely. + networking.firewall.extraInputRules = '' + iifname "${nasLanInterface}" tcp dport ${toString apiPort} accept comment "kube API, LAN leg only, NEVER in the tunnel" + ''; + }) + + # ── THE AGENTS: coordinator and worker ─────────────────────────────── + (lib.mkIf (cfg.enable && cfg.role == "agent") { + services.k3s = { + inherit serverAddr; + containerdConfigTemplate = containerdTemplate; + }; + + # runsc on PATH for the k3s unit. Not strictly needed for the + # BinaryName form above, which uses an absolute store path, but it is + # what makes switching to runtime_type "io.containerd.runsc.v1" a + # one-line edit rather than a second debugging session. + systemd.services.k3s.path = [ pkgs.gvisor ]; + }) + ]; +} diff --git a/secrets.nix b/secrets.nix index 03f43d3aa..5ab9b0af9 100644 --- a/secrets.nix +++ b/secrets.nix @@ -92,6 +92,17 @@ in # NAS private media HTTPS: zone-limited DNS-01 token, no broad Wrangler OAuth # authority. Ciphertext is provisioned before enabling personal-https.nix. "secrets/nas-cloudflare-dns.age".publicKeys = editors ++ nasOnly; + # The k3s cluster join token (2026-09-20 sandbox spike, modules/k3s-fleet.nix). + # Read by all three cluster hosts: the NAS server mints the cluster from it + # and both Strix agents present it to join, so this is the one secret whose + # tier is genuinely "the appliance AND the twins". Written out rather than + # reusing `delivered`, which deliberately excludes the nas. + # + # THE CIPHERTEXT DOES NOT EXIST YET -- this is the recipient ACL, which is + # what `agenix -e` needs to mint it. A k3s token is any sufficiently long + # opaque string; `openssl rand -hex 32` is fine. Mint with: + # nix develop -c agenix -e secrets/k3s-token.age + "secrets/k3s-token.age".publicKeys = editors ++ delivered ++ nasOnly; # --- wifi PSK tier: the coordinator, whose Freebox uplink # (wlp192s0) is now declarative too (migrated from an imperative profile on # flash night — refs #37). Rekey after this change: nix develop -c agenix -r From 7ebb7400fe0eff4f7518634103350e91f8e69d7b Mon Sep 17 00:00:00 2001 From: mecattaf Date: Sun, 20 Sep 2026 23:39:34 +0200 Subject: [PATCH 03/37] k3s: fold in A5a's three measured corrections from the live 1.35.6 guest A5a stood the pinned k3s up inside a throwaway guest built from this exact nixpkgs revision the same night and measured three things this file had wrong. Report: ~/today/review/2026-09-20/sandbox-spike/U9-U13-U15-K3S.md. 1. THE THREE GATES ALONE SERVE NOTHING. Boot 5 held the gates on and dropped the runtime-config flag: all three kubernetes_feature_enabled metrics read 1 while /apis/certificates.k8s.io listed only v1 and api-resources returned NO_RESOURCES. A check that read only the metric would have reported a false pass. --kube-apiserver-arg=runtime-config=certificates.k8s.io/v1beta1=true is what turns the group version on, and podcertcontroller has nothing to talk to without it. It now sits beside the gates, and the flag values inside a -arg= carry no leading dashes of their own, which is A5a's measured working form. U9 is therefore ANSWERED YES on the pin: v1beta1 serves clustertrustbundles and podcertificaterequests. Appendix J section 9's fallback does not have to be taken and no newer k3s is needed. kubelet also accepts all three gate names, so the cautious subset this file used to hand it is gone. 2. THE NIXPKGS EXAMPLE'S CONTAINERD KEY PATH IS WRONG FOR THIS k3s. The bundled containerd is 2.2.5-k3s2 and the config it generates starts `version = 3` with runtimes under [plugins.'io.containerd.cri.v1.runtime'.containerd.runtimes.runc]. A grpc.v1.cri block would be parsed, accepted and silently ignored, which is the worst of the three outcomes. A5a's exact working table is used here. And runtime_path, NOT options.BinaryName. The BinaryName shape this file used is a trap that looks right: a pod under it reaches Running and stays 1/1 Running in kubelet's view, produces no logs at all, and kubectl exec fails with "in state stopped". The generic runc shim starts runsc but carries neither its stdio nor its state. containerd-shim-runsc-v1 is what goes here, and nixpkgs' gvisor builds it. 3. k3s DOES NOT FIND runsc ON ITS OWN. Not by auto-detection (with gvisor in systemPackages the generated config still carried only runc and runhcs-wcow-process) and not from the unit PATH, which the nixpkgs rancher module leaves empty but for zfs. Without the path line every sandbox dies at creation with `exec: "runsc": executable file not found in $PATH`. The line was already here; it now carries why, and A5a's caution that the assignment replaces rather than extends. Also corrected: this branch's earlier claim that identical gate-off derivation paths proved the module inert. They do not prove it. The flake embeds the configuration revision, so the toplevel tracks the commit. The sound check is that with the gate off, services.k3s.enable is false, no 6443 rule is emitted and no registries.yaml exists on any of the three hosts, which was measured directly. Co-Authored-By: Claude Opus 5 (1M context) --- modules/k3s-fleet.nix | 167 +++++++++++++++++++++++++++++++----------- 1 file changed, 123 insertions(+), 44 deletions(-) diff --git a/modules/k3s-fleet.nix b/modules/k3s-fleet.nix index d6aef8239..1c0e5a53e 100644 --- a/modules/k3s-fleet.nix +++ b/modules/k3s-fleet.nix @@ -52,7 +52,7 @@ # never installed, nothing here was wasted; if it is, nothing here has to # change. # -# ── THE FEATURE GATES, AND THE HONEST PART ─────────────────────────────── +# ── THE FEATURE GATES, AND WHAT IS NOW MEASURED ────────────────────────── # MEASURED from ~/Downloads/substrate, hack/create-kind-cluster.sh:104-112, # which is upstream's own comment on why they are not optional: # @@ -69,13 +69,33 @@ # clusterTrustBundle volume (manifests/ate-install/atelet.yaml:271-289), and # every Substrate component's mTLS identity comes from that signer. # -# WHAT IS NOT MEASURED, and must not be claimed until the first switch: which -# of these three gates each component actually accepts on k3s 1.35.6+k3s1. -# ClusterTrustBundle is an apiserver-side gate and a kubelet that is handed an -# unrecognised gate name refuses to start, so the kubelet below is given only -# the two that are its own. If the first switch produces a kubelet that will -# not start, the fix is in the kubeletGates list below and nowhere else. This -# is U9 and it is the single measurement that gates all of Track S. +# U9 IS ANSWERED, AND THE ANSWER IS YES. The first draft of this file said the +# opposite: that nothing about these gates was measured and that the first +# switch would tell us. A5a measured it the same night, inside a k3s +# 1.35.6+k3s1 guest built from this exact pin, and the answer is that the +# pinned k3s serves the group. MEASURED there: +# +# kubectl get --raw /apis/certificates.k8s.io/v1beta1 | jq -r .resources[].name +# clustertrustbundles +# podcertificaterequests +# podcertificaterequests/status +# kubectl api-resources | grep -i "trustbundle\|podcertificate" +# clustertrustbundles certificates.k8s.io/v1beta1 false ClusterTrustBundle +# podcertificaterequests certificates.k8s.io/v1beta1 true PodCertificateRequest +# +# So Appendix J section 9's step 1 success criterion is met on the pin, its +# fallback paragraph does not have to be taken, and no newer k3s is needed for +# this reason. k3s_1_36 (1.36.2+k3s1) is in both stable and unstable if one is +# ever wanted for another. +# +# TWO THINGS THAT WOULD HAVE BEEN WRONG WITHOUT THAT MEASUREMENT, both fixed +# in this file and both worth knowing before editing it: +# 1. The three gates ALONE serve nothing. See runtimeConfigFlag below: the +# metrics read 1 while the group version stays unserved, so a check that +# read only the metric would have reported a false pass. +# 2. kubelet accepts all three gate names, including ClusterTrustBundle. +# This file used to hand kubelet a subset out of caution. It no longer +# needs to. # # ── GATE OFF ───────────────────────────────────────────────────────────── # Every host lands with `enable = false`. secrets/k3s-token.age does not exist @@ -129,18 +149,38 @@ let x.lo <= y.hi && y.lo <= x.hi; # ── The feature gates Substrate needs (see the header) ── - apiserverGates = [ + # All three, on all three components. The earlier draft of this file gave + # kubelet only two of them, on the reasoning that ClusterTrustBundle is + # apiserver-side and an unrecognised gate name is fatal to kubelet. A5a + # MEASURED otherwise on 2026-09-20, inside a k3s 1.35.6+k3s1 guest built from + # this exact pin: "the apiserver, the controller manager and the kubelet all + # accept all three gate names. None of the three components refused an + # unknown gate, and the node reached Ready in about ten seconds." + substrateGates = [ "ClusterTrustBundle=true" "ClusterTrustBundleProjection=true" "PodCertificateRequest=true" ]; - controllerManagerGates = apiserverGates; - # Deliberately a SUBSET: see the honest part in the header. ClusterTrustBundle - # itself is apiserver-side, and an unrecognised gate name is fatal to kubelet. - kubeletGates = [ - "ClusterTrustBundleProjection=true" - "PodCertificateRequest=true" - ]; + + # ── THE FLAG THAT ACTUALLY DECIDES IT ───────────────────────────────── + # The three gates alone serve NOTHING. A5a's boot 5 held the gates on and + # dropped this line; MEASURED result: + # + # kubernetes_feature_enabled{name="ClusterTrustBundle",stage="BETA"} 1 + # kubernetes_feature_enabled{name="ClusterTrustBundleProjection"...} 1 + # kubernetes_feature_enabled{name="PodCertificateRequest",stage="BETA"} 1 + # kubectl get --raw /apis/certificates.k8s.io | jq -c .versions + # -> [{"groupVersion":"certificates.k8s.io/v1","version":"v1"}] + # kubectl api-resources | grep -i "trustbundle\|podcertificate" + # -> NO_RESOURCES + # + # So the gates flip to 1 while the group version stays unserved, and a check + # that read only the metric would have reported a false pass. The group + # version is turned on separately, by this flag, and podcertcontroller has + # nothing to talk to without it. It is quoted straight out of upstream's own + # kind config (hack/create-kind-cluster.sh:111-112, the `runtimeConfig` + # block, which is the part a reader skips). + runtimeConfigFlag = "runtime-config=certificates.k8s.io/v1beta1=true"; serverFlags = [ "--cluster-cidr=${podCidr}" @@ -153,40 +193,55 @@ let # dial `nas`, resolved by the static pins in modules/common.nix:130 and # modules/fleet-hosts.nix, not by DNS. "--tls-san=nas" - "--kube-apiserver-arg=--feature-gates=${lib.concatStringsSep "," apiserverGates}" - "--kube-apiserver-arg=--runtime-config=certificates.k8s.io/v1beta1=true" - "--kube-controller-manager-arg=--feature-gates=${lib.concatStringsSep "," controllerManagerGates}" - "--kubelet-arg=--feature-gates=${lib.concatStringsSep "," kubeletGates}" + # A5a's MEASURED working form: the value inside a `-arg=` carries NO leading + # dashes of its own. Both flag families are required and neither is + # sufficient alone; see runtimeConfigFlag above. + "--kube-apiserver-arg=feature-gates=${lib.concatStringsSep "," substrateGates}" + "--kube-apiserver-arg=${runtimeConfigFlag}" + "--kube-controller-manager-arg=feature-gates=${lib.concatStringsSep "," substrateGates}" + "--kubelet-arg=feature-gates=${lib.concatStringsSep "," substrateGates}" ]; agentFlags = [ - "--kubelet-arg=--feature-gates=${lib.concatStringsSep "," kubeletGates}" + "--kubelet-arg=feature-gates=${lib.concatStringsSep "," substrateGates}" ]; # ── containerd: add runsc WITHOUT losing the stock config ────────────── - # `{{ template "base" . }}` is the module's own documented way to keep - # k3s's generated containerd configuration and append to it (the option's - # example in nixos/modules/services/cluster/rancher/default.nix:628-646 is - # literally "Add a custom runtime"). Dropping that line replaces the whole - # config and the node loses its CNI, its snapshotter and its registry - # mirrors at once. It is one line and it is load-bearing. + # `{{ template "base" . }}` is the module's own documented way to keep k3s's + # generated containerd configuration and append to it. Dropping that line + # replaces the whole config and the node loses its CNI, its snapshotter and + # its registry mirrors at once. It is one line and it is load-bearing. + # + # ── THE KEY PATH BELOW IS NOT THE ONE THE NIXPKGS EXAMPLE SHOWS ─────── + # The option's example (nixos/modules/services/cluster/rancher/default.nix + # 628-646) documents + # [plugins."io.containerd.grpc.v1.cri".containerd.runtimes."custom"] + # and that path is WRONG for this k3s. A5a MEASURED, 2026-09-20, reading the + # config that k3s 1.35.6+k3s1 actually generates at + # /var/lib/rancher/k3s/agent/etc/containerd/config.toml: the file starts + # `version = 3` and its runtime table is + # [plugins.'io.containerd.cri.v1.runtime'.containerd.runtimes.runc] + # That is containerd 2.x config v3 (the node reports containerd://2.2.5-k3s2). + # A `grpc.v1.cri` block would be parsed, accepted and silently ignored, which + # is the worst of the three outcomes. Use the path below, and check it again + # the day the k3s pin moves a major version. # - # runc.v2 with BinaryName pointing at runsc, rather than the runsc shim: - # runsc is an OCI runtime, the runc.v2 shim is already in k3s, and this is - # the shape the module documents. The alternative (runtime_type - # "io.containerd.runsc.v1", which needs containerd-shim-runsc-v1 on k3s's - # PATH -- the nixpkgs gvisor package does build it) is the upstream-preferred - # path and is the first thing to try if a sandbox refuses to start. Whether - # Claude Code itself survives inside gVisor at all is U7 and is measured - # separately tonight; this module only makes the class available. + # ── AND runtime_path, NOT options.BinaryName ───────────────────────── + # The first draft of this file used runtime_type "io.containerd.runc.v2" with + # options.BinaryName pointing at an absolute runsc, which is the shape the + # nixpkgs example suggests and which looks right. A5a MEASURED it and it is a + # trap: a pod under that RuntimeClass reaches Running and STAYS "1/1 Running" + # in kubelet's view, produces NO LOGS AT ALL, and `kubectl exec` into it + # fails with `cannot execute in container ...: in state stopped`. The generic + # runc shim starts runsc but carries neither its stdio nor its state. Do not + # use BinaryName for gVisor. The nixpkgs gvisor package builds + # containerd-shim-runsc-v1 beside runsc, and that shim is what goes here. containerdTemplate = '' {{ template "base" . }} - [plugins."io.containerd.grpc.v1.cri".containerd.runtimes."runsc"] - runtime_type = "io.containerd.runc.v2" - [plugins."io.containerd.grpc.v1.cri".containerd.runtimes."runsc".options] - BinaryName = "${pkgs.gvisor}/bin/runsc" - SystemdCgroup = true + [plugins.'io.containerd.cri.v1.runtime'.containerd.runtimes.runsc] + runtime_type = "io.containerd.runsc.v1" + runtime_path = "${pkgs.gvisor}/bin/containerd-shim-runsc-v1" ''; # The NAS registry is plain HTTP on the LAN (hosts/nas/state-services.nix @@ -404,10 +459,34 @@ in containerdConfigTemplate = containerdTemplate; }; - # runsc on PATH for the k3s unit. Not strictly needed for the - # BinaryName form above, which uses an absolute store path, but it is - # what makes switching to runtime_type "io.containerd.runsc.v1" a - # one-line edit rather than a second debugging session. + # ── runsc ON THE k3s UNIT'S PATH: REQUIRED, NOT A CONVENIENCE ────── + # `runtime_path` above tells containerd where the SHIM is. The shim then + # execs `runsc` from its OWN $PATH, and the k3s unit has essentially + # none: A5a MEASURED that the nixpkgs rancher module sets + # `path = lib.optional config.boot.zfs.enabled config.boot.zfs.package` + # and nothing else (default.nix:918), so the unit PATH is empty by + # default and k3s relies on its own wrapper for iptables and friends. + # Without this line every sandbox fails at creation, MEASURED: + # + # Failed to create pod sandbox: rpc error: code = Unknown desc = + # failed to start sandbox "...": failed to create containerd task: + # failed to create shim task: OCI runtime create failed: + # exec: "runsc": executable file not found in $PATH + # + # Note also that gVisor is NOT auto-detected. A5a MEASURED that with + # `gvisor` in environment.systemPackages and runsc resolvable at + # /run/current-system/sw/bin/runsc, the generated containerd config + # still contained only `runc` and `runhcs-wcow-process`. k3s ships + # RuntimeClasses for crun, lunatic, nvidia, slight, spin, wasmedge, + # wasmer, wasmtime and wws out of the box, and none for gVisor. The + # runtime has to be declared, which is what this module does. + # + # CAUTION FOR THE NEXT EDITOR: this assignment REPLACES the unit PATH + # rather than extending a populated one. A5a MEASURED `systemctl show + # k3s -p Environment` afterwards containing only the two gvisor + # directories. It did not break kube-proxy or flannel there because the + # nixpkgs k3s package wraps its own binary with the tools it needs, but + # anyone adding a second entry should append rather than assume. systemd.services.k3s.path = [ pkgs.gvisor ]; }) ]; From 643a4196fb661279fa26bcd7db12a3da5b4a3122 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Mon, 21 Sep 2026 12:39:13 +0200 Subject: [PATCH 04/37] secrets: add k3s-token.age (encrypted, no plaintext in git) Co-Authored-By: Claude Opus 5 (1M context) --- secrets/k3s-token.age | 13 +++++++++++++ 1 file changed, 13 insertions(+) create mode 100644 secrets/k3s-token.age diff --git a/secrets/k3s-token.age b/secrets/k3s-token.age new file mode 100644 index 000000000..e186668d6 --- /dev/null +++ b/secrets/k3s-token.age @@ -0,0 +1,13 @@ +age-encryption.org/v1 +-> X25519 y+gqc8HC/ZfWgnn9o8usBYyz46P008G4aLATi5Yhwxs +APBAungjujrBhT1RdhkYk19c7SSF0+yJ6Nw5vF9vA8A +-> ssh-ed25519 60zAgA e7LN3K6I5q4UauscCH0kayD4sRgA/iDUYRnxJRZ4eDM +uz7fB+BFAz4jtzvlIiIrxD3pkFNj3TKAPwZ2twhcU6k +-> ssh-ed25519 b4VnIg Jq/hXKs8BWavUOUA/K3J79+D8Fs5zsCSD4psuaQRyzk +bn0yo1YA9YfAZNTfz0y9bTbnCdO8YKGYkwk1uV7+iiw +-> ssh-ed25519 TKMZIQ +mhfynCuc2kc80KiS+JWyMXtT9rw+akJTaE+nysgnkQ +zv2a3s/LeAL6ebGrex0KYlZXqLNm/eBk7lfU7Jm6pvk +-> ssh-ed25519 sKUETw t1+9AuwS24H6IK215BXRLA5zxxS/rkzkfH9ScKxymwo +bJN7c35gAZkaudCtRN6djvs6dk8y69DoZwfNC4U1Ezw +--- id+7Z5DGCuTJwOwmVDtMp++Eq3/x/B6db/3hkL7JFKs +|À™a=7ý¬| fò(·:X„8dP„Úø}29⪚ )=§Á!ç‚M_¯Zœ=$¨ÿŸ5û¡7ò¥Ñ_¤h�ÉtÚÛ fnþÊQíärLã‹jBÂX�ø Ñ39Úw­¥-d \ No newline at end of file From 13c947bc318888a3a66af5b2e1bd7a6093994308 Mon Sep 17 00:00:00 2001 From: Tom Mecattaf Date: Wed, 23 Sep 2026 01:03:40 +0200 Subject: [PATCH 05/37] ax: package google/ax v0.3.0, and gate kubectl behind myAxClient (OFF) pkgs/ax builds google/ax v0.3.0 pinned by commit d8ed0fe (tag v0.3.0), wired through overlays/default.nix and the flake's packages. list like every sibling package. modules/ax-client.nix puts kubectl and ax into environment.systemPackages on coordinator, worker and client when myAxClient.enable is true; it lands false on all three. Refs #453. The one judgement call: ax's go.mod opens `go 1.27.1` and Go refuses to build a module whose go directive is newer than the running toolchain (`go: go.mod requires go >= 1.27.1 (running go 1.27.0; GOTOOLCHAIN=local)`), with no network in the sandbox to fetch one. MEASURED 2026-09-23: nixpkgs has go_1_27 = 1.27rc2 and nixpkgs-fresh has 1.27.0, so neither pin already in this flake can build it. Adds one input, nixpkgs-go, pinned BY REVISION and supplying exactly one attribute to exactly one package - the same shape as the existing nixpkgs-paperless. Bumping nixpkgs-fresh instead was rejected on purpose: that input also carries the twins' linux 7.2 kernel. checks.ax-client-topology asserts the option exists on the three hosts that import the module, that it is false on all three, that kubectl and ax are absent from all three systemPackages, and that hosts/nas carries no such option at all. It goes red on the flip by design. No switch, no rebuild, no gate flipped, no cluster, no Substrate, no patches - this builds pristine v0.3.0. Co-Authored-By: Claude Opus 5 --- flake.lock | 17 ++++++++ flake.nix | 65 ++++++++++++++++++++++++++++ hosts/client/default.nix | 11 +++++ hosts/coordinator/default.nix | 11 +++++ hosts/worker/default.nix | 11 +++++ modules/ax-client.nix | 66 ++++++++++++++++++++++++++++ overlays/default.nix | 11 ++++- pkgs/ax/default.nix | 81 +++++++++++++++++++++++++++++++++++ 8 files changed, 272 insertions(+), 1 deletion(-) create mode 100644 modules/ax-client.nix create mode 100644 pkgs/ax/default.nix diff --git a/flake.lock b/flake.lock index bbddca031..91cc9f24c 100644 --- a/flake.lock +++ b/flake.lock @@ -737,6 +737,22 @@ "type": "github" } }, + "nixpkgs-go": { + "locked": { + "lastModified": 1790096505, + "narHash": "sha256-UbSc81aqsHU4mOficKbMitf3S7gFnJkYDAKfprVS1R8=", + "owner": "NixOS", + "repo": "nixpkgs", + "rev": "a251c42236bbff9f870fcdc513dac5873009c304", + "type": "github" + }, + "original": { + "owner": "NixOS", + "repo": "nixpkgs", + "rev": "a251c42236bbff9f870fcdc513dac5873009c304", + "type": "github" + } + }, "nixpkgs-lib": { "locked": { "lastModified": 1774748309, @@ -880,6 +896,7 @@ "nixos-hardware": "nixos-hardware", "nixpkgs": "nixpkgs_5", "nixpkgs-fresh": "nixpkgs-fresh", + "nixpkgs-go": "nixpkgs-go", "nixpkgs-paperless": "nixpkgs-paperless", "nixpkgs-stable": "nixpkgs-stable", "piri": "piri", diff --git a/flake.nix b/flake.nix index fcae9c295..2c252e0c9 100644 --- a/flake.nix +++ b/flake.nix @@ -27,6 +27,26 @@ # writing the lock. A plain local build uses the reviewed fallback revision. nixpkgs-fresh.url = "github:NixOS/nixpkgs/nixos-unstable-small"; + # nixpkgs-go — pins ONE attribute, `go_1_27`, for ONE package, pkgs/ax. + # Same shape and same reasoning as nixpkgs-paperless below: a single + # upstream that needs a version no pin this flake already carries can + # supply, named here so it is reviewable rather than hidden in the package. + # + # google/ax v0.3.0's go.mod opens `go 1.27.1`, and Go refuses outright to + # build a module whose `go` directive is newer than the running toolchain + # (`go: go.mod requires go >= 1.27.1 (running go 1.27.0; GOTOOLCHAIN=local)`), + # with no network in the sandbox to fetch one. MEASURED 2026-09-23: + # nixpkgs go 1.26.5, go_1_27 1.27rc2 too old + # nixpkgs-fresh go 1.26.7, go_1_27 1.27.0 too old, by one patch release + # this input go_1_27 1.27.1 exact + # + # Pinned BY REVISION, not by branch, deliberately: nixpkgs-fresh is a rolling + # resolver whose whole job is to advance, and a toolchain pin has the opposite + # job. Retire this input the moment nixpkgs-fresh's go_1_27 reaches 1.27.1 — + # `nix eval .#inputs.nixpkgs-fresh.legacyPackages.x86_64-linux.go_1_27.version` + # is the whole test — and point overlays/default.nix back at it. + nixpkgs-go.url = "github:NixOS/nixpkgs/a251c42236bbff9f870fcdc513dac5873009c304"; + # nixpkgs-stable — pins ONLY nixosConfigurations.nas (issue #135 ruling): # the NAS is a frozen self-sustaining appliance on standard stable nixpkgs, # maintained manually every few years. It never rides the unstable @@ -571,6 +591,9 @@ overlays.default = import ./overlays { torchRocm = inputs.nix-strix-halo.packages.${system}.torch-rocm; + # One attribute out of the nixpkgs-go input, for pkgs/ax only. An + # overlay cannot read `inputs`, so it is passed like torchRocm above. + go127 = inputs.nixpkgs-go.legacyPackages.${system}.go_1_27; }; nixosConfigurations = { @@ -646,6 +669,11 @@ speech-session parakeet-service academic-ocr + # `nix build .#ax` — the ax control plane's four binaries. Exposed + # because nothing installs it by default (modules/ax-client.nix + # lands with its gate OFF on every host), so this is the only way + # to build or inspect it without flipping a gate first. + ax brother-print-text call-diarize browser-desktop @@ -715,6 +743,43 @@ python3 repo/tests/qwen-speech/test_speech.py touch "$out" ''; + # ax-client-topology — the gate's rendered shape (modules/ax-client.nix, + # #453). `nix flake check --no-build` on its own proves only that + # the tree EVALUATES, and it would stay green through a merge resolution + # that dropped ../../modules/ax-client.nix from a host's imports, that + # flipped a gate, or that let the module reach hosts/nas. Every assertion + # below is eval-time, so each runs under --no-build: + # - the option EXISTS on the three interactive hosts, which is what + # proves the import survived (a dropped import makes the option + # undefined, not false); + # - it is FALSE on all three. This is the line that goes red on the + # flip, deliberately: the flip edits this check in the same commit, + # so no gate on this fleet can move without a reviewer seeing it; + # - kubectl and ax are therefore absent from all three systemPackages, + # asserted directly rather than inferred from the gate; + # - the NAS carries no myAxClient option at all — it does not import + # the module, it is pinned to nixpkgs-stable, and it is an appliance. + ax-client-topology = + let + hostCfg = host: self.nixosConfigurations.${host}.config; + gated = [ + "coordinator" + "worker" + "client" + ]; + in + assert builtins.all (host: (hostCfg host) ? myAxClient) gated; + assert builtins.all (host: (hostCfg host).myAxClient.enable == false) gated; + assert builtins.all ( + host: + !(builtins.elem pkgs.kubectl (hostCfg host).environment.systemPackages) + && !(builtins.elem pkgs.ax (hostCfg host).environment.systemPackages) + ) gated; + assert !((hostCfg "nas") ? myAxClient); + pkgs.runCommand "ax-client-topology" { } '' + touch "$out" + ''; + qwen-speech-topology = let coord = self.nixosConfigurations.coordinator.config; diff --git a/hosts/client/default.nix b/hosts/client/default.nix index 17b326c0e..f30969747 100644 --- a/hosts/client/default.nix +++ b/hosts/client/default.nix @@ -79,6 +79,10 @@ inputs.nixos-hardware.nixosModules.common-pc-laptop inputs.nixos-hardware.nixosModules.common-pc-laptop-ssd ../../modules/zenbook-duo-daemon.nix + # kubectl + the google/ax binaries, behind myAxClient.enable. Imported on + # all three interactive hosts, OFF on all three; read that module's header + # for the runbook and for what it deliberately does not declare. + ../../modules/ax-client.nix ]; networking.hostName = "client"; @@ -94,6 +98,13 @@ # 2026-08-21 return. mySecrets.enable = true; + # OFF, and it lands OFF (modules/ax-client.nix). There is no cluster on this + # fleet to point kubectl at and no Agent Substrate for ax to delegate to, so + # flipping this today installs two binaries with nothing to talk to. The flip + # is Tom's, one host at a time, and ax-client-topology in flake.nix goes red + # on it by design. + myAxClient.enable = false; + # ── names ────────────────────────────────────────────────────────────────── # `nas` is fleet-wide (modules/common.nix). The two twins are pinned here # by hand because modules/fleet-hosts.nix is twins-only and also deletes diff --git a/hosts/coordinator/default.nix b/hosts/coordinator/default.nix index 9bc36001e..6e004cde5 100644 --- a/hosts/coordinator/default.nix +++ b/hosts/coordinator/default.nix @@ -79,6 +79,10 @@ # user-bus tally-daemon.service (U-D13). Declared here, installed by U-D19's # switch — never hand-started (DEFERRED.md DF-U-D13-1). ../../modules/tally-b.nix + # kubectl + the google/ax binaries, behind myAxClient.enable. Imported on + # all three interactive hosts, OFF on all three; read that module's header + # for the runbook and for what it deliberately does not declare. + ../../modules/ax-client.nix ]; networking.hostName = "coordinator"; @@ -110,6 +114,13 @@ # backend; flips with the NAS's myNas.paperless.enable (2026-09-13). myNasClient.relayPaperless = true; + # OFF, and it lands OFF (modules/ax-client.nix). There is no cluster on this + # fleet to point kubectl at and no Agent Substrate for ax to delegate to, so + # flipping this today installs two binaries with nothing to talk to. The flip + # is Tom's, one host at a time, and ax-client-topology in flake.nix goes red + # on it by design. + myAxClient.enable = false; + # The rewrite's served kernel: ONE kernel, on the coordinator (spec §2.4 Q2 — # the worker twin is a ROW this kernel serves, not a second kernel), on the # system bus, against the rewrite's own state root. The live daemon on tom's diff --git a/hosts/worker/default.nix b/hosts/worker/default.nix index c18052618..8da7d470b 100644 --- a/hosts/worker/default.nix +++ b/hosts/worker/default.nix @@ -79,6 +79,10 @@ # resolves to loopback, which every distributed library happily binds — the # rank-1-hangs-forever failure. The NAS must NOT import this. ../../modules/fleet-hosts.nix + # kubectl + the google/ax binaries, behind myAxClient.enable. Imported on + # all three interactive hosts, OFF on all three; read that module's header + # for the runbook and for what it deliberately does not declare. + ../../modules/ax-client.nix ]; networking.hostName = "worker"; @@ -318,6 +322,13 @@ # first boot of the new closure — no flash, no host-key dance. mySecrets.enable = true; + # OFF, and it lands OFF (modules/ax-client.nix). There is no cluster on this + # fleet to point kubectl at and no Agent Substrate for ax to delegate to, so + # flipping this today installs two binaries with nothing to talk to. The flip + # is Tom's, one host at a time, and ax-client-topology in flake.nix goes red + # on it by design. + myAxClient.enable = false; + # ── Fleet candidate adoption (#354, 2026-09-13): ROLLING ───────────────── # The worker adopts the NAS's signed nightly candidate on its own when it is # safe to disturb (modules/update-adopt.nix). "Safe" here is one fact: the diff --git a/modules/ax-client.nix b/modules/ax-client.nix new file mode 100644 index 000000000..2c977169f --- /dev/null +++ b/modules/ax-client.nix @@ -0,0 +1,66 @@ +{ + config, + lib, + pkgs, + ... +}: +# ─── ax-client: kubectl and the ax binaries, on a host, behind a gate ─────── +# +# Issue #453, 2026-09-23. Puts `kubectl` and `pkgs.ax` (google/ax v0.3.0, +# pkgs/ax) into environment.systemPackages on the host that imports this module +# and sets `myAxClient.enable = true`. +# +# THE GATE LANDS OFF, on all three hosts that import it. Flipping it is a +# separate act and it is Tom's. Nothing here starts, schedules or contacts +# anything: both are client binaries that do nothing until invoked with a target. +# +# WHY IT EXISTS. `kubectl` is absent from every host on this fleet (MEASURED +# 2026-09-22), so the ax CLI has no way to reach a cluster even once ax itself is +# installed, and the two are useless apart. They are host-scoped rather than +# fleet-wide user packages because kubectl is: it reads /etc/kubernetes and a +# host kube context, and the NAS must never grow either. +# +# WHAT IS DELIBERATELY NOT DONE HERE, and each reason: +# - No cluster. No k3s, no kubelet, no API server, no control plane of any +# kind is declared by this module or anywhere else in this tree. The +# k3s-in-microVM route was NOT run on 2026-09-23; issue #453 records +# why (the 2026-09-20 precedent needed a MODIFIED copy of +# home/dot_local/bin/runtime-test carrying `--dev-bind /dev/kvm`, and the +# private global rules forbid running that experiment without the declared +# wrapper being changed first, reviewed, in its own change). +# - No Agent Substrate and no Redis. ax delegates every sandbox to Agent +# Substrate and is inert without it; Substrate has no flake and is not +# packaged (see pkgs/ax's header). This module installs the client half +# only, knowingly. +# - No kubeconfig, no context, no credential, no secret. Nothing in this +# module writes to /etc or to a home directory. +# - No service, no unit, no timer, no socket, no firewall hole. +# - hosts/nas does NOT import this module and must not: it is pinned to +# nixpkgs-stable with no home-manager, and it is an appliance. +# +# RUNBOOK, for the flip. On the host whose turn it is, one host at a time: +# 1. Set `myAxClient.enable = true;` in that host's hosts//default.nix, +# beside the `false` this module landed with. +# 2. `nix build --no-link .#ax` and `nix flake check --no-build` from the repo +# root. The ax-client-topology check in flake.nix asserts the gate's value +# per host, so it goes RED on the flip by design: update its expectation in +# the same commit, which is the point — the flip cannot be silent. +# 3. Tom switches. Agents do not. +# 4. `kubectl version --client` and `ax --help` on the box. Neither needs a +# cluster to answer, so both are safe first probes. +# 5. There is still no cluster to point either at. That is a later change and +# it is gated on issue #453 being resolved first. +let + cfg = config.myAxClient; +in +{ + options.myAxClient.enable = + lib.mkEnableOption "the ax control-plane client: kubectl plus the google/ax binaries, on this host"; + + config = lib.mkIf cfg.enable { + environment.systemPackages = [ + pkgs.kubectl + pkgs.ax + ]; + }; +} diff --git a/overlays/default.nix b/overlays/default.nix index 7586ddccf..4e2289d1a 100644 --- a/overlays/default.nix +++ b/overlays/default.nix @@ -1,4 +1,4 @@ -{ torchRocm }: +{ torchRocm, go127 }: final: prev: { qwentts = final.callPackage ../pkgs/qwentts.nix { }; parakeet-service = final.callPackage ../pkgs/parakeet-service { }; @@ -34,6 +34,15 @@ final: prev: { mactahoe-gtk-theme = final.callPackage ../pkgs/mactahoe-gtk-theme.nix { }; mactahoe-icon-theme = final.callPackage ../pkgs/mactahoe-icon-theme.nix { }; + # google/ax — the Kubernetes control plane for agent Tasks (v0.3.0, pinned by + # commit). Not in nixpkgs under any name, in any channel: upstream ships no nix + # packaging, no flake and no published image, so there is nothing to take. + # go_1_27 is threaded in from the nixpkgs-go input rather than resolved from + # this fixpoint because ax's go.mod requires exactly 1.27.1 and neither the + # main pin (1.27rc2) nor nixpkgs-fresh (1.27.0) has it — see that input's + # comment in flake.nix, and the `let` in pkgs/ax/default.nix. + ax = final.callPackage ../pkgs/ax { go_1_27 = go127; }; + # Backlog.md — markdown-native task manager CLI (`backlog`). Not in nixpkgs; # packaged from the upstream release binary (Bun compile). See pkgs/backlog-md.nix. backlog-md = final.callPackage ../pkgs/backlog-md.nix { }; diff --git a/pkgs/ax/default.nix b/pkgs/ax/default.nix new file mode 100644 index 000000000..2594a5b90 --- /dev/null +++ b/pkgs/ax/default.nix @@ -0,0 +1,81 @@ +{ + lib, + buildGoModule, + fetchFromGitHub, + git, + # The Go toolchain, passed in by overlays/default.nix rather than taken from + # this pkgs fixpoint. See the `go` binding below for why. + go_1_27, +}: +# google/ax — a Kubernetes control plane for agent Tasks, which delegates the +# sandbox to Agent Substrate rather than running workloads itself. Four commands +# ship in cmd/: `ax` (CLI), `ax-controller`, `ax-server`, `ax-task-runner`. +# +# NOT in nixpkgs, in any channel, under any name (checked 2026-09-22): upstream +# tagged v0.3.0 and has no nix packaging of its own, no flake, and no container +# publishing path a fleet could consume. This expression is the whole of it. +# +# Pinned by commit, not by tag, so a retag upstream cannot move what this fleet +# builds. d8ed0fe38bceb7842d3c47817d53d16ccdfcb601 IS tag v0.3.0 as of 2026-09-22. +# +# NO PATCHES. This builds pristine v0.3.0. ax hardcodes +# SandboxClass_SANDBOX_CLASS_GVISOR at internal/substrate/client.go:273 (the only +# SANDBOX_CLASS occurrence in the tree, MEASURED 2026-09-22 by driving a real +# control plane against a mock Substrate), and making the sandbox class per-Task +# is the first of the nix-side patches on the list. It belongs here as a +# `patches = [ ... ]` entry when it is written, so the upstream clone stays clean. +let + # go.mod's first directive is `go 1.27.1` (MEASURED). Go refuses to build a + # module whose `go` line is newer than the running toolchain, and the sandbox + # has no network to fetch one, so the toolchain must be at least 1.27.1: + # this flake's main nixpkgs pin go = 1.26.5, go_1_27 = 1.27rc2 too old + # nixpkgs-fresh (flake.nix:28) go = 1.26.7, go_1_27 = 1.27.0 too old + # so overlays/default.nix resolves go_1_27 from the `nixpkgs-go` input, which + # exists for this one attribute and nothing else. All three versions MEASURED + # 2026-09-23; see that input's comment in flake.nix. + buildGo127Module = buildGoModule.override { go = go_1_27; }; +in +buildGo127Module { + pname = "ax"; + version = "0.3.0"; + + src = fetchFromGitHub { + owner = "google"; + repo = "ax"; + rev = "d8ed0fe38bceb7842d3c47817d53d16ccdfcb601"; # = tag v0.3.0 + hash = "sha256-mGSQ4QsYLdeKDtVMBODCulqQQ0Ze0NjeADPhB6edaYU="; + }; + + # Obtained the ordinary way: build once with lib.fakeHash, read the "got:" + # line off the failure, paste it back. + vendorHash = "sha256-iC/X6Bg1M7Pn3dT1zWs2YxuPfgl9ZKNEYQsBisIQguY="; + + # subPackages left unset so all four commands build, matching upstream's + # `make build-binaries` plus the cross-compiled runner. -s -w mirrors the + # Makefile's ldflags. + ldflags = [ + "-s" + "-w" + ]; + + # Upstream's `make test` is `go test ./...`, and per docs/development.md it + # "Runs everything, including the mock Substrate gRPC server, in-memory store + # validation, and API server tests". It needs no cluster and no network, so it + # is the one cluster-free regression proof this package has. Kept ON. + doCheck = true; + + # MEASURED 2026-09-22: without git on PATH, two internal/workspace tests fail + # with `exit status 127 ... git: command not found` (TestSetupWorkspace_GitSubdir + # at setup_test.go:104, TestSetupWorkspace_GitDepth at setup_test.go:227). They + # shell out to git to build a throwaway repo. This is a packaging fix, not a + # source patch: nothing upstream is wrong. + nativeCheckInputs = [ git ]; + + meta = { + description = "Kubernetes control plane for agent Tasks, delegating sandboxes to Agent Substrate"; + homepage = "https://github.com/google/ax"; + license = lib.licenses.asl20; + platforms = lib.platforms.linux; + mainProgram = "ax"; + }; +} From 307ee686d5e174b329c2997939c7327bacaa1820 Mon Sep 17 00:00:00 2001 From: tom Date: Wed, 23 Sep 2026 01:29:16 +0200 Subject: [PATCH 06/37] pkgs/ax: carry sandbox-class.patch, a per-Task sandbox class Adds `string sandbox_class = 11` to TaskSpec, regenerates ax.pb.go with the same protoc-gen-go v1.36.11 upstream used, refuses unknown values in ValidateTask, and threads the value from the reconciler through BuildActorTemplate, replacing the SandboxClass_SANDBOX_CLASS_GVISOR hardcode at internal/substrate/client.go:273 (the only SANDBOX_CLASS occurrence in the tree). Empty means gvisor, so existing manifests behave exactly as before. The generated Go is in the patch on purpose: ax bridges YAML through protojson with unknown fields rejected, so a .proto-only edit would make every manifest naming sandboxClass fail strict decode. This does NOT deliver a workerd sandbox. Agent Substrate's SandboxClass enum has three members (UNSPECIFIED, GVISOR, MICROVM), so "gvisor" and "microvm" are the only values that can reach a real substrate. A workerd class needs an upstream Agent Substrate change that does not exist. vendorHash is unchanged: the patch touches no go.mod or go.sum line. Measured on the patched scratch tree with go 1.27.1: `go build ./...` rc 0, `go test ./...` rc 0. In this clone: `nix build --no-link .#ax` rc 0, with the patch applying to all six files and checkPhase passing. Co-Authored-By: Claude Opus 5 --- pkgs/ax/default.nix | 32 +++- pkgs/ax/sandbox-class.patch | 304 ++++++++++++++++++++++++++++++++++++ 2 files changed, 329 insertions(+), 7 deletions(-) create mode 100644 pkgs/ax/sandbox-class.patch diff --git a/pkgs/ax/default.nix b/pkgs/ax/default.nix index 2594a5b90..b1a5e9d93 100644 --- a/pkgs/ax/default.nix +++ b/pkgs/ax/default.nix @@ -18,12 +18,10 @@ # Pinned by commit, not by tag, so a retag upstream cannot move what this fleet # builds. d8ed0fe38bceb7842d3c47817d53d16ccdfcb601 IS tag v0.3.0 as of 2026-09-22. # -# NO PATCHES. This builds pristine v0.3.0. ax hardcodes -# SandboxClass_SANDBOX_CLASS_GVISOR at internal/substrate/client.go:273 (the only -# SANDBOX_CLASS occurrence in the tree, MEASURED 2026-09-22 by driving a real -# control plane against a mock Substrate), and making the sandbox class per-Task -# is the first of the nix-side patches on the list. It belongs here as a -# `patches = [ ... ]` entry when it is written, so the upstream clone stays clean. +# ONE PATCH, sandbox-class.patch, described at the `patches` entry below. It is +# the one the previous revision of this comment predicted: it makes the sandbox +# class per-Task instead of hardcoded. The upstream clone stays clean; the patch +# was extracted from a scratch copy of the fetched source. let # go.mod's first directive is `go 1.27.1` (MEASURED). Go refuses to build a # module whose `go` line is newer than the running toolchain, and the sandbox @@ -47,9 +45,29 @@ buildGo127Module { }; # Obtained the ordinary way: build once with lib.fakeHash, read the "got:" - # line off the failure, paste it back. + # line off the failure, paste it back. UNCHANGED by sandbox-class.patch, which + # touches no go.mod or go.sum line and so vendors the same module set + # (MEASURED 2026-09-23: the patched build reuses this hash). vendorHash = "sha256-iC/X6Bg1M7Pn3dT1zWs2YxuPfgl9ZKNEYQsBisIQguY="; + # sandbox-class.patch adds `string sandbox_class = 11` to TaskSpec, regenerates + # ax.pb.go with the same protoc-gen-go v1.36.11 upstream used, validates the + # value in ValidateTask, and threads it from the reconciler through + # BuildActorTemplate, replacing the SandboxClass_SANDBOX_CLASS_GVISOR hardcode + # at internal/substrate/client.go:273. Empty means gvisor, so every existing + # manifest behaves exactly as before. + # + # The generated Go is part of the patch on purpose: ax bridges YAML through + # protojson with unknown fields REJECTED, so a .proto-only edit would make + # every manifest naming sandboxClass fail strict decode. + # + # It does NOT give ax a workerd sandbox. Agent Substrate's SandboxClass enum + # has exactly three members (UNSPECIFIED, GVISOR, MICROVM; MEASURED 2026-09-23 + # from the vendored ateapipb), so "gvisor" and "microvm" are the only values + # that can reach a real substrate. A workerd class needs an upstream Agent + # Substrate change that does not exist, and ax cannot invent the enum member. + patches = [ ./sandbox-class.patch ]; + # subPackages left unset so all four commands build, matching upstream's # `make build-binaries` plus the cross-compiled runner. -s -w mirrors the # Makefile's ldflags. diff --git a/pkgs/ax/sandbox-class.patch b/pkgs/ax/sandbox-class.patch new file mode 100644 index 000000000..68675ea94 --- /dev/null +++ b/pkgs/ax/sandbox-class.patch @@ -0,0 +1,304 @@ +diff --git a/internal/controller/reconciler.go b/internal/controller/reconciler.go +index 31c399b..c4aebda 100644 +--- a/internal/controller/reconciler.go ++++ b/internal/controller/reconciler.go +@@ -164,11 +164,11 @@ func (r *TaskReconciler) Reconcile(ctx context.Context, task *v1alpha1.Task, gat + } + + // If a custom image, workspace, or extra environment is specified, provision or use a dedicated ActorTemplate +- if task.Spec != nil && (task.Spec.Image != "" || len(extraEnv) > 0) { ++ if task.Spec != nil && (task.Spec.Image != "" || task.Spec.GetSandboxClass() != "" || len(extraEnv) > 0) { + slog.Info("ensuring custom ActorTemplate for task", "image", task.Spec.Image) +- customTemplateName := taskTemplateName(task.Metadata.Name, task.Spec.Image, extraEnv) ++ customTemplateName := taskTemplateName(task.Metadata.Name, task.Spec.Image, task.Spec.GetSandboxClass(), extraEnv) + +- tmpl, err := r.client.EnsureActorTemplateWithImage(ctx, templateAtespace, templateName, atespace, customTemplateName, task.Spec.Image, extraEnv) ++ tmpl, err := r.client.EnsureActorTemplateWithImage(ctx, templateAtespace, templateName, atespace, customTemplateName, task.Spec.Image, task.Spec.GetSandboxClass(), extraEnv) + if err != nil { + slog.Warn("could not create custom ActorTemplate, falling back to default template", "error", err) + } else if tmpl != nil && tmpl.Metadata != nil { +@@ -386,9 +386,14 @@ func (r *TaskReconciler) lookupGeminiKey(ctx context.Context, atespace string) s + + // taskTemplateName derives the per-task ActorTemplate name from the task name and a + // digest of the image and container environment, so a spec change yields a new template. +-func taskTemplateName(taskName, image string, env map[string]string) string { ++func taskTemplateName(taskName, image, sandboxClass string, env map[string]string) string { + h := sha256.New() + h.Write([]byte(image)) ++ // Only mixed in when set, so that a task that does not name a sandbox class ++ // keeps the template name it had before the field existed. ++ if sandboxClass != "" { ++ h.Write([]byte("sandboxClass=" + sandboxClass + ";")) ++ } + keys := make([]string, 0, len(env)) + for k := range env { + keys = append(keys, k) +diff --git a/internal/substrate/client.go b/internal/substrate/client.go +index bacb078..2840ce2 100644 +--- a/internal/substrate/client.go ++++ b/internal/substrate/client.go +@@ -210,7 +210,7 @@ const ( + ) + + // BuildActorTemplate constructs a Substrate ActorTemplate based on the standard ate-env specification. +-func BuildActorTemplate(atespace, name, image string, envMap map[string]string, command []string, snapshotsBucket string) *ateapipb.ActorTemplate { ++func BuildActorTemplate(atespace, name, image string, envMap map[string]string, command []string, snapshotsBucket, sandboxClass string) *ateapipb.ActorTemplate { + if atespace == "" { + atespace = "default" + } +@@ -269,15 +269,38 @@ func BuildActorTemplate(atespace, name, image string, envMap map[string]string, + FromData: ateapipb.ResumeSource_RESUME_SOURCE_GOLDEN, + }, + }, +- SandboxConfig: &ateapipb.SandboxConfig{ +- SandboxClass: ateapipb.SandboxClass_SANDBOX_CLASS_GVISOR, +- ConfigName: "gvisor-default", +- }, ++ SandboxConfig: sandboxConfigFor(sandboxClass), ++ } ++} ++ ++// sandboxConfigFor maps an ax spec.sandboxClass onto the substrate's own ++// SandboxConfig. The empty string means gVisor, which is what this function's ++// caller hardcoded before the field existed, so the default is unchanged. ++// ++// config_name names a cluster-scoped substrate SandboxConfig object and must ++// match the class. "gvisor-default" is the name ax has always used; the microVM ++// name follows the same convention, and ax neither creates nor verifies it. ++// ++// An unrecognised value should never reach here: v1alpha1.ValidateTask refuses ++// it at admission. If one does, gVisor is the fallback because it is the ++// historical default and the only class whose config object ax has ever named ++// successfully. The substrate enum has no workerd member, so no value of ++// sandboxClass can select one. ++func sandboxConfigFor(sandboxClass string) *ateapipb.SandboxConfig { ++ if sandboxClass == v1alpha1.SandboxClassMicroVM { ++ return &ateapipb.SandboxConfig{ ++ SandboxClass: ateapipb.SandboxClass_SANDBOX_CLASS_MICROVM, ++ ConfigName: "microvm-default", ++ } ++ } ++ return &ateapipb.SandboxConfig{ ++ SandboxClass: ateapipb.SandboxClass_SANDBOX_CLASS_GVISOR, ++ ConfigName: "gvisor-default", + } + } + + // EnsureActorTemplateWithImage creates an ActorTemplate using the specified container image and optional environment variables. +-func (c *Client) EnsureActorTemplateWithImage(ctx context.Context, baseAtespace, baseTemplate, targetAtespace, targetTemplate, image string, extraEnv ...map[string]string) (*ateapipb.ActorTemplate, error) { ++func (c *Client) EnsureActorTemplateWithImage(ctx context.Context, baseAtespace, baseTemplate, targetAtespace, targetTemplate, image, sandboxClass string, extraEnv ...map[string]string) (*ateapipb.ActorTemplate, error) { + existing, err := c.GetActorTemplate(ctx, targetAtespace, targetTemplate) + if err == nil && existing != nil { + return existing, nil +@@ -290,7 +313,7 @@ func (c *Client) EnsureActorTemplateWithImage(ctx context.Context, baseAtespace, + } + } + +- tmpl := BuildActorTemplate(targetAtespace, targetTemplate, image, envMap, nil, "") ++ tmpl := BuildActorTemplate(targetAtespace, targetTemplate, image, envMap, nil, "", sandboxClass) + req := &ateapipb.CreateActorTemplateRequest{ + ActorTemplate: tmpl, + } +diff --git a/pkg/apis/v1alpha1/ax.pb.go b/pkg/apis/v1alpha1/ax.pb.go +index b1adb26..9a71dc0 100644 +--- a/pkg/apis/v1alpha1/ax.pb.go ++++ b/pkg/apis/v1alpha1/ax.pb.go +@@ -15,7 +15,7 @@ + // Code generated by protoc-gen-go. DO NOT EDIT. + // versions: + // protoc-gen-go v1.36.11 +-// protoc v7.34.1 ++// protoc v7.35.1 + // source: pkg/apis/v1alpha1/ax.proto + + package v1alpha1 +@@ -187,7 +187,23 @@ type TaskSpec struct { + Gateway *GatewayRef `protobuf:"bytes,8,opt,name=gateway,proto3" json:"gateway,omitempty"` + // debug enables the in-container guest services (process execution and file + // access) that back `ax ssh`. Off by default. +- Debug bool `protobuf:"varint,10,opt,name=debug,proto3" json:"debug,omitempty"` ++ Debug bool `protobuf:"varint,10,opt,name=debug,proto3" json:"debug,omitempty"` ++ // sandbox_class selects the sandbox runtime family the task's actor runs in. ++ // Empty means "gvisor", which is what every task got before this field ++ // existed, so the default is unchanged. ++ // ++ // The accepted values are exactly the members Agent Substrate's own ++ // SandboxClass enum offers, lowercased: "gvisor" and "microvm". Any other ++ // value is refused by ValidateTask rather than downgraded, because a task ++ // that silently runs in a weaker sandbox than it asked for is worse than a ++ // task that does not start. In particular there is no "workerd" value: the ++ // substrate enum has no such member, and ax cannot add one. ++ // ++ // "microvm" additionally requires a cluster-scoped substrate SandboxConfig ++ // object named "microvm-default", by analogy with the "gvisor-default" object ++ // the gVisor path has always named. ax neither creates nor verifies it; if it ++ // is absent the substrate rejects the ActorTemplate. ++ SandboxClass string `protobuf:"bytes,11,opt,name=sandbox_class,json=sandboxClass,proto3" json:"sandbox_class,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache + } +@@ -278,6 +294,13 @@ func (x *TaskSpec) GetDebug() bool { + return false + } + ++func (x *TaskSpec) GetSandboxClass() string { ++ if x != nil { ++ return x.SandboxClass ++ } ++ return "" ++} ++ + type EnvVar struct { + state protoimpl.MessageState `protogen:"open.v1"` + Name string `protobuf:"bytes,1,opt,name=name,proto3" json:"name,omitempty"` +@@ -3165,7 +3188,7 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + + "\x04kind\x18\x02 \x01(\tR\x04kind\x123\n" + + "\bmetadata\x18\x03 \x01(\v2\x17.ax.v1alpha1.ObjectMetaR\bmetadata\x12)\n" + + "\x04spec\x18\x04 \x01(\v2\x15.ax.v1alpha1.TaskSpecR\x04spec\x12/\n" + +- "\x06status\x18\x05 \x01(\v2\x17.ax.v1alpha1.TaskStatusR\x06status\"\xd4\x02\n" + ++ "\x06status\x18\x05 \x01(\v2\x17.ax.v1alpha1.TaskStatusR\x06status\"\xf9\x02\n" + + "\bTaskSpec\x12\x18\n" + + "\asuspend\x18\x02 \x01(\bR\asuspend\x12\x14\n" + + "\x05image\x18\x03 \x01(\tR\x05image\x12\x18\n" + +@@ -3177,7 +3200,8 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + + "workspaces\x121\n" + + "\agateway\x18\b \x01(\v2\x17.ax.v1alpha1.GatewayRefR\agateway\x12\x14\n" + + "\x05debug\x18\n" + +- " \x01(\bR\x05debugJ\x04\b\x01\x10\x02J\x04\b\t\x10\n" + ++ " \x01(\bR\x05debug\x12#\n" + ++ "\rsandbox_class\x18\v \x01(\tR\fsandboxClassJ\x04\b\x01\x10\x02J\x04\b\t\x10\n" + + "R\x04goalR\bpolicies\"2\n" + + "\x06EnvVar\x12\x12\n" + + "\x04name\x18\x01 \x01(\tR\x04name\x12\x14\n" + +diff --git a/pkg/apis/v1alpha1/ax.proto b/pkg/apis/v1alpha1/ax.proto +index 62bd57a..cf35316 100644 +--- a/pkg/apis/v1alpha1/ax.proto ++++ b/pkg/apis/v1alpha1/ax.proto +@@ -91,6 +91,22 @@ message TaskSpec { + // debug enables the in-container guest services (process execution and file + // access) that back `ax ssh`. Off by default. + bool debug = 10; ++ // sandbox_class selects the sandbox runtime family the task's actor runs in. ++ // Empty means "gvisor", which is what every task got before this field ++ // existed, so the default is unchanged. ++ // ++ // The accepted values are exactly the members Agent Substrate's own ++ // SandboxClass enum offers, lowercased: "gvisor" and "microvm". Any other ++ // value is refused by ValidateTask rather than downgraded, because a task ++ // that silently runs in a weaker sandbox than it asked for is worse than a ++ // task that does not start. In particular there is no "workerd" value: the ++ // substrate enum has no such member, and ax cannot add one. ++ // ++ // "microvm" additionally requires a cluster-scoped substrate SandboxConfig ++ // object named "microvm-default", by analogy with the "gvisor-default" object ++ // the gVisor path has always named. ax neither creates nor verifies it; if it ++ // is absent the substrate rejects the ActorTemplate. ++ string sandbox_class = 11; + } + + message EnvVar { +diff --git a/pkg/apis/v1alpha1/types.go b/pkg/apis/v1alpha1/types.go +index a65e626..820c19b 100644 +--- a/pkg/apis/v1alpha1/types.go ++++ b/pkg/apis/v1alpha1/types.go +@@ -36,6 +36,16 @@ const ( + + DefaultTaskImage = "gcr.io/ax-substrate/ate-images/ax-task-runner" + ++ // The sandbox classes TaskSpec.SandboxClass accepts. They are exactly the ++ // members of Agent Substrate's own SandboxClass enum, lowercased; ax does ++ // not own that enum and cannot add to it. ++ SandboxClassGVisor = "gvisor" ++ SandboxClassMicroVM = "microvm" ++ ++ // DefaultSandboxClass is what an empty spec.sandboxClass resolves to. It is ++ // the class every task ran in before the field existed. ++ DefaultSandboxClass = SandboxClassGVisor ++ + // PhaseTerminating marks a task whose deletion has been requested and whose + // actor is being torn down. The record disappears once cleanup completes. + PhaseTerminating = "Terminating" +@@ -252,6 +262,15 @@ func ValidateTask(t *Task) error { + if spec == nil { + return nil + } ++ switch spec.GetSandboxClass() { ++ case "", SandboxClassGVisor, SandboxClassMicroVM: ++ default: ++ // Refused, not downgraded: a task that quietly runs in a different ++ // sandbox class than it asked for is worse than a task that does not ++ // start. Agent Substrate offers no other class today. ++ return fmt.Errorf("spec.sandboxClass: %q is not a sandbox class Agent Substrate offers; supported values are %q and %q", ++ spec.GetSandboxClass(), SandboxClassGVisor, SandboxClassMicroVM) ++ } + refs := spec.WorkspaceRefs() + paths := spec.WorkspacePaths() + names := make(map[string]bool, len(refs)) +diff --git a/pkg/apis/v1alpha1/types_test.go b/pkg/apis/v1alpha1/types_test.go +index 83deaa4..2431b97 100644 +--- a/pkg/apis/v1alpha1/types_test.go ++++ b/pkg/apis/v1alpha1/types_test.go +@@ -137,6 +137,43 @@ func TestStrictDecoding_RejectsUnknownFields(t *testing.T) { + } + } + ++func TestTask_SandboxClass_YAML(t *testing.T) { ++ // The point of this test is that the field is real to protojson, not just ++ // present in the .proto: strict decoding rejects anything it does not know. ++ var task v1alpha1.Task ++ if err := yaml.Unmarshal([]byte("kind: Task\nspec:\n sandboxClass: microvm\n"), &task); err != nil { ++ t.Fatalf("decoding a manifest naming sandboxClass: %v", err) ++ } ++ if got := task.GetSpec().GetSandboxClass(); got != v1alpha1.SandboxClassMicroVM { ++ t.Errorf("spec.sandboxClass = %q, want %q", got, v1alpha1.SandboxClassMicroVM) ++ } ++ ++ out, err := yaml.Marshal(&task) ++ if err != nil { ++ t.Fatalf("marshalling: %v", err) ++ } ++ if !strings.Contains(string(out), "sandboxClass: microvm") { ++ t.Errorf("round trip lost the field, got:\n%s", out) ++ } ++ ++ // Empty stays empty rather than being rendered as the default, so existing ++ // manifests keep their existing YAML shape. ++ var plain v1alpha1.Task ++ if err := yaml.Unmarshal([]byte("kind: Task\nspec:\n image: img\n"), &plain); err != nil { ++ t.Fatalf("decoding a manifest without sandboxClass: %v", err) ++ } ++ if got := plain.GetSpec().GetSandboxClass(); got != "" { ++ t.Errorf("spec.sandboxClass = %q, want empty", got) ++ } ++ out, err = yaml.Marshal(&plain) ++ if err != nil { ++ t.Fatalf("marshalling: %v", err) ++ } ++ if strings.Contains(string(out), "sandboxClass") { ++ t.Errorf("unset field should not be rendered, got:\n%s", out) ++ } ++} ++ + func TestGateway_RoundTrip(t *testing.T) { + want := &v1alpha1.Gateway{ + ApiVersion: v1alpha1.APIVersion, +@@ -410,6 +447,19 @@ func TestValidateTask(t *testing.T) { + wantErr string + }{ + {name: "nil spec"}, ++ {name: "sandbox class unset", spec: &v1alpha1.TaskSpec{}}, ++ {name: "sandbox class gvisor", spec: &v1alpha1.TaskSpec{SandboxClass: v1alpha1.SandboxClassGVisor}}, ++ {name: "sandbox class microvm", spec: &v1alpha1.TaskSpec{SandboxClass: v1alpha1.SandboxClassMicroVM}}, ++ { ++ name: "sandbox class workerd", ++ spec: &v1alpha1.TaskSpec{SandboxClass: "workerd"}, ++ wantErr: `spec.sandboxClass: "workerd" is not a sandbox class Agent Substrate offers`, ++ }, ++ { ++ name: "sandbox class wrong case", ++ spec: &v1alpha1.TaskSpec{SandboxClass: "gVisor"}, ++ wantErr: `spec.sandboxClass: "gVisor" is not a sandbox class Agent Substrate offers`, ++ }, + {name: "no workspaces", spec: &v1alpha1.TaskSpec{}}, + {name: "list of one", spec: &v1alpha1.TaskSpec{Workspaces: []*v1alpha1.WorkspaceRef{{Name: "a"}}}}, + { From fdfe15b49ea7742b8629b43a796eadc8205de217 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 01:43:16 +0200 Subject: [PATCH 07/37] home/ax-conwip: a CONWIP scheduler user service, declared and OFF home/ax-conwip.nix declares `myAxConwip`, a home-manager module that WOULD run the CONWIP scheduler at /home/tom/mecattaf/ax-conwip as one long-running user service against a declared ax server, with the rewrite's seat meter directory as a read-only input. It is imported from home/home.nix and gated OFF, so it defines no unit on any host today. Seven options, every one of them data in this repository rather than a command line somebody types: enable (false), serverUrl (127.0.0.1:8080, the loopback the mock stack listens on, so an accidental enable reaches nothing real), metersDir (the REWRITE's meters, ~/.local/state/tally-rewrite/meters, never branch (a)'s, read only, no tmpfiles rule over it), wipCap (1, stricter than the program's own default of 2), sourceDir, recordsDir (a path that does not exist, so an accidental enable fails legibly instead of passing over an empty directory) and stateDir. Four things deliberately not done, and the module header says all four out loud: - no flake input, because the CONWIP repository has NO REMOTE. A fetchGit of a path on one box is not a declaration, it is a machine-local accident that would break every other host's eval; - no package, because packaging follows the input. The unit runs the checkout in place through `pnpm exec tsx`. That is a development shape, not a delivered one, and it is the biggest single reason the gate stays off; - no timer: the release signal is a WatchTask stream, not a poll; - nothing enabled and nothing armed. `enable` is set nowhere, and even with the gate flipped the unit carries NO Install section, so a rebuild declares it and a `systemctl --user start` is a second, separate act. checks.x86_64-linux.ax-conwip-topology asserts the option EXISTS on all three home-manager hosts (a dropped import makes it undefined, not false, which is the mutation hint), that it is false on all three, that no ax-conwip service, timer or socket is rendered anywhere, that there is no system-bus twin on any of the four hosts, and that no tmpfiles rule naming ax-conwip exists while the gate is off. The `enable == false` line goes red on the flip, deliberately. MEASURED in this tree: nix build --no-link .#checks.x86_64-linux.ax-conwip-topology -> rc 0, so every assertion above ran and passed nix eval .#nixosConfigurations.{coordinator,worker,client}.config .home-manager.users.tom.systemd.user.services --apply builtins.attrNames -> rc 0, no ax-conwip on any of the three the same eval under extendModules with the gate flipped renders the unit with no Install section, so the OFF branch is not a dead branch docs/ax-conwip.md carries what the CONWIP is, what the unit would run, the nine-step sequence to enable it, and the line that says the repository has no remote so nothing is packaged. Nothing here is enabled, nothing was started and nothing was switched. The flip is a separate act and it is Tom's. Refs #455 (the gate's own pre-existing bug: nix flake check does not reach a verdict on this repository). Co-Authored-By: Claude Opus 5 --- docs/ax-conwip.md | 229 +++++++++++++++++++++++++++++++++++++++++++++ flake.nix | 65 +++++++++++++ home/ax-conwip.nix | 225 ++++++++++++++++++++++++++++++++++++++++++++ home/home.nix | 1 + 4 files changed, 520 insertions(+) create mode 100644 docs/ax-conwip.md create mode 100644 home/ax-conwip.nix diff --git a/docs/ax-conwip.md b/docs/ax-conwip.md new file mode 100644 index 000000000..9754f4f98 --- /dev/null +++ b/docs/ax-conwip.md @@ -0,0 +1,229 @@ +# ax-conwip: the CONWIP scheduler as a user service + +Module: `home/ax-conwip.nix`. Option namespace: `myAxConwip`. Check: +`checks.x86_64-linux.ax-conwip-topology` in `flake.nix`. Written 2026-09-23. + +**Status: declared and OFF. `myAxConwip.enable` is set nowhere on this fleet, so +this module defines no unit on any host today.** The flake check asserts exactly +that, and it is written so that it goes red on the flip. Flipping it is Tom's. + +## What the CONWIP is + +`/home/tom/mecattaf/ax-conwip` is a CONWIP scheduler. CONWIP is constant +work-in-progress: a fixed number of slots, and a new item is admitted only when +a finished one gives its slot back. It is a cap on how much is in flight, not a +rate limit and not a queue discipline. + +This one: + +1. reads Claude ultracode workflow run records (`wf_*.json`); +2. derives one work item per `workflow_agent` entry of `workflowProgress`, never + one per workflow. Phases become dependency edges between items; +3. resolves each item's full prompt out of the record's own `script` string, by + scanning for the matching `agent(...)` call. It never parses and never + evaluates that script: the dialect opens with `export const meta`, closes + with a top-level `return`, and uses top-level `await`, which together mean no + single JavaScript goal symbol will load it. A label that matches zero calls, + or two, is a refusal with a named reason, not a guess; +4. admits items under a fixed work-in-progress cap, only on a free slot and only + when every dependency edge into the item is satisfied; +5. dispatches each admitted item to an ax server as a Task over gRPC, using + `UpdateTask`, which is an upsert. ax v0.3.0 has no `CreateTask` rpc at all; +6. gives the slot back when the Task reaches a phase the CONWIP calls terminal, + which is `{Completed, Failed, Terminating}`. That is deliberately NOT ax's own + terminal set, `{Running, Completed, Failed}`: a Running Task still occupies + work-in-progress, so the end of the `WatchTask` stream is a signal to read the + phase again, not a release; +7. writes an append-only ledger, one JSON object per line, deterministically + serialized. + +Read that repository's `DESIGN.md` first. It is the authority on all of the +above; this page only says what the module would run. + +## Nothing is packaged, and why + +**The `ax-conwip` repository has no remote.** It is a local git repository on the +coordinator, created with `git init`, and nothing has been pushed anywhere. + +The consequences are the reason this module has the shape it has: + +- **There is no flake input.** There is nothing for `inputs.ax-conwip` to point + at. A `builtins.fetchGit` of a path that exists on one box is not a + declaration; it is a machine-local accident that would break every other + host's evaluation. +- **There is therefore no package.** There is no `pkgs/ax-conwip.nix` and no + `.#ax-conwip`, because packaging follows the input. +- **So the unit runs a checkout in place.** `WorkingDirectory` is the + `sourceDir` option, and `ExecStart` is `pnpm exec tsx src/cli.ts`, resolved + against the `node_modules` that checkout already carries. That is a + development shape, not a delivered one, and it is the single biggest reason + `enable` must stay false. + +This is the first thing to fix if the CONWIP is ever to be a real lane: give the +repository a remote, declare it as a flake input, package it, and change +`ExecStart` to name a store path. Until then, do not flip the gate on a host you +care about. + +## What the module would run + +One long-running user service, `ax-conwip.service`, on whichever host has the +gate set. No timer: the release signal is a `WatchTask` stream, not a poll, so +there is nothing to wake on a cadence. + +`Type=simple`, `Restart=no`, `Nice=10`. `Restart=no` is deliberate: the +scheduler holds slots for the life of the process, so restarting it silently +would re-derive and re-admit work that is already in flight. If it dies, that is +a fact to read in the journal. + +**The unit carries no `Install` section.** Flipping `enable` DECLARES the unit; +it does not arm it and it does not start it. Starting it is a second, separate, +deliberate act. There are two gates here, not one, and that is on purpose. + +### The options, and their defaults + +| option | type | default | what it is | +|---|---|---|---| +| `enable` | bool | `false` | the gate. Set nowhere on this fleet. | +| `serverUrl` | str | `"127.0.0.1:8080"` | the ax server to dispatch to, as `host:port` | +| `metersDir` | path | `~/.local/state/tally-rewrite/meters` | the seat meters, READ ONLY | +| `wipCap` | positive int | `1` | how many Tasks may be admitted at once | +| `sourceDir` | path | `/home/tom/mecattaf/ax-conwip` | where the program lives, because it is not packaged | +| `recordsDir` | path | `~/.local/state/ax-conwip/records` | the `wf_*.json` run records to derive from | +| `stateDir` | path | `~/.local/state/ax-conwip` | this module's own state root | + +Three of those defaults are choices worth defending: + +- **`serverUrl` defaults to loopback**, specifically the address the mock stack + listens on by default (`ax-mockstack -addr 127.0.0.1:8080`). No default here + points at a live host and none ever should: an accidental enable on a box with + no mock stack running reaches nothing at all. Note that this is a gRPC target + and not a URL. The transport is plain h2c with insecure credentials, so there + is no scheme to write. +- **`wipCap` defaults to 1**, which is stricter than the program's own default + of 2 and stricter than the seat dry run's 3. A cap is the one number where the + conservative default costs only throughput. +- **`recordsDir` defaults to a path that does not exist**, so an accidental + enable fails legibly instead of passing silently over an empty directory. + +### The seat meters are an input and only an input + +`metersDir` defaults to the REWRITE's meters directory, +`~/.local/state/tally-rewrite/meters` — the one `home/seat-feeder.nix` declares +and its three timers write. Not `~/.local/state/tally/meters`, which is branch +(a)'s and is pinned by `SHA256SUMS`. The two estates never share a path. + +This module **never writes into either**. It declares no tmpfiles rule over the +meters directory and does not create it; `seat-feeder` owns that directory's +existence and its mode. The scheduler opens those files read-only. The flake +check asserts that no tmpfiles rule naming `ax-conwip` exists while the gate is +off. + +## The sequence to enable it + +Each step is separate and each is reversible. Do them in order. + +1. **Decide the host.** The seat meters are per-user and the feeder timers that + write them run on the coordinator only, so the coordinator is the only host + where the default `metersDir` has anything in it. + +2. **Put run records where the scheduler will look**, or point `recordsDir` at + where they already are. The default path does not exist; `src/cli.ts` exits 2 + without `--records`. + +3. **Set the option**, in the host's home-manager configuration: + + ```nix + myAxConwip = { + enable = true; + serverUrl = "127.0.0.1:8099"; # or wherever the ax server actually listens + wipCap = 1; + }; + ``` + +4. **Edit `checks.x86_64-linux.ax-conwip-topology` in `flake.nix` in the same + commit.** Its `enable == false` assertion goes red on the flip, deliberately, + so that no gate on this fleet moves without a reviewer seeing it. Do not + delete the check; narrow it to the hosts that are still off. + +5. **Check it evaluates before rebuilding anything:** + + ``` + nix build --no-link .#checks.x86_64-linux.ax-conwip-topology + nix eval .#nixosConfigurations.coordinator.config.home-manager.users.tom.systemd.user.services \ + --apply builtins.attrNames + ``` + + The second should now list `ax-conwip`. While the gate is off it does not, + which is this module's whole present claim. + +6. **Rebuild.** The usual switch, which is Tom's. + +7. **Look at the unit before starting it.** It has no `Install` section, so the + rebuild declares it and nothing wants it: + + ``` + systemctl --user cat ax-conwip.service + systemctl --user status ax-conwip.service # inactive (dead), as expected + ``` + +8. **Start it by hand, once, and watch it:** + + ``` + systemctl --user start ax-conwip.service + journalctl --user -u ax-conwip.service -f + ``` + + The ledger is printed to stdout and therefore lands in the journal. v1 keeps + it in memory and writes nothing to `stateDir`. + +9. **Stop it when you are done looking.** Nothing wants it, so a stop is final + until the next manual start. + +## What is deliberately not here + +- No flake input, because the repository has no remote. +- No package, because packaging follows the input. +- No timer, because the release signal is a stream. +- Nothing enabled and nothing armed: the gate is off, and even with the gate on + the unit has no `Install` section. +- No system-bus twin. This is a per-user scheduler reading per-user seat meters + and it must never acquire a system unit. The check asserts that on all four + hosts, the NAS included. +- No write to any meters directory, by this module or by the program. + +## Related + +- `home/seat-feeder.nix` — owns the rewrite's meters directory and its three + feeder timers. R44: read-only from here. +- `modules/ax-client.nix` — the `myAxClient` gate that puts `kubectl` and the + `ax` binaries on a host. Also off, also #454. +- dotfiles#455 — `nix flake check --no-build` aborts at + `nixosConfigurations.client` before the `checks` output is reached. Until that + is fixed, the check on this page must be built by hand: + `nix build --no-link .#checks.x86_64-linux.ax-conwip-topology`. + +## Unknowns and proposed defaults + +- **Whether the scheduler should read the seat meters at all in v1.** The + program's `DESIGN.md` section 10 lists "seat meters as an admission input" + under what was DROPPED for v1: it admits on slots and edges only, and the + meter refusal rule lives in the seat dry run, not in the loop. The module + passes `metersDir` in as `AX_CONWIP_METERS` anyway. Proposed default: keep + passing it, because the path is the thing worth declaring and reviewing, and + an unread environment variable costs nothing. Revisit when the loop grows a + meter gate. +- **Whether `recordsDir` should default to a real directory.** Proposed + default: no. A path that does not exist is a legible failure; an empty + directory that does exist is a silent pass. +- **Whether the ledger should be written to `stateDir` rather than printed.** + Proposed default: printed, for as long as v1 keeps it in memory. When it is + written, `stateDir` is where it goes and the tmpfiles rule already declared + under the gate is what creates the directory. +- **Whether `Restart` should stay `no`.** Proposed default: yes, until the + scheduler can reconcile against Tasks already in flight on the ax server. A + restart that re-derives and re-admits is worse than a process that stays dead + and legible. +- **Which host, if the gate is ever flipped.** Proposed default: the + coordinator, because it is the only host whose feeder timers write the meters + directory. Not measured against any real need; nobody has asked for this to + run yet. diff --git a/flake.nix b/flake.nix index 2c252e0c9..24f72682b 100644 --- a/flake.nix +++ b/flake.nix @@ -780,6 +780,71 @@ touch "$out" ''; + # ax-conwip-topology — the CONWIP gate's rendered shape + # (home/ax-conwip.nix, PR #454). Home Manager gives no `assertions` + # option, so the invariants over the RENDERED user units live here, + # the same reasoning as tally-filler-topology and tally-pump-topology + # further down this file. + # + # Every assertion below is eval-time and sits in front of the + # runCommand, so each runs under `--no-build`. What each one is for: + # - the option EXISTS on all three home-manager hosts, which is what + # proves the import survived: dropping ./ax-conwip.nix from + # home/home.nix's imports makes the option UNDEFINED, not false, + # and turns the first assert red. That is the mutation hint; + # - it is FALSE on all three, and this is the line that goes red on + # the flip, deliberately, so the flip edits this check in the same + # commit and no gate on this fleet moves without a reviewer; + # - NO `ax-conwip` user unit is rendered anywhere — service, timer or + # socket — asserted directly over the rendered attrsets rather than + # inferred from the gate, because `lib.mkIf false` removing the key + # is the property under test, not an assumption; + # - no system-bus twin: this is a per-user scheduler reading per-user + # seat meters, and it must never acquire a system unit; + # - it writes no tmpfiles rule while off, and in particular declares + # nothing over the REWRITE's meters directory, which belongs to + # home/seat-feeder.nix (R44) and is an input to this module and + # never an output; + # - the NAS carries no home-manager at all (flake.nix:603, + # `withHomeManager = false`), so it cannot carry this option; the + # assert pins that rather than leaving it to be rediscovered. + ax-conwip-topology = + let + homeHosts = [ + "coordinator" + "worker" + "client" + ]; + homeCfg = host: self.nixosConfigurations.${host}.config.home-manager.users.tom; + hostCfg = host: self.nixosConfigurations.${host}.config; + in + # the import survived, on every host that has home-manager. + assert builtins.all (host: (homeCfg host) ? myAxConwip) homeHosts; + # and the gate is OFF on every one of them. + assert builtins.all (host: (homeCfg host).myAxConwip.enable == false) homeHosts; + # therefore NOTHING is rendered: no service, no timer, no socket. + assert builtins.all (host: !((homeCfg host).systemd.user.services ? ax-conwip)) homeHosts; + assert builtins.all (host: !((homeCfg host).systemd.user.timers ? ax-conwip)) homeHosts; + assert builtins.all (host: !((homeCfg host).systemd.user.sockets ? ax-conwip)) homeHosts; + # no system-bus twin, on any host, including the NAS. + assert builtins.all ( + host: + !((hostCfg host).systemd.services ? ax-conwip) && !((hostCfg host).systemd.timers ? ax-conwip) + ) (homeHosts ++ [ "nas" ]); + # no tmpfiles rule of its own while off, and nothing at all naming + # the meters directory it only ever reads. + assert builtins.all ( + host: + !(builtins.any ( + r: nixpkgs.lib.hasInfix "ax-conwip" r + ) (homeCfg host).systemd.user.tmpfiles.rules) + ) homeHosts; + # the NAS has no home-manager, so it cannot carry the option. + assert !((hostCfg "nas") ? home-manager); + pkgs.runCommand "ax-conwip-topology" { } '' + touch "$out" + ''; + qwen-speech-topology = let coord = self.nixosConfigurations.coordinator.config; diff --git a/home/ax-conwip.nix b/home/ax-conwip.nix new file mode 100644 index 000000000..c6d3b98bc --- /dev/null +++ b/home/ax-conwip.nix @@ -0,0 +1,225 @@ +{ + config, + lib, + osConfig, + pkgs, + ... +}: +# ax-conwip — the CONWIP scheduler as ONE long-running user service, declared +# and OFF. +# +# UNIT: none yet; this is a skeleton, not a lane. ISSUE: dotfiles#455 is the +# gate's own bug (nix flake check never reaches `checks`), filed alongside this +# file; the module itself carries no issue of its own and rides PR #454. +# DATE: 2026-09-23. DOC: docs/ax-conwip.md. +# +# WHAT IT IS. `/home/tom/mecattaf/ax-conwip` is a CONWIP scheduler: it reads +# Claude ultracode workflow run records, derives one work item per agent entry, +# admits items under a fixed work-in-progress cap, dispatches each admitted item +# to an ax server as a Task over gRPC (`UpdateTask`, an upsert — ax v0.3.0 has +# no `CreateTask`), and gives the slot back when the Task reaches a phase the +# CONWIP calls terminal, which is `{Completed, Failed, Terminating}` and NOT +# ax's own `{Running, Completed, Failed}`. The output is an append-only ledger. +# Read that repository's DESIGN.md before touching anything here. +# +# WHY THIS FILE EXISTS. So the shape of running it is declared and reviewable +# BEFORE anything runs it, and so the argument list is data in this repository +# rather than a command line somebody types. Every number a running scheduler +# would obey — the cap, the server it dispatches to, the meters it reads — is an +# option below, set in one place, visible in one diff. +# +# THE SEAT METERS ARE AN INPUT AND ONLY AN INPUT. `metersDir` defaults to the +# REWRITE's meters directory, %h/.local/state/tally-rewrite/meters, the one +# home/seat-feeder.nix declares and its three timers write. Not +# ~/.local/state/tally/meters, which is branch (a)'s and is pinned by +# SHA256SUMS; the two estates never share a path. This module never writes into +# either: it declares no tmpfiles rule over the meters directory, it does not +# create it, and the scheduler opens those files read-only. seat-feeder owns +# that directory's existence and its mode, and R44 puts that file out of reach. +# +# FOUR THINGS THIS FILE DELIBERATELY DOES NOT DO. +# +# 1. NO FLAKE INPUT. `/home/tom/mecattaf/ax-conwip` is a local git repository +# with NO remote (MEASURED 2026-09-23: `git remote -v` is empty). There is +# nothing for `inputs.ax-conwip` to point at, and a `builtins.fetchGit` of +# a path on one box is not a declaration — it is a machine-local accident +# that would break every other host's eval. So the module names a +# DIRECTORY, `sourceDir`, and says out loud that it is a directory. +# 2. NO PACKAGE. There is no `pkgs/ax-conwip.nix` and no `.#ax-conwip`, +# because packaging follows the input, and there is no input. The unit +# below runs the checkout in place, through `pnpm exec tsx`, against the +# `node_modules` that checkout already carries. That is honest about what +# it is: a development shape, not a delivered one. It is the single +# biggest reason `enable` must stay false. +# 3. NO TIMER. The release signal is a `WatchTask` stream, not a poll, so the +# scheduler is a long-running service and there is nothing to wake on a +# cadence. Compare home/seat-feeder.nix, which is all timers and no +# long-running anything, because its job is freshness. +# 4. NOTHING ENABLED, AND NOTHING ARMED EITHER. `enable` defaults to false and +# is set nowhere, so this module defines NO unit on any host today; the +# flake check `ax-conwip-topology` asserts exactly that. And even with the +# gate flipped the unit carries NO `Install` section, so it is declared and +# not wanted by anything: `systemctl --user start ax-conwip` is a second, +# separate, deliberate act. Both flips are Tom's. docs/ax-conwip.md carries +# the runbook. +# +# ONE OPTION BEYOND THE SIX THE BRIEF NAMED, said plainly: `recordsDir`. The +# scheduler's entry point refuses to start without `--records ` +# (src/cli.ts prints usage and exits 2), so a skeleton without it could not form +# a command line at all, and a skeleton that cannot form a command line is not a +# skeleton, it is a comment. Its default is deliberately a path that does not +# exist, so an accidental enable produces a legible failure and not a silent +# pass over an empty directory. +let + cfg = config.myAxConwip; + + # A user unit inherits no interactive PATH and no profile. `node`, `pnpm` and + # `npm` are absent from PATH on both hosts anyway (MEASURED 2026-09-22), so + # naming the store paths here is not belt-and-braces, it is the only way the + # unit could ever run. + runtimePath = lib.makeBinPath [ + pkgs.nodejs_22 + pkgs.pnpm + pkgs.coreutils + ]; +in +{ + options.myAxConwip = { + enable = lib.mkEnableOption '' + the ax CONWIP scheduler as a user service. OFF, and set nowhere. Read + docs/ax-conwip.md before flipping this: the program is not packaged and + the unit runs a checkout in place + ''; + + serverUrl = lib.mkOption { + type = lib.types.str; + default = "127.0.0.1:8080"; + example = "127.0.0.1:8099"; + description = '' + The ax server the scheduler dispatches Tasks to, as `host:port`. This + is the program's own `--addr`, which is a gRPC target and NOT a URL: + the transport is plain h2c with insecure credentials, so there is no + scheme to write. The default is the LOOPBACK address the mock stack + listens on by default (`ax-mockstack -addr 127.0.0.1:8080`), chosen so + that an accidental enable on a box with no mock stack running reaches + nothing at all, and on a box with one running reaches only the mock. + No default here points at a live host, and none ever should. + ''; + }; + + metersDir = lib.mkOption { + type = lib.types.path; + default = "${config.home.homeDirectory}/.local/state/tally-rewrite/meters"; + description = '' + The seat meter directory, read-only, the rewrite's and not branch + (a)'s. home/seat-feeder.nix declares it and its timers write it; this + module only reads it and declares no tmpfiles rule over it. + ''; + }; + + wipCap = lib.mkOption { + type = lib.types.ints.positive; + default = 1; + description = '' + The work-in-progress cap: how many Tasks may be admitted at once. The + cap is data and is never derived by the scheduler. One, deliberately, + which is stricter than the program's own default of 2 and stricter + than the seat dry run's 3: a cap is the one number where the + conservative default costs only throughput, and the expensive + direction is the other one. + ''; + }; + + sourceDir = lib.mkOption { + type = lib.types.path; + default = "/home/tom/mecattaf/ax-conwip"; + description = '' + Where the CONWIP program lives, as a DIRECTORY on this box, because it + is not packaged and there is no flake input to package it from: the + repository has no remote. The unit runs `pnpm exec tsx src/cli.ts` + with this as its working directory, against the `node_modules` that + checkout already carries. Nothing in the Nix store is built from it + and nothing here pretends otherwise. + ''; + }; + + recordsDir = lib.mkOption { + type = lib.types.path; + default = "${config.home.homeDirectory}/.local/state/ax-conwip/records"; + description = '' + The directory of `wf_*.json` ultracode run records the scheduler + derives work items from. `src/cli.ts` exits 2 without it. The default + points at a path that does not exist today, so an accidental enable + fails legibly instead of passing silently over nothing. + ''; + }; + + stateDir = lib.mkOption { + type = lib.types.path; + default = "${config.home.homeDirectory}/.local/state/ax-conwip"; + description = '' + This module's own state root, under %h/.local/state/. v1's ledger is + in memory and printed to the journal, so nothing is written here yet; + the directory is declared because the ledger will land in it and + because a missing directory turns a first run into a silent no-op + rather than a legible failure (the same reasoning as dotfiles#292). + ''; + }; + }; + + # Gate OFF: mkIf false removes the attribute outright, so with `enable` + # unset this module contributes NO `systemd.user.services.ax-conwip` key at + # all, on any host. That is what `checks.ax-conwip-topology` asserts, and it + # is why adding this file to home/home.nix's imports changes no rendered + # byte on any host today. + config = lib.mkIf cfg.enable { + systemd.user.services.ax-conwip = { + Unit = { + Description = "ax CONWIP scheduler: admit ultracode work items under a fixed WIP cap"; + # Read back by `nix eval` the way home/seat-feeder.nix's X-Tally* keys + # are, so the declared cap and target cannot drift from whatever a + # future oracle reads. X- is systemd's extension space; the manager + # ignores these. + X-AxConwipCap = toString cfg.wipCap; + X-AxConwipServer = cfg.serverUrl; + X-AxConwipHost = osConfig.networking.hostName; + }; + + Service = { + # Long-running, not a oneshot: the release signal is a WatchTask + # stream. The scheduler holds slots for the whole life of the process, + # so restarting it silently would re-derive and re-admit. If it dies, + # that is a fact to read in the journal, not a thing to paper over. + Type = "simple"; + Restart = "no"; + Nice = 10; + WorkingDirectory = cfg.sourceDir; + Environment = [ + "PATH=${runtimePath}" + "AX_CONWIP_STATE=${cfg.stateDir}" + "AX_CONWIP_METERS=${cfg.metersDir}" + ]; + # The whole argument list, as data. Every value is an option above. + ExecStart = lib.escapeShellArgs [ + "${pkgs.pnpm}/bin/pnpm" + "exec" + "tsx" + "src/cli.ts" + "--records" + "${cfg.recordsDir}" + "--addr" + "${cfg.serverUrl}" + "--cap" + "${toString cfg.wipCap}" + ]; + }; + + # NO Install section, deliberately. See point 4 of the header: flipping + # the gate declares this unit; starting it is a separate act. + }; + + systemd.user.tmpfiles.rules = [ + "d ${cfg.stateDir} 0700 - - -" + ]; + }; +} diff --git a/home/home.nix b/home/home.nix index 9d1c0988b..519862679 100644 --- a/home/home.nix +++ b/home/home.nix @@ -122,6 +122,7 @@ in imports = [ ./browser-trust.nix ./ai-memory.nix + ./ax-conwip.nix # gate OFF; defines no unit on any host. docs/ax-conwip.md ./client-apps.nix ./harness-records.nix ./herdr.nix From b11b996250a9fa84a71cfdd7704df7310c066df1 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 03:51:56 +0200 Subject: [PATCH 08/37] ax-conwip: point the module at the serve entry point and pass the meters directory Co-Authored-By: Claude Opus 5 --- docs/ax-conwip.md | 40 +++++++++++++++++++---- flake.nix | 35 ++++++++++++++++++++ home/ax-conwip.nix | 81 +++++++++++++++++++++++++++++++++------------- 3 files changed, 127 insertions(+), 29 deletions(-) diff --git a/docs/ax-conwip.md b/docs/ax-conwip.md index 9754f4f98..509f54220 100644 --- a/docs/ax-conwip.md +++ b/docs/ax-conwip.md @@ -54,7 +54,7 @@ The consequences are the reason this module has the shape it has: - **There is therefore no package.** There is no `pkgs/ax-conwip.nix` and no `.#ax-conwip`, because packaging follows the input. - **So the unit runs a checkout in place.** `WorkingDirectory` is the - `sourceDir` option, and `ExecStart` is `pnpm exec tsx src/cli.ts`, resolved + `sourceDir` option, and `ExecStart` is `pnpm exec tsx src/serve.ts`, resolved against the `node_modules` that checkout already carries. That is a development shape, not a delivered one, and it is the single biggest reason `enable` must stay false. @@ -67,7 +67,27 @@ care about. ## What the module would run One long-running user service, `ax-conwip.service`, on whichever host has the -gate set. No timer: the release signal is a `WatchTask` stream, not a poll, so +gate set, running `src/serve.ts`. + +Until 2026-09-23 the unit's `ExecStart` ran `src/cli.ts`, which reads the records +directory once, runs the loop over what it found, prints its ledger and exits: a +`Type = oneshot` in everything but name. `src/serve.ts` is the program that makes +the unit's own claim true. It polls the records directory every +`--poll-interval-ms` (2000 by default, and this module passes no other value), +derives each new `wf_*.json` exactly once keyed by absolute path, admits under a +cap that HOLDS FOR THE LIFE OF THE PROCESS rather than per tick, reads +`--meters` for seat admission through the same pure refusal rule the seat dry run +uses, and exits 0 on SIGTERM after appending a final `stop` line to its ledger. +Dispatch is DRY RUN by default; the live path needs all three of `--live`, +`AX_CONWIP_LIVE_HALOGEN=1` and a seat on the `["halogen"]` allow list in the +program's own `src/seats.ts`, and this module never passes `--live`. + +Two startup refusals are worth knowing before flipping anything: a `recordsDir` +that does not exist is exit 2, not a watcher that polls nothing forever, and an +ax server unreachable at startup is exit 4, because `Restart = no` means a dead +process is a fact to read in the journal rather than a thing to paper over. + +No timer: the release signal is a `WatchTask` stream, not a poll, so there is nothing to wake on a cadence. `Type=simple`, `Restart=no`, `Nice=10`. `Restart=no` is deliberate: the @@ -85,10 +105,10 @@ deliberate act. There are two gates here, not one, and that is on purpose. |---|---|---|---| | `enable` | bool | `false` | the gate. Set nowhere on this fleet. | | `serverUrl` | str | `"127.0.0.1:8080"` | the ax server to dispatch to, as `host:port` | -| `metersDir` | path | `~/.local/state/tally-rewrite/meters` | the seat meters, READ ONLY | +| `metersDir` | path | `~/.local/state/tally-rewrite/meters` | the seat meters, READ ONLY. Passed both as `AX_CONWIP_METERS` and as the program's `--meters` | | `wipCap` | positive int | `1` | how many Tasks may be admitted at once | | `sourceDir` | path | `/home/tom/mecattaf/ax-conwip` | where the program lives, because it is not packaged | -| `recordsDir` | path | `~/.local/state/ax-conwip/records` | the `wf_*.json` run records to derive from | +| `recordsDir` | path | `~/.local/state/ax-conwip/records` | the `wf_*.json` run records to derive from, and to WATCH | | `stateDir` | path | `~/.local/state/ax-conwip` | this module's own state root | Three of those defaults are choices worth defending: @@ -127,8 +147,11 @@ Each step is separate and each is reversible. Do them in order. where the default `metersDir` has anything in it. 2. **Put run records where the scheduler will look**, or point `recordsDir` at - where they already are. The default path does not exist; `src/cli.ts` exits 2 - without `--records`. + where they already are. The default path does not exist; `src/serve.ts` exits + 2 without `--records`, and exits 2 again if the directory it is pointed at + does not exist. Records may also be dropped in AFTER the service is running: + that is the whole point of the serve entry point, and a file is derived + exactly once, so a record that is edited in place is not re-admitted. 3. **Set the option**, in the host's home-manager configuration: @@ -143,7 +166,10 @@ Each step is separate and each is reversible. Do them in order. 4. **Edit `checks.x86_64-linux.ax-conwip-topology` in `flake.nix` in the same commit.** Its `enable == false` assertion goes red on the flip, deliberately, so that no gate on this fleet moves without a reviewer seeing it. Do not - delete the check; narrow it to the hosts that are still off. + delete the check; narrow it to the hosts that are still off. Keep the + flipped-unit assertions that follow it: they pin that `ExecStart` names + `src/serve.ts`, carries `--meters`, does NOT carry `--live`, and that the + flipped unit still has no `Install` section. 5. **Check it evaluates before rebuilding anything:** diff --git a/flake.nix b/flake.nix index 24f72682b..ff2f38e00 100644 --- a/flake.nix +++ b/flake.nix @@ -841,6 +841,41 @@ ) homeHosts; # the NAS has no home-manager, so it cannot carry the option. assert !((hostCfg "nas") ? home-manager); + # AND, with the gate flipped IN MEMORY through extendModules — which + # changes no rendered byte on any host and writes nothing anywhere — + # the argument list the unit would run is the one the program accepts. + # + # Until 2026-09-23 `ExecStart` ran `src/cli.ts`, which reads the + # records directory once, prints its ledger and exits: this file + # declared a long-running service whose program was a oneshot in + # everything but name. It now runs `src/serve.ts`, which polls, + # admits under a cap that holds for the life of the process, reads + # the meters directory, and exits 0 on SIGTERM. These asserts are + # what keep the module's argument list and the program's contract + # from drifting apart again, and they are cheap and eval-time: + # - the entry point IS src/serve.ts and is NOT src/cli.ts; + # - `--meters` is passed, so the AX_CONWIP_METERS this unit already + # set is no longer read by nothing; + # - `--live` is ABSENT. The live dispatch path is gated three ways + # inside the program and this module must never be one of the + # ways in; + # - there is still NO Install section on the flipped unit, so + # declaring it and arming it stay two separate acts. + assert ( + let + flipped = self.nixosConfigurations.coordinator.extendModules { + modules = [ { home-manager.users.tom.myAxConwip.enable = true; } ]; + }; + unit = flipped.config.home-manager.users.tom.systemd.user.services.ax-conwip; + raw = unit.Service.ExecStart; + exec = if builtins.isList raw then builtins.concatStringsSep " " raw else raw; + in + nixpkgs.lib.hasInfix "src/serve.ts" exec + && !(nixpkgs.lib.hasInfix "src/cli.ts" exec) + && nixpkgs.lib.hasInfix "--meters" exec + && !(nixpkgs.lib.hasInfix "--live" exec) + && !(unit ? Install) + ); pkgs.runCommand "ax-conwip-topology" { } '' touch "$out" ''; diff --git a/home/ax-conwip.nix b/home/ax-conwip.nix index c6d3b98bc..1b1695c4d 100644 --- a/home/ax-conwip.nix +++ b/home/ax-conwip.nix @@ -51,9 +51,15 @@ # `node_modules` that checkout already carries. That is honest about what # it is: a development shape, not a delivered one. It is the single # biggest reason `enable` must stay false. -# 3. NO TIMER. The release signal is a `WatchTask` stream, not a poll, so the -# scheduler is a long-running service and there is nothing to wake on a -# cadence. Compare home/seat-feeder.nix, which is all timers and no +# 3. NO TIMER, and as of 2026-09-23 that is a statement about the PROGRAM and +# not only about this module. `ExecStart` runs `src/serve.ts`, which polls +# the records directory on its own interval and holds each admitted slot on +# a `WatchTask` stream until the CONWIP's release rule gives it back, so the +# cap holds for the life of the process. There is nothing for a timer to +# wake, and restarting on a cadence would re-derive and re-admit. Until this +# date the unit ran `src/cli.ts`, which reads the directory once and exits: +# a `Type = oneshot` in everything but name, and the mismatch this change +# closes. Compare home/seat-feeder.nix, which is all timers and no # long-running anything, because its job is freshness. # 4. NOTHING ENABLED, AND NOTHING ARMED EITHER. `enable` defaults to false and # is set nowhere, so this module defines NO unit on any host today; the @@ -65,11 +71,20 @@ # # ONE OPTION BEYOND THE SIX THE BRIEF NAMED, said plainly: `recordsDir`. The # scheduler's entry point refuses to start without `--records ` -# (src/cli.ts prints usage and exits 2), so a skeleton without it could not form -# a command line at all, and a skeleton that cannot form a command line is not a -# skeleton, it is a comment. Its default is deliberately a path that does not -# exist, so an accidental enable produces a legible failure and not a silent -# pass over an empty directory. +# (src/serve.ts prints usage and exits 2), so a skeleton without it could not +# form a command line at all, and a skeleton that cannot form a command line is +# not a skeleton, it is a comment. Its default is deliberately a path that does +# not exist, and `src/serve.ts` exits 2 on a records directory that is absent +# rather than polling it forever, so an accidental enable produces a legible +# failure and not a silent pass over nothing. +# +# NO SEVENTH OPTION WAS ADDED. `src/serve.ts` also takes `--poll-interval-ms`, +# `--max-ticks`, `--staleness-bound-seconds` and `--live`, and this module +# passes none of them: every one has a conservative default in the program, and +# an option is a thing Tom has to review. `--live` in particular is not passed +# and must never grow a way to be. Dispatch is dry run by default, and the live +# path needs all three of `--live`, `AX_CONWIP_LIVE_HALOGEN=1` and a seat on the +# `["halogen"]` allow list in the program's own `src/seats.ts`. let cfg = config.myAxConwip; @@ -114,6 +129,15 @@ in The seat meter directory, read-only, the rewrite's and not branch (a)'s. home/seat-feeder.nix declares it and its timers write it; this module only reads it and declares no tmpfiles rule over it. + + It is passed BOTH as `AX_CONWIP_METERS` below and as the program's + `--meters` flag. Until 2026-09-23 only the environment variable was + set and nothing read it: `src/cli.ts` took no `--meters` and read no + variable. `src/serve.ts` reads `/.json` + once per tick and lets the pure refusal rule decide admission, and a + refusal costs throughput and never a slot. The scheduler opens those + files read-only and its ledger writer refuses any path under this + directory outright. ''; }; @@ -136,7 +160,7 @@ in description = '' Where the CONWIP program lives, as a DIRECTORY on this box, because it is not packaged and there is no flake input to package it from: the - repository has no remote. The unit runs `pnpm exec tsx src/cli.ts` + repository has no remote. The unit runs `pnpm exec tsx src/serve.ts` with this as its working directory, against the `node_modules` that checkout already carries. Nothing in the Nix store is built from it and nothing here pretends otherwise. @@ -148,9 +172,12 @@ in default = "${config.home.homeDirectory}/.local/state/ax-conwip/records"; description = '' The directory of `wf_*.json` ultracode run records the scheduler - derives work items from. `src/cli.ts` exits 2 without it. The default - points at a path that does not exist today, so an accidental enable - fails legibly instead of passing silently over nothing. + derives work items from, and then WATCHES: `src/serve.ts` re-lists it + every poll interval and derives each new file exactly once, keyed by + absolute path. `src/serve.ts` exits 2 without this flag, and exits 2 + again if the directory does not exist, so the default below — a path + that does not exist today — makes an accidental enable fail legibly + instead of passing silently over nothing. ''; }; @@ -158,11 +185,13 @@ in type = lib.types.path; default = "${config.home.homeDirectory}/.local/state/ax-conwip"; description = '' - This module's own state root, under %h/.local/state/. v1's ledger is - in memory and printed to the journal, so nothing is written here yet; - the directory is declared because the ledger will land in it and - because a missing directory turns a first run into a silent no-op - rather than a legible failure (the same reasoning as dotfiles#292). + This module's own state root, under %h/.local/state/. Nothing is + written here yet: `src/serve.ts` puts its jsonl ledgers under the + checkout's own gitignored `out/ledger/`, so run output never becomes + source. The directory is declared because the ledger belongs here once + the program is packaged, and because a missing directory turns a first + run into a silent no-op rather than a legible failure (the same + reasoning as dotfiles#292). ''; }; }; @@ -186,10 +215,13 @@ in }; Service = { - # Long-running, not a oneshot: the release signal is a WatchTask - # stream. The scheduler holds slots for the whole life of the process, - # so restarting it silently would re-derive and re-admit. If it dies, - # that is a fact to read in the journal, not a thing to paper over. + # Long-running, not a oneshot, and `src/serve.ts` is the program that + # makes that true: it polls the records directory, admits under a cap + # that holds for the LIFE OF THE PROCESS, and gives a slot back only + # when the release rule says so. Restarting it silently would re-derive + # and re-admit. If it dies, that is a fact to read in the journal, not + # a thing to paper over. It exits 0 on SIGTERM, so a deliberate stop is + # not reported as a signal death. Type = "simple"; Restart = "no"; Nice = 10; @@ -200,17 +232,22 @@ in "AX_CONWIP_METERS=${cfg.metersDir}" ]; # The whole argument list, as data. Every value is an option above. + # Every flag here is one the program prints under `--print-flags`, and + # `checks.ax-conwip-topology` asserts the entry point and `--meters`. + # `--live` is absent and must stay absent. ExecStart = lib.escapeShellArgs [ "${pkgs.pnpm}/bin/pnpm" "exec" "tsx" - "src/cli.ts" + "src/serve.ts" "--records" "${cfg.recordsDir}" "--addr" "${cfg.serverUrl}" "--cap" "${toString cfg.wipCap}" + "--meters" + "${cfg.metersDir}" ]; }; From f2f339e276575531dd25097a4d4214f4043f051f Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 08:37:35 +0200 Subject: [PATCH 09/37] pkgs/ax: carry P1, a finished command frees its worker Stock ax v0.3.0 never learns that a Task's command exited, so the Task stays Running and its actor keeps a worker. The 2026-09-23 Substrate probe measured the consequence: finished Tasks hold every worker of a small pool and the next Task fails ResourceExhausted. p1-completion.patch (on top of sandbox-class.patch, both now under pkgs/ax/patches/): - runner serves /metadata/v1alpha1/ax/{exit,result,usage}; result capped 1 MiB - controller: terminal guard, exit read after resume, --running-resync (15s), Completed / Failed ExitCode=N write-back, result and usage copy, then SuspendActor; a CRASHED actor with no exit report is Failed ActorCrashed - API: TaskStatus.command, UsageStats.tool_calls, rpc GetTaskResult and `ax result task `; generated Go regenerated Tests run in checkPhase (doCheck stays true), including a four-Task floor test on a 2-worker mock pool and the no-exit-report contrast. vendorHash and the vendored Substrate client pin 672533541dbf are unchanged. Co-Authored-By: Claude Opus 5.5 --- pkgs/ax/default.nix | 47 +- pkgs/ax/patches/p1-completion.patch | 3545 +++++++++++++++++++++ pkgs/ax/{ => patches}/sandbox-class.patch | 0 3 files changed, 3585 insertions(+), 7 deletions(-) create mode 100644 pkgs/ax/patches/p1-completion.patch rename pkgs/ax/{ => patches}/sandbox-class.patch (100%) diff --git a/pkgs/ax/default.nix b/pkgs/ax/default.nix index b1a5e9d93..244a1ffea 100644 --- a/pkgs/ax/default.nix +++ b/pkgs/ax/default.nix @@ -18,10 +18,12 @@ # Pinned by commit, not by tag, so a retag upstream cannot move what this fleet # builds. d8ed0fe38bceb7842d3c47817d53d16ccdfcb601 IS tag v0.3.0 as of 2026-09-22. # -# ONE PATCH, sandbox-class.patch, described at the `patches` entry below. It is -# the one the previous revision of this comment predicted: it makes the sandbox -# class per-Task instead of hardcoded. The upstream clone stays clean; the patch -# was extracted from a scratch copy of the fetched source. +# TWO CARRIED PATCHES in ./patches, applied in order, described at the `patches` +# entry below: sandbox-class.patch (a per-Task sandbox class) and +# p1-completion.patch (a finished command frees its worker). The upstream clone +# stays clean; each patch was extracted from a scratch copy of the fetched +# source, one commit per patch. The vendored Substrate client stays at +# 672533541dbf: no patch touches go.mod or go.sum. let # go.mod's first directive is `go 1.27.1` (MEASURED). Go refuses to build a # module whose `go` line is newer than the running toolchain, and the sandbox @@ -45,8 +47,8 @@ buildGo127Module { }; # Obtained the ordinary way: build once with lib.fakeHash, read the "got:" - # line off the failure, paste it back. UNCHANGED by sandbox-class.patch, which - # touches no go.mod or go.sum line and so vendors the same module set + # line off the failure, paste it back. UNCHANGED by both patches, which + # touch no go.mod or go.sum line and so vendors the same module set # (MEASURED 2026-09-23: the patched build reuses this hash). vendorHash = "sha256-iC/X6Bg1M7Pn3dT1zWs2YxuPfgl9ZKNEYQsBisIQguY="; @@ -66,7 +68,38 @@ buildGo127Module { # from the vendored ateapipb), so "gvisor" and "microvm" are the only values # that can reach a real substrate. A workerd class needs an upstream Agent # Substrate change that does not exist, and ax cannot invent the enum member. - patches = [ ./sandbox-class.patch ]; + # + # p1-completion.patch (REQUIRED for ax on the fleet). Stock v0.3.0 never learns + # that a Task's command exited: the Task stays Running, its actor keeps its + # worker, and a small WorkerPool is exhausted after a few finished Tasks + # (MEASURED by the 2026-09-23 Substrate probe: the third Task on a 3-worker + # pool failed ResourceExhausted; evals-2026-09-23/substrate/probe-build.md 4). + # The patch: + # - runner: records the command's own exit and serves + # /metadata/v1alpha1/ax/{exit,result,usage} on the metadata port; result + # is the file at AX_RESULT_PATH (default /.ax/result.json), + # capped at 1 MiB (413 above that, never truncated); + # - controller: a terminal guard (Completed, or Failed with Ready reason + # CommandExited, is never resumed again), an exit read after each resume, + # and a --running-resync loop (default 15s) that re-checks Running Tasks, + # because nothing publishes an event when a command exits. On exit it + # writes Completed (0) or Failed (Ready False CommandExited, ExitCode=N), + # TaskStatus.command, usage, stores the result, then SuspendActor frees + # the worker. A CRASHED actor with no exit report becomes Failed + # ActorCrashed instead of being silently recreated; + # - API: TaskStatus.command = 8, UsageStats.tool_calls = 3, rpc + # GetTaskResult (store key task-result::), and + # `ax result task `. ax.pb.go and ax_grpc.pb.go are regenerated + # with protoc-gen-go v1.36.11 and protoc-gen-go-grpc v1.6.2. + # Its tests run in checkPhase below: the exit write-back for 0 and 3, the + # terminal guard (the flipped probe TestProbe_CompletedWriteBackIsOverwritten), + # the resync, a crashed actor, the runner's endpoints, and the floor test: + # four Tasks in a row on a 2-worker pool all Completed, next to the contrast + # that without an exit report the third is refused ResourceExhausted. + patches = [ + ./patches/sandbox-class.patch + ./patches/p1-completion.patch + ]; # subPackages left unset so all four commands build, matching upstream's # `make build-binaries` plus the cross-compiled runner. -s -w mirrors the diff --git a/pkgs/ax/patches/p1-completion.patch b/pkgs/ax/patches/p1-completion.patch new file mode 100644 index 000000000..9176e8464 --- /dev/null +++ b/pkgs/ax/patches/p1-completion.patch @@ -0,0 +1,3545 @@ +diff --git a/cmd/ax-controller/main.go b/cmd/ax-controller/main.go +index 9d1020e..f882e2d 100644 +--- a/cmd/ax-controller/main.go ++++ b/cmd/ax-controller/main.go +@@ -21,6 +21,7 @@ import ( + "os" + "os/signal" + "syscall" ++ "time" + + "github.com/google/ax/internal/controller" + "github.com/google/ax/internal/store/redis" +@@ -42,6 +43,7 @@ func main() { + redisPassword string + redisGroup string + redisConsumer string ++ runningResync time.Duration + ) + + flag.StringVar(&redisAddr, "redis-addr", "localhost:6379", "Redis server address (e.g. localhost:6379)") +@@ -56,6 +58,7 @@ func main() { + flag.BoolVar(&substratePlaintext, "substrate-plaintext", false, "Use insecure plaintext gRPC connection to Substrate") + flag.StringVar(&defaultTemplate, "template", "default-template", "Default Substrate ActorTemplate name") + flag.StringVar(&defaultTemplateAtespace, "template-atespace", "ax-system", "Default Substrate ActorTemplate atespace") ++ flag.DurationVar(&runningResync, "running-resync", controller.DefaultRunningResync, "How often Running tasks are re-checked for a command exit (0 disables)") + flag.Parse() + + if envRedis := os.Getenv("REDIS_ADDR"); envRedis != "" { +@@ -104,6 +107,7 @@ func main() { + + rStore := redis.NewStore(rClient, redis.Options{}) + worker := controller.NewWorker(rStore, reconciler, redisGroup, redisConsumer) ++ worker.RunningResync = runningResync + if err := worker.Run(ctx); err != nil && err != context.Canceled { + slog.Error("redis worker stopped with error", "error", err) + os.Exit(1) +diff --git a/cmd/ax/main.go b/cmd/ax/main.go +index 7b2922e..fff6379 100644 +--- a/cmd/ax/main.go ++++ b/cmd/ax/main.go +@@ -147,6 +147,8 @@ func main() { + err = runDelete(serverURL, atespace, cleanArgs) + case "ssh": + err = runSSH(serverURL, atespace, kubeContext, cleanArgs) ++ case "result": ++ err = runResult(serverURL, atespace, cleanArgs) + default: + fmt.Fprintf(os.Stderr, "unknown command: %s\n", cmd) + printUsage() +@@ -183,6 +185,7 @@ Available Commands: + ssh [-- cmd] Run a command or shell inside the running task container + suspend task Suspend execution of a task and checkpoint state + resume task Resume execution of a suspended task ++ result task Print the result file a finished task's command wrote + delete task Delete a task + delete gateway Delete a gateway + delete workspace Delete a workspace +@@ -1228,3 +1231,33 @@ func runSSH(serverURL, atespace, kubeContext string, args []string) error { + + return nil + } ++ ++// runResult prints the result content the controller copied from a finished ++// task's sandbox, byte for byte, to stdout. ++func runResult(serverURL, atespace string, args []string) error { ++ name := "" ++ switch { ++ case len(args) == 1: ++ name = args[0] ++ case len(args) >= 2 && (args[0] == "task" || args[0] == "tasks"): ++ name = args[1] ++ default: ++ return fmt.Errorf("usage: ax result task ") ++ } ++ ++ client, conn, err := getAXClient(serverURL) ++ if err != nil { ++ return err ++ } ++ defer conn.Close() ++ ++ ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) ++ defer cancel() ++ ++ res, err := client.GetTaskResult(ctx, &v1alpha1.GetTaskResultRequest{Atespace: atespace, Name: name}) ++ if err != nil { ++ return fmt.Errorf("getting task result: %w", err) ++ } ++ _, err = os.Stdout.Write(res.GetContent()) ++ return err ++} +diff --git a/internal/controller/completion_test.go b/internal/controller/completion_test.go +new file mode 100644 +index 0000000..d286b42 +--- /dev/null ++++ b/internal/controller/completion_test.go +@@ -0,0 +1,396 @@ ++// Copyright 2026 Google LLC ++// ++// Licensed under the Apache License, Version 2.0 (the "License"); ++// you may not use this file except in compliance with the License. ++// You may obtain a copy of the License at ++// ++// http://www.apache.org/licenses/LICENSE-2.0 ++// ++// Unless required by applicable law or agreed to in writing, software ++// distributed under the License is distributed on an "AS IS" BASIS, ++// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. ++// See the License for the specific language governing permissions and ++// limitations under the License. ++ ++package controller_test ++ ++import ( ++ "context" ++ "encoding/json" ++ "fmt" ++ "net" ++ "net/http" ++ "net/http/httptest" ++ "sync" ++ "sync/atomic" ++ "testing" ++ "time" ++ ++ "github.com/agent-substrate/substrate/pkg/proto/ateapipb" ++ "github.com/google/ax/internal/controller" ++ "github.com/google/ax/internal/metadata" ++ "github.com/google/ax/internal/store" ++ "github.com/google/ax/internal/store/memory" ++ "github.com/google/ax/internal/substrate" ++ "github.com/google/ax/pkg/apis/v1alpha1" ++ "google.golang.org/grpc" ++ "google.golang.org/grpc/codes" ++ "google.golang.org/grpc/credentials/insecure" ++ "google.golang.org/grpc/status" ++) ++ ++// poolControl is a Substrate mock with a fixed number of workers. A resumed ++// actor holds a worker until it is suspended or deleted; with every worker ++// held, ResumeActor fails ResourceExhausted, as the real scheduler does. ++type poolControl struct { ++ ateapipb.UnimplementedControlServer ++ mu sync.Mutex ++ workers int ++ workerIP string ++ running map[string]bool ++ crashed map[string]bool ++ resumes map[string]int ++ suspends map[string]int ++} ++ ++func newPoolControl(workers int, workerIP string) *poolControl { ++ return &poolControl{workers: workers, workerIP: workerIP, running: map[string]bool{}, crashed: map[string]bool{}, resumes: map[string]int{}, suspends: map[string]int{}} ++} ++ ++func (m *poolControl) CreateAtespace(ctx context.Context, req *ateapipb.CreateAtespaceRequest) (*ateapipb.Atespace, error) { ++ return &ateapipb.Atespace{Metadata: req.GetAtespace().GetMetadata()}, nil ++} ++ ++func (m *poolControl) CreateActorTemplate(ctx context.Context, req *ateapipb.CreateActorTemplateRequest) (*ateapipb.ActorTemplate, error) { ++ return req.GetActorTemplate(), nil ++} ++ ++func (m *poolControl) GetActorTemplate(ctx context.Context, req *ateapipb.GetActorTemplateRequest) (*ateapipb.ActorTemplate, error) { ++ return &ateapipb.ActorTemplate{Metadata: &ateapipb.ResourceMetadata{Atespace: req.GetActorTemplate().GetAtespace(), Name: req.GetActorTemplate().GetName()}}, nil ++} ++ ++func (m *poolControl) CreateActor(ctx context.Context, req *ateapipb.CreateActorRequest) (*ateapipb.Actor, error) { ++ return &ateapipb.Actor{Metadata: req.GetActor().GetMetadata(), Status: &ateapipb.ActorStatus{State: ateapipb.ActorState_ACTOR_STATE_SUSPENDED}}, nil ++} ++ ++func (m *poolControl) CreateActorEgressPolicy(ctx context.Context, req *ateapipb.CreateActorEgressPolicyRequest) (*ateapipb.EgressPolicy, error) { ++ return &ateapipb.EgressPolicy{}, nil ++} ++ ++func (m *poolControl) GetActor(ctx context.Context, req *ateapipb.GetActorRequest) (*ateapipb.Actor, error) { ++ m.mu.Lock() ++ defer m.mu.Unlock() ++ name := req.GetActor().GetName() ++ state := ateapipb.ActorState_ACTOR_STATE_SUSPENDED ++ switch { ++ case m.crashed[name]: ++ state = ateapipb.ActorState_ACTOR_STATE_CRASHED ++ case m.running[name]: ++ state = ateapipb.ActorState_ACTOR_STATE_RUNNING ++ } ++ return &ateapipb.Actor{Metadata: &ateapipb.ResourceMetadata{Name: name}, Status: &ateapipb.ActorStatus{State: state}}, nil ++} ++ ++func (m *poolControl) ResumeActor(ctx context.Context, req *ateapipb.ResumeActorRequest) (*ateapipb.ResumeActorResponse, error) { ++ m.mu.Lock() ++ defer m.mu.Unlock() ++ name := req.GetActor().GetName() ++ m.resumes[name]++ ++ if !m.running[name] { ++ if len(m.running) >= m.workers { ++ return nil, status.Error(codes.ResourceExhausted, "no free workers available") ++ } ++ m.running[name] = true ++ } ++ return &ateapipb.ResumeActorResponse{Actor: &ateapipb.Actor{ ++ Metadata: &ateapipb.ResourceMetadata{Name: name}, ++ Status: &ateapipb.ActorStatus{ ++ State: ateapipb.ActorState_ACTOR_STATE_RUNNING, ++ WorkerAssignment: &ateapipb.WorkerAssignment{WorkerPod: "w", WorkerPodIp: m.workerIP}, ++ }, ++ }, Resumed: true}, nil ++} ++ ++func (m *poolControl) SuspendActor(ctx context.Context, req *ateapipb.SuspendActorRequest) (*ateapipb.SuspendActorResponse, error) { ++ m.mu.Lock() ++ defer m.mu.Unlock() ++ name := req.GetActor().GetName() ++ m.suspends[name]++ ++ delete(m.running, name) ++ return &ateapipb.SuspendActorResponse{}, nil ++} ++ ++func (m *poolControl) resumeCount(name string) int { ++ m.mu.Lock() ++ defer m.mu.Unlock() ++ return m.resumes[name] ++} ++ ++// fakeRunner serves the runner's metadata endpoints. exited toggles whether ++// the command has exited; exitCode is what it reports. ++type fakeRunner struct { ++ exited atomic.Bool ++ exitCode atomic.Int32 ++ result string ++} ++ ++func (f *fakeRunner) handler() http.Handler { ++ mux := http.NewServeMux() ++ mux.HandleFunc("/readyz", func(w http.ResponseWriter, r *http.Request) { w.WriteHeader(http.StatusOK) }) ++ mux.HandleFunc("/metadata/v1alpha1/ax/exit", func(w http.ResponseWriter, r *http.Request) { ++ _ = json.NewEncoder(w).Encode(metadata.CommandExitStatus{Exited: f.exited.Load(), ExitCode: int(f.exitCode.Load()), FinishedAt: time.Now()}) ++ }) ++ mux.HandleFunc("/metadata/v1alpha1/ax/result", func(w http.ResponseWriter, r *http.Request) { ++ if !f.exited.Load() { ++ http.Error(w, "not exited", http.StatusConflict) ++ return ++ } ++ if f.result == "" { ++ http.NotFound(w, r) ++ return ++ } ++ _, _ = w.Write([]byte(f.result)) ++ }) ++ mux.HandleFunc("/metadata/v1alpha1/ax/usage", func(w http.ResponseWriter, r *http.Request) { ++ _, _ = w.Write([]byte(`{"prompt_tokens":11,"completion_tokens":7,"tool_calls":2}`)) ++ }) ++ return mux ++} ++ ++type harness struct { ++ pool *poolControl ++ runner *fakeRunner ++ reconciler *controller.TaskReconciler ++ store *memory.MemoryStore ++} ++ ++func newHarness(t *testing.T, workers int) *harness { ++ t.Helper() ++ fr := &fakeRunner{result: `{"answer":42}`} ++ hs := httptest.NewServer(fr.handler()) ++ t.Cleanup(hs.Close) ++ pool := newPoolControl(workers, hs.Listener.Addr().String()) ++ ++ lis, err := net.Listen("tcp", "127.0.0.1:0") ++ if err != nil { ++ t.Fatal(err) ++ } ++ gs := grpc.NewServer() ++ ateapipb.RegisterControlServer(gs, pool) ++ go gs.Serve(lis) ++ t.Cleanup(gs.Stop) ++ client, err := substrate.NewClient(lis.Addr().String(), grpc.WithTransportCredentials(insecure.NewCredentials())) ++ if err != nil { ++ t.Fatal(err) ++ } ++ t.Cleanup(func() { client.Close() }) ++ r := controller.NewTaskReconciler(client, "default-template", "ax-system") ++ r.SecretResolver = noSecrets ++ r.WorkspaceReadyTimeout = 100 * time.Millisecond ++ return &harness{pool: pool, runner: fr, reconciler: r, store: memory.NewStore()} ++} ++ ++func newTask(name string) *v1alpha1.Task { ++ return &v1alpha1.Task{ ++ Metadata: &v1alpha1.ObjectMeta{Name: name, Atespace: "fleet"}, ++ Spec: &v1alpha1.TaskSpec{Image: "localhost:5000/ax/ax-agent@sha256:00", Command: []string{"true"}}, ++ } ++} ++ ++func readyReason(task *v1alpha1.Task) string { ++ for _, c := range task.GetStatus().GetConditions() { ++ if c.GetType() == "Ready" { ++ return c.GetReason() ++ } ++ } ++ return "" ++} ++ ++func TestReconcile_ExitWriteBack(t *testing.T) { ++ ctx := context.Background() ++ for _, tc := range []struct { ++ code int32 ++ wantPhase string ++ }{{0, "Completed"}, {3, "Failed"}} { ++ t.Run(tc.wantPhase, func(t *testing.T) { ++ h := newHarness(t, 2) ++ h.runner.exited.Store(true) ++ h.runner.exitCode.Store(tc.code) ++ var saved []byte ++ h.reconciler.SaveResult = func(ctx context.Context, atespace, name string, content []byte) error { ++ saved = content ++ return nil ++ } ++ task, err := h.reconciler.Reconcile(ctx, newTask("t"), nil) ++ if err != nil { ++ t.Fatal(err) ++ } ++ if task.Status.Phase != tc.wantPhase { ++ t.Fatalf("phase = %q, want %q", task.Status.Phase, tc.wantPhase) ++ } ++ if readyReason(task) != "CommandExited" { ++ t.Fatalf("Ready reason = %q", readyReason(task)) ++ } ++ if want := fmt.Sprintf("ExitCode=%d", tc.code); task.Status.Conditions[len(task.Status.Conditions)-1].GetMessage() != want && !hasMessage(task, want) { ++ t.Fatalf("no condition message %q", want) ++ } ++ if c := task.Status.GetCommand(); !c.GetExited() || c.GetExitCode() != tc.code || c.GetResultBytes() != int64(len(`{"answer":42}`)) || c.GetResultSha256() == "" { ++ t.Fatalf("command status = %v", c) ++ } ++ if string(saved) != `{"answer":42}` { ++ t.Fatalf("saved result = %q", saved) ++ } ++ if u := task.Status.GetUsage(); u.GetPromptTokens() != 11 || u.GetToolCalls() != 2 { ++ t.Fatalf("usage = %v", u) ++ } ++ if h.pool.suspends["t"] != 1 || task.Status.WorkerIp != "" { ++ t.Fatalf("worker not freed: suspends=%d workerIP=%q", h.pool.suspends["t"], task.Status.WorkerIp) ++ } ++ if !controller.IsTerminal(task) { ++ t.Fatal("finished task is not terminal") ++ } ++ }) ++ } ++} ++ ++func hasMessage(task *v1alpha1.Task, msg string) bool { ++ for _, c := range task.GetStatus().GetConditions() { ++ if c.GetMessage() == msg { ++ return true ++ } ++ } ++ return false ++} ++ ++// The probe measured that stock v0.3.0 turns a Completed write-back into ++// Running and resumes the actor again. With the terminal guard it stays put. ++func TestWorker_CompletedWriteBackIsNotOverwritten(t *testing.T) { ++ ctx, cancel := context.WithCancel(context.Background()) ++ defer cancel() ++ h := newHarness(t, 2) ++ task := newTask("done") ++ task.Status = &v1alpha1.TaskStatus{Phase: "Completed"} ++ if err := h.store.SaveTask(ctx, task); err != nil { ++ t.Fatal(err) ++ } ++ w := controller.NewWorker(h.store, h.reconciler, "g", "c") ++ go func() { _ = w.Run(ctx) }() ++ time.Sleep(300 * time.Millisecond) ++ got, err := h.store.GetTask(ctx, "fleet", "done") ++ if err != nil { ++ t.Fatal(err) ++ } ++ if got.Status.Phase != "Completed" { ++ t.Fatalf("phase = %q, want Completed", got.Status.Phase) ++ } ++ if n := h.pool.resumeCount("done"); n != 0 { ++ t.Fatalf("terminal task resumed %d times", n) ++ } ++} ++ ++// runSequence applies n tasks one after another through a worker, each command ++// exiting after its first resume, and returns the final phase of each. ++func runSequence(t *testing.T, h *harness, n int) []*v1alpha1.Task { ++ t.Helper() ++ ctx, cancel := context.WithCancel(context.Background()) ++ defer cancel() ++ w := controller.NewWorker(h.store, h.reconciler, "g", "c") ++ w.RunningResync = 50 * time.Millisecond ++ go func() { _ = w.Run(ctx) }() ++ var out []*v1alpha1.Task ++ for i := 0; i < n; i++ { ++ name := fmt.Sprintf("floor-%d", i) ++ if err := h.store.SaveTask(ctx, newTask(name)); err != nil { ++ t.Fatal(err) ++ } ++ var got *v1alpha1.Task ++ deadline := time.Now().Add(5 * time.Second) ++ for time.Now().Before(deadline) { ++ tk, err := h.store.GetTask(ctx, "fleet", name) ++ if err == nil && tk.Status.GetPhase() != "Pending" && tk.Status.GetPhase() != "Running" { ++ got = tk ++ break ++ } ++ if err == nil && tk.Status.GetPhase() == "Running" && !h.runner.exited.Load() { ++ // Without an exit report a Running task never finishes. ++ got = tk ++ break ++ } ++ time.Sleep(20 * time.Millisecond) ++ } ++ if got == nil { ++ t.Fatalf("%s did not settle", name) ++ } ++ out = append(out, got) ++ } ++ return out ++} ++ ++// Four Tasks in a row on a 2-worker pool: every one reaches a terminal phase ++// and none fails ResourceExhausted, because each finished Task frees its worker. ++func TestWorker_FloorFourOnTwoWorkers(t *testing.T) { ++ h := newHarness(t, 2) ++ h.runner.exited.Store(true) ++ for _, tk := range runSequence(t, h, 4) { ++ if tk.Status.Phase != "Completed" { ++ t.Fatalf("%s: phase %q reason %q, want Completed", tk.Metadata.Name, tk.Status.Phase, readyReason(tk)) ++ } ++ } ++} ++ ++// The contrast the probe measured on stock v0.3.0: when no exit is reported, ++// finished work keeps its worker and the third Task on 2 workers is refused. ++func TestWorker_NoExitReportExhaustsPool(t *testing.T) { ++ h := newHarness(t, 2) ++ tasks := runSequence(t, h, 3) ++ if tasks[2].Status.Phase != "Failed" || readyReason(tasks[2]) != "ActorResumeFailed" { ++ t.Fatalf("third task: phase %q reason %q, want Failed ActorResumeFailed", tasks[2].Status.Phase, readyReason(tasks[2])) ++ } ++} ++ ++func TestWorker_ResyncFinishesRunningTask(t *testing.T) { ++ ctx := context.Background() ++ h := newHarness(t, 2) ++ w := controller.NewWorker(h.store, h.reconciler, "g", "c") ++ task, err := h.reconciler.Reconcile(ctx, newTask("slow"), nil) ++ if err != nil || task.Status.Phase != "Running" { ++ t.Fatalf("reconcile: %v phase %q", err, task.GetStatus().GetPhase()) ++ } ++ if err := h.store.SaveTask(ctx, task); err != nil { ++ t.Fatal(err) ++ } ++ if n := w.ResyncOnce(ctx); n != 0 { ++ t.Fatalf("resync changed %d tasks before exit", n) ++ } ++ h.runner.exited.Store(true) ++ if n := w.ResyncOnce(ctx); n != 1 { ++ t.Fatalf("resync changed %d tasks after exit, want 1", n) ++ } ++ got, _ := h.store.GetTask(ctx, "fleet", "slow") ++ if got.Status.Phase != "Completed" { ++ t.Fatalf("phase = %q", got.Status.Phase) ++ } ++ if b, err := h.store.GetTaskResult(ctx, "fleet", "slow"); err != nil || string(b) != `{"answer":42}` { ++ t.Fatalf("stored result = %q, %v", b, err) ++ } ++ if _, err := h.store.GetTaskResult(ctx, "fleet", "nope"); err != store.ErrNotFound { ++ t.Fatalf("missing result err = %v", err) ++ } ++} ++ ++func TestResync_CrashedActorIsFailed(t *testing.T) { ++ ctx := context.Background() ++ h := newHarness(t, 2) ++ task, err := h.reconciler.Reconcile(ctx, newTask("boom"), nil) ++ if err != nil { ++ t.Fatal(err) ++ } ++ h.pool.mu.Lock() ++ h.pool.crashed["boom"] = true ++ h.pool.mu.Unlock() ++ task.Status.WorkerIp = "127.0.0.1:1" // runner unreachable ++ changed, _ := h.reconciler.ResyncRunning(ctx, task) ++ if !changed || task.Status.Phase != "Failed" || readyReason(task) != "ActorCrashed" { ++ t.Fatalf("changed=%v phase=%q reason=%q", changed, task.Status.Phase, readyReason(task)) ++ } ++} +diff --git a/internal/controller/reconciler.go b/internal/controller/reconciler.go +index c4aebda..a4692ca 100644 +--- a/internal/controller/reconciler.go ++++ b/internal/controller/reconciler.go +@@ -17,8 +17,11 @@ package controller + import ( + "context" + "crypto/sha256" ++ "encoding/hex" ++ "encoding/json" + "errors" + "fmt" ++ "io" + "log/slog" + "net" + "net/http" +@@ -28,6 +31,8 @@ import ( + "strings" + "time" + ++ "github.com/agent-substrate/substrate/pkg/proto/ateapipb" ++ "github.com/google/ax/internal/metadata" + "github.com/google/ax/internal/model" + "github.com/google/ax/internal/substrate" + "github.com/google/ax/pkg/apis/v1alpha1" +@@ -70,6 +75,10 @@ type TaskReconciler struct { + // WorkspaceReadyTimeout bounds how long Reconcile waits for the actor's workspace + // to report ready before recording it as still initializing. + WorkspaceReadyTimeout time.Duration ++ ++ // SaveResult stores the result content copied from a finished task's ++ // sandbox. Nil drops the content; the digest and size still reach status. ++ SaveResult func(ctx context.Context, atespace, name string, content []byte) error + } + + // NewTaskReconciler creates a new TaskReconciler. +@@ -111,6 +120,14 @@ func (r *TaskReconciler) Reconcile(ctx context.Context, task *v1alpha1.Task, gat + task.Spec.Image = v1alpha1.DefaultTaskImage + } + ++ // Terminal guard: a task whose command already exited is never resumed ++ // again. Without it every later event (an apply, a status write-back that ++ // republishes) would resume the actor and flip the phase back to Running. ++ if IsTerminal(task) { ++ slog.Info("task is terminal, not reconciling", "name", task.Metadata.Name, "atespace", atespace, "phase", task.Status.Phase) ++ return task, nil ++ } ++ + slog.Info("reconciling task", "name", task.Metadata.Name, "atespace", atespace, "image", task.Spec.Image) + + now := time.Now() +@@ -293,6 +310,15 @@ func (r *TaskReconciler) Reconcile(ctx context.Context, task *v1alpha1.Task, gat + } + DonePolling: + ++ // A short command may already have finished; if so, record it now. ++ if workerIP != "" { ++ if done, err := r.finishIfExited(ctx, task, host, port); err != nil { ++ slog.Warn("could not read command exit", "name", task.Metadata.Name, "error", err) ++ } else if done { ++ return task, nil ++ } ++ } ++ + // The task is Ready only once its actor is running and the workspace inside it is set up. + if workspaceReady { + if !r.conditionTrue(task, condWorkspaceReady) { +@@ -317,6 +343,11 @@ DonePolling: + + // Condition types reported on Task status. + const ( ++ // reasonCommandExited is the Ready reason on a task whose command exited. ++ reasonCommandExited = "CommandExited" ++ // reasonActorCrashed is the Ready reason on a task whose actor crashed. ++ reasonActorCrashed = "ActorCrashed" ++ + // condReady reports whether the task as a whole is ready to do work: its actor is + // running and the workspace inside it has finished setting up. + condReady = "Ready" +@@ -483,3 +514,172 @@ func marshalWorkspaces(workspaces []*v1alpha1.Workspace) (string, error) { + } + return sb.String(), nil + } ++ ++// IsTerminal reports whether the task's command has finished: phase Completed, ++// or Failed with the Ready reason CommandExited. Such a task is never resumed. ++func IsTerminal(task *v1alpha1.Task) bool { ++ st := task.GetStatus() ++ switch st.GetPhase() { ++ case "Completed": ++ return true ++ case "Failed": ++ for _, c := range st.GetConditions() { ++ if c.GetType() == condReady && c.GetReason() == reasonCommandExited { ++ return true ++ } ++ } ++ } ++ return false ++} ++ ++// ResyncRunning re-checks a Running task, because no event fires when its ++// command exits. It returns true when it changed the task's status: the command ++// exited (Completed or Failed, worker freed) or the actor crashed (Failed). ++func (r *TaskReconciler) ResyncRunning(ctx context.Context, task *v1alpha1.Task) (bool, error) { ++ if task.GetStatus().GetPhase() != "Running" || task.GetMetadata() == nil || IsTerminal(task) { ++ return false, nil ++ } ++ atespace := task.Metadata.Atespace ++ if atespace == "" { ++ atespace = "default" ++ } ++ host, port := task.Status.WorkerIp, "80" ++ if h, p, err := net.SplitHostPort(host); err == nil { ++ host, port = h, p ++ } ++ done, exitErr := r.finishIfExited(ctx, task, host, port) ++ if done { ++ return true, nil ++ } ++ // No exit report: a crashed actor is Failed, not restarted behind the ++ // caller's back. ++ state, err := r.client.ActorState(ctx, atespace, task.Metadata.Name) ++ if err == nil && state == ateapipb.ActorState_ACTOR_STATE_CRASHED { ++ task.Status.Phase = "Failed" ++ task.Status.WorkerIp = "" ++ r.setNotReady(task, reasonActorCrashed, "Substrate reports the actor CRASHED", time.Now()) ++ return true, nil ++ } ++ if exitErr != nil { ++ return false, exitErr ++ } ++ return false, err ++} ++ ++// finishIfExited asks the runner whether the command exited. If it did, it ++// copies the result and usage, writes the outcome into status, and suspends ++// the actor so its worker is free for the next task. ++func (r *TaskReconciler) finishIfExited(ctx context.Context, task *v1alpha1.Task, host, port string) (bool, error) { ++ atespace := task.Metadata.Atespace ++ actorName := task.Metadata.Name ++ body, err := r.metadataGet(ctx, atespace, actorName, host, port, "/metadata/v1alpha1/ax/exit") ++ if err != nil { ++ return false, err ++ } ++ var exit metadata.CommandExitStatus ++ if err := json.Unmarshal(body, &exit); err != nil { ++ return false, fmt.Errorf("decoding exit status: %w", err) ++ } ++ if !exit.Exited { ++ return false, nil ++ } ++ ++ now := time.Now() ++ cmdStatus := &v1alpha1.CommandStatus{Exited: true, ExitCode: int32(exit.ExitCode)} ++ if !exit.FinishedAt.IsZero() { ++ cmdStatus.FinishedAt = timestamppb.New(exit.FinishedAt) ++ } else { ++ cmdStatus.FinishedAt = timestamppb.New(now) ++ } ++ if result, err := r.metadataGet(ctx, atespace, actorName, host, port, "/metadata/v1alpha1/ax/result"); err == nil { ++ sum := sha256.Sum256(result) ++ cmdStatus.ResultBytes = int64(len(result)) ++ cmdStatus.ResultSha256 = hex.EncodeToString(sum[:]) ++ if r.SaveResult != nil { ++ if err := r.SaveResult(ctx, atespace, actorName, result); err != nil { ++ slog.Warn("could not store task result", "name", actorName, "error", err) ++ } ++ } ++ } else { ++ slog.Info("task produced no result", "name", actorName, "reason", err) ++ } ++ if usageBody, err := r.metadataGet(ctx, atespace, actorName, host, port, "/metadata/v1alpha1/ax/usage"); err == nil { ++ var u struct { ++ PromptTokens int32 `json:"prompt_tokens"` ++ CompletionTokens int32 `json:"completion_tokens"` ++ ToolCalls int32 `json:"tool_calls"` ++ } ++ if json.Unmarshal(usageBody, &u) == nil { ++ task.Status.Usage = &v1alpha1.UsageStats{PromptTokens: u.PromptTokens, CompletionTokens: u.CompletionTokens, ToolCalls: u.ToolCalls} ++ } ++ } ++ task.Status.Command = cmdStatus ++ ++ msg := fmt.Sprintf("ExitCode=%d", exit.ExitCode) ++ if exit.ExitCode == 0 { ++ task.Status.Phase = "Completed" ++ } else { ++ task.Status.Phase = "Failed" ++ } ++ r.setCondition(task, condReady, "False", reasonCommandExited, msg, now) ++ ++ // Free the worker. The DATA snapshot keeps /workspace for inspection. ++ if err := r.client.SuspendActor(ctx, atespace, actorName); err != nil { ++ slog.Warn("could not suspend finished task's actor", "name", actorName, "error", err) ++ r.setCondition(task, condReady, "False", reasonCommandExited, msg+"; suspend failed: "+err.Error(), now) ++ } else { ++ task.Status.WorkerIp = "" ++ } ++ slog.Info("task command exited", "name", actorName, "atespace", atespace, "exitCode", exit.ExitCode, "phase", task.Status.Phase) ++ return true, nil ++} ++ ++// metadataGet reads a runner metadata path, directly from the worker first and ++// then through the atenet router when ATENET_ROUTER_ADDR is set. ++func (r *TaskReconciler) metadataGet(ctx context.Context, atespace, actorName, host, port, path string) ([]byte, error) { ++ var errs []error ++ if host != "" { ++ b, err := r.fetch(ctx, fmt.Sprintf("http://%s%s", net.JoinHostPort(host, port), path), "") ++ if err == nil { ++ return b, nil ++ } ++ errs = append(errs, err) ++ } ++ if routerAddr := os.Getenv("ATENET_ROUTER_ADDR"); routerAddr != "" { ++ b, err := r.fetch(ctx, fmt.Sprintf("http://%s%s", routerAddr, path), fmt.Sprintf("%s/%s", atespace, actorName)) ++ if err == nil { ++ return b, nil ++ } ++ errs = append(errs, err) ++ } ++ if len(errs) == 0 { ++ return nil, errors.New("no route to the task's metadata server") ++ } ++ return nil, errors.Join(errs...) ++} ++ ++func (r *TaskReconciler) fetch(ctx context.Context, url, targetActor string) ([]byte, error) { ++ req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil) ++ if err != nil { ++ return nil, err ++ } ++ if targetActor != "" { ++ req.Header.Set("ate-target-actor", targetActor) ++ } ++ resp, err := r.httpClient.Do(req) ++ if err != nil { ++ return nil, err ++ } ++ defer resp.Body.Close() ++ body, err := io.ReadAll(io.LimitReader(resp.Body, metadata.MaxResultBytes+1)) ++ if err != nil { ++ return nil, err ++ } ++ if resp.StatusCode != http.StatusOK { ++ return nil, fmt.Errorf("GET %s: %s", url, resp.Status) ++ } ++ if int64(len(body)) > metadata.MaxResultBytes { ++ return nil, fmt.Errorf("GET %s: body over %d bytes", url, metadata.MaxResultBytes) ++ } ++ return body, nil ++} +diff --git a/internal/controller/worker.go b/internal/controller/worker.go +index 398b090..c783e8a 100644 +--- a/internal/controller/worker.go ++++ b/internal/controller/worker.go +@@ -20,6 +20,7 @@ import ( + "fmt" + "log/slog" + "os" ++ "sync" + "time" + + "github.com/google/ax/internal/store" +@@ -31,6 +32,9 @@ const ( + // readRetryDelay is how long the worker waits after a transient error from the + // event queue before trying again. + readRetryDelay = time.Second ++ // DefaultRunningResync is how often Running tasks are re-checked for a ++ // command exit, since nothing publishes an event when a command exits. ++ DefaultRunningResync = 15 * time.Second + ) + + // Worker consumes task events from the store's event queue and reconciles each +@@ -41,6 +45,14 @@ type Worker struct { + reconciler *TaskReconciler + group string + consumer string ++ ++ // RunningResync is the period of the Running-task re-check. Zero or less ++ // disables it. ++ RunningResync time.Duration ++ ++ // mu serializes event handling and the resync, so one task is never ++ // reconciled twice at once by this worker. ++ mu sync.Mutex + } + + // NewWorker creates a worker that joins group as consumer. An empty group uses the +@@ -53,11 +65,15 @@ func NewWorker(s store.Store, reconciler *TaskReconciler, group, consumer string + hostname, _ := os.Hostname() + consumer = fmt.Sprintf("%s-%d", hostname, time.Now().UnixNano()%10000) + } ++ if reconciler != nil && reconciler.SaveResult == nil { ++ reconciler.SaveResult = s.SaveTaskResult ++ } + return &Worker{ +- store: s, +- reconciler: reconciler, +- group: group, +- consumer: consumer, ++ store: s, ++ reconciler: reconciler, ++ group: group, ++ consumer: consumer, ++ RunningResync: DefaultRunningResync, + } + } + +@@ -73,6 +89,10 @@ func (w *Worker) Run(ctx context.Context) error { + } + defer sub.Close() + ++ if w.RunningResync > 0 { ++ go w.resyncLoop(ctx) ++ } ++ + for { + ev, err := sub.Next(ctx) + if err != nil { +@@ -89,7 +109,10 @@ func (w *Worker) Run(ctx context.Context) error { + continue + } + +- if err := w.processEvent(ctx, ev); err != nil { ++ w.mu.Lock() ++ err = w.processEvent(ctx, ev) ++ w.mu.Unlock() ++ if err != nil { + slog.Error("error processing task event", + "id", ev.ID, + "atespace", ev.Atespace, +@@ -165,3 +188,53 @@ func (w *Worker) processEvent(ctx context.Context, ev store.TaskEvent) error { + + return nil + } ++ ++// resyncLoop re-checks every Running task each RunningResync until ctx is done. ++func (w *Worker) resyncLoop(ctx context.Context) { ++ ticker := time.NewTicker(w.RunningResync) ++ defer ticker.Stop() ++ for { ++ select { ++ case <-ctx.Done(): ++ return ++ case <-ticker.C: ++ w.ResyncOnce(ctx) ++ } ++ } ++} ++ ++// ResyncOnce re-checks every Running task once and writes back any that ++// finished. It returns how many tasks changed. ++func (w *Worker) ResyncOnce(ctx context.Context) int { ++ w.mu.Lock() ++ defer w.mu.Unlock() ++ tasks, err := w.store.ListTasks(ctx, "", 0, 0) ++ if err != nil { ++ slog.Warn("resync: listing tasks", "error", err) ++ return 0 ++ } ++ changed := 0 ++ for _, t := range tasks { ++ if t.GetStatus().GetPhase() != "Running" { ++ continue ++ } ++ // Re-read so a status written since the list is not overwritten. ++ task, err := w.store.GetTask(ctx, t.Metadata.Atespace, t.Metadata.Name) ++ if err != nil || task.GetStatus().GetPhase() != "Running" { ++ continue ++ } ++ ok, err := w.reconciler.ResyncRunning(ctx, task) ++ if err != nil { ++ slog.Debug("resync: task not finished or unreachable", "atespace", task.Metadata.Atespace, "name", task.Metadata.Name, "error", err) ++ } ++ if !ok { ++ continue ++ } ++ if err := w.store.UpdateTaskStatus(ctx, task.Metadata.Atespace, task.Metadata.Name, task.Status); err != nil { ++ slog.Error("resync: writing task status", "name", task.Metadata.Name, "error", err) ++ continue ++ } ++ changed++ ++ } ++ return changed ++} +diff --git a/internal/metadata/exit_test.go b/internal/metadata/exit_test.go +new file mode 100644 +index 0000000..85da1c1 +--- /dev/null ++++ b/internal/metadata/exit_test.go +@@ -0,0 +1,118 @@ ++// Copyright 2026 Google LLC ++// ++// Licensed under the Apache License, Version 2.0 (the "License"); ++// you may not use this file except in compliance with the License. ++// You may obtain a copy of the License at ++// ++// http://www.apache.org/licenses/LICENSE-2.0 ++// ++// Unless required by applicable law or agreed to in writing, software ++// distributed under the License is distributed on an "AS IS" BASIS, ++// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. ++// See the License for the specific language governing permissions and ++// limitations under the License. ++ ++package metadata_test ++ ++import ( ++ "context" ++ "encoding/json" ++ "fmt" ++ "io" ++ "net" ++ "net/http" ++ "os" ++ "path/filepath" ++ "strings" ++ "testing" ++ "time" ++ ++ "github.com/google/ax/internal/metadata" ++ "github.com/google/ax/pkg/apis/v1alpha1" ++) ++ ++func freePort(t *testing.T) int { ++ t.Helper() ++ l, err := net.Listen("tcp", "127.0.0.1:0") ++ if err != nil { ++ t.Fatal(err) ++ } ++ defer l.Close() ++ return l.Addr().(*net.TCPAddr).Port ++} ++ ++func get(t *testing.T, url string) (int, string) { ++ t.Helper() ++ var resp *http.Response ++ var err error ++ for i := 0; i < 50; i++ { ++ resp, err = http.Get(url) ++ if err == nil { ++ break ++ } ++ time.Sleep(20 * time.Millisecond) ++ } ++ if err != nil { ++ t.Fatalf("GET %s: %v", url, err) ++ } ++ defer resp.Body.Close() ++ b, _ := io.ReadAll(resp.Body) ++ return resp.StatusCode, string(b) ++} ++ ++func TestMetadataServer_ExitAndResult(t *testing.T) { ++ dir := t.TempDir() ++ resultPath := filepath.Join(dir, "result.json") ++ usagePath := filepath.Join(dir, "usage.json") ++ port := freePort(t) ++ task := &v1alpha1.Task{Metadata: &v1alpha1.ObjectMeta{Name: "t1"}, Spec: &v1alpha1.TaskSpec{}} ++ s := metadata.NewServer(port, task, nil, metadata.ServerOptions{ResultPath: resultPath, UsagePath: usagePath}) ++ if err := s.Start(); err != nil { ++ t.Fatal(err) ++ } ++ defer s.Stop(context.Background()) ++ base := fmt.Sprintf("http://127.0.0.1:%d/metadata/v1alpha1/ax", port) ++ ++ // Before exit: exit says not exited, result is refused. ++ code, body := get(t, base+"/exit") ++ var st metadata.CommandExitStatus ++ if code != http.StatusOK || json.Unmarshal([]byte(body), &st) != nil || st.Exited { ++ t.Fatalf("exit before command exit = %d %q", code, body) ++ } ++ if code, _ := get(t, base+"/result"); code != http.StatusConflict { ++ t.Fatalf("result before exit = %d, want 409", code) ++ } ++ ++ if err := os.WriteFile(resultPath, []byte(`{"ok":true}`), 0o644); err != nil { ++ t.Fatal(err) ++ } ++ if err := os.WriteFile(usagePath, []byte(`{"prompt_tokens":3}`), 0o644); err != nil { ++ t.Fatal(err) ++ } ++ s.SetCommandExit(3, time.Now()) ++ s.SetCommandExit(0, time.Now()) // only the first report counts ++ ++ code, body = get(t, base+"/exit") ++ if err := json.Unmarshal([]byte(body), &st); err != nil || code != http.StatusOK || !st.Exited || st.ExitCode != 3 { ++ t.Fatalf("exit after command exit = %d %q", code, body) ++ } ++ if code, body := get(t, base+"/result"); code != http.StatusOK || body != `{"ok":true}` { ++ t.Fatalf("result = %d %q", code, body) ++ } ++ if code, body := get(t, base+"/usage"); code != http.StatusOK || !strings.Contains(body, "prompt_tokens") { ++ t.Fatalf("usage = %d %q", code, body) ++ } ++ ++ // Over the cap: refused, not truncated. ++ if err := os.WriteFile(resultPath, make([]byte, metadata.MaxResultBytes+1), 0o644); err != nil { ++ t.Fatal(err) ++ } ++ if code, _ := get(t, base+"/result"); code != http.StatusRequestEntityTooLarge { ++ t.Fatalf("oversized result = %d, want 413", code) ++ } ++ // Missing: 404. ++ _ = os.Remove(resultPath) ++ if code, _ := get(t, base+"/result"); code != http.StatusNotFound { ++ t.Fatalf("missing result = %d, want 404", code) ++ } ++} +diff --git a/internal/metadata/server.go b/internal/metadata/server.go +index fb4feba..909e879 100644 +--- a/internal/metadata/server.go ++++ b/internal/metadata/server.go +@@ -16,13 +16,17 @@ package metadata + + import ( + "context" ++ "encoding/json" + "errors" + "fmt" ++ "io" + "log/slog" + "net" + "net/http" ++ "os" + "strings" + "sync" ++ "time" + + "github.com/agent-substrate/env/guest" + "github.com/google/ax/pkg/apis/v1alpha1" +@@ -41,12 +45,31 @@ type Server struct { + task *v1alpha1.Task + workspaces []*v1alpha1.Workspace + workspaceReady bool ++ resultPath string ++ usagePath string ++ exit *CommandExitStatus ++} ++ ++// MaxResultBytes caps the result file served at /metadata/v1alpha1/ax/result. ++// A larger file is refused with 413 rather than truncated. ++const MaxResultBytes = 1 << 20 ++ ++// CommandExitStatus is the JSON body of /metadata/v1alpha1/ax/exit. ++type CommandExitStatus struct { ++ Exited bool `json:"exited"` ++ ExitCode int `json:"exitCode"` ++ FinishedAt time.Time `json:"finishedAt,omitempty"` + } + + // ServerOptions configures optional settings for the metadata and guest server. + type ServerOptions struct { + WorkspacePath string + LogDir string ++ // ResultPath is the file served at /metadata/v1alpha1/ax/result once the ++ // command has exited. Empty disables the endpoint. ++ ResultPath string ++ // UsagePath is the JSON usage file served at /metadata/v1alpha1/ax/usage. ++ UsagePath string + } + + // NewServer creates a new metadata and guest server serving the task and its +@@ -65,6 +88,8 @@ func NewServer(port int, task *v1alpha1.Task, workspaces []*v1alpha1.Workspace, + if len(opts) > 0 { + opt = opts[0] + } ++ s.resultPath = opt.ResultPath ++ s.usagePath = opt.UsagePath + + // Guest services expose process execution and file access inside the container, + // so they are only served when the task opts in via spec.debug. +@@ -98,6 +123,12 @@ func NewServer(port int, task *v1alpha1.Task, workspaces []*v1alpha1.Workspace, + // /metadata/v1alpha1/ax/workspaces every bound Workspace, as a YAML stream + mux.HandleFunc("/metadata/v1alpha1/ax/task", s.handleTask) + mux.HandleFunc("/metadata/v1alpha1/ax/workspaces", s.handleWorkspaces) ++ // /metadata/v1alpha1/ax/exit whether the command exited, and its code ++ // /metadata/v1alpha1/ax/result the command's result file, after exit ++ // /metadata/v1alpha1/ax/usage the command's usage file, after exit ++ mux.HandleFunc("/metadata/v1alpha1/ax/exit", s.handleExit) ++ mux.HandleFunc("/metadata/v1alpha1/ax/result", s.handleResult) ++ mux.HandleFunc("/metadata/v1alpha1/ax/usage", s.handleUsage) + + handler := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if s.grpcServer != nil && r.ProtoMajor == 2 && strings.HasPrefix(r.Header.Get("Content-Type"), "application/grpc") { +@@ -236,3 +267,72 @@ func (s *Server) handleWorkspaces(w http.ResponseWriter, r *http.Request) { + http.Error(w, err.Error(), http.StatusInternalServerError) + } + } ++ ++// SetCommandExit records that the task command exited on its own with code. ++// Only the first call counts. ++func (s *Server) SetCommandExit(code int, at time.Time) { ++ s.mu.Lock() ++ defer s.mu.Unlock() ++ if s.exit == nil { ++ s.exit = &CommandExitStatus{Exited: true, ExitCode: code, FinishedAt: at.UTC()} ++ } ++} ++ ++func (s *Server) commandExit() *CommandExitStatus { ++ s.mu.RLock() ++ defer s.mu.RUnlock() ++ if s.exit == nil { ++ return nil ++ } ++ cp := *s.exit ++ return &cp ++} ++ ++// handleExit reports whether the command has exited. It always answers 200 ++// with a JSON body, so a caller can tell "still running" from "unreachable". ++func (s *Server) handleExit(w http.ResponseWriter, r *http.Request) { ++ st := s.commandExit() ++ if st == nil { ++ st = &CommandExitStatus{} ++ } ++ w.Header().Set("Content-Type", "application/json") ++ _ = json.NewEncoder(w).Encode(st) ++} ++ ++func (s *Server) handleResult(w http.ResponseWriter, r *http.Request) { ++ s.serveAfterExit(w, s.resultPath, MaxResultBytes) ++} ++ ++func (s *Server) handleUsage(w http.ResponseWriter, r *http.Request) { ++ s.serveAfterExit(w, s.usagePath, 64<<10) ++} ++ ++// serveAfterExit serves the file at path once the command has exited: 409 ++// before that, 404 when there is no file, 413 when it exceeds limit. ++func (s *Server) serveAfterExit(w http.ResponseWriter, path string, limit int64) { ++ if s.commandExit() == nil { ++ http.Error(w, "command has not exited", http.StatusConflict) ++ return ++ } ++ if path == "" { ++ http.Error(w, "not configured", http.StatusNotFound) ++ return ++ } ++ f, err := os.Open(path) ++ if err != nil { ++ http.Error(w, "no file", http.StatusNotFound) ++ return ++ } ++ defer f.Close() ++ fi, err := f.Stat() ++ if err != nil || !fi.Mode().IsRegular() { ++ http.Error(w, "no file", http.StatusNotFound) ++ return ++ } ++ if fi.Size() > limit { ++ http.Error(w, fmt.Sprintf("file is %d bytes, over the %d byte cap", fi.Size(), limit), http.StatusRequestEntityTooLarge) ++ return ++ } ++ w.Header().Set("Content-Type", "application/octet-stream") ++ _, _ = io.Copy(w, io.LimitReader(f, limit)) ++} +diff --git a/internal/server/server.go b/internal/server/server.go +index 27e89ab..669cbac 100644 +--- a/internal/server/server.go ++++ b/internal/server/server.go +@@ -16,6 +16,8 @@ package server + + import ( + "context" ++ "crypto/sha256" ++ "encoding/hex" + "errors" + "net/http" + "strings" +@@ -88,6 +90,27 @@ func (s *Server) GetTask(ctx context.Context, req *v1alpha1.GetTaskRequest) (*v1 + return task, nil + } + ++// GetTaskResult returns the result content the controller copied from the ++// task's sandbox when its command exited. ++func (s *Server) GetTaskResult(ctx context.Context, req *v1alpha1.GetTaskResultRequest) (*v1alpha1.TaskResult, error) { ++ if req == nil || req.Name == "" { ++ return nil, status.Error(codes.InvalidArgument, "missing task name") ++ } ++ atespace := req.Atespace ++ if atespace == "" { ++ atespace = "default" ++ } ++ content, err := s.store.GetTaskResult(ctx, atespace, req.Name) ++ if err != nil { ++ if errors.Is(err, store.ErrNotFound) { ++ return nil, status.Errorf(codes.NotFound, "no result for task %q in atespace %q", req.Name, atespace) ++ } ++ return nil, status.Errorf(codes.Internal, "getting task result: %v", err) ++ } ++ sum := sha256.Sum256(content) ++ return &v1alpha1.TaskResult{Content: content, Sha256: hex.EncodeToString(sum[:])}, nil ++} ++ + func (s *Server) ListTasks(ctx context.Context, req *v1alpha1.ListTasksRequest) (*v1alpha1.ListTasksResponse, error) { + atespace := "" + limit := int64(50) +diff --git a/internal/store/memory/store.go b/internal/store/memory/store.go +index 8a0b13d..215f9a4 100644 +--- a/internal/store/memory/store.go ++++ b/internal/store/memory/store.go +@@ -47,6 +47,7 @@ type MemoryStore struct { + workspaces map[string]*v1alpha1.Workspace + events chan store.TaskEvent + watchers map[string][]chan *v1alpha1.Task ++ results map[string][]byte + } + + // NewStore creates a new in-memory Store. +@@ -58,9 +59,29 @@ func NewStore() *MemoryStore { + workspaces: make(map[string]*v1alpha1.Workspace), + events: make(chan store.TaskEvent, 1000), + watchers: make(map[string][]chan *v1alpha1.Task), ++ results: make(map[string][]byte), + } + } + ++// SaveTaskResult stores a copy of the task's result content. ++func (s *MemoryStore) SaveTaskResult(ctx context.Context, atespace, name string, content []byte) error { ++ s.mu.Lock() ++ defer s.mu.Unlock() ++ s.results[taskKey(atespace, name)] = append([]byte(nil), content...) ++ return nil ++} ++ ++// GetTaskResult returns a copy of the stored result content. ++func (s *MemoryStore) GetTaskResult(ctx context.Context, atespace, name string) ([]byte, error) { ++ s.mu.RLock() ++ defer s.mu.RUnlock() ++ content, ok := s.results[taskKey(atespace, name)] ++ if !ok { ++ return nil, store.ErrNotFound ++ } ++ return append([]byte(nil), content...), nil ++} ++ + func taskKey(atespace, name string) string { + if atespace == "" { + atespace = "default" +@@ -216,6 +237,7 @@ func (s *MemoryStore) DeleteTask(ctx context.Context, atespace, name string) err + s.mu.Lock() + defer s.mu.Unlock() + delete(s.tasks, taskKey(atespace, name)) ++ delete(s.results, taskKey(atespace, name)) + return nil + } + +diff --git a/internal/store/redis/store.go b/internal/store/redis/store.go +index 0bbf7aa..34da35b 100644 +--- a/internal/store/redis/store.go ++++ b/internal/store/redis/store.go +@@ -82,6 +82,37 @@ func (s *Store) taskKey(atespace, name string) string { + return fmt.Sprintf("%s:task:%s:%s", s.opts.KeyPrefix, atespace, name) + } + ++func (s *Store) taskResultKey(atespace, name string) string { ++ return fmt.Sprintf("%s:task-result:%s:%s", s.opts.KeyPrefix, atespace, name) ++} ++ ++// SaveTaskResult stores the task's result content under its own key, so task ++// records, lists and watches stay small. It publishes no event. ++func (s *Store) SaveTaskResult(ctx context.Context, atespace, name string, content []byte) error { ++ if atespace == "" { ++ atespace = "default" ++ } ++ if err := s.client.Set(ctx, s.taskResultKey(atespace, name), content, s.opts.TTL).Err(); err != nil { ++ return fmt.Errorf("saving task result in redis: %w", err) ++ } ++ return nil ++} ++ ++// GetTaskResult returns the stored result content, or store.ErrNotFound. ++func (s *Store) GetTaskResult(ctx context.Context, atespace, name string) ([]byte, error) { ++ if atespace == "" { ++ atespace = "default" ++ } ++ content, err := s.client.Get(ctx, s.taskResultKey(atespace, name)).Bytes() ++ if err != nil { ++ if errors.Is(err, redis.Nil) { ++ return nil, store.ErrNotFound ++ } ++ return nil, fmt.Errorf("getting task result from redis: %w", err) ++ } ++ return content, nil ++} ++ + func (s *Store) gwKey(atespace, name string) string { + return fmt.Sprintf("%s:gw:%s:%s", s.opts.KeyPrefix, atespace, name) + } +@@ -335,6 +366,7 @@ func (s *Store) DeleteTask(ctx context.Context, atespace, name string) error { + + pipe := s.client.TxPipeline() + pipe.Del(ctx, s.taskKey(atespace, name)) ++ pipe.Del(ctx, s.taskResultKey(atespace, name)) + pipe.ZRem(ctx, s.taskIndexKey(), member) + pipe.ZRem(ctx, s.taskAtespaceIndexKey(atespace), name) + if _, err := pipe.Exec(ctx); err != nil { +diff --git a/internal/store/store.go b/internal/store/store.go +index 1e84a0f..684f140 100644 +--- a/internal/store/store.go ++++ b/internal/store/store.go +@@ -86,5 +86,11 @@ type Store interface { + DeleteModel(ctx context.Context, atespace, name string) error + + WatchTask(ctx context.Context, atespace, name string) (<-chan *v1alpha1.Task, io.Closer, error) ++ ++ // SaveTaskResult stores the result file a task's command wrote, as copied ++ // by the controller once the command exited. It publishes no event. ++ SaveTaskResult(ctx context.Context, atespace, name string, content []byte) error ++ // GetTaskResult returns the stored result. ErrNotFound when none was saved. ++ GetTaskResult(ctx context.Context, atespace, name string) ([]byte, error) + Close() error + } +diff --git a/internal/substrate/client.go b/internal/substrate/client.go +index 2840ce2..3693e99 100644 +--- a/internal/substrate/client.go ++++ b/internal/substrate/client.go +@@ -389,6 +389,17 @@ func (c *Client) EnsureActor(ctx context.Context, atespace, actorName, templateA + return actor, nil + } + ++// ActorState returns the actor's current state as Substrate reports it. ++func (c *Client) ActorState(ctx context.Context, atespace, actorName string) (ateapipb.ActorState, error) { ++ actor, err := c.control.GetActor(ctx, &ateapipb.GetActorRequest{ ++ Actor: &ateapipb.ObjectRef{Atespace: atespace, Name: actorName}, ++ }) ++ if err != nil { ++ return ateapipb.ActorState_ACTOR_STATE_UNSPECIFIED, fmt.Errorf("getting actor %s/%s: %w", atespace, actorName, err) ++ } ++ return actor.GetStatus().GetState(), nil ++} ++ + // ResumeActor resumes the specified actor onto a worker and returns the worker details. + func (c *Client) ResumeActor(ctx context.Context, atespace, actorName string) (*ateapipb.Actor, string, error) { + req := &ateapipb.ResumeActorRequest{ +diff --git a/pkg/apis/v1alpha1/ax.pb.go b/pkg/apis/v1alpha1/ax.pb.go +index 9a71dc0..0ce2bf0 100644 +--- a/pkg/apis/v1alpha1/ax.pb.go ++++ b/pkg/apis/v1alpha1/ax.pb.go +@@ -571,8 +571,10 @@ type TaskStatus struct { + PendingApproval *PendingApproval `protobuf:"bytes,5,opt,name=pending_approval,json=pendingApproval,proto3" json:"pending_approval,omitempty"` + Usage *UsageStats `protobuf:"bytes,6,opt,name=usage,proto3" json:"usage,omitempty"` + Conditions []*Condition `protobuf:"bytes,7,rep,name=conditions,proto3" json:"conditions,omitempty"` +- unknownFields protoimpl.UnknownFields +- sizeCache protoimpl.SizeCache ++ // command reports how the task's command finished. Unset while it runs. ++ Command *CommandStatus `protobuf:"bytes,8,opt,name=command,proto3" json:"command,omitempty"` ++ unknownFields protoimpl.UnknownFields ++ sizeCache protoimpl.SizeCache + } + + func (x *TaskStatus) Reset() { +@@ -654,6 +656,92 @@ func (x *TaskStatus) GetConditions() []*Condition { + return nil + } + ++func (x *TaskStatus) GetCommand() *CommandStatus { ++ if x != nil { ++ return x.Command ++ } ++ return nil ++} ++ ++// CommandStatus is written by the controller once the runner reports that the ++// task's command exited. The result content itself is served by GetTaskResult. ++type CommandStatus struct { ++ state protoimpl.MessageState `protogen:"open.v1"` ++ Exited bool `protobuf:"varint,1,opt,name=exited,proto3" json:"exited,omitempty"` ++ ExitCode int32 `protobuf:"varint,2,opt,name=exit_code,json=exitCode,proto3" json:"exit_code,omitempty"` ++ FinishedAt *timestamppb.Timestamp `protobuf:"bytes,3,opt,name=finished_at,json=finishedAt,proto3" json:"finished_at,omitempty"` ++ // result_bytes is the size of the copied result, 0 when there was none. ++ ResultBytes int64 `protobuf:"varint,4,opt,name=result_bytes,json=resultBytes,proto3" json:"result_bytes,omitempty"` ++ ResultSha256 string `protobuf:"bytes,5,opt,name=result_sha256,json=resultSha256,proto3" json:"result_sha256,omitempty"` ++ unknownFields protoimpl.UnknownFields ++ sizeCache protoimpl.SizeCache ++} ++ ++func (x *CommandStatus) Reset() { ++ *x = CommandStatus{} ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[9] ++ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ++ ms.StoreMessageInfo(mi) ++} ++ ++func (x *CommandStatus) String() string { ++ return protoimpl.X.MessageStringOf(x) ++} ++ ++func (*CommandStatus) ProtoMessage() {} ++ ++func (x *CommandStatus) ProtoReflect() protoreflect.Message { ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[9] ++ if x != nil { ++ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ++ if ms.LoadMessageInfo() == nil { ++ ms.StoreMessageInfo(mi) ++ } ++ return ms ++ } ++ return mi.MessageOf(x) ++} ++ ++// Deprecated: Use CommandStatus.ProtoReflect.Descriptor instead. ++func (*CommandStatus) Descriptor() ([]byte, []int) { ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{9} ++} ++ ++func (x *CommandStatus) GetExited() bool { ++ if x != nil { ++ return x.Exited ++ } ++ return false ++} ++ ++func (x *CommandStatus) GetExitCode() int32 { ++ if x != nil { ++ return x.ExitCode ++ } ++ return 0 ++} ++ ++func (x *CommandStatus) GetFinishedAt() *timestamppb.Timestamp { ++ if x != nil { ++ return x.FinishedAt ++ } ++ return nil ++} ++ ++func (x *CommandStatus) GetResultBytes() int64 { ++ if x != nil { ++ return x.ResultBytes ++ } ++ return 0 ++} ++ ++func (x *CommandStatus) GetResultSha256() string { ++ if x != nil { ++ return x.ResultSha256 ++ } ++ return "" ++} ++ + type PendingApproval struct { + state protoimpl.MessageState `protogen:"open.v1"` + Id string `protobuf:"bytes,1,opt,name=id,proto3" json:"id,omitempty"` +@@ -665,7 +753,7 @@ type PendingApproval struct { + + func (x *PendingApproval) Reset() { + *x = PendingApproval{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[9] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[10] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -677,7 +765,7 @@ func (x *PendingApproval) String() string { + func (*PendingApproval) ProtoMessage() {} + + func (x *PendingApproval) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[9] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[10] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -690,7 +778,7 @@ func (x *PendingApproval) ProtoReflect() protoreflect.Message { + + // Deprecated: Use PendingApproval.ProtoReflect.Descriptor instead. + func (*PendingApproval) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{9} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{10} + } + + func (x *PendingApproval) GetId() string { +@@ -718,13 +806,14 @@ type UsageStats struct { + state protoimpl.MessageState `protogen:"open.v1"` + PromptTokens int32 `protobuf:"varint,1,opt,name=prompt_tokens,json=promptTokens,proto3" json:"prompt_tokens,omitempty"` + CompletionTokens int32 `protobuf:"varint,2,opt,name=completion_tokens,json=completionTokens,proto3" json:"completion_tokens,omitempty"` ++ ToolCalls int32 `protobuf:"varint,3,opt,name=tool_calls,json=toolCalls,proto3" json:"tool_calls,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache + } + + func (x *UsageStats) Reset() { + *x = UsageStats{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[10] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[11] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -736,7 +825,7 @@ func (x *UsageStats) String() string { + func (*UsageStats) ProtoMessage() {} + + func (x *UsageStats) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[10] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[11] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -749,7 +838,7 @@ func (x *UsageStats) ProtoReflect() protoreflect.Message { + + // Deprecated: Use UsageStats.ProtoReflect.Descriptor instead. + func (*UsageStats) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{10} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{11} + } + + func (x *UsageStats) GetPromptTokens() int32 { +@@ -766,6 +855,13 @@ func (x *UsageStats) GetCompletionTokens() int32 { + return 0 + } + ++func (x *UsageStats) GetToolCalls() int32 { ++ if x != nil { ++ return x.ToolCalls ++ } ++ return 0 ++} ++ + type Condition struct { + state protoimpl.MessageState `protogen:"open.v1"` + Type string `protobuf:"bytes,1,opt,name=type,proto3" json:"type,omitempty"` +@@ -779,7 +875,7 @@ type Condition struct { + + func (x *Condition) Reset() { + *x = Condition{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[11] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[12] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -791,7 +887,7 @@ func (x *Condition) String() string { + func (*Condition) ProtoMessage() {} + + func (x *Condition) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[11] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[12] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -804,7 +900,7 @@ func (x *Condition) ProtoReflect() protoreflect.Message { + + // Deprecated: Use Condition.ProtoReflect.Descriptor instead. + func (*Condition) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{11} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{12} + } + + func (x *Condition) GetType() string { +@@ -854,7 +950,7 @@ type Gateway struct { + + func (x *Gateway) Reset() { + *x = Gateway{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[12] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[13] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -866,7 +962,7 @@ func (x *Gateway) String() string { + func (*Gateway) ProtoMessage() {} + + func (x *Gateway) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[12] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[13] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -879,7 +975,7 @@ func (x *Gateway) ProtoReflect() protoreflect.Message { + + // Deprecated: Use Gateway.ProtoReflect.Descriptor instead. + func (*Gateway) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{12} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{13} + } + + func (x *Gateway) GetApiVersion() string { +@@ -920,7 +1016,7 @@ type GatewaySpec struct { + + func (x *GatewaySpec) Reset() { + *x = GatewaySpec{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[13] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[14] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -932,7 +1028,7 @@ func (x *GatewaySpec) String() string { + func (*GatewaySpec) ProtoMessage() {} + + func (x *GatewaySpec) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[13] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[14] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -945,7 +1041,7 @@ func (x *GatewaySpec) ProtoReflect() protoreflect.Message { + + // Deprecated: Use GatewaySpec.ProtoReflect.Descriptor instead. + func (*GatewaySpec) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{13} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{14} + } + + func (x *GatewaySpec) GetListeners() []*Listener { +@@ -973,7 +1069,7 @@ type Listener struct { + + func (x *Listener) Reset() { + *x = Listener{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[14] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[15] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -985,7 +1081,7 @@ func (x *Listener) String() string { + func (*Listener) ProtoMessage() {} + + func (x *Listener) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[14] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[15] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -998,7 +1094,7 @@ func (x *Listener) ProtoReflect() protoreflect.Message { + + // Deprecated: Use Listener.ProtoReflect.Descriptor instead. + func (*Listener) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{14} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{15} + } + + func (x *Listener) GetName() string { +@@ -1031,7 +1127,7 @@ type EgressConfig struct { + + func (x *EgressConfig) Reset() { + *x = EgressConfig{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[15] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[16] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1043,7 +1139,7 @@ func (x *EgressConfig) String() string { + func (*EgressConfig) ProtoMessage() {} + + func (x *EgressConfig) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[15] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[16] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1056,7 +1152,7 @@ func (x *EgressConfig) ProtoReflect() protoreflect.Message { + + // Deprecated: Use EgressConfig.ProtoReflect.Descriptor instead. + func (*EgressConfig) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{15} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{16} + } + + func (x *EgressConfig) GetAllowlist() *EgressAllowlist { +@@ -1075,7 +1171,7 @@ type EgressAllowlist struct { + + func (x *EgressAllowlist) Reset() { + *x = EgressAllowlist{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[16] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[17] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1087,7 +1183,7 @@ func (x *EgressAllowlist) String() string { + func (*EgressAllowlist) ProtoMessage() {} + + func (x *EgressAllowlist) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[16] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[17] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1100,7 +1196,7 @@ func (x *EgressAllowlist) ProtoReflect() protoreflect.Message { + + // Deprecated: Use EgressAllowlist.ProtoReflect.Descriptor instead. + func (*EgressAllowlist) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{16} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{17} + } + + func (x *EgressAllowlist) GetHosts() []*HostRule { +@@ -1120,7 +1216,7 @@ type HostRule struct { + + func (x *HostRule) Reset() { + *x = HostRule{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[17] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[18] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1132,7 +1228,7 @@ func (x *HostRule) String() string { + func (*HostRule) ProtoMessage() {} + + func (x *HostRule) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[17] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[18] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1145,7 +1241,7 @@ func (x *HostRule) ProtoReflect() protoreflect.Message { + + // Deprecated: Use HostRule.ProtoReflect.Descriptor instead. + func (*HostRule) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{17} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{18} + } + + func (x *HostRule) GetHost() string { +@@ -1174,7 +1270,7 @@ type Workspace struct { + + func (x *Workspace) Reset() { + *x = Workspace{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[18] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[19] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1186,7 +1282,7 @@ func (x *Workspace) String() string { + func (*Workspace) ProtoMessage() {} + + func (x *Workspace) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[18] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[19] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1199,7 +1295,7 @@ func (x *Workspace) ProtoReflect() protoreflect.Message { + + // Deprecated: Use Workspace.ProtoReflect.Descriptor instead. + func (*Workspace) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{18} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{19} + } + + func (x *Workspace) GetApiVersion() string { +@@ -1241,7 +1337,7 @@ type WorkspaceSpec struct { + + func (x *WorkspaceSpec) Reset() { + *x = WorkspaceSpec{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[19] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[20] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1253,7 +1349,7 @@ func (x *WorkspaceSpec) String() string { + func (*WorkspaceSpec) ProtoMessage() {} + + func (x *WorkspaceSpec) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[19] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[20] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1266,7 +1362,7 @@ func (x *WorkspaceSpec) ProtoReflect() protoreflect.Message { + + // Deprecated: Use WorkspaceSpec.ProtoReflect.Descriptor instead. + func (*WorkspaceSpec) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{19} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{20} + } + + func (x *WorkspaceSpec) GetGit() []*GitRepo { +@@ -1303,7 +1399,7 @@ type GitRepo struct { + + func (x *GitRepo) Reset() { + *x = GitRepo{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[20] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[21] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1315,7 +1411,7 @@ func (x *GitRepo) String() string { + func (*GitRepo) ProtoMessage() {} + + func (x *GitRepo) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[20] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[21] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1328,7 +1424,7 @@ func (x *GitRepo) ProtoReflect() protoreflect.Message { + + // Deprecated: Use GitRepo.ProtoReflect.Descriptor instead. + func (*GitRepo) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{20} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{21} + } + + func (x *GitRepo) GetName() string { +@@ -1376,7 +1472,7 @@ type MCPConfig struct { + + func (x *MCPConfig) Reset() { + *x = MCPConfig{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[21] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[22] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1388,7 +1484,7 @@ func (x *MCPConfig) String() string { + func (*MCPConfig) ProtoMessage() {} + + func (x *MCPConfig) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[21] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[22] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1401,7 +1497,7 @@ func (x *MCPConfig) ProtoReflect() protoreflect.Message { + + // Deprecated: Use MCPConfig.ProtoReflect.Descriptor instead. + func (*MCPConfig) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{21} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{22} + } + + func (x *MCPConfig) GetRegistries() []*MCPRegistry { +@@ -1430,7 +1526,7 @@ type MCPRegistry struct { + + func (x *MCPRegistry) Reset() { + *x = MCPRegistry{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[22] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[23] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1442,7 +1538,7 @@ func (x *MCPRegistry) String() string { + func (*MCPRegistry) ProtoMessage() {} + + func (x *MCPRegistry) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[22] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[23] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1455,7 +1551,7 @@ func (x *MCPRegistry) ProtoReflect() protoreflect.Message { + + // Deprecated: Use MCPRegistry.ProtoReflect.Descriptor instead. + func (*MCPRegistry) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{22} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{23} + } + + func (x *MCPRegistry) GetProvider() string { +@@ -1498,7 +1594,7 @@ type MCPServer struct { + + func (x *MCPServer) Reset() { + *x = MCPServer{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[23] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[24] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1510,7 +1606,7 @@ func (x *MCPServer) String() string { + func (*MCPServer) ProtoMessage() {} + + func (x *MCPServer) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[23] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[24] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1523,7 +1619,7 @@ func (x *MCPServer) ProtoReflect() protoreflect.Message { + + // Deprecated: Use MCPServer.ProtoReflect.Descriptor instead. + func (*MCPServer) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{23} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{24} + } + + func (x *MCPServer) GetName() string { +@@ -1564,7 +1660,7 @@ type SkillsConfig struct { + + func (x *SkillsConfig) Reset() { + *x = SkillsConfig{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[24] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[25] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1576,7 +1672,7 @@ func (x *SkillsConfig) String() string { + func (*SkillsConfig) ProtoMessage() {} + + func (x *SkillsConfig) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[24] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[25] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1589,7 +1685,7 @@ func (x *SkillsConfig) ProtoReflect() protoreflect.Message { + + // Deprecated: Use SkillsConfig.ProtoReflect.Descriptor instead. + func (*SkillsConfig) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{24} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{25} + } + + func (x *SkillsConfig) GetRegistries() []*SkillRegistry { +@@ -1617,7 +1713,7 @@ type SkillRegistry struct { + + func (x *SkillRegistry) Reset() { + *x = SkillRegistry{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[25] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[26] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1629,7 +1725,7 @@ func (x *SkillRegistry) String() string { + func (*SkillRegistry) ProtoMessage() {} + + func (x *SkillRegistry) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[25] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[26] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1642,7 +1738,7 @@ func (x *SkillRegistry) ProtoReflect() protoreflect.Message { + + // Deprecated: Use SkillRegistry.ProtoReflect.Descriptor instead. + func (*SkillRegistry) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{25} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{26} + } + + func (x *SkillRegistry) GetProvider() string { +@@ -1678,7 +1774,7 @@ type Model struct { + + func (x *Model) Reset() { + *x = Model{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[26] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[27] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1690,7 +1786,7 @@ func (x *Model) String() string { + func (*Model) ProtoMessage() {} + + func (x *Model) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[26] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[27] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1703,7 +1799,7 @@ func (x *Model) ProtoReflect() protoreflect.Message { + + // Deprecated: Use Model.ProtoReflect.Descriptor instead. + func (*Model) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{26} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{27} + } + + func (x *Model) GetApiVersion() string { +@@ -1748,7 +1844,7 @@ type ModelSpec struct { + + func (x *ModelSpec) Reset() { + *x = ModelSpec{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[27] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[28] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1760,7 +1856,7 @@ func (x *ModelSpec) String() string { + func (*ModelSpec) ProtoMessage() {} + + func (x *ModelSpec) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[27] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[28] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1773,7 +1869,7 @@ func (x *ModelSpec) ProtoReflect() protoreflect.Message { + + // Deprecated: Use ModelSpec.ProtoReflect.Descriptor instead. + func (*ModelSpec) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{27} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{28} + } + + func (x *ModelSpec) GetProvider() string { +@@ -1814,7 +1910,7 @@ type SecretKeyRef struct { + + func (x *SecretKeyRef) Reset() { + *x = SecretKeyRef{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[28] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[29] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1826,7 +1922,7 @@ func (x *SecretKeyRef) String() string { + func (*SecretKeyRef) ProtoMessage() {} + + func (x *SecretKeyRef) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[28] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[29] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1839,7 +1935,7 @@ func (x *SecretKeyRef) ProtoReflect() protoreflect.Message { + + // Deprecated: Use SecretKeyRef.ProtoReflect.Descriptor instead. + func (*SecretKeyRef) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{28} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{29} + } + + func (x *SecretKeyRef) GetName() string { +@@ -1867,7 +1963,7 @@ type GetTaskRequest struct { + + func (x *GetTaskRequest) Reset() { + *x = GetTaskRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[29] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[30] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1879,7 +1975,7 @@ func (x *GetTaskRequest) String() string { + func (*GetTaskRequest) ProtoMessage() {} + + func (x *GetTaskRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[29] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[30] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1892,7 +1988,7 @@ func (x *GetTaskRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use GetTaskRequest.ProtoReflect.Descriptor instead. + func (*GetTaskRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{29} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{30} + } + + func (x *GetTaskRequest) GetAtespace() string { +@@ -1909,6 +2005,110 @@ func (x *GetTaskRequest) GetName() string { + return "" + } + ++type GetTaskResultRequest struct { ++ state protoimpl.MessageState `protogen:"open.v1"` ++ Atespace string `protobuf:"bytes,1,opt,name=atespace,proto3" json:"atespace,omitempty"` ++ Name string `protobuf:"bytes,2,opt,name=name,proto3" json:"name,omitempty"` ++ unknownFields protoimpl.UnknownFields ++ sizeCache protoimpl.SizeCache ++} ++ ++func (x *GetTaskResultRequest) Reset() { ++ *x = GetTaskResultRequest{} ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[31] ++ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ++ ms.StoreMessageInfo(mi) ++} ++ ++func (x *GetTaskResultRequest) String() string { ++ return protoimpl.X.MessageStringOf(x) ++} ++ ++func (*GetTaskResultRequest) ProtoMessage() {} ++ ++func (x *GetTaskResultRequest) ProtoReflect() protoreflect.Message { ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[31] ++ if x != nil { ++ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ++ if ms.LoadMessageInfo() == nil { ++ ms.StoreMessageInfo(mi) ++ } ++ return ms ++ } ++ return mi.MessageOf(x) ++} ++ ++// Deprecated: Use GetTaskResultRequest.ProtoReflect.Descriptor instead. ++func (*GetTaskResultRequest) Descriptor() ([]byte, []int) { ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{31} ++} ++ ++func (x *GetTaskResultRequest) GetAtespace() string { ++ if x != nil { ++ return x.Atespace ++ } ++ return "" ++} ++ ++func (x *GetTaskResultRequest) GetName() string { ++ if x != nil { ++ return x.Name ++ } ++ return "" ++} ++ ++type TaskResult struct { ++ state protoimpl.MessageState `protogen:"open.v1"` ++ Content []byte `protobuf:"bytes,1,opt,name=content,proto3" json:"content,omitempty"` ++ Sha256 string `protobuf:"bytes,2,opt,name=sha256,proto3" json:"sha256,omitempty"` ++ unknownFields protoimpl.UnknownFields ++ sizeCache protoimpl.SizeCache ++} ++ ++func (x *TaskResult) Reset() { ++ *x = TaskResult{} ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[32] ++ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ++ ms.StoreMessageInfo(mi) ++} ++ ++func (x *TaskResult) String() string { ++ return protoimpl.X.MessageStringOf(x) ++} ++ ++func (*TaskResult) ProtoMessage() {} ++ ++func (x *TaskResult) ProtoReflect() protoreflect.Message { ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[32] ++ if x != nil { ++ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) ++ if ms.LoadMessageInfo() == nil { ++ ms.StoreMessageInfo(mi) ++ } ++ return ms ++ } ++ return mi.MessageOf(x) ++} ++ ++// Deprecated: Use TaskResult.ProtoReflect.Descriptor instead. ++func (*TaskResult) Descriptor() ([]byte, []int) { ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{32} ++} ++ ++func (x *TaskResult) GetContent() []byte { ++ if x != nil { ++ return x.Content ++ } ++ return nil ++} ++ ++func (x *TaskResult) GetSha256() string { ++ if x != nil { ++ return x.Sha256 ++ } ++ return "" ++} ++ + type ListTasksRequest struct { + state protoimpl.MessageState `protogen:"open.v1"` + Atespace string `protobuf:"bytes,1,opt,name=atespace,proto3" json:"atespace,omitempty"` +@@ -1920,7 +2120,7 @@ type ListTasksRequest struct { + + func (x *ListTasksRequest) Reset() { + *x = ListTasksRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[30] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[33] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1932,7 +2132,7 @@ func (x *ListTasksRequest) String() string { + func (*ListTasksRequest) ProtoMessage() {} + + func (x *ListTasksRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[30] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[33] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -1945,7 +2145,7 @@ func (x *ListTasksRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use ListTasksRequest.ProtoReflect.Descriptor instead. + func (*ListTasksRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{30} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{33} + } + + func (x *ListTasksRequest) GetAtespace() string { +@@ -1978,7 +2178,7 @@ type ListTasksResponse struct { + + func (x *ListTasksResponse) Reset() { + *x = ListTasksResponse{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[31] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[34] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -1990,7 +2190,7 @@ func (x *ListTasksResponse) String() string { + func (*ListTasksResponse) ProtoMessage() {} + + func (x *ListTasksResponse) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[31] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[34] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2003,7 +2203,7 @@ func (x *ListTasksResponse) ProtoReflect() protoreflect.Message { + + // Deprecated: Use ListTasksResponse.ProtoReflect.Descriptor instead. + func (*ListTasksResponse) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{31} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{34} + } + + func (x *ListTasksResponse) GetTasks() []*Task { +@@ -2022,7 +2222,7 @@ type UpdateTaskRequest struct { + + func (x *UpdateTaskRequest) Reset() { + *x = UpdateTaskRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[32] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[35] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2034,7 +2234,7 @@ func (x *UpdateTaskRequest) String() string { + func (*UpdateTaskRequest) ProtoMessage() {} + + func (x *UpdateTaskRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[32] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[35] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2047,7 +2247,7 @@ func (x *UpdateTaskRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use UpdateTaskRequest.ProtoReflect.Descriptor instead. + func (*UpdateTaskRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{32} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{35} + } + + func (x *UpdateTaskRequest) GetTask() *Task { +@@ -2067,7 +2267,7 @@ type DeleteTaskRequest struct { + + func (x *DeleteTaskRequest) Reset() { + *x = DeleteTaskRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[33] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[36] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2079,7 +2279,7 @@ func (x *DeleteTaskRequest) String() string { + func (*DeleteTaskRequest) ProtoMessage() {} + + func (x *DeleteTaskRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[33] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[36] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2092,7 +2292,7 @@ func (x *DeleteTaskRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use DeleteTaskRequest.ProtoReflect.Descriptor instead. + func (*DeleteTaskRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{33} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{36} + } + + func (x *DeleteTaskRequest) GetAtespace() string { +@@ -2117,7 +2317,7 @@ type DeleteTaskResponse struct { + + func (x *DeleteTaskResponse) Reset() { + *x = DeleteTaskResponse{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[34] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[37] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2129,7 +2329,7 @@ func (x *DeleteTaskResponse) String() string { + func (*DeleteTaskResponse) ProtoMessage() {} + + func (x *DeleteTaskResponse) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[34] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[37] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2142,7 +2342,7 @@ func (x *DeleteTaskResponse) ProtoReflect() protoreflect.Message { + + // Deprecated: Use DeleteTaskResponse.ProtoReflect.Descriptor instead. + func (*DeleteTaskResponse) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{34} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{37} + } + + type SuspendTaskRequest struct { +@@ -2155,7 +2355,7 @@ type SuspendTaskRequest struct { + + func (x *SuspendTaskRequest) Reset() { + *x = SuspendTaskRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[35] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[38] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2167,7 +2367,7 @@ func (x *SuspendTaskRequest) String() string { + func (*SuspendTaskRequest) ProtoMessage() {} + + func (x *SuspendTaskRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[35] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[38] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2180,7 +2380,7 @@ func (x *SuspendTaskRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use SuspendTaskRequest.ProtoReflect.Descriptor instead. + func (*SuspendTaskRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{35} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{38} + } + + func (x *SuspendTaskRequest) GetAtespace() string { +@@ -2207,7 +2407,7 @@ type ResumeTaskRequest struct { + + func (x *ResumeTaskRequest) Reset() { + *x = ResumeTaskRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[36] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[39] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2219,7 +2419,7 @@ func (x *ResumeTaskRequest) String() string { + func (*ResumeTaskRequest) ProtoMessage() {} + + func (x *ResumeTaskRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[36] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[39] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2232,7 +2432,7 @@ func (x *ResumeTaskRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use ResumeTaskRequest.ProtoReflect.Descriptor instead. + func (*ResumeTaskRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{36} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{39} + } + + func (x *ResumeTaskRequest) GetAtespace() string { +@@ -2259,7 +2459,7 @@ type WatchTaskRequest struct { + + func (x *WatchTaskRequest) Reset() { + *x = WatchTaskRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[37] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[40] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2271,7 +2471,7 @@ func (x *WatchTaskRequest) String() string { + func (*WatchTaskRequest) ProtoMessage() {} + + func (x *WatchTaskRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[37] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[40] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2284,7 +2484,7 @@ func (x *WatchTaskRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use WatchTaskRequest.ProtoReflect.Descriptor instead. + func (*WatchTaskRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{37} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{40} + } + + func (x *WatchTaskRequest) GetAtespace() string { +@@ -2311,7 +2511,7 @@ type WatchTaskResponse struct { + + func (x *WatchTaskResponse) Reset() { + *x = WatchTaskResponse{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[38] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[41] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2323,7 +2523,7 @@ func (x *WatchTaskResponse) String() string { + func (*WatchTaskResponse) ProtoMessage() {} + + func (x *WatchTaskResponse) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[38] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[41] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2336,7 +2536,7 @@ func (x *WatchTaskResponse) ProtoReflect() protoreflect.Message { + + // Deprecated: Use WatchTaskResponse.ProtoReflect.Descriptor instead. + func (*WatchTaskResponse) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{38} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{41} + } + + func (x *WatchTaskResponse) GetTask() *Task { +@@ -2364,7 +2564,7 @@ type GetGatewayRequest struct { + + func (x *GetGatewayRequest) Reset() { + *x = GetGatewayRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[39] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[42] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2376,7 +2576,7 @@ func (x *GetGatewayRequest) String() string { + func (*GetGatewayRequest) ProtoMessage() {} + + func (x *GetGatewayRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[39] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[42] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2389,7 +2589,7 @@ func (x *GetGatewayRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use GetGatewayRequest.ProtoReflect.Descriptor instead. + func (*GetGatewayRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{39} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{42} + } + + func (x *GetGatewayRequest) GetAtespace() string { +@@ -2415,7 +2615,7 @@ type ListGatewaysRequest struct { + + func (x *ListGatewaysRequest) Reset() { + *x = ListGatewaysRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[40] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[43] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2427,7 +2627,7 @@ func (x *ListGatewaysRequest) String() string { + func (*ListGatewaysRequest) ProtoMessage() {} + + func (x *ListGatewaysRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[40] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[43] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2440,7 +2640,7 @@ func (x *ListGatewaysRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use ListGatewaysRequest.ProtoReflect.Descriptor instead. + func (*ListGatewaysRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{40} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{43} + } + + func (x *ListGatewaysRequest) GetAtespace() string { +@@ -2459,7 +2659,7 @@ type ListGatewaysResponse struct { + + func (x *ListGatewaysResponse) Reset() { + *x = ListGatewaysResponse{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[41] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[44] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2471,7 +2671,7 @@ func (x *ListGatewaysResponse) String() string { + func (*ListGatewaysResponse) ProtoMessage() {} + + func (x *ListGatewaysResponse) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[41] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[44] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2484,7 +2684,7 @@ func (x *ListGatewaysResponse) ProtoReflect() protoreflect.Message { + + // Deprecated: Use ListGatewaysResponse.ProtoReflect.Descriptor instead. + func (*ListGatewaysResponse) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{41} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{44} + } + + func (x *ListGatewaysResponse) GetGateways() []*Gateway { +@@ -2503,7 +2703,7 @@ type UpdateGatewayRequest struct { + + func (x *UpdateGatewayRequest) Reset() { + *x = UpdateGatewayRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[42] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[45] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2515,7 +2715,7 @@ func (x *UpdateGatewayRequest) String() string { + func (*UpdateGatewayRequest) ProtoMessage() {} + + func (x *UpdateGatewayRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[42] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[45] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2528,7 +2728,7 @@ func (x *UpdateGatewayRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use UpdateGatewayRequest.ProtoReflect.Descriptor instead. + func (*UpdateGatewayRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{42} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{45} + } + + func (x *UpdateGatewayRequest) GetGateway() *Gateway { +@@ -2548,7 +2748,7 @@ type DeleteGatewayRequest struct { + + func (x *DeleteGatewayRequest) Reset() { + *x = DeleteGatewayRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[43] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[46] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2560,7 +2760,7 @@ func (x *DeleteGatewayRequest) String() string { + func (*DeleteGatewayRequest) ProtoMessage() {} + + func (x *DeleteGatewayRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[43] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[46] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2573,7 +2773,7 @@ func (x *DeleteGatewayRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use DeleteGatewayRequest.ProtoReflect.Descriptor instead. + func (*DeleteGatewayRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{43} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{46} + } + + func (x *DeleteGatewayRequest) GetAtespace() string { +@@ -2598,7 +2798,7 @@ type DeleteGatewayResponse struct { + + func (x *DeleteGatewayResponse) Reset() { + *x = DeleteGatewayResponse{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[44] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[47] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2610,7 +2810,7 @@ func (x *DeleteGatewayResponse) String() string { + func (*DeleteGatewayResponse) ProtoMessage() {} + + func (x *DeleteGatewayResponse) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[44] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[47] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2623,7 +2823,7 @@ func (x *DeleteGatewayResponse) ProtoReflect() protoreflect.Message { + + // Deprecated: Use DeleteGatewayResponse.ProtoReflect.Descriptor instead. + func (*DeleteGatewayResponse) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{44} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{47} + } + + // Workspaces +@@ -2637,7 +2837,7 @@ type GetWorkspaceRequest struct { + + func (x *GetWorkspaceRequest) Reset() { + *x = GetWorkspaceRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[45] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[48] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2649,7 +2849,7 @@ func (x *GetWorkspaceRequest) String() string { + func (*GetWorkspaceRequest) ProtoMessage() {} + + func (x *GetWorkspaceRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[45] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[48] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2662,7 +2862,7 @@ func (x *GetWorkspaceRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use GetWorkspaceRequest.ProtoReflect.Descriptor instead. + func (*GetWorkspaceRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{45} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{48} + } + + func (x *GetWorkspaceRequest) GetAtespace() string { +@@ -2688,7 +2888,7 @@ type ListWorkspacesRequest struct { + + func (x *ListWorkspacesRequest) Reset() { + *x = ListWorkspacesRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[46] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[49] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2700,7 +2900,7 @@ func (x *ListWorkspacesRequest) String() string { + func (*ListWorkspacesRequest) ProtoMessage() {} + + func (x *ListWorkspacesRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[46] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[49] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2713,7 +2913,7 @@ func (x *ListWorkspacesRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use ListWorkspacesRequest.ProtoReflect.Descriptor instead. + func (*ListWorkspacesRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{46} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{49} + } + + func (x *ListWorkspacesRequest) GetAtespace() string { +@@ -2732,7 +2932,7 @@ type ListWorkspacesResponse struct { + + func (x *ListWorkspacesResponse) Reset() { + *x = ListWorkspacesResponse{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[47] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[50] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2744,7 +2944,7 @@ func (x *ListWorkspacesResponse) String() string { + func (*ListWorkspacesResponse) ProtoMessage() {} + + func (x *ListWorkspacesResponse) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[47] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[50] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2757,7 +2957,7 @@ func (x *ListWorkspacesResponse) ProtoReflect() protoreflect.Message { + + // Deprecated: Use ListWorkspacesResponse.ProtoReflect.Descriptor instead. + func (*ListWorkspacesResponse) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{47} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{50} + } + + func (x *ListWorkspacesResponse) GetWorkspaces() []*Workspace { +@@ -2776,7 +2976,7 @@ type UpdateWorkspaceRequest struct { + + func (x *UpdateWorkspaceRequest) Reset() { + *x = UpdateWorkspaceRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[48] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[51] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2788,7 +2988,7 @@ func (x *UpdateWorkspaceRequest) String() string { + func (*UpdateWorkspaceRequest) ProtoMessage() {} + + func (x *UpdateWorkspaceRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[48] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[51] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2801,7 +3001,7 @@ func (x *UpdateWorkspaceRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use UpdateWorkspaceRequest.ProtoReflect.Descriptor instead. + func (*UpdateWorkspaceRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{48} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{51} + } + + func (x *UpdateWorkspaceRequest) GetWorkspace() *Workspace { +@@ -2821,7 +3021,7 @@ type DeleteWorkspaceRequest struct { + + func (x *DeleteWorkspaceRequest) Reset() { + *x = DeleteWorkspaceRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[49] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[52] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2833,7 +3033,7 @@ func (x *DeleteWorkspaceRequest) String() string { + func (*DeleteWorkspaceRequest) ProtoMessage() {} + + func (x *DeleteWorkspaceRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[49] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[52] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2846,7 +3046,7 @@ func (x *DeleteWorkspaceRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use DeleteWorkspaceRequest.ProtoReflect.Descriptor instead. + func (*DeleteWorkspaceRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{49} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{52} + } + + func (x *DeleteWorkspaceRequest) GetAtespace() string { +@@ -2871,7 +3071,7 @@ type DeleteWorkspaceResponse struct { + + func (x *DeleteWorkspaceResponse) Reset() { + *x = DeleteWorkspaceResponse{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[50] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[53] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2883,7 +3083,7 @@ func (x *DeleteWorkspaceResponse) String() string { + func (*DeleteWorkspaceResponse) ProtoMessage() {} + + func (x *DeleteWorkspaceResponse) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[50] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[53] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2896,7 +3096,7 @@ func (x *DeleteWorkspaceResponse) ProtoReflect() protoreflect.Message { + + // Deprecated: Use DeleteWorkspaceResponse.ProtoReflect.Descriptor instead. + func (*DeleteWorkspaceResponse) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{50} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{53} + } + + // Models +@@ -2910,7 +3110,7 @@ type GetModelRequest struct { + + func (x *GetModelRequest) Reset() { + *x = GetModelRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[51] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[54] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2922,7 +3122,7 @@ func (x *GetModelRequest) String() string { + func (*GetModelRequest) ProtoMessage() {} + + func (x *GetModelRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[51] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[54] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2935,7 +3135,7 @@ func (x *GetModelRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use GetModelRequest.ProtoReflect.Descriptor instead. + func (*GetModelRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{51} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{54} + } + + func (x *GetModelRequest) GetAtespace() string { +@@ -2961,7 +3161,7 @@ type ListModelsRequest struct { + + func (x *ListModelsRequest) Reset() { + *x = ListModelsRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[52] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[55] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -2973,7 +3173,7 @@ func (x *ListModelsRequest) String() string { + func (*ListModelsRequest) ProtoMessage() {} + + func (x *ListModelsRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[52] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[55] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -2986,7 +3186,7 @@ func (x *ListModelsRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use ListModelsRequest.ProtoReflect.Descriptor instead. + func (*ListModelsRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{52} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{55} + } + + func (x *ListModelsRequest) GetAtespace() string { +@@ -3005,7 +3205,7 @@ type ListModelsResponse struct { + + func (x *ListModelsResponse) Reset() { + *x = ListModelsResponse{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[53] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[56] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -3017,7 +3217,7 @@ func (x *ListModelsResponse) String() string { + func (*ListModelsResponse) ProtoMessage() {} + + func (x *ListModelsResponse) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[53] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[56] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -3030,7 +3230,7 @@ func (x *ListModelsResponse) ProtoReflect() protoreflect.Message { + + // Deprecated: Use ListModelsResponse.ProtoReflect.Descriptor instead. + func (*ListModelsResponse) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{53} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{56} + } + + func (x *ListModelsResponse) GetModels() []*Model { +@@ -3049,7 +3249,7 @@ type UpdateModelRequest struct { + + func (x *UpdateModelRequest) Reset() { + *x = UpdateModelRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[54] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[57] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -3061,7 +3261,7 @@ func (x *UpdateModelRequest) String() string { + func (*UpdateModelRequest) ProtoMessage() {} + + func (x *UpdateModelRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[54] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[57] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -3074,7 +3274,7 @@ func (x *UpdateModelRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use UpdateModelRequest.ProtoReflect.Descriptor instead. + func (*UpdateModelRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{54} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{57} + } + + func (x *UpdateModelRequest) GetModel() *Model { +@@ -3094,7 +3294,7 @@ type DeleteModelRequest struct { + + func (x *DeleteModelRequest) Reset() { + *x = DeleteModelRequest{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[55] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[58] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -3106,7 +3306,7 @@ func (x *DeleteModelRequest) String() string { + func (*DeleteModelRequest) ProtoMessage() {} + + func (x *DeleteModelRequest) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[55] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[58] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -3119,7 +3319,7 @@ func (x *DeleteModelRequest) ProtoReflect() protoreflect.Message { + + // Deprecated: Use DeleteModelRequest.ProtoReflect.Descriptor instead. + func (*DeleteModelRequest) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{55} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{58} + } + + func (x *DeleteModelRequest) GetAtespace() string { +@@ -3144,7 +3344,7 @@ type DeleteModelResponse struct { + + func (x *DeleteModelResponse) Reset() { + *x = DeleteModelResponse{} +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[56] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[59] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) + } +@@ -3156,7 +3356,7 @@ func (x *DeleteModelResponse) String() string { + func (*DeleteModelResponse) ProtoMessage() {} + + func (x *DeleteModelResponse) ProtoReflect() protoreflect.Message { +- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[56] ++ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[59] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { +@@ -3169,7 +3369,7 @@ func (x *DeleteModelResponse) ProtoReflect() protoreflect.Message { + + // Deprecated: Use DeleteModelResponse.ProtoReflect.Descriptor instead. + func (*DeleteModelResponse) Descriptor() ([]byte, []int) { +- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{56} ++ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{59} + } + + var File_pkg_apis_v1alpha1_ax_proto protoreflect.FileDescriptor +@@ -3218,7 +3418,7 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + + "\x04goal\x18\x03 \x01(\tR\x04goal\" \n" + + "\n" + + "GatewayRef\x12\x12\n" + +- "\x04name\x18\x01 \x01(\tR\x04name\"\x95\x02\n" + ++ "\x04name\x18\x01 \x01(\tR\x04name\"\xcb\x02\n" + + "\n" + + "TaskStatus\x12\x14\n" + + "\x05phase\x18\x01 \x01(\tR\x05phase\x12\x0e\n" + +@@ -3229,15 +3429,25 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + + "\x05usage\x18\x06 \x01(\v2\x17.ax.v1alpha1.UsageStatsR\x05usage\x126\n" + + "\n" + + "conditions\x18\a \x03(\v2\x16.ax.v1alpha1.ConditionR\n" + +- "conditions\"x\n" + ++ "conditions\x124\n" + ++ "\acommand\x18\b \x01(\v2\x1a.ax.v1alpha1.CommandStatusR\acommand\"\xc9\x01\n" + ++ "\rCommandStatus\x12\x16\n" + ++ "\x06exited\x18\x01 \x01(\bR\x06exited\x12\x1b\n" + ++ "\texit_code\x18\x02 \x01(\x05R\bexitCode\x12;\n" + ++ "\vfinished_at\x18\x03 \x01(\v2\x1a.google.protobuf.TimestampR\n" + ++ "finishedAt\x12!\n" + ++ "\fresult_bytes\x18\x04 \x01(\x03R\vresultBytes\x12#\n" + ++ "\rresult_sha256\x18\x05 \x01(\tR\fresultSha256\"x\n" + + "\x0fPendingApproval\x12\x0e\n" + + "\x02id\x18\x01 \x01(\tR\x02id\x12\x16\n" + + "\x06action\x18\x02 \x01(\tR\x06action\x12=\n" + +- "\frequested_at\x18\x03 \x01(\v2\x1a.google.protobuf.TimestampR\vrequestedAt\"^\n" + ++ "\frequested_at\x18\x03 \x01(\v2\x1a.google.protobuf.TimestampR\vrequestedAt\"}\n" + + "\n" + + "UsageStats\x12#\n" + + "\rprompt_tokens\x18\x01 \x01(\x05R\fpromptTokens\x12+\n" + +- "\x11completion_tokens\x18\x02 \x01(\x05R\x10completionTokens\"\xb7\x01\n" + ++ "\x11completion_tokens\x18\x02 \x01(\x05R\x10completionTokens\x12\x1d\n" + ++ "\n" + ++ "tool_calls\x18\x03 \x01(\x05R\ttoolCalls\"\xb7\x01\n" + + "\tCondition\x12\x12\n" + + "\x04type\x18\x01 \x01(\tR\x04type\x12\x16\n" + + "\x06status\x18\x02 \x01(\tR\x06status\x12L\n" + +@@ -3324,7 +3534,14 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + + "\x03key\x18\x02 \x01(\tR\x03key\"@\n" + + "\x0eGetTaskRequest\x12\x1a\n" + + "\batespace\x18\x01 \x01(\tR\batespace\x12\x12\n" + +- "\x04name\x18\x02 \x01(\tR\x04name\"\\\n" + ++ "\x04name\x18\x02 \x01(\tR\x04name\"F\n" + ++ "\x14GetTaskResultRequest\x12\x1a\n" + ++ "\batespace\x18\x01 \x01(\tR\batespace\x12\x12\n" + ++ "\x04name\x18\x02 \x01(\tR\x04name\">\n" + ++ "\n" + ++ "TaskResult\x12\x18\n" + ++ "\acontent\x18\x01 \x01(\fR\acontent\x12\x16\n" + ++ "\x06sha256\x18\x02 \x01(\tR\x06sha256\"\\\n" + + "\x10ListTasksRequest\x12\x1a\n" + + "\batespace\x18\x01 \x01(\tR\batespace\x12\x14\n" + + "\x05limit\x18\x02 \x01(\x03R\x05limit\x12\x16\n" + +@@ -3389,7 +3606,7 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + + "\x12DeleteModelRequest\x12\x1a\n" + + "\batespace\x18\x01 \x01(\tR\batespace\x12\x12\n" + + "\x04name\x18\x02 \x01(\tR\x04name\"\x15\n" + +- "\x13DeleteModelResponse2\x9e\v\n" + ++ "\x13DeleteModelResponse2\xeb\v\n" + + "\x02AX\x129\n" + + "\aGetTask\x12\x1b.ax.v1alpha1.GetTaskRequest\x1a\x11.ax.v1alpha1.Task\x12J\n" + + "\tListTasks\x12\x1d.ax.v1alpha1.ListTasksRequest\x1a\x1e.ax.v1alpha1.ListTasksResponse\x12?\n" + +@@ -3400,7 +3617,8 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + + "\vSuspendTask\x12\x1f.ax.v1alpha1.SuspendTaskRequest\x1a\x11.ax.v1alpha1.Task\x12?\n" + + "\n" + + "ResumeTask\x12\x1e.ax.v1alpha1.ResumeTaskRequest\x1a\x11.ax.v1alpha1.Task\x12L\n" + +- "\tWatchTask\x12\x1d.ax.v1alpha1.WatchTaskRequest\x1a\x1e.ax.v1alpha1.WatchTaskResponse0\x01\x12B\n" + ++ "\tWatchTask\x12\x1d.ax.v1alpha1.WatchTaskRequest\x1a\x1e.ax.v1alpha1.WatchTaskResponse0\x01\x12K\n" + ++ "\rGetTaskResult\x12!.ax.v1alpha1.GetTaskResultRequest\x1a\x17.ax.v1alpha1.TaskResult\x12B\n" + + "\n" + + "GetGateway\x12\x1e.ax.v1alpha1.GetGatewayRequest\x1a\x14.ax.v1alpha1.Gateway\x12S\n" + + "\fListGateways\x12 .ax.v1alpha1.ListGatewaysRequest\x1a!.ax.v1alpha1.ListGatewaysResponse\x12H\n" + +@@ -3428,7 +3646,7 @@ func file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP() []byte { + return file_pkg_apis_v1alpha1_ax_proto_rawDescData + } + +-var file_pkg_apis_v1alpha1_ax_proto_msgTypes = make([]protoimpl.MessageInfo, 57) ++var file_pkg_apis_v1alpha1_ax_proto_msgTypes = make([]protoimpl.MessageInfo, 60) + var file_pkg_apis_v1alpha1_ax_proto_goTypes = []any{ + (*ObjectMeta)(nil), // 0: ax.v1alpha1.ObjectMeta + (*Task)(nil), // 1: ax.v1alpha1.Task +@@ -3439,59 +3657,62 @@ var file_pkg_apis_v1alpha1_ax_proto_goTypes = []any{ + (*WorkspaceRef)(nil), // 6: ax.v1alpha1.WorkspaceRef + (*GatewayRef)(nil), // 7: ax.v1alpha1.GatewayRef + (*TaskStatus)(nil), // 8: ax.v1alpha1.TaskStatus +- (*PendingApproval)(nil), // 9: ax.v1alpha1.PendingApproval +- (*UsageStats)(nil), // 10: ax.v1alpha1.UsageStats +- (*Condition)(nil), // 11: ax.v1alpha1.Condition +- (*Gateway)(nil), // 12: ax.v1alpha1.Gateway +- (*GatewaySpec)(nil), // 13: ax.v1alpha1.GatewaySpec +- (*Listener)(nil), // 14: ax.v1alpha1.Listener +- (*EgressConfig)(nil), // 15: ax.v1alpha1.EgressConfig +- (*EgressAllowlist)(nil), // 16: ax.v1alpha1.EgressAllowlist +- (*HostRule)(nil), // 17: ax.v1alpha1.HostRule +- (*Workspace)(nil), // 18: ax.v1alpha1.Workspace +- (*WorkspaceSpec)(nil), // 19: ax.v1alpha1.WorkspaceSpec +- (*GitRepo)(nil), // 20: ax.v1alpha1.GitRepo +- (*MCPConfig)(nil), // 21: ax.v1alpha1.MCPConfig +- (*MCPRegistry)(nil), // 22: ax.v1alpha1.MCPRegistry +- (*MCPServer)(nil), // 23: ax.v1alpha1.MCPServer +- (*SkillsConfig)(nil), // 24: ax.v1alpha1.SkillsConfig +- (*SkillRegistry)(nil), // 25: ax.v1alpha1.SkillRegistry +- (*Model)(nil), // 26: ax.v1alpha1.Model +- (*ModelSpec)(nil), // 27: ax.v1alpha1.ModelSpec +- (*SecretKeyRef)(nil), // 28: ax.v1alpha1.SecretKeyRef +- (*GetTaskRequest)(nil), // 29: ax.v1alpha1.GetTaskRequest +- (*ListTasksRequest)(nil), // 30: ax.v1alpha1.ListTasksRequest +- (*ListTasksResponse)(nil), // 31: ax.v1alpha1.ListTasksResponse +- (*UpdateTaskRequest)(nil), // 32: ax.v1alpha1.UpdateTaskRequest +- (*DeleteTaskRequest)(nil), // 33: ax.v1alpha1.DeleteTaskRequest +- (*DeleteTaskResponse)(nil), // 34: ax.v1alpha1.DeleteTaskResponse +- (*SuspendTaskRequest)(nil), // 35: ax.v1alpha1.SuspendTaskRequest +- (*ResumeTaskRequest)(nil), // 36: ax.v1alpha1.ResumeTaskRequest +- (*WatchTaskRequest)(nil), // 37: ax.v1alpha1.WatchTaskRequest +- (*WatchTaskResponse)(nil), // 38: ax.v1alpha1.WatchTaskResponse +- (*GetGatewayRequest)(nil), // 39: ax.v1alpha1.GetGatewayRequest +- (*ListGatewaysRequest)(nil), // 40: ax.v1alpha1.ListGatewaysRequest +- (*ListGatewaysResponse)(nil), // 41: ax.v1alpha1.ListGatewaysResponse +- (*UpdateGatewayRequest)(nil), // 42: ax.v1alpha1.UpdateGatewayRequest +- (*DeleteGatewayRequest)(nil), // 43: ax.v1alpha1.DeleteGatewayRequest +- (*DeleteGatewayResponse)(nil), // 44: ax.v1alpha1.DeleteGatewayResponse +- (*GetWorkspaceRequest)(nil), // 45: ax.v1alpha1.GetWorkspaceRequest +- (*ListWorkspacesRequest)(nil), // 46: ax.v1alpha1.ListWorkspacesRequest +- (*ListWorkspacesResponse)(nil), // 47: ax.v1alpha1.ListWorkspacesResponse +- (*UpdateWorkspaceRequest)(nil), // 48: ax.v1alpha1.UpdateWorkspaceRequest +- (*DeleteWorkspaceRequest)(nil), // 49: ax.v1alpha1.DeleteWorkspaceRequest +- (*DeleteWorkspaceResponse)(nil), // 50: ax.v1alpha1.DeleteWorkspaceResponse +- (*GetModelRequest)(nil), // 51: ax.v1alpha1.GetModelRequest +- (*ListModelsRequest)(nil), // 52: ax.v1alpha1.ListModelsRequest +- (*ListModelsResponse)(nil), // 53: ax.v1alpha1.ListModelsResponse +- (*UpdateModelRequest)(nil), // 54: ax.v1alpha1.UpdateModelRequest +- (*DeleteModelRequest)(nil), // 55: ax.v1alpha1.DeleteModelRequest +- (*DeleteModelResponse)(nil), // 56: ax.v1alpha1.DeleteModelResponse +- (*timestamppb.Timestamp)(nil), // 57: google.protobuf.Timestamp +- (*structpb.Struct)(nil), // 58: google.protobuf.Struct ++ (*CommandStatus)(nil), // 9: ax.v1alpha1.CommandStatus ++ (*PendingApproval)(nil), // 10: ax.v1alpha1.PendingApproval ++ (*UsageStats)(nil), // 11: ax.v1alpha1.UsageStats ++ (*Condition)(nil), // 12: ax.v1alpha1.Condition ++ (*Gateway)(nil), // 13: ax.v1alpha1.Gateway ++ (*GatewaySpec)(nil), // 14: ax.v1alpha1.GatewaySpec ++ (*Listener)(nil), // 15: ax.v1alpha1.Listener ++ (*EgressConfig)(nil), // 16: ax.v1alpha1.EgressConfig ++ (*EgressAllowlist)(nil), // 17: ax.v1alpha1.EgressAllowlist ++ (*HostRule)(nil), // 18: ax.v1alpha1.HostRule ++ (*Workspace)(nil), // 19: ax.v1alpha1.Workspace ++ (*WorkspaceSpec)(nil), // 20: ax.v1alpha1.WorkspaceSpec ++ (*GitRepo)(nil), // 21: ax.v1alpha1.GitRepo ++ (*MCPConfig)(nil), // 22: ax.v1alpha1.MCPConfig ++ (*MCPRegistry)(nil), // 23: ax.v1alpha1.MCPRegistry ++ (*MCPServer)(nil), // 24: ax.v1alpha1.MCPServer ++ (*SkillsConfig)(nil), // 25: ax.v1alpha1.SkillsConfig ++ (*SkillRegistry)(nil), // 26: ax.v1alpha1.SkillRegistry ++ (*Model)(nil), // 27: ax.v1alpha1.Model ++ (*ModelSpec)(nil), // 28: ax.v1alpha1.ModelSpec ++ (*SecretKeyRef)(nil), // 29: ax.v1alpha1.SecretKeyRef ++ (*GetTaskRequest)(nil), // 30: ax.v1alpha1.GetTaskRequest ++ (*GetTaskResultRequest)(nil), // 31: ax.v1alpha1.GetTaskResultRequest ++ (*TaskResult)(nil), // 32: ax.v1alpha1.TaskResult ++ (*ListTasksRequest)(nil), // 33: ax.v1alpha1.ListTasksRequest ++ (*ListTasksResponse)(nil), // 34: ax.v1alpha1.ListTasksResponse ++ (*UpdateTaskRequest)(nil), // 35: ax.v1alpha1.UpdateTaskRequest ++ (*DeleteTaskRequest)(nil), // 36: ax.v1alpha1.DeleteTaskRequest ++ (*DeleteTaskResponse)(nil), // 37: ax.v1alpha1.DeleteTaskResponse ++ (*SuspendTaskRequest)(nil), // 38: ax.v1alpha1.SuspendTaskRequest ++ (*ResumeTaskRequest)(nil), // 39: ax.v1alpha1.ResumeTaskRequest ++ (*WatchTaskRequest)(nil), // 40: ax.v1alpha1.WatchTaskRequest ++ (*WatchTaskResponse)(nil), // 41: ax.v1alpha1.WatchTaskResponse ++ (*GetGatewayRequest)(nil), // 42: ax.v1alpha1.GetGatewayRequest ++ (*ListGatewaysRequest)(nil), // 43: ax.v1alpha1.ListGatewaysRequest ++ (*ListGatewaysResponse)(nil), // 44: ax.v1alpha1.ListGatewaysResponse ++ (*UpdateGatewayRequest)(nil), // 45: ax.v1alpha1.UpdateGatewayRequest ++ (*DeleteGatewayRequest)(nil), // 46: ax.v1alpha1.DeleteGatewayRequest ++ (*DeleteGatewayResponse)(nil), // 47: ax.v1alpha1.DeleteGatewayResponse ++ (*GetWorkspaceRequest)(nil), // 48: ax.v1alpha1.GetWorkspaceRequest ++ (*ListWorkspacesRequest)(nil), // 49: ax.v1alpha1.ListWorkspacesRequest ++ (*ListWorkspacesResponse)(nil), // 50: ax.v1alpha1.ListWorkspacesResponse ++ (*UpdateWorkspaceRequest)(nil), // 51: ax.v1alpha1.UpdateWorkspaceRequest ++ (*DeleteWorkspaceRequest)(nil), // 52: ax.v1alpha1.DeleteWorkspaceRequest ++ (*DeleteWorkspaceResponse)(nil), // 53: ax.v1alpha1.DeleteWorkspaceResponse ++ (*GetModelRequest)(nil), // 54: ax.v1alpha1.GetModelRequest ++ (*ListModelsRequest)(nil), // 55: ax.v1alpha1.ListModelsRequest ++ (*ListModelsResponse)(nil), // 56: ax.v1alpha1.ListModelsResponse ++ (*UpdateModelRequest)(nil), // 57: ax.v1alpha1.UpdateModelRequest ++ (*DeleteModelRequest)(nil), // 58: ax.v1alpha1.DeleteModelRequest ++ (*DeleteModelResponse)(nil), // 59: ax.v1alpha1.DeleteModelResponse ++ (*timestamppb.Timestamp)(nil), // 60: google.protobuf.Timestamp ++ (*structpb.Struct)(nil), // 61: google.protobuf.Struct + } + var file_pkg_apis_v1alpha1_ax_proto_depIdxs = []int32{ +- 57, // 0: ax.v1alpha1.ObjectMeta.creation_timestamp:type_name -> google.protobuf.Timestamp ++ 60, // 0: ax.v1alpha1.ObjectMeta.creation_timestamp:type_name -> google.protobuf.Timestamp + 0, // 1: ax.v1alpha1.Task.metadata:type_name -> ax.v1alpha1.ObjectMeta + 2, // 2: ax.v1alpha1.Task.spec:type_name -> ax.v1alpha1.TaskSpec + 8, // 3: ax.v1alpha1.Task.status:type_name -> ax.v1alpha1.TaskStatus +@@ -3501,82 +3722,86 @@ var file_pkg_apis_v1alpha1_ax_proto_depIdxs = []int32{ + 7, // 7: ax.v1alpha1.TaskSpec.gateway:type_name -> ax.v1alpha1.GatewayRef + 5, // 8: ax.v1alpha1.ResourceReqs.requests:type_name -> ax.v1alpha1.ResourceList + 5, // 9: ax.v1alpha1.ResourceReqs.limits:type_name -> ax.v1alpha1.ResourceList +- 9, // 10: ax.v1alpha1.TaskStatus.pending_approval:type_name -> ax.v1alpha1.PendingApproval +- 10, // 11: ax.v1alpha1.TaskStatus.usage:type_name -> ax.v1alpha1.UsageStats +- 11, // 12: ax.v1alpha1.TaskStatus.conditions:type_name -> ax.v1alpha1.Condition +- 57, // 13: ax.v1alpha1.PendingApproval.requested_at:type_name -> google.protobuf.Timestamp +- 57, // 14: ax.v1alpha1.Condition.last_transition_time:type_name -> google.protobuf.Timestamp +- 0, // 15: ax.v1alpha1.Gateway.metadata:type_name -> ax.v1alpha1.ObjectMeta +- 13, // 16: ax.v1alpha1.Gateway.spec:type_name -> ax.v1alpha1.GatewaySpec +- 14, // 17: ax.v1alpha1.GatewaySpec.listeners:type_name -> ax.v1alpha1.Listener +- 15, // 18: ax.v1alpha1.GatewaySpec.egress:type_name -> ax.v1alpha1.EgressConfig +- 16, // 19: ax.v1alpha1.EgressConfig.allowlist:type_name -> ax.v1alpha1.EgressAllowlist +- 17, // 20: ax.v1alpha1.EgressAllowlist.hosts:type_name -> ax.v1alpha1.HostRule +- 0, // 21: ax.v1alpha1.Workspace.metadata:type_name -> ax.v1alpha1.ObjectMeta +- 19, // 22: ax.v1alpha1.Workspace.spec:type_name -> ax.v1alpha1.WorkspaceSpec +- 20, // 23: ax.v1alpha1.WorkspaceSpec.git:type_name -> ax.v1alpha1.GitRepo +- 21, // 24: ax.v1alpha1.WorkspaceSpec.mcp:type_name -> ax.v1alpha1.MCPConfig +- 24, // 25: ax.v1alpha1.WorkspaceSpec.skills:type_name -> ax.v1alpha1.SkillsConfig +- 22, // 26: ax.v1alpha1.MCPConfig.registries:type_name -> ax.v1alpha1.MCPRegistry +- 23, // 27: ax.v1alpha1.MCPConfig.servers:type_name -> ax.v1alpha1.MCPServer +- 23, // 28: ax.v1alpha1.MCPRegistry.servers:type_name -> ax.v1alpha1.MCPServer +- 25, // 29: ax.v1alpha1.SkillsConfig.registries:type_name -> ax.v1alpha1.SkillRegistry +- 0, // 30: ax.v1alpha1.Model.metadata:type_name -> ax.v1alpha1.ObjectMeta +- 27, // 31: ax.v1alpha1.Model.spec:type_name -> ax.v1alpha1.ModelSpec +- 28, // 32: ax.v1alpha1.ModelSpec.secret_key:type_name -> ax.v1alpha1.SecretKeyRef +- 58, // 33: ax.v1alpha1.ModelSpec.parameters:type_name -> google.protobuf.Struct +- 1, // 34: ax.v1alpha1.ListTasksResponse.tasks:type_name -> ax.v1alpha1.Task +- 1, // 35: ax.v1alpha1.UpdateTaskRequest.task:type_name -> ax.v1alpha1.Task +- 1, // 36: ax.v1alpha1.WatchTaskResponse.task:type_name -> ax.v1alpha1.Task +- 12, // 37: ax.v1alpha1.ListGatewaysResponse.gateways:type_name -> ax.v1alpha1.Gateway +- 12, // 38: ax.v1alpha1.UpdateGatewayRequest.gateway:type_name -> ax.v1alpha1.Gateway +- 18, // 39: ax.v1alpha1.ListWorkspacesResponse.workspaces:type_name -> ax.v1alpha1.Workspace +- 18, // 40: ax.v1alpha1.UpdateWorkspaceRequest.workspace:type_name -> ax.v1alpha1.Workspace +- 26, // 41: ax.v1alpha1.ListModelsResponse.models:type_name -> ax.v1alpha1.Model +- 26, // 42: ax.v1alpha1.UpdateModelRequest.model:type_name -> ax.v1alpha1.Model +- 29, // 43: ax.v1alpha1.AX.GetTask:input_type -> ax.v1alpha1.GetTaskRequest +- 30, // 44: ax.v1alpha1.AX.ListTasks:input_type -> ax.v1alpha1.ListTasksRequest +- 32, // 45: ax.v1alpha1.AX.UpdateTask:input_type -> ax.v1alpha1.UpdateTaskRequest +- 33, // 46: ax.v1alpha1.AX.DeleteTask:input_type -> ax.v1alpha1.DeleteTaskRequest +- 35, // 47: ax.v1alpha1.AX.SuspendTask:input_type -> ax.v1alpha1.SuspendTaskRequest +- 36, // 48: ax.v1alpha1.AX.ResumeTask:input_type -> ax.v1alpha1.ResumeTaskRequest +- 37, // 49: ax.v1alpha1.AX.WatchTask:input_type -> ax.v1alpha1.WatchTaskRequest +- 39, // 50: ax.v1alpha1.AX.GetGateway:input_type -> ax.v1alpha1.GetGatewayRequest +- 40, // 51: ax.v1alpha1.AX.ListGateways:input_type -> ax.v1alpha1.ListGatewaysRequest +- 42, // 52: ax.v1alpha1.AX.UpdateGateway:input_type -> ax.v1alpha1.UpdateGatewayRequest +- 43, // 53: ax.v1alpha1.AX.DeleteGateway:input_type -> ax.v1alpha1.DeleteGatewayRequest +- 45, // 54: ax.v1alpha1.AX.GetWorkspace:input_type -> ax.v1alpha1.GetWorkspaceRequest +- 46, // 55: ax.v1alpha1.AX.ListWorkspaces:input_type -> ax.v1alpha1.ListWorkspacesRequest +- 48, // 56: ax.v1alpha1.AX.UpdateWorkspace:input_type -> ax.v1alpha1.UpdateWorkspaceRequest +- 49, // 57: ax.v1alpha1.AX.DeleteWorkspace:input_type -> ax.v1alpha1.DeleteWorkspaceRequest +- 51, // 58: ax.v1alpha1.AX.GetModel:input_type -> ax.v1alpha1.GetModelRequest +- 52, // 59: ax.v1alpha1.AX.ListModels:input_type -> ax.v1alpha1.ListModelsRequest +- 54, // 60: ax.v1alpha1.AX.UpdateModel:input_type -> ax.v1alpha1.UpdateModelRequest +- 55, // 61: ax.v1alpha1.AX.DeleteModel:input_type -> ax.v1alpha1.DeleteModelRequest +- 1, // 62: ax.v1alpha1.AX.GetTask:output_type -> ax.v1alpha1.Task +- 31, // 63: ax.v1alpha1.AX.ListTasks:output_type -> ax.v1alpha1.ListTasksResponse +- 1, // 64: ax.v1alpha1.AX.UpdateTask:output_type -> ax.v1alpha1.Task +- 34, // 65: ax.v1alpha1.AX.DeleteTask:output_type -> ax.v1alpha1.DeleteTaskResponse +- 1, // 66: ax.v1alpha1.AX.SuspendTask:output_type -> ax.v1alpha1.Task +- 1, // 67: ax.v1alpha1.AX.ResumeTask:output_type -> ax.v1alpha1.Task +- 38, // 68: ax.v1alpha1.AX.WatchTask:output_type -> ax.v1alpha1.WatchTaskResponse +- 12, // 69: ax.v1alpha1.AX.GetGateway:output_type -> ax.v1alpha1.Gateway +- 41, // 70: ax.v1alpha1.AX.ListGateways:output_type -> ax.v1alpha1.ListGatewaysResponse +- 12, // 71: ax.v1alpha1.AX.UpdateGateway:output_type -> ax.v1alpha1.Gateway +- 44, // 72: ax.v1alpha1.AX.DeleteGateway:output_type -> ax.v1alpha1.DeleteGatewayResponse +- 18, // 73: ax.v1alpha1.AX.GetWorkspace:output_type -> ax.v1alpha1.Workspace +- 47, // 74: ax.v1alpha1.AX.ListWorkspaces:output_type -> ax.v1alpha1.ListWorkspacesResponse +- 18, // 75: ax.v1alpha1.AX.UpdateWorkspace:output_type -> ax.v1alpha1.Workspace +- 50, // 76: ax.v1alpha1.AX.DeleteWorkspace:output_type -> ax.v1alpha1.DeleteWorkspaceResponse +- 26, // 77: ax.v1alpha1.AX.GetModel:output_type -> ax.v1alpha1.Model +- 53, // 78: ax.v1alpha1.AX.ListModels:output_type -> ax.v1alpha1.ListModelsResponse +- 26, // 79: ax.v1alpha1.AX.UpdateModel:output_type -> ax.v1alpha1.Model +- 56, // 80: ax.v1alpha1.AX.DeleteModel:output_type -> ax.v1alpha1.DeleteModelResponse +- 62, // [62:81] is the sub-list for method output_type +- 43, // [43:62] is the sub-list for method input_type +- 43, // [43:43] is the sub-list for extension type_name +- 43, // [43:43] is the sub-list for extension extendee +- 0, // [0:43] is the sub-list for field type_name ++ 10, // 10: ax.v1alpha1.TaskStatus.pending_approval:type_name -> ax.v1alpha1.PendingApproval ++ 11, // 11: ax.v1alpha1.TaskStatus.usage:type_name -> ax.v1alpha1.UsageStats ++ 12, // 12: ax.v1alpha1.TaskStatus.conditions:type_name -> ax.v1alpha1.Condition ++ 9, // 13: ax.v1alpha1.TaskStatus.command:type_name -> ax.v1alpha1.CommandStatus ++ 60, // 14: ax.v1alpha1.CommandStatus.finished_at:type_name -> google.protobuf.Timestamp ++ 60, // 15: ax.v1alpha1.PendingApproval.requested_at:type_name -> google.protobuf.Timestamp ++ 60, // 16: ax.v1alpha1.Condition.last_transition_time:type_name -> google.protobuf.Timestamp ++ 0, // 17: ax.v1alpha1.Gateway.metadata:type_name -> ax.v1alpha1.ObjectMeta ++ 14, // 18: ax.v1alpha1.Gateway.spec:type_name -> ax.v1alpha1.GatewaySpec ++ 15, // 19: ax.v1alpha1.GatewaySpec.listeners:type_name -> ax.v1alpha1.Listener ++ 16, // 20: ax.v1alpha1.GatewaySpec.egress:type_name -> ax.v1alpha1.EgressConfig ++ 17, // 21: ax.v1alpha1.EgressConfig.allowlist:type_name -> ax.v1alpha1.EgressAllowlist ++ 18, // 22: ax.v1alpha1.EgressAllowlist.hosts:type_name -> ax.v1alpha1.HostRule ++ 0, // 23: ax.v1alpha1.Workspace.metadata:type_name -> ax.v1alpha1.ObjectMeta ++ 20, // 24: ax.v1alpha1.Workspace.spec:type_name -> ax.v1alpha1.WorkspaceSpec ++ 21, // 25: ax.v1alpha1.WorkspaceSpec.git:type_name -> ax.v1alpha1.GitRepo ++ 22, // 26: ax.v1alpha1.WorkspaceSpec.mcp:type_name -> ax.v1alpha1.MCPConfig ++ 25, // 27: ax.v1alpha1.WorkspaceSpec.skills:type_name -> ax.v1alpha1.SkillsConfig ++ 23, // 28: ax.v1alpha1.MCPConfig.registries:type_name -> ax.v1alpha1.MCPRegistry ++ 24, // 29: ax.v1alpha1.MCPConfig.servers:type_name -> ax.v1alpha1.MCPServer ++ 24, // 30: ax.v1alpha1.MCPRegistry.servers:type_name -> ax.v1alpha1.MCPServer ++ 26, // 31: ax.v1alpha1.SkillsConfig.registries:type_name -> ax.v1alpha1.SkillRegistry ++ 0, // 32: ax.v1alpha1.Model.metadata:type_name -> ax.v1alpha1.ObjectMeta ++ 28, // 33: ax.v1alpha1.Model.spec:type_name -> ax.v1alpha1.ModelSpec ++ 29, // 34: ax.v1alpha1.ModelSpec.secret_key:type_name -> ax.v1alpha1.SecretKeyRef ++ 61, // 35: ax.v1alpha1.ModelSpec.parameters:type_name -> google.protobuf.Struct ++ 1, // 36: ax.v1alpha1.ListTasksResponse.tasks:type_name -> ax.v1alpha1.Task ++ 1, // 37: ax.v1alpha1.UpdateTaskRequest.task:type_name -> ax.v1alpha1.Task ++ 1, // 38: ax.v1alpha1.WatchTaskResponse.task:type_name -> ax.v1alpha1.Task ++ 13, // 39: ax.v1alpha1.ListGatewaysResponse.gateways:type_name -> ax.v1alpha1.Gateway ++ 13, // 40: ax.v1alpha1.UpdateGatewayRequest.gateway:type_name -> ax.v1alpha1.Gateway ++ 19, // 41: ax.v1alpha1.ListWorkspacesResponse.workspaces:type_name -> ax.v1alpha1.Workspace ++ 19, // 42: ax.v1alpha1.UpdateWorkspaceRequest.workspace:type_name -> ax.v1alpha1.Workspace ++ 27, // 43: ax.v1alpha1.ListModelsResponse.models:type_name -> ax.v1alpha1.Model ++ 27, // 44: ax.v1alpha1.UpdateModelRequest.model:type_name -> ax.v1alpha1.Model ++ 30, // 45: ax.v1alpha1.AX.GetTask:input_type -> ax.v1alpha1.GetTaskRequest ++ 33, // 46: ax.v1alpha1.AX.ListTasks:input_type -> ax.v1alpha1.ListTasksRequest ++ 35, // 47: ax.v1alpha1.AX.UpdateTask:input_type -> ax.v1alpha1.UpdateTaskRequest ++ 36, // 48: ax.v1alpha1.AX.DeleteTask:input_type -> ax.v1alpha1.DeleteTaskRequest ++ 38, // 49: ax.v1alpha1.AX.SuspendTask:input_type -> ax.v1alpha1.SuspendTaskRequest ++ 39, // 50: ax.v1alpha1.AX.ResumeTask:input_type -> ax.v1alpha1.ResumeTaskRequest ++ 40, // 51: ax.v1alpha1.AX.WatchTask:input_type -> ax.v1alpha1.WatchTaskRequest ++ 31, // 52: ax.v1alpha1.AX.GetTaskResult:input_type -> ax.v1alpha1.GetTaskResultRequest ++ 42, // 53: ax.v1alpha1.AX.GetGateway:input_type -> ax.v1alpha1.GetGatewayRequest ++ 43, // 54: ax.v1alpha1.AX.ListGateways:input_type -> ax.v1alpha1.ListGatewaysRequest ++ 45, // 55: ax.v1alpha1.AX.UpdateGateway:input_type -> ax.v1alpha1.UpdateGatewayRequest ++ 46, // 56: ax.v1alpha1.AX.DeleteGateway:input_type -> ax.v1alpha1.DeleteGatewayRequest ++ 48, // 57: ax.v1alpha1.AX.GetWorkspace:input_type -> ax.v1alpha1.GetWorkspaceRequest ++ 49, // 58: ax.v1alpha1.AX.ListWorkspaces:input_type -> ax.v1alpha1.ListWorkspacesRequest ++ 51, // 59: ax.v1alpha1.AX.UpdateWorkspace:input_type -> ax.v1alpha1.UpdateWorkspaceRequest ++ 52, // 60: ax.v1alpha1.AX.DeleteWorkspace:input_type -> ax.v1alpha1.DeleteWorkspaceRequest ++ 54, // 61: ax.v1alpha1.AX.GetModel:input_type -> ax.v1alpha1.GetModelRequest ++ 55, // 62: ax.v1alpha1.AX.ListModels:input_type -> ax.v1alpha1.ListModelsRequest ++ 57, // 63: ax.v1alpha1.AX.UpdateModel:input_type -> ax.v1alpha1.UpdateModelRequest ++ 58, // 64: ax.v1alpha1.AX.DeleteModel:input_type -> ax.v1alpha1.DeleteModelRequest ++ 1, // 65: ax.v1alpha1.AX.GetTask:output_type -> ax.v1alpha1.Task ++ 34, // 66: ax.v1alpha1.AX.ListTasks:output_type -> ax.v1alpha1.ListTasksResponse ++ 1, // 67: ax.v1alpha1.AX.UpdateTask:output_type -> ax.v1alpha1.Task ++ 37, // 68: ax.v1alpha1.AX.DeleteTask:output_type -> ax.v1alpha1.DeleteTaskResponse ++ 1, // 69: ax.v1alpha1.AX.SuspendTask:output_type -> ax.v1alpha1.Task ++ 1, // 70: ax.v1alpha1.AX.ResumeTask:output_type -> ax.v1alpha1.Task ++ 41, // 71: ax.v1alpha1.AX.WatchTask:output_type -> ax.v1alpha1.WatchTaskResponse ++ 32, // 72: ax.v1alpha1.AX.GetTaskResult:output_type -> ax.v1alpha1.TaskResult ++ 13, // 73: ax.v1alpha1.AX.GetGateway:output_type -> ax.v1alpha1.Gateway ++ 44, // 74: ax.v1alpha1.AX.ListGateways:output_type -> ax.v1alpha1.ListGatewaysResponse ++ 13, // 75: ax.v1alpha1.AX.UpdateGateway:output_type -> ax.v1alpha1.Gateway ++ 47, // 76: ax.v1alpha1.AX.DeleteGateway:output_type -> ax.v1alpha1.DeleteGatewayResponse ++ 19, // 77: ax.v1alpha1.AX.GetWorkspace:output_type -> ax.v1alpha1.Workspace ++ 50, // 78: ax.v1alpha1.AX.ListWorkspaces:output_type -> ax.v1alpha1.ListWorkspacesResponse ++ 19, // 79: ax.v1alpha1.AX.UpdateWorkspace:output_type -> ax.v1alpha1.Workspace ++ 53, // 80: ax.v1alpha1.AX.DeleteWorkspace:output_type -> ax.v1alpha1.DeleteWorkspaceResponse ++ 27, // 81: ax.v1alpha1.AX.GetModel:output_type -> ax.v1alpha1.Model ++ 56, // 82: ax.v1alpha1.AX.ListModels:output_type -> ax.v1alpha1.ListModelsResponse ++ 27, // 83: ax.v1alpha1.AX.UpdateModel:output_type -> ax.v1alpha1.Model ++ 59, // 84: ax.v1alpha1.AX.DeleteModel:output_type -> ax.v1alpha1.DeleteModelResponse ++ 65, // [65:85] is the sub-list for method output_type ++ 45, // [45:65] is the sub-list for method input_type ++ 45, // [45:45] is the sub-list for extension type_name ++ 45, // [45:45] is the sub-list for extension extendee ++ 0, // [0:45] is the sub-list for field type_name + } + + func init() { file_pkg_apis_v1alpha1_ax_proto_init() } +@@ -3590,7 +3815,7 @@ func file_pkg_apis_v1alpha1_ax_proto_init() { + GoPackagePath: reflect.TypeOf(x{}).PkgPath(), + RawDescriptor: unsafe.Slice(unsafe.StringData(file_pkg_apis_v1alpha1_ax_proto_rawDesc), len(file_pkg_apis_v1alpha1_ax_proto_rawDesc)), + NumEnums: 0, +- NumMessages: 57, ++ NumMessages: 60, + NumExtensions: 0, + NumServices: 1, + }, +diff --git a/pkg/apis/v1alpha1/ax.proto b/pkg/apis/v1alpha1/ax.proto +index cf35316..072305b 100644 +--- a/pkg/apis/v1alpha1/ax.proto ++++ b/pkg/apis/v1alpha1/ax.proto +@@ -34,6 +34,9 @@ service AX { + rpc SuspendTask(SuspendTaskRequest) returns (Task); + rpc ResumeTask(ResumeTaskRequest) returns (Task); + rpc WatchTask(WatchTaskRequest) returns (stream WatchTaskResponse); ++ // GetTaskResult returns the result file the task's command wrote, as copied ++ // by the controller when the command exited. NotFound until then. ++ rpc GetTaskResult(GetTaskResultRequest) returns (TaskResult); + + // Gateways + rpc GetGateway(GetGatewayRequest) returns (Gateway); +@@ -143,6 +146,19 @@ message TaskStatus { + PendingApproval pending_approval = 5; + UsageStats usage = 6; + repeated Condition conditions = 7; ++ // command reports how the task's command finished. Unset while it runs. ++ CommandStatus command = 8; ++} ++ ++// CommandStatus is written by the controller once the runner reports that the ++// task's command exited. The result content itself is served by GetTaskResult. ++message CommandStatus { ++ bool exited = 1; ++ int32 exit_code = 2; ++ google.protobuf.Timestamp finished_at = 3; ++ // result_bytes is the size of the copied result, 0 when there was none. ++ int64 result_bytes = 4; ++ string result_sha256 = 5; + } + + message PendingApproval { +@@ -154,6 +170,7 @@ message PendingApproval { + message UsageStats { + int32 prompt_tokens = 1; + int32 completion_tokens = 2; ++ int32 tool_calls = 3; + } + + message Condition { +@@ -285,6 +302,16 @@ message GetTaskRequest { + string name = 2; + } + ++message GetTaskResultRequest { ++ string atespace = 1; ++ string name = 2; ++} ++ ++message TaskResult { ++ bytes content = 1; ++ string sha256 = 2; ++} ++ + message ListTasksRequest { + string atespace = 1; + int64 limit = 2; +diff --git a/pkg/apis/v1alpha1/ax_grpc.pb.go b/pkg/apis/v1alpha1/ax_grpc.pb.go +index d106157..a8b2ca8 100644 +--- a/pkg/apis/v1alpha1/ax_grpc.pb.go ++++ b/pkg/apis/v1alpha1/ax_grpc.pb.go +@@ -14,8 +14,8 @@ + + // Code generated by protoc-gen-go-grpc. DO NOT EDIT. + // versions: +-// - protoc-gen-go-grpc v1.6.0 +-// - protoc v7.34.1 ++// - protoc-gen-go-grpc v1.6.2 ++// - protoc v7.35.1 + // source: pkg/apis/v1alpha1/ax.proto + + package v1alpha1 +@@ -40,6 +40,7 @@ const ( + AX_SuspendTask_FullMethodName = "/ax.v1alpha1.AX/SuspendTask" + AX_ResumeTask_FullMethodName = "/ax.v1alpha1.AX/ResumeTask" + AX_WatchTask_FullMethodName = "/ax.v1alpha1.AX/WatchTask" ++ AX_GetTaskResult_FullMethodName = "/ax.v1alpha1.AX/GetTaskResult" + AX_GetGateway_FullMethodName = "/ax.v1alpha1.AX/GetGateway" + AX_ListGateways_FullMethodName = "/ax.v1alpha1.AX/ListGateways" + AX_UpdateGateway_FullMethodName = "/ax.v1alpha1.AX/UpdateGateway" +@@ -68,6 +69,9 @@ type AXClient interface { + SuspendTask(ctx context.Context, in *SuspendTaskRequest, opts ...grpc.CallOption) (*Task, error) + ResumeTask(ctx context.Context, in *ResumeTaskRequest, opts ...grpc.CallOption) (*Task, error) + WatchTask(ctx context.Context, in *WatchTaskRequest, opts ...grpc.CallOption) (grpc.ServerStreamingClient[WatchTaskResponse], error) ++ // GetTaskResult returns the result file the task's command wrote, as copied ++ // by the controller when the command exited. NotFound until then. ++ GetTaskResult(ctx context.Context, in *GetTaskResultRequest, opts ...grpc.CallOption) (*TaskResult, error) + // Gateways + GetGateway(ctx context.Context, in *GetGatewayRequest, opts ...grpc.CallOption) (*Gateway, error) + ListGateways(ctx context.Context, in *ListGatewaysRequest, opts ...grpc.CallOption) (*ListGatewaysResponse, error) +@@ -172,6 +176,16 @@ func (c *aXClient) WatchTask(ctx context.Context, in *WatchTaskRequest, opts ... + // This type alias is provided for backwards compatibility with existing code that references the prior non-generic stream type by name. + type AX_WatchTaskClient = grpc.ServerStreamingClient[WatchTaskResponse] + ++func (c *aXClient) GetTaskResult(ctx context.Context, in *GetTaskResultRequest, opts ...grpc.CallOption) (*TaskResult, error) { ++ cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...) ++ out := new(TaskResult) ++ err := c.cc.Invoke(ctx, AX_GetTaskResult_FullMethodName, in, out, cOpts...) ++ if err != nil { ++ return nil, err ++ } ++ return out, nil ++} ++ + func (c *aXClient) GetGateway(ctx context.Context, in *GetGatewayRequest, opts ...grpc.CallOption) (*Gateway, error) { + cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...) + out := new(Gateway) +@@ -306,6 +320,9 @@ type AXServer interface { + SuspendTask(context.Context, *SuspendTaskRequest) (*Task, error) + ResumeTask(context.Context, *ResumeTaskRequest) (*Task, error) + WatchTask(*WatchTaskRequest, grpc.ServerStreamingServer[WatchTaskResponse]) error ++ // GetTaskResult returns the result file the task's command wrote, as copied ++ // by the controller when the command exited. NotFound until then. ++ GetTaskResult(context.Context, *GetTaskResultRequest) (*TaskResult, error) + // Gateways + GetGateway(context.Context, *GetGatewayRequest) (*Gateway, error) + ListGateways(context.Context, *ListGatewaysRequest) (*ListGatewaysResponse, error) +@@ -352,6 +369,9 @@ func (UnimplementedAXServer) ResumeTask(context.Context, *ResumeTaskRequest) (*T + func (UnimplementedAXServer) WatchTask(*WatchTaskRequest, grpc.ServerStreamingServer[WatchTaskResponse]) error { + return status.Error(codes.Unimplemented, "method WatchTask not implemented") + } ++func (UnimplementedAXServer) GetTaskResult(context.Context, *GetTaskResultRequest) (*TaskResult, error) { ++ return nil, status.Error(codes.Unimplemented, "method GetTaskResult not implemented") ++} + func (UnimplementedAXServer) GetGateway(context.Context, *GetGatewayRequest) (*Gateway, error) { + return nil, status.Error(codes.Unimplemented, "method GetGateway not implemented") + } +@@ -528,6 +548,24 @@ func _AX_WatchTask_Handler(srv interface{}, stream grpc.ServerStream) error { + // This type alias is provided for backwards compatibility with existing code that references the prior non-generic stream type by name. + type AX_WatchTaskServer = grpc.ServerStreamingServer[WatchTaskResponse] + ++func _AX_GetTaskResult_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { ++ in := new(GetTaskResultRequest) ++ if err := dec(in); err != nil { ++ return nil, err ++ } ++ if interceptor == nil { ++ return srv.(AXServer).GetTaskResult(ctx, in) ++ } ++ info := &grpc.UnaryServerInfo{ ++ Server: srv, ++ FullMethod: AX_GetTaskResult_FullMethodName, ++ } ++ handler := func(ctx context.Context, req interface{}) (interface{}, error) { ++ return srv.(AXServer).GetTaskResult(ctx, req.(*GetTaskResultRequest)) ++ } ++ return interceptor(ctx, in, info, handler) ++} ++ + func _AX_GetGateway_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(GetGatewayRequest) + if err := dec(in); err != nil { +@@ -775,6 +813,10 @@ var AX_ServiceDesc = grpc.ServiceDesc{ + MethodName: "ResumeTask", + Handler: _AX_ResumeTask_Handler, + }, ++ { ++ MethodName: "GetTaskResult", ++ Handler: _AX_GetTaskResult_Handler, ++ }, + { + MethodName: "GetGateway", + Handler: _AX_GetGateway_Handler, +diff --git a/runner/exit_test.go b/runner/exit_test.go +new file mode 100644 +index 0000000..a340da9 +--- /dev/null ++++ b/runner/exit_test.go +@@ -0,0 +1,65 @@ ++// Copyright 2026 Google LLC ++// ++// Licensed under the Apache License, Version 2.0 (the "License"); ++// you may not use this file except in compliance with the License. ++// You may obtain a copy of the License at ++// ++// http://www.apache.org/licenses/LICENSE-2.0 ++// ++// Unless required by applicable law or agreed to in writing, software ++// distributed under the License is distributed on an "AS IS" BASIS, ++// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. ++// See the License for the specific language governing permissions and ++// limitations under the License. ++ ++package runner_test ++ ++import ( ++ "encoding/json" ++ "fmt" ++ "io" ++ "net/http" ++ "testing" ++ "time" ++ ++ "github.com/google/ax/internal/metadata" ++) ++ ++// The runner reports the command's exit and serves its result file from the ++// default path under the workspace, which is what the controller reads back. ++func TestRun_ServesExitAndResult(t *testing.T) { ++ h := newHarness(t) ++ h.task.Spec.Command = []string{"/bin/sh", "-c", `mkdir -p .ax && printf '{"ok":1}' > .ax/result.json && exit 5`} ++ h.start(t) ++ ++ select { ++ case e := <-h.exited: ++ if e.ExitCode != 5 { ++ t.Fatalf("exit code = %d, want 5", e.ExitCode) ++ } ++ case <-time.After(10 * time.Second): ++ t.Fatal("command did not exit") ++ } ++ ++ base := fmt.Sprintf("http://127.0.0.1:%d/metadata/v1alpha1/ax", h.cfg.Port) ++ resp, err := http.Get(base + "/exit") ++ if err != nil { ++ t.Fatal(err) ++ } ++ var st metadata.CommandExitStatus ++ err = json.NewDecoder(resp.Body).Decode(&st) ++ resp.Body.Close() ++ if err != nil || !st.Exited || st.ExitCode != 5 { ++ t.Fatalf("exit status = %+v, %v", st, err) ++ } ++ ++ resp, err = http.Get(base + "/result") ++ if err != nil { ++ t.Fatal(err) ++ } ++ body, _ := io.ReadAll(resp.Body) ++ resp.Body.Close() ++ if resp.StatusCode != http.StatusOK || string(body) != `{"ok":1}` { ++ t.Fatalf("result = %d %q", resp.StatusCode, body) ++ } ++} +diff --git a/runner/runner.go b/runner/runner.go +index 4eaadb0..7ec8da9 100644 +--- a/runner/runner.go ++++ b/runner/runner.go +@@ -28,6 +28,7 @@ import ( + "log/slog" + "os" + "os/exec" ++ "path/filepath" + "syscall" + "time" + +@@ -44,6 +45,12 @@ const ( + // binds no workspaces. + DefaultWorkspacePath = v1alpha1.DefaultWorkspacePath + ++ // DefaultResultPath is where the task command writes its result, relative ++ // to the first workspace, unless AX_RESULT_PATH says otherwise. ++ DefaultResultPath = ".ax/result.json" ++ // DefaultUsagePath is where the task command writes its usage JSON. ++ DefaultUsagePath = ".ax/usage.json" ++ + // stopGracePeriod is how long a running task command gets to exit after + // SIGTERM before it is killed during shutdown. + stopGracePeriod = 10 * time.Second +@@ -143,7 +150,11 @@ func Run(ctx context.Context, cfg Config) error { + _ = os.Setenv(e.GetName(), e.GetValue()) + } + +- metaServer := metadata.NewServer(port, cfg.Task, workspaces, metadata.ServerOptions{WorkspacePath: wsPath}) ++ metaServer := metadata.NewServer(port, cfg.Task, workspaces, metadata.ServerOptions{ ++ WorkspacePath: wsPath, ++ ResultPath: outputPath(wsPath, DefaultResultPath, "AX_RESULT_PATH", "AX_CONWIP_RESULT_PATH"), ++ UsagePath: outputPath(wsPath, DefaultUsagePath, "AX_USAGE_PATH", "AX_CONWIP_USAGE_PATH"), ++ }) + if err := metaServer.Start(); err != nil { + return fmt.Errorf("starting metadata server: %w", err) + } +@@ -191,6 +202,13 @@ func Run(ctx context.Context, cfg Config) error { + + select { + case err := <-exited: ++ // Only a command that exited on its own is reported as finished; a ++ // command stopped at shutdown is not, so a suspend never reads as exit. ++ code := -1 ++ if cmd.ProcessState != nil { ++ code = cmd.ProcessState.ExitCode() ++ } ++ metaServer.SetCommandExit(code, time.Now()) + reportExit(cfg, cmd, err) + // Keep the sandbox up and inspectable until told to stop. + <-ctx.Done() +@@ -244,3 +262,19 @@ func commandEnv(task *v1alpha1.Task, port int) []string { + } + return env + } ++ ++// outputPath resolves a command output file: the first of envVars that is set, ++// else def; a relative path is taken relative to the workspace. ++func outputPath(wsPath, def string, envVars ...string) string { ++ p := def ++ for _, v := range envVars { ++ if val := os.Getenv(v); val != "" { ++ p = val ++ break ++ } ++ } ++ if !filepath.IsAbs(p) { ++ p = filepath.Join(wsPath, p) ++ } ++ return p ++} diff --git a/pkgs/ax/sandbox-class.patch b/pkgs/ax/patches/sandbox-class.patch similarity index 100% rename from pkgs/ax/sandbox-class.patch rename to pkgs/ax/patches/sandbox-class.patch From 9a521c2ba39486ac9937ef0ba4f2042cc146a810 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 08:38:08 +0200 Subject: [PATCH 10/37] pkgs/substrate: upstream Agent Substrate d277088b, images and gVisor asset buildGoModule with the nixpkgs-go toolchain (go 1.27.1), vendored tree, CGO off, version ldflags d277088b: ate-setup plus ateapi, atecontroller, atelet, atenet, podcertcontroller and ateom-gvisor. images.nix: the six component images as OCI layouts with a digest file, the eight third-party images the kind install pulls as digest-preserving fixed-output copies of their linux/amd64 manifests, and the gVisor nightly tarball SandboxConfig gvisor-default names (sha256 d547d814). Patches touch manifests only: 0001 points pauseImage at the NAS registry through atelet's localhost rewrite; 0002 pins third-party images to their linux/amd64 child digests so the NAS store holds about 0.7 GB instead of every platform. Co-Authored-By: Claude Opus 5.5 --- pkgs/substrate/default.nix | 107 +++++++ pkgs/substrate/images.nix | 299 ++++++++++++++++++ .../0001-sandboxconfig-pause-localhost.patch | 15 + .../0002-images-linux-amd64-digests.patch | 122 +++++++ 4 files changed, 543 insertions(+) create mode 100644 pkgs/substrate/default.nix create mode 100644 pkgs/substrate/images.nix create mode 100644 pkgs/substrate/patches/0001-sandboxconfig-pause-localhost.patch create mode 100644 pkgs/substrate/patches/0002-images-linux-amd64-digests.patch diff --git a/pkgs/substrate/default.nix b/pkgs/substrate/default.nix new file mode 100644 index 000000000..ebc063978 --- /dev/null +++ b/pkgs/substrate/default.nix @@ -0,0 +1,107 @@ +{ + lib, + buildGoModule, + fetchFromGitHub, + applyPatches, + # Passed in by the caller from the nixpkgs-go input, exactly as pkgs/ax gets + # it (see overlays/default.nix and the nixpkgs-go comment in flake.nix). + # Substrate's go.mod says `go 1.27.0`; the probe built it with 1.27.1 + # (MEASURED, evals-2026-09-23/substrate/probe-build.md section 2), and one + # toolchain for ax and Substrate keeps the fleet on one Go. + go_1_27, +}: +# Agent Substrate, the sandbox control plane ax delegates to. Upstream +# github.com/agent-substrate/substrate, pinned by commit to d277088b: the +# SERVER version the 2026-09-23 version rule names. ax keeps its own vendored +# CLIENT pin (672533541dbf); wire compatibility between the two is MEASURED +# (probe-build.md section 4). +# +# Not in nixpkgs. Upstream publishes no release images the fleet may use (the +# kagent-dev fork's ghcr images are ruled out), so every component image is +# built here and seeded into the NAS registry by pkgs/substrate/images.nix. +# +# The two patches touch only manifests/, never Go code: +# 0001 pauseImage -> localhost:5000/pause (atelet pulls it itself; the fleet +# must not depend on registry.k8s.io at sandbox start). +# 0002 third-party images -> their linux/amd64 child digests, so the NAS +# seeds ~0.7 GB instead of every platform (see the patch header). +# ate-setup reads the manifests from its working directory's repository root +# (it walks up to go.mod), so the patched tree is passthru.source and the +# bootstrap runs ate-setup from there. +let + version = "d277088b"; + rev = "d277088bc1d081ef716d81dd7986d05d0a36ad3a"; + + source = applyPatches { + name = "substrate-source-${version}"; + src = fetchFromGitHub { + owner = "agent-substrate"; + repo = "substrate"; + inherit rev; + # nix-prefetch-url --unpack of the GitHub archive; the unpacked tree is + # byte-identical to /home/tom/Downloads/substrate at this rev (MEASURED + # diff -rq, 2026-09-23). + hash = "sha256-/KY4vYgHbeiVnpRUsVFOVfoN3SVEtWJrXTj0Pb4Zeqw="; + }; + patches = [ + ./patches/0001-sandboxconfig-pause-localhost.patch + ./patches/0002-images-linux-amd64-digests.patch + ]; + }; + + buildGo127Module = buildGoModule.override { go = go_1_27; }; + + # The six components the kind install and the WorkerPool run, plus the + # installer. Image names are the import paths' last element, which is what + # ate-setup's --image-repo mode looks up (cmd/ate-setup/internal/images). + components = [ + "ateapi" + "atecontroller" + "atelet" + "atenet" + "podcertcontroller" + "ateom-gvisor" + ]; +in +buildGo127Module { + pname = "substrate"; + inherit version; + src = source; + + # The tree is vendored (vendor/modules.txt), so no module download. + vendorHash = null; + + subPackages = [ "cmd/ate-setup" ] ++ map (c: "cmd/${c}") components; + + env.CGO_ENABLED = "0"; + + # The Makefile's LDFLAGS (Makefile:44-45), with the version the node label, + # the image tag and ate-setup's VERSION all share. + ldflags = [ + "-s" + "-w" + "-X=github.com/agent-substrate/substrate/internal/version.Version=${version}" + ]; + + # Upstream's suite needs Docker (225 PostgreSQL testcontainer tests), root + # (19 tests) and an FHS /bin/sleep; the probe ran it outside nix (1422 pass, + # 2 environment-only failures, MEASURED probe-build.md section 3). Inside the + # nix sandbox it can only fail for the same environmental reasons. + doCheck = false; + + passthru = { + inherit + source + rev + components + ; + }; + + meta = { + description = "Agent Substrate (upstream main ${version}): ate-setup and the gVisor control plane"; + homepage = "https://github.com/agent-substrate/substrate"; + license = lib.licenses.asl20; + platforms = [ "x86_64-linux" ]; + mainProgram = "ate-setup"; + }; +} diff --git a/pkgs/substrate/images.nix b/pkgs/substrate/images.nix new file mode 100644 index 000000000..6b4b70a0e --- /dev/null +++ b/pkgs/substrate/images.nix @@ -0,0 +1,299 @@ +{ + lib, + dockerTools, + runCommand, + linkFarm, + fetchurl, + skopeo, + jq, + cacert, + tzdata, + # pkgs/substrate/default.nix, built with the fleet's Go 1.27.1. + substrate, +}: +# Every image the Substrate install and the gVisor WorkerPool pull, as OCI +# layouts the NAS seeds into its registry (modules/ax-fleet/substrate.nix, +# myAxFleet.registrySeed). Each layout carries a `digest` file holding the +# manifest digest, so references are written by digest without IFD: whatever +# needs the digest reads the file at build or run time. +# +# components the six Substrate components, nix-built from d277088b, tagged +# with the version (ate-setup --image-repo/--image-tag resolves +# each tag to a digest with a HEAD request, then pins it). +# thirdParty the images the kind install names by digest, fetched as the +# linux/amd64 child manifest of the upstream-pinned index +# (patch 0002), bytes and digest preserved. Fixed-output. +# gvisor the runsc tarball SandboxConfig gvisor-default names, fetched +# with the sha256 that manifest carries. +# +# The whole set is one derivation (a linkFarm) so `nix build` of the flake's +# substrate-images output proves every image; the parts are in passthru. +let + inherit (substrate) version; + + # ko's default base is gcr.io/distroless/static-debian13 (.ko.yaml). What a + # static Go binary needs from it: CA roots, zoneinfo, a passwd naming root and + # nonroot 65532 (atenet runs as 65532, atelet as 0; MEASURED + # manifests/ate-install), and a world-writable /tmp. + staticBase = '' + mkdir -p etc tmp home/nonroot + chmod 1777 tmp + cat > etc/passwd <<'EOF' + root:x:0:0:root:/root:/sbin/nologin + nonroot:x:65532:65532:nonroot:/home/nonroot:/sbin/nologin + nobody:x:65534:65534:nobody:/nonexistent:/sbin/nologin + EOF + cat > etc/group <<'EOF' + root:x:0: + nonroot:x:65532: + nobody:x:65534: + EOF + sed -i 's/^ //' etc/passwd etc/group + ''; + + # docker-archive -> OCI layout plus its manifest digest. Layers are gzipped + # so the registry and containerd hold compressed blobs, as they would for a + # ko push. + toOci = + { + name, + image, + tag, + repo, + }: + runCommand "${lib.replaceStrings [ "/" ] [ "-" ] repo}-${tag}-oci" + { + nativeBuildInputs = [ + skopeo + jq + ]; + passthru = { + inherit repo tag image; + }; + } + '' + export HOME=$TMPDIR + skopeo --insecure-policy --tmpdir "$TMPDIR" copy \ + --dest-compress --dest-compress-format gzip \ + docker-archive:${image} oci:$out:${tag} + jq -r '.manifests | if length == 1 then .[0].digest else error("expected one manifest") end' \ + $out/index.json > $out/digest + grep -Eq '^sha256:[0-9a-f]{64}$' $out/digest + ''; + + component = + name: + let + image = dockerTools.buildLayeredImage { + name = "substrate/${name}"; + tag = version; + contents = [ + cacert + tzdata + ]; + extraCommands = staticBase + '' + mkdir -p ko-app + cp ${substrate}/bin/${name} ko-app/${name} + ''; + config = { + # ko's layout: the binary at /ko-app/ is the entrypoint. + Entrypoint = [ "/ko-app/${name}" ]; + Env = [ + "SSL_CERT_FILE=${cacert}/etc/ssl/certs/ca-bundle.crt" + "ZONEINFO=${tzdata}/share/zoneinfo" + "PATH=/ko-app" + ]; + WorkingDir = "/"; + Labels = { + "org.opencontainers.image.source" = "https://github.com/agent-substrate/substrate"; + "org.opencontainers.image.revision" = substrate.rev; + "org.opencontainers.image.version" = version; + }; + }; + }; + in + toOci { + inherit name image; + tag = version; + repo = "substrate/${name}"; + }; + + components = lib.genAttrs substrate.components component; + + # A digest-preserving copy of one upstream manifest. The build fetches the + # upstream-pinned index, checks that it lists `digest` for linux/amd64, then + # copies that manifest and its blobs byte for byte. outputHash pins the result. + thirdPartyImage = + { + name, + upstream, + index, + digest, + repo, + tag, + hash, + }: + runCommand "${name}-${tag}-oci" + { + nativeBuildInputs = [ + skopeo + jq + ]; + SSL_CERT_FILE = "${cacert}/etc/ssl/certs/ca-bundle.crt"; + impureEnvVars = lib.fetchers.proxyImpureEnvVars; + outputHashMode = "recursive"; + outputHashAlgo = "sha256"; + outputHash = hash; + passthru = { + inherit + repo + tag + digest + index + upstream + ; + }; + } + '' + export HOME=$TMPDIR + skopeo --insecure-policy --tmpdir "$TMPDIR" inspect --raw docker://${upstream}@${index} > index.json + echo "${lib.removePrefix "sha256:" index} index.json" | sha256sum -c - + jq -e --arg d ${digest} \ + '[.manifests[] | select(.digest == $d and .platform.os == "linux" and .platform.architecture == "amd64")] | length == 1' \ + index.json + skopeo --insecure-policy --tmpdir "$TMPDIR" copy --preserve-digests \ + docker://${upstream}@${digest} oci:$out:${tag} + jq -e --arg d ${digest} '.manifests[0].digest == $d' $out/index.json + echo ${digest} > $out/digest + ''; + + # upstream index digests: manifests/ate-install at d277088b (MEASURED grep). + # linux/amd64 children: `skopeo inspect --raw` of each index (MEASURED + # 2026-09-23); the same values are in patches/0002. + thirdParty = lib.mapAttrs (name: a: thirdPartyImage (a // { inherit name; })) { + envoy = { + upstream = "docker.io/envoyproxy/envoy"; + repo = "envoyproxy/envoy"; + tag = "v1.39-latest"; + index = "sha256:57e14a549d7bd43c8d3f6d03e8cfa653e037d4b38e133acd9b54f38c524401b4"; + digest = "sha256:be87c8b52663c1164a5bdf3c5419017a269cb3d8c74be1ec93638a71f1ffbd4b"; + hash = "sha256-8eny55hV9lIrOBFpRt9Zw563sJHEf5IsJsAgvyYrSNs="; + }; + postgres = { + upstream = "docker.io/library/postgres"; + repo = "library/postgres"; + tag = "18-alpine"; + index = "sha256:9a8afca54e7861fd90fab5fdf4c42477a6b1cb7d293595148e674e0a3181de15"; + digest = "sha256:b6a16ed0eb96e2c362811f7eeb951eac8b459e7b40be4149ea5444aa7c65569b"; + hash = "sha256-s0f5LZJr0644wB7RDApe03GXwwNIvvUZD3JXta5qcQQ="; + }; + rustfs = { + upstream = "docker.io/rustfs/rustfs"; + repo = "rustfs/rustfs"; + tag = "1.0.0-beta.3"; + index = "sha256:378642b05b7dcb4849fb77ebe6aca4ced1c3f66e7e504247df95a5c9018d3358"; + digest = "sha256:d441111efe3af5bbd6e29eba7cd124f96bb7d49a0612390907f9216d5f970382"; + hash = "sha256-TKKJLC0H9huZDQQNgx8s9xGvpre2bSa2GGArAFfv53s="; + }; + aws-cli = { + upstream = "docker.io/amazon/aws-cli"; + repo = "amazon/aws-cli"; + tag = "2.17.0"; + index = "sha256:643507c10ada7964ca6157b3d799f030b90577643da9955d319a77399ed80d73"; + digest = "sha256:7b7edf789765c22d75e61ad6f307c06950e15357b79ea1749104641ce3a11fec"; + hash = "sha256-kZF9g+VfTk5gmtNsZhJNBaX9KceOvDl2s/eWD/ucY00="; + }; + otel-collector = { + upstream = "docker.io/otel/opentelemetry-collector-contrib"; + repo = "otel/opentelemetry-collector-contrib"; + tag = "0.157.0"; + index = "sha256:f2f01157055a9b2aab9df7118e1f1c9abf345e99b23bc7a2bc791db374a7d0f6"; + digest = "sha256:4eb842091c796156d4d3c994eb22ba793590f5723719dbf6b8436cb4dfc17f48"; + hash = "sha256-/btIX9XM2O9wkPwAiN6mLJw8JZ2qoB1yzChWcYiEaGY="; + }; + jaeger = { + upstream = "docker.io/jaegertracing/all-in-one"; + repo = "jaegertracing/all-in-one"; + tag = "1.55"; + index = "sha256:f6b5d09073f14f76873d300f565a6691d815e81bea8e07e1dc3ff67e0596dd4e"; + digest = "sha256:d5bbf80eb37e3a0d1b1644f17d1c3a7b88abd74177ec06198d4a289b58b41798"; + hash = "sha256-Ag47jLKCMmMczbW0muNZuaOYkMtEHxVYEt9+BkgYWQ0="; + }; + prometheus = { + upstream = "docker.io/prom/prometheus"; + repo = "prom/prometheus"; + tag = "v3.5.3"; + index = "sha256:ddc2493835a1509976d5e4e0c94199c4f843ce1f42dd6bcfc8231ba734a93ff7"; + digest = "sha256:442634af681c5988a3ccbd4c6e6ab57e077dc89eead4a31e2abf410501874a92"; + hash = "sha256-yf0l7RsqNY+X6muqRkaK+PlfuUP8w2+2UMaM8I1oUxs="; + }; + # Named by the patched SandboxConfig as localhost:5000/pause:3.10.2@...; + # atelet rewrites localhost:5000 to kind-registry:5000, so repo `pause`. + pause = { + upstream = "registry.k8s.io/pause"; + repo = "pause"; + tag = "3.10.2"; + index = "sha256:f548e0e8e3dc1896ca956272154dde3314e8cc4fde0a57577ee9fa1c63f5baf4"; + digest = "sha256:412c4a7219cb8a299a37337f3d87810c5340095322e15594a1637785adad0f17"; + hash = "sha256-wQV0nZBm2siOhrsElV7IS1dyk3ppsRzC0TtugyGGxA0="; + }; + }; + + # SandboxConfig gvisor-default's amd64 asset (sandboxconfig-gvisor.yaml:34-35 + # at d277088b). atelet tries anonymous GCS, then its S3 client with the same + # bucket and key against the in-cluster RustFS; bootstrap step 40 puts this + # file there, so a node without internet still gets runsc. + gvisor = + let + bucket = "gvisor"; + key = "releases/nightly/2026-09-02/x86_64/gvisor.tar.zstd"; + sha256 = "d547d81401461fd1c679c5c4fa0a6c2b8ef7dc3c22ce23c9e25dcc4c69cfd06f"; + in + fetchurl { + name = "gvisor-nightly-2026-09-02-x86_64.tar.zstd"; + url = "https://storage.googleapis.com/${bucket}/${key}"; + inherit sha256; + passthru = { + inherit bucket key sha256; + }; + }; + + # The shape myAxFleet.registrySeed takes: name -> { oci, repo, tag }. + seed = + lib.mapAttrs' ( + n: v: + lib.nameValuePair "substrate-${n}" { + oci = v; + inherit (v) repo tag; + } + ) components + // lib.mapAttrs (_: v: { + oci = v; + inherit (v) repo tag; + }) thirdParty; +in +linkFarm "substrate-images-${version}" ( + lib.mapAttrsToList (n: v: { + name = "components/${n}"; + path = v; + }) components + ++ lib.mapAttrsToList (n: v: { + name = "third-party/${n}"; + path = v; + }) thirdParty + ++ [ + { + name = "gvisor/gvisor.tar.zstd"; + path = gvisor; + } + ] +) +// { + inherit + components + thirdParty + gvisor + seed + version + ; +} diff --git a/pkgs/substrate/patches/0001-sandboxconfig-pause-localhost.patch b/pkgs/substrate/patches/0001-sandboxconfig-pause-localhost.patch new file mode 100644 index 000000000..94eab02b6 --- /dev/null +++ b/pkgs/substrate/patches/0001-sandboxconfig-pause-localhost.patch @@ -0,0 +1,15 @@ +--- a/manifests/ate-install/sandboxconfig-gvisor.yaml ++++ b/manifests/ate-install/sandboxconfig-gvisor.yaml +@@ -27,7 +27,11 @@ + sandboxClass: gvisor + # The root sandbox container's image. On GCP, prefer the in-project mirror + # gcr.io/gke-release/pause@sha256:bcbd57ba5653580ec647b16d8163cdd1112df3609129b01f912a8032e48265da. +- pauseImage: "registry.k8s.io/pause:3.10.2@sha256:f548e0e8e3dc1896ca956272154dde3314e8cc4fde0a57577ee9fa1c63f5baf4" ++ # mecattaf fleet: the same upstream image, seeded digest-preserved into the ++ # NAS registry as its linux/amd64 manifest (a child of the upstream index ++ # f548e0e8...). atelet pulls it itself and rewrites localhost:5000 through ++ # --localhost-registry-replacement, so no node needs registry.k8s.io. ++ pauseImage: "localhost:5000/pause:3.10.2@sha256:412c4a7219cb8a299a37337f3d87810c5340095322e15594a1637785adad0f17" + assets: + amd64: + gvisor: diff --git a/pkgs/substrate/patches/0002-images-linux-amd64-digests.patch b/pkgs/substrate/patches/0002-images-linux-amd64-digests.patch new file mode 100644 index 000000000..ec5c4ea82 --- /dev/null +++ b/pkgs/substrate/patches/0002-images-linux-amd64-digests.patch @@ -0,0 +1,122 @@ +Pin every third-party image the kind install pulls to its linux/amd64 manifest. + +Upstream pins each image by its multi-arch index digest. Seeding an index +digest-preserved means copying every platform it lists (postgres:18-alpine +lists 16 manifests), several GB that would sit in the NAS's /nix/store on its +57 GB eMMC root. Each digest below is the linux/amd64 child listed in the +upstream-pinned index (MEASURED 2026-09-23 with `skopeo inspect --raw`), so +the content pin is the same Merkle chain, one level down. The fleet is +x86_64-only. pkgs/substrate/images.nix fetches each child by this digest and +checks at fetch time that the upstream index lists it for linux/amd64. + +diff -ru a/manifests/ate-install/atenet-egress-with-sdsmint.yaml b/manifests/ate-install/atenet-egress-with-sdsmint.yaml +--- a/manifests/ate-install/atenet-egress-with-sdsmint.yaml ++++ b/manifests/ate-install/atenet-egress-with-sdsmint.yaml +@@ -1111,7 +1111,7 @@ + readOnly: true + containers: + - name: envoy +- image: envoyproxy/envoy:v1.39-latest@sha256:57e14a549d7bd43c8d3f6d03e8cfa653e037d4b38e133acd9b54f38c524401b4 ++ image: envoyproxy/envoy:v1.39-latest@sha256:be87c8b52663c1164a5bdf3c5419017a269cb3d8c74be1ec93638a71f1ffbd4b + securityContext: + allowPrivilegeEscalation: false + capabilities: +diff -ru a/manifests/ate-install/atenet-egress.yaml b/manifests/ate-install/atenet-egress.yaml +--- a/manifests/ate-install/atenet-egress.yaml ++++ b/manifests/ate-install/atenet-egress.yaml +@@ -601,7 +601,7 @@ + terminationGracePeriodSeconds: 60 + containers: + - name: envoy +- image: envoyproxy/envoy:v1.39-latest@sha256:57e14a549d7bd43c8d3f6d03e8cfa653e037d4b38e133acd9b54f38c524401b4 ++ image: envoyproxy/envoy:v1.39-latest@sha256:be87c8b52663c1164a5bdf3c5419017a269cb3d8c74be1ec93638a71f1ffbd4b + securityContext: + allowPrivilegeEscalation: false + capabilities: +diff -ru a/manifests/ate-install/atenet-router.yaml b/manifests/ate-install/atenet-router.yaml +--- a/manifests/ate-install/atenet-router.yaml ++++ b/manifests/ate-install/atenet-router.yaml +@@ -243,7 +243,7 @@ + - name: "drain-signal" + mountPath: "/var/run/atenet" + - name: envoy +- image: envoyproxy/envoy:v1.39-latest@sha256:57e14a549d7bd43c8d3f6d03e8cfa653e037d4b38e133acd9b54f38c524401b4 ++ image: envoyproxy/envoy:v1.39-latest@sha256:be87c8b52663c1164a5bdf3c5419017a269cb3d8c74be1ec93638a71f1ffbd4b + command: + - "/usr/local/bin/envoy" + - "-c" +diff -ru a/manifests/ate-install/kind/otel-collector.yaml b/manifests/ate-install/kind/otel-collector.yaml +--- a/manifests/ate-install/kind/otel-collector.yaml ++++ b/manifests/ate-install/kind/otel-collector.yaml +@@ -177,7 +177,7 @@ + spec: + containers: + - name: otel-collector +- image: otel/opentelemetry-collector-contrib:0.157.0@sha256:f2f01157055a9b2aab9df7118e1f1c9abf345e99b23bc7a2bc791db374a7d0f6 ++ image: otel/opentelemetry-collector-contrib:0.157.0@sha256:4eb842091c796156d4d3c994eb22ba793590f5723719dbf6b8436cb4dfc17f48 + args: + - --config=/conf/otel-collector-config.yaml + volumeMounts: +@@ -242,7 +242,7 @@ + spec: + containers: + - name: jaeger +- image: jaegertracing/all-in-one:1.55@sha256:f6b5d09073f14f76873d300f565a6691d815e81bea8e07e1dc3ff67e0596dd4e ++ image: jaegertracing/all-in-one:1.55@sha256:d5bbf80eb37e3a0d1b1644f17d1c3a7b88abd74177ec06198d4a289b58b41798 + ports: + - name: otlp-grpc + containerPort: 4317 +diff -ru a/manifests/ate-install/kind/prometheus.yaml b/manifests/ate-install/kind/prometheus.yaml +--- a/manifests/ate-install/kind/prometheus.yaml ++++ b/manifests/ate-install/kind/prometheus.yaml +@@ -172,7 +172,7 @@ + type: RuntimeDefault + containers: + - name: prometheus +- image: prom/prometheus:v3.5.3@sha256:ddc2493835a1509976d5e4e0c94199c4f843ce1f42dd6bcfc8231ba734a93ff7 ++ image: prom/prometheus:v3.5.3@sha256:442634af681c5988a3ccbd4c6e6ab57e077dc89eead4a31e2abf410501874a92 + args: + - --config.file=/etc/prometheus/prometheus.yml + - --storage.tsdb.path=/prometheus +diff -ru a/manifests/ate-install/kind/rustfs.yaml b/manifests/ate-install/kind/rustfs.yaml +--- a/manifests/ate-install/kind/rustfs.yaml ++++ b/manifests/ate-install/kind/rustfs.yaml +@@ -62,7 +62,7 @@ + fsGroup: 10001 + containers: + - name: rustfs +- image: rustfs/rustfs:1.0.0-beta.3@sha256:378642b05b7dcb4849fb77ebe6aca4ced1c3f66e7e504247df95a5c9018d3358 ++ image: rustfs/rustfs:1.0.0-beta.3@sha256:d441111efe3af5bbd6e29eba7cd124f96bb7d49a0612390907f9216d5f970382 + imagePullPolicy: IfNotPresent + ports: + - containerPort: 9000 +@@ -102,7 +102,7 @@ + restartPolicy: OnFailure + containers: + - name: create-bucket +- image: amazon/aws-cli:2.17.0@sha256:643507c10ada7964ca6157b3d799f030b90577643da9955d319a77399ed80d73 ++ image: amazon/aws-cli:2.17.0@sha256:7b7edf789765c22d75e61ad6f307c06950e15357b79ea1749104641ce3a11fec + env: + - name: AWS_ACCESS_KEY_ID + value: rustfsadmin +diff -ru a/manifests/ate-install/postgres/postgres.yaml b/manifests/ate-install/postgres/postgres.yaml +--- a/manifests/ate-install/postgres/postgres.yaml ++++ b/manifests/ate-install/postgres/postgres.yaml +@@ -120,7 +120,7 @@ + initContainers: + - name: tls-reloader + restartPolicy: Always +- image: postgres:18-alpine@sha256:9a8afca54e7861fd90fab5fdf4c42477a6b1cb7d293595148e674e0a3181de15 ++ image: postgres:18-alpine@sha256:b6a16ed0eb96e2c362811f7eeb951eac8b459e7b40be4149ea5444aa7c65569b + securityContext: + runAsUser: 70 + command: +@@ -143,7 +143,7 @@ + memory: 32Mi + containers: + - name: postgres +- image: postgres:18-alpine@sha256:9a8afca54e7861fd90fab5fdf4c42477a6b1cb7d293595148e674e0a3181de15 ++ image: postgres:18-alpine@sha256:b6a16ed0eb96e2c362811f7eeb951eac8b459e7b40be4149ea5444aa7c65569b + lifecycle: + postStart: + exec: From 94ce9265aa28bde40ca18489172bc40303438f1f Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 08:41:04 +0200 Subject: [PATCH 11/37] ax images: ax-server, ax-controller, ax-redis and the ax-agent Task image OCI layouts with a `digest` file (pkgs/ax/oci-layout.nix), the shape myAxFleet.registrySeed takes. The digest is read at build time; nothing imports from a derivation. - pkgs/ax/images.nix: the patched ax's server and controller (uid 65532, cacert) and nixpkgs redis for ax-redis. - pkgs/ax-agent-image: ax-task-runner at /usr/local/bin (PID 1), pi from llm-agents, and the ax-agent adapter with modes halogen-smoke, pi (the probe's adapter-pi.sh and validate.py), fetch URL and exit N. pi-models.json names Halogen only, no secret; the endpoint comes from $HALOGEN_URL at run time so one digest serves every host. No claude-code. Co-Authored-By: Claude Opus 5.5 --- pkgs/ax-agent-image/ax-agent.sh | 84 +++++++++++++++++ pkgs/ax-agent-image/default.nix | 144 +++++++++++++++++++++++++++++ pkgs/ax-agent-image/pi-models.json | 26 ++++++ pkgs/ax-agent-image/validate.py | 36 ++++++++ pkgs/ax/images.nix | 64 +++++++++++++ pkgs/ax/oci-layout.nix | 36 ++++++++ 6 files changed, 390 insertions(+) create mode 100644 pkgs/ax-agent-image/ax-agent.sh create mode 100644 pkgs/ax-agent-image/default.nix create mode 100644 pkgs/ax-agent-image/pi-models.json create mode 100755 pkgs/ax-agent-image/validate.py create mode 100644 pkgs/ax/images.nix create mode 100644 pkgs/ax/oci-layout.nix diff --git a/pkgs/ax-agent-image/ax-agent.sh b/pkgs/ax-agent-image/ax-agent.sh new file mode 100644 index 000000000..77c25c0ed --- /dev/null +++ b/pkgs/ax-agent-image/ax-agent.sh @@ -0,0 +1,84 @@ +#!/usr/bin/env bash +# ax-agent: the command an ax Task runs in the fleet's task image. +# +# ax-agent halogen-smoke one chat completion against Halogen with curl +# ax-agent pi pi against Halogen, result validated against a +# JSON Schema (the probe's adapter-pi.sh, e1) +# ax-agent fetch URL GET URL; exits with curl's code (egress checks) +# ax-agent exit N exit with code N +# +# Every mode writes its result to $AX_RESULT_PATH (default .ax/result.json +# under the workspace), which P1's runner serves to the controller once the +# command exits. No mode reads or needs a secret: Halogen has no auth, and no +# Claude credential exists in this image (DESIGN.md section 11). +set -uo pipefail + +mode="${1:-}" +[ $# -gt 0 ] && shift +out="${AX_RESULT_PATH:-${AX_CONWIP_RESULT_PATH:-.ax/result.json}}" +mkdir -p "$(dirname "$out")" +halogen="${HALOGEN_URL:-http://10.42.0.5:8731}" +model="${AX_AGENT_MODEL:-${AX_CONWIP_MODEL:-halogen-qwen3.8-flash-next}}" +share="@share@" + +case "$mode" in +halogen-smoke) + body=$(jq -nc --arg m "$model" \ + '{model:$m, max_tokens:32, messages:[{role:"user", content:"Reply with the single word: ok"}]}') + curl -sS --max-time 300 -H 'content-type: application/json' \ + -d "$body" "$halogen/v1/chat/completions" >"$out.raw" + rc=$? + content=$(jq -r '.choices[0].message.content // empty' "$out.raw" 2>/dev/null) + jq -n --arg mode "$mode" --arg model "$model" --arg url "$halogen" --argjson rc "$rc" \ + --arg content "$content" \ + '{mode:$mode, model:$model, halogen:$url, curl_rc:$rc, ok:($rc == 0 and ($content | length) > 0), content:$content}' >"$out" + echo "ax-agent halogen-smoke: curl rc=$rc content_bytes=${#content}" + [ "$rc" -eq 0 ] && [ -n "$content" ] + ;; + +pi) + # pi reads its providers from ~/.pi/agent/models.json (pi README). HOME is + # /workspace/.home, which survives a suspend. + mkdir -p "$HOME/.pi/agent" + sed "s#@HALOGEN_URL@#${halogen}#" "$share/pi-models.json" >"$HOME/.pi/agent/models.json" + prompt="${AX_CONWIP_PROMPT:-What is 6 times 7? Answer with the number.}" + default_schema='{"type":"object","required":["answer"],"properties":{"answer":{"type":"integer"}}}' + schema="${AX_CONWIP_SCHEMA_JSON:-$default_schema}" + sys="Return only one JSON object, no prose and no code fence, that validates against this JSON Schema: ${schema}" + t0=$(date +%s.%N) + raw=$(pi -p --provider halogen --model "$model" --thinking "${AX_CONWIP_EFFORT:-low}" \ + --no-session --no-tools --no-context-files --no-skills --no-extensions --no-prompt-templates --offline \ + --append-system-prompt "$sys" "$prompt" 2>"$out.stderr") + rc=$? + t1=$(date +%s.%N) + printf '%s' "$raw" >"$out.raw" + verdict=$(printf '%s' "$raw" | python3 "$share/validate.py" "$schema") + vrc=$? + jq -n --arg label "${AX_CONWIP_LABEL:-}" --arg model "$model" --arg effort "${AX_CONWIP_EFFORT:-low}" \ + --argjson rc "$rc" --argjson v "$verdict" --arg secs "$(python3 -c "print($t1-$t0)")" --arg cwd "$PWD" \ + '{label:$label, model:$model, effort:$effort, harness:"pi", harness_rc:$rc, + valid:$v.valid, errors:$v.errors, result:$v.value, seconds:($secs|tonumber), cwd:$cwd}' >"$out" + echo "ax-agent pi: valid=$(jq .valid "$out") rc=$rc vrc=$vrc" + [ "$rc" -eq 0 ] && [ "$vrc" -eq 0 ] + ;; + +fetch) + url="${1:?usage: ax-agent fetch URL}" + code=$(curl -sS -o /dev/null -w '%{http_code}' --max-time 20 "$url") + rc=$? + jq -n --arg url "$url" --argjson rc "$rc" --arg code "$code" '{mode:"fetch", url:$url, curl_rc:$rc, http_code:$code}' >"$out" + echo "ax-agent fetch: $url rc=$rc http=$code" + exit "$rc" + ;; + +exit) + n="${1:-0}" + jq -n --argjson n "$n" '{mode:"exit", code:$n}' >"$out" + exit "$n" + ;; + +*) + echo "usage: ax-agent halogen-smoke | pi | fetch URL | exit N" >&2 + exit 64 + ;; +esac diff --git a/pkgs/ax-agent-image/default.nix b/pkgs/ax-agent-image/default.nix new file mode 100644 index 000000000..d264c2de8 --- /dev/null +++ b/pkgs/ax-agent-image/default.nix @@ -0,0 +1,144 @@ +{ + lib, + callPackage, + dockerTools, + runCommand, + writeTextDir, + buildEnv, + bashInteractive, + coreutils, + findutils, + gnugrep, + gnused, + gawk, + curl, + jq, + python3, + cacert, + gnutar, + gzip, + git, + shellcheck, + # The patched ax (sandbox-class + P1): its ax-task-runner is PID 1. + ax, + # pi from llm-agents. Halogen only on day one; no claude-code in this image + # until Tom rules on Claude credentials in sandboxes (DESIGN.md section 11). + pi, +}: +# The ax Task image for the fleet (DESIGN.md section 10.2): `ax-agent`. +# +# Substrate starts /usr/local/bin/ax-task-runner (ax's DefaultGuestCommand) as +# PID 1; the runner sets up /workspace and runs the Task's spec.command, here +# `ax-agent `. Output: an OCI layout plus `digest`, seeded into the NAS +# registry as ax/ax-agent and named by Tasks as +# localhost:5000/ax/ax-agent@sha256: (atelet rewrites localhost:5000). +# +# No secret is baked in: pi-models.json names Halogen, which has no auth, and +# the endpoint is filled at run time from $HALOGEN_URL (default the worker, +# http://10.42.0.5:8731), so one image digest serves every host. +let + ociLayout = callPackage ../ax/oci-layout.nix { }; + + share = runCommand "ax-agent-share" { } '' + mkdir -p $out/share/ax-agent + cp ${./pi-models.json} $out/share/ax-agent/pi-models.json + cp ${./validate.py} $out/share/ax-agent/validate.py + ''; + + ax-agent = runCommand "ax-agent" { nativeBuildInputs = [ shellcheck ]; } '' + mkdir -p $out/bin + substitute ${./ax-agent.sh} $out/bin/ax-agent \ + --replace-fail '@share@' '${share}/share/ax-agent' \ + --replace-fail '#!/usr/bin/env bash' '#!${bashInteractive}/bin/bash' + chmod +x $out/bin/ax-agent + shellcheck -S warning $out/bin/ax-agent + ''; + + # The runner at the path Substrate's template names. + runner = runCommand "ax-task-runner-usr-local" { } '' + mkdir -p $out/usr/local/bin + ln -s ${ax}/bin/ax-task-runner $out/usr/local/bin/ax-task-runner + ln -s ${ax-agent}/bin/ax-agent $out/usr/local/bin/ax-agent + ''; + + etc = [ + (writeTextDir "etc/passwd" '' + root:x:0:0:root:/root:/bin/bash + agent:x:1000:1000:agent:/workspace/.home:/bin/bash + nobody:x:65534:65534:nobody:/var/empty:/bin/false + '') + (writeTextDir "etc/group" '' + root:x:0: + agent:x:1000: + nobody:x:65534: + '') + ]; + + env = buildEnv { + name = "ax-agent-env"; + paths = [ + bashInteractive + coreutils + findutils + gnugrep + gnused + gawk + curl + jq + python3 + cacert + gnutar + gzip + git + pi + ax-agent + ]; + pathsToLink = [ + "/bin" + "/etc/ssl" + "/share/ax-agent" + ]; + }; + + image = dockerTools.buildLayeredImage { + name = "ax/ax-agent"; + tag = "v${ax.version}-p1"; + contents = [ + env + runner + share + ] + ++ etc; + extraCommands = '' + mkdir -p tmp workspace usr/bin etc/ax-agent/pi + chmod 1777 tmp + ln -s ${coreutils}/bin/env usr/bin/env + cp ${./pi-models.json} etc/ax-agent/pi/models.json + ''; + config = { + Cmd = [ "/usr/local/bin/ax-task-runner" ]; + WorkingDir = "/workspace"; + Env = [ + "PATH=/usr/local/bin:/bin" + "HOME=/workspace/.home" + "SSL_CERT_FILE=${cacert}/etc/ssl/certs/ca-bundle.crt" + "HALOGEN_URL=http://10.42.0.5:8731" + ]; + }; + }; +in +(ociLayout { + name = "ax/ax-agent"; + tag = "v${ax.version}-p1"; + inherit image; +}).overrideAttrs + (old: { + passthru = old.passthru // { + inherit ax-agent image; + }; + meta = { + description = "ax Task image for the fleet: ax-task-runner, pi, and the ax-agent adapter (OCI layout)"; + platforms = [ "x86_64-linux" ]; + license = lib.licenses.asl20; + }; + }) diff --git a/pkgs/ax-agent-image/pi-models.json b/pkgs/ax-agent-image/pi-models.json new file mode 100644 index 000000000..282b8fa89 --- /dev/null +++ b/pkgs/ax-agent-image/pi-models.json @@ -0,0 +1,26 @@ +{ + "providers": { + "halogen": { + "api": "openai-completions", + "apiKey": "none", + "authHeader": false, + "baseUrl": "@HALOGEN_URL@/v1", + "compat": { + "supportsDeveloperRole": false, + "supportsReasoningEffort": false, + "supportsStore": false + }, + "models": [ + { + "id": "halogen-qwen3.8-flash-next", + "name": "Qwen3.8-Flash-Next (Halogen, worker)", + "contextWindow": 262144, + "maxTokens": 32768, + "input": ["text", "image"], + "reasoning": true, + "cost": { "input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0 } + } + ] + } + } +} diff --git a/pkgs/ax-agent-image/validate.py b/pkgs/ax-agent-image/validate.py new file mode 100755 index 000000000..7a7fe3d1b --- /dev/null +++ b/pkgs/ax-agent-image/validate.py @@ -0,0 +1,36 @@ +#!/usr/bin/env python3 +"""Minimal JSON Schema subset check (type, required, properties, enum, items, +additionalProperties:false) for the probe. Reads schema from $1 (JSON text) and +candidate text from stdin; extracts the first JSON object; prints a verdict JSON.""" +import json, sys, re +schema = json.loads(sys.argv[1]) +raw = sys.stdin.read() +T = {"object": dict, "array": list, "string": str, "integer": int, "number": (int, float), "boolean": bool, "null": type(None)} +def check(s, v, p="$"): + errs = [] + t = s.get("type") + if t and not isinstance(v, T[t]) or (t == "integer" and isinstance(v, bool)): + return [f"{p}: expected {t}"] + if "enum" in s and v not in s["enum"]: + errs.append(f"{p}: not in enum") + if isinstance(v, dict): + for r in s.get("required", []): + if r not in v: errs.append(f"{p}.{r}: missing") + props = s.get("properties", {}) + for k, sub in props.items(): + if k in v: errs += check(sub, v[k], f"{p}.{k}") + if s.get("additionalProperties") is False: + errs += [f"{p}.{k}: extra" for k in v if k not in props] + if isinstance(v, list) and "items" in s: + for i, x in enumerate(v): errs += check(s["items"], x, f"{p}[{i}]") + return errs +m = re.search(r"\{.*\}", raw, re.S) +if not m: + print(json.dumps({"valid": False, "errors": ["no JSON object in output"], "value": None})); sys.exit(3) +try: + val = json.loads(m.group(0)) +except Exception as e: + print(json.dumps({"valid": False, "errors": [f"parse: {e}"], "value": None})); sys.exit(3) +errs = check(schema, val) +print(json.dumps({"valid": not errs, "errors": errs, "value": val})) +sys.exit(0 if not errs else 4) diff --git a/pkgs/ax/images.nix b/pkgs/ax/images.nix new file mode 100644 index 000000000..ff146a41d --- /dev/null +++ b/pkgs/ax/images.nix @@ -0,0 +1,64 @@ +{ + callPackage, + dockerTools, + cacert, + redis, + # The patched ax (pkgs/ax: sandbox-class + P1). Passed in so every host seeds + # and references the same derivation. + ax, +}: +# The ax control-plane images for the fleet (DESIGN.md section 8): ax-server, +# ax-controller and ax-redis, each an OCI layout with a `digest` file, seeded +# into the NAS registry as ax/: and referenced by digest in the +# ax-fleet-40-ax manifest (modules/ax-fleet/ax.nix). +# +# Upstream publishes no image (ko:// references only, deploy/*.yaml), so these +# are built here from the same pkgs/ax the CLI uses. +let + ociLayout = callPackage ./oci-layout.nix { }; + tag = "v${ax.version}-p1"; + + axImage = + cmd: + ociLayout { + name = "ax/${cmd}"; + inherit tag; + image = dockerTools.buildLayeredImage { + name = "ax/${cmd}"; + inherit tag; + contents = [ cacert ]; + config = { + Entrypoint = [ "${ax}/bin/${cmd}" ]; + Env = [ "SSL_CERT_FILE=${cacert}/etc/ssl/certs/ca-bundle.crt" ]; + User = "65532:65532"; + }; + }; + }; +in +{ + inherit tag ociLayout; + + ax-server = axImage "ax-server"; + ax-controller = axImage "ax-controller"; + + # nixpkgs redis, not upstream's redis:7-alpine. Upstream's image patches + # protected mode off; a stock redis-server refuses non-loopback clients when + # it has no password, so the manifest passes --protected-mode no. It is + # reachable only as a ClusterIP (DESIGN.md section 9). + ax-redis = ociLayout { + name = "ax/ax-redis"; + tag = "${redis.version}"; + image = dockerTools.buildLayeredImage { + name = "ax/ax-redis"; + tag = "${redis.version}"; + contents = [ ]; + extraCommands = '' + mkdir -p data + ''; + config = { + Entrypoint = [ "${redis}/bin/redis-server" ]; + WorkingDir = "/data"; + }; + }; + }; +} diff --git a/pkgs/ax/oci-layout.nix b/pkgs/ax/oci-layout.nix new file mode 100644 index 000000000..06ab9b92c --- /dev/null +++ b/pkgs/ax/oci-layout.nix @@ -0,0 +1,36 @@ +{ + lib, + runCommand, + skopeo, + jq, +}: +# Turn a dockerTools image (a docker-archive tarball) into an OCI image layout +# at the output root (oci-layout, index.json, blobs/) plus a `digest` file with +# the manifest digest (sha256:...). That is the shape myAxFleet.registrySeed +# takes: the NAS pushes it with `skopeo copy --preserve-digests`, so the digest +# a manifest or a Task names is the digest the registry serves. +# +# The digest is read at BUILD time, inside this derivation, and consumers read +# it from "${layout}/digest" in their own build steps. No import-from-derivation. +{ + name, + image, + tag ? "latest", +}: +runCommand "${lib.replaceStrings [ "/" ] [ "-" ] name}-oci" + { + nativeBuildInputs = [ + skopeo + jq + ]; + passthru = { inherit image tag; }; + } + '' + export HOME=$TMPDIR + # gzip layers: these images cross the coordinator's wifi leg when pulled. + skopeo --insecure-policy --tmpdir "$TMPDIR" copy --quiet \ + --dest-compress --dest-compress-format gzip \ + docker-archive:${image} oci:$out:${tag} + jq -r '.manifests[0].digest' $out/index.json > $out/digest + grep -Eq '^sha256:[0-9a-f]{64}$' $out/digest + '' From 7170e6dc6fbf83bb6f5bc0a5580d980fb0c1d7f2 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 08:45:54 +0200 Subject: [PATCH 12/37] ax-fleet: k3s control on the NAS, gVisor harness on the coordinator (modules/ax-fleet) One module, one kill switch per host (myAxFleet.enable), three roles: control (nas: k3s server with its kubelet, untainted; registry; seed and bootstrap runners), harness (coordinator: k3s agent tainted ate.dev/sandboxClass=gvisor:NoSchedule, labelled with the Substrate version at registration; guard chain; NetworkManager conf.d drop-in and a config reload, never a restart), inference (worker: one assertion). Folds #447's CIDRs, assertions, feature gates and runtime-config; drops its Cilium, containerd template and runsc RuntimeClass (Substrate runs its own runsc). Keeps #446's registry only, on the data pool. flannel VXLAN and kube-proxy bound to the LAN leg. k3s state bind-mounted from /mnt/fast, PersistentVolumes under /mnt/nas/services/ax-fleet/local-path. pkgs/ax-fleet-teardown wraps the pinned k3s-killall.sh, removes the guard chain and restores the sysctls snapshotted before k3s first ran. substrate.nix and ax.nix are empty for their tracks. Co-Authored-By: Claude Opus 5.5 --- modules/ax-fleet/ax.nix | 4 + modules/ax-fleet/control.nix | 289 +++++++++++++++++++++++++++++ modules/ax-fleet/default.nix | 129 +++++++++++++ modules/ax-fleet/harness.nix | 188 +++++++++++++++++++ modules/ax-fleet/inference.nix | 31 ++++ modules/ax-fleet/interface.nix | 220 ++++++++++++++++++++++ modules/ax-fleet/k3s.nix | 161 ++++++++++++++++ modules/ax-fleet/substrate.nix | 4 + pkgs/ax-fleet-teardown/default.nix | 68 +++++++ 9 files changed, 1094 insertions(+) create mode 100644 modules/ax-fleet/ax.nix create mode 100644 modules/ax-fleet/control.nix create mode 100644 modules/ax-fleet/default.nix create mode 100644 modules/ax-fleet/harness.nix create mode 100644 modules/ax-fleet/inference.nix create mode 100644 modules/ax-fleet/interface.nix create mode 100644 modules/ax-fleet/k3s.nix create mode 100644 modules/ax-fleet/substrate.nix create mode 100644 pkgs/ax-fleet-teardown/default.nix diff --git a/modules/ax-fleet/ax.nix b/modules/ax-fleet/ax.nix new file mode 100644 index 000000000..8616f8389 --- /dev/null +++ b/modules/ax-fleet/ax.nix @@ -0,0 +1,4 @@ +# Track ax fills this (DESIGN.md section 14, A3): the ax-fleet-40-ax manifest, +# registrySeed entries, ax-fleet-image-ref, ax-fleet-smoke and AX_SERVER. It +# writes only the myAxFleet extension points and coordinator scripts. +_: { } diff --git a/modules/ax-fleet/control.nix b/modules/ax-fleet/control.nix new file mode 100644 index 000000000..ae97d842f --- /dev/null +++ b/modules/ax-fleet/control.nix @@ -0,0 +1,289 @@ +{ + config, + lib, + pkgs, + ... +}: +# The control node: the NAS. "Hypervisor on NAS" (Tom, 2026-09-23) read as +# everything that schedules or remembers: the k3s server WITH its kubelet and +# NO taint (so CoreDNS, local-path, Substrate's and ax's control pods and every +# PersistentVolume land here by elimination, because the coordinator is +# tainted), the registry, and the two runners that seed it and bootstrap the +# cluster. DESIGN.md sections 6.2, 6.3, 7, 8, 9. +# +# Nothing churns on the 57 GB eMMC root: k3s's datastore, containerd and +# kubelet state are bind-mounted from the fast tier (stateRoot), and every +# PersistentVolume and the registry live on the data pool. The NAS's shared +# PostgreSQL (Paperless, Immich) is not touched. +let + cfg = config.myAxFleet; + on = cfg.enable && cfg.role == "control"; + + gates = "ClusterTrustBundle=true,ClusterTrustBundleProjection=true,PodCertificateRequest=true"; + + registryPort = lib.toInt (lib.last (lib.splitString ":" cfg.registry)); + registryHost = lib.head (lib.splitString ":" cfg.registry); + + serverFlags = [ + "--cluster-cidr=${cfg.podCidr}" + "--service-cidr=${cfg.serviceCidr}" + "--cluster-dns=${cfg.clusterDns}" + "--flannel-backend=vxlan" + "--advertise-address=${cfg.lan.address}" + "--tls-san=${cfg.lan.address}" + "--tls-san=nas" + "--disable-helm-controller" + # k3s's network-policy controller off: Substrate ships no NetworkPolicy + # (MEASURED at d277088b). Confining worker pods by policy is a follow-up. + "--disable-network-policy" + # local-storage stays ON: the kind install's PostgreSQL and RustFS claim + # volumes (MEASURED kind/rustfs.yaml:16, postgres/postgres.yaml:242). + "--default-local-storage-path=${cfg.localPathRoot}" + "--secrets-encryption" + "--service-node-port-range=${cfg.nodePortRange}" + "--kube-apiserver-arg=feature-gates=${gates}" + # The gates alone serve nothing; this is what serves the group (A5a, + # MEASURED; upstream's hack/create-kind-cluster.sh runtimeConfig). + "--kube-apiserver-arg=runtime-config=certificates.k8s.io/v1beta1=true" + "--kube-controller-manager-arg=feature-gates=${gates}" + ]; + + # k3s hot state on the fast tier, by bind mount (not --data-dir: the NixOS + # module links manifests and images into fixed /var/lib/rancher/k3s paths). + binds = { + "/var/lib/rancher" = "${cfg.stateRoot}/k3s/rancher"; + "/var/lib/kubelet" = "${cfg.stateRoot}/k3s/kubelet"; + "/var/log/pods" = "${cfg.stateRoot}/k3s/pod-logs"; + }; + + kubectl = "${cfg.k3sPackage}/bin/kubectl"; + + # ── the bootstrap steps this track owns ── + bootstrapApi = '' + # 10-api: wait for the apiserver and the admin kubeconfig. + for _ in $(seq 1 300); do + if [ -s /etc/rancher/k3s/k3s.yaml ] && ${kubectl} get --raw /readyz >/dev/null 2>&1; then + echo "apiserver ready" + exit 0 + fi + sleep 2 + done + echo "apiserver not ready" >&2 + exit 1 + ''; + + bootstrapKubeconfig = '' + # 60-kubeconfig: the admin kubeconfig for ax-fleet-kubeconfig on the + # coordinator (read over the existing ssh trust). 0640 root:wheel, never + # printed. + install -d -m 0755 /etc/ax-fleet + tmp=$(mktemp /etc/ax-fleet/.admin.kubeconfig.XXXXXX) + sed 's#https://127.0.0.1:6443#https://${cfg.serverAddress}:6443#' /etc/rancher/k3s/k3s.yaml > "$tmp" + chown root:wheel "$tmp" + chmod 0640 "$tmp" + mv "$tmp" /etc/ax-fleet/admin.kubeconfig + ''; + + stepNames = lib.sort (a: b: a < b) (lib.attrNames cfg.bootstrap); + stepScript = name: pkgs.writeShellScript "ax-fleet-step-${name}" '' + set -euo pipefail + ${cfg.bootstrap.${name}} + ''; + + seedScript = '' + set -euo pipefail + for _ in $(seq 1 120); do + curl -fsS -o /dev/null http://${cfg.registry}/v2/ && break + sleep 1 + done + curl -fsS -o /dev/null http://${cfg.registry}/v2/ + ${lib.concatStrings ( + lib.mapAttrsToList (name: s: '' + echo "seed ${name}: ${s.repo}:${s.tag}" + digest=$(tr -d '[:space:]' < ${s.oci}/digest) + skopeo --insecure-policy copy --all --preserve-digests --dest-tls-verify=false \ + oci:${s.oci} docker://${cfg.registry}/${s.repo}:${s.tag} + skopeo --insecure-policy inspect --raw --tls-verify=false \ + docker://${cfg.registry}/${s.repo}@"$digest" >/dev/null + echo "seeded ${s.repo}@$digest" + '') cfg.registrySeed + )} + echo "registry seed complete: ${toString (lib.length (lib.attrNames cfg.registrySeed))} image(s)" + ''; +in +{ + config = lib.mkIf on { + myAxFleet.kubelet = { + # Protects DNS, DHCP, headscale, Paperless and Immich on 8 cores / 22 GiB. + systemReserved = lib.mkDefault "cpu=2,memory=8Gi"; + }; + + myAxFleet.bootstrap = { + "10-api" = bootstrapApi; + "60-kubeconfig" = bootstrapKubeconfig; + }; + + services.k3s = { + role = "server"; + disable = [ + "traefik" + "servicelb" + "metrics-server" + ]; + # No nodeTaint: the NAS is untainted (DESIGN D5). The `none` version + # value keeps Substrate's version-keyed atelet DaemonSet off the NAS + # (ate-setup only labels nodes that lack the key). + nodeLabel = [ + "ax.mecattaf.dev/role=control" + "ate.dev/substrate-version=none" + ]; + extraFlags = serverFlags; + manifests = lib.mapAttrs (name: m: { + inherit (m) source; + target = "${name}.yaml"; + }) cfg.manifests; + }; + + # ── bind mounts: k3s state on the fast tier ── + # systemd mount units, not fileSystems: they stay out of local-fs.target + # (a failed bind cannot drop the router into emergency mode at boot; only + # k3s, which RequiresMountsFor them, would fail), and the VM test runs the + # exact same units (qemu-vm replaces `fileSystems` wholesale). + systemd.mounts = lib.mapAttrsToList (where: what: { + inherit what where; + type = "none"; + options = "bind"; + requires = [ "ax-fleet-dirs.service" ]; + after = [ "ax-fleet-dirs.service" ]; + wantedBy = [ "k3s.service" ]; + before = [ "k3s.service" ]; + }) binds; + + systemd.services.ax-fleet-dirs = { + description = "ax-fleet: create the k3s state and data-pool directories"; + unitConfig = { + DefaultDependencies = false; + RequiresMountsFor = [ + cfg.stateRoot + cfg.localPathRoot + cfg.registryRoot + ]; + }; + serviceConfig = { + Type = "oneshot"; + RemainAfterExit = true; + }; + script = '' + ${lib.concatMapStringsSep "\n" (d: "${pkgs.coreutils}/bin/install -d -m 0711 ${d}") ( + lib.attrValues binds + )} + ${pkgs.coreutils}/bin/install -d -m 0711 ${cfg.localPathRoot} + ${pkgs.coreutils}/bin/install -d -m 0750 -o docker-registry -g docker-registry ${cfg.registryRoot} + ''; + }; + + systemd.services.k3s = { + wants = [ "ax-fleet-dirs.service" ]; + after = [ "ax-fleet-dirs.service" ]; + unitConfig.RequiresMountsFor = (lib.attrNames binds) ++ [ cfg.localPathRoot ]; + }; + + # ── the registry (moved from #446), on the data pool ── + services.dockerRegistry = { + enable = true; + listenAddress = registryHost; + port = registryPort; + storagePath = cfg.registryRoot; + enableDelete = true; + enableGarbageCollect = true; + garbageCollectDates = "weekly"; + # openFirewall NOT used: the source-scoped rule below is the access control. + }; + systemd.services.docker-registry = { + wants = [ "ax-fleet-dirs.service" ]; + after = [ "ax-fleet-dirs.service" ]; + unitConfig.RequiresMountsFor = [ cfg.registryRoot ]; + }; + + # ── firewall: only what the coordinator and the pods need ── + # #447 opened 6443 to the whole LAN and #446 opened 5432/9000/5000 to the + # whole LAN. Now: 6443, 5000 and VXLAN only from the coordinator; DNS and + # the apiserver from pods on cni0. 5432, 6379 and 9000 are never opened on + # the host. Nothing on tailscale0. + networking.firewall.extraInputRules = '' + iifname "${cfg.lan.interface}" ip saddr { ${lib.concatStringsSep ", " cfg.harnessAddresses} } tcp dport { 6443, ${toString registryPort} } accept comment "ax-fleet: kube API and registry, coordinator only" + iifname "${cfg.lan.interface}" ip saddr { ${lib.concatStringsSep ", " cfg.harnessAddresses} } udp dport 8472 accept comment "ax-fleet: flannel VXLAN, coordinator only" + iifname "cni0" ip saddr ${cfg.podCidr} tcp dport { 53, 6443 } accept comment "ax-fleet: pods to AdGuard and the apiserver" + iifname "cni0" ip saddr ${cfg.podCidr} udp dport 53 accept comment "ax-fleet: pods to AdGuard" + ''; + + # ── the registry seed: from store paths in the NAS closure ── + systemd.services.ax-fleet-registry-seed = { + description = "ax-fleet: seed the NAS registry from the store, digests preserved"; + wantedBy = [ "multi-user.target" ]; + requires = [ "docker-registry.service" ]; + after = [ "docker-registry.service" ]; + path = [ + pkgs.skopeo + pkgs.curl + pkgs.coreutils + ]; + environment.HOME = "/var/lib/ax-fleet"; + serviceConfig = { + Type = "oneshot"; + RemainAfterExit = true; + StateDirectory = "ax-fleet"; + }; + script = seedScript; + }; + + # ── the bootstrap: myAxFleet.bootstrap, in name order ── + systemd.services.ax-fleet-bootstrap = { + description = "ax-fleet: bootstrap the cluster (idempotent steps, in name order)"; + wantedBy = [ "multi-user.target" ]; + wants = [ + "k3s.service" + "docker-registry.service" + "ax-fleet-registry-seed.service" + ]; + after = [ + "k3s.service" + "docker-registry.service" + "ax-fleet-registry-seed.service" + ]; + path = [ + cfg.k3sPackage + pkgs.coreutils + pkgs.gnugrep + pkgs.gnused + pkgs.gawk + pkgs.findutils + pkgs.jq + pkgs.curl + pkgs.skopeo + pkgs.util-linux + pkgs.bash + ]; + environment = { + KUBECONFIG = "/etc/rancher/k3s/k3s.yaml"; + HOME = "/var/lib/ax-fleet"; + }; + serviceConfig = { + Type = "oneshot"; + RemainAfterExit = true; + Restart = "on-failure"; + RestartSec = 30; + TimeoutStartSec = "45min"; + StateDirectory = "ax-fleet"; + }; + script = '' + set -euo pipefail + ${lib.concatMapStringsSep "\n" (n: '' + echo "== ax-fleet-bootstrap step ${n}" + ${stepScript n} + '') stepNames} + echo "== ax-fleet-bootstrap complete: ${toString (lib.length stepNames)} step(s)" + ''; + }; + }; +} diff --git a/modules/ax-fleet/default.nix b/modules/ax-fleet/default.nix new file mode 100644 index 000000000..3ea282549 --- /dev/null +++ b/modules/ax-fleet/default.nix @@ -0,0 +1,129 @@ +{ + config, + lib, + options, + ... +}: +# ax on the fleet (DESIGN.md, ~/today/evals-2026-09-23/ax-fleet/). One import +# per host, one kill switch per host (`myAxFleet.enable`), three roles: +# +# control nas k3s server + kubelet (untainted), Substrate and ax +# control planes, the registry. "Hypervisor on NAS." +# harness coordinator k3s agent tainted ate.dev/sandboxClass=gvisor, the +# atelet DaemonSet and the gVisor WorkerPool. "Agent +# harnesses on coordinator." +# inference worker Halogen as today, as a host service. Nothing from +# this PR runs there; one evaluation assertion only. +# +# Folded in and deleted: modules/k3s-fleet.nix (#447; its CIDRs, assertions, +# feature gates and runtime-config are kept below and in ./k3s.nix; its Cilium, +# containerd template and runsc RuntimeClass are dropped because Substrate runs +# its own runsc inside the worker pods) and hosts/nas/state-services.nix (#446; +# only its registry is kept, in ./control.nix). +let + cfg = config.myAxFleet; + + # ── CIDR arithmetic (from #447), so the overlap check is arithmetic ── + ipToInt = + s: + let + o = map lib.toInt (lib.splitString "." s); + at = builtins.elemAt o; + in + (at 0) * 16777216 + (at 1) * 65536 + (at 2) * 256 + (at 3); + + cidrRange = + c: + let + parts = lib.splitString "/" c; + base = ipToInt (builtins.head parts); + bits = lib.toInt (builtins.elemAt parts 1); + size = builtins.foldl' (a: _: a * 2) 1 (lib.range 1 (32 - bits)); + in + { + lo = base; + hi = base + size - 1; + }; + + overlaps = + a: b: + let + x = cidrRange a; + y = cidrRange b; + in + x.lo <= y.hi && y.lo <= x.hi; + + inCidr = ip: c: overlaps "${ip}/32" c; + + roleHost = { + control = "nas"; + harness = "coordinator"; + inference = "worker"; + }; +in +{ + imports = [ + ./interface.nix + ./k3s.nix + ./control.nix + ./harness.nix + ./inference.nix + ./substrate.nix + ./ax.nix + ]; + + config = lib.mkMerge [ + { + # ── OUTSIDE THE GATE, ON PURPOSE (kept from #447) ────────────────── + # k3s's own default pod range (10.42.0.0/16) contains the house LAN. + # These evaluate on every host whether or not the fleet is enabled, so + # an edit to a CIDR fails `nix eval`, not the house's DNS. + assertions = [ + { + assertion = !(overlaps cfg.podCidr cfg.lan.cidr); + message = "modules/ax-fleet: the pod CIDR ${cfg.podCidr} overlaps the house LAN ${cfg.lan.cidr}."; + } + { + assertion = !(overlaps cfg.serviceCidr cfg.lan.cidr); + message = "modules/ax-fleet: the service CIDR ${cfg.serviceCidr} overlaps the house LAN ${cfg.lan.cidr}."; + } + { + assertion = !(overlaps cfg.podCidr cfg.serviceCidr); + message = "modules/ax-fleet: the pod CIDR ${cfg.podCidr} and the service CIDR ${cfg.serviceCidr} overlap."; + } + { + assertion = inCidr cfg.clusterDns cfg.serviceCidr && inCidr cfg.axServerClusterIP cfg.serviceCidr; + message = "modules/ax-fleet: clusterDns and axServerClusterIP must sit inside the service CIDR ${cfg.serviceCidr}."; + } + ]; + } + + (lib.mkIf cfg.enable { + assertions = [ + { + # The test VMs carry the production hostnames, so this holds there too. + assertion = config.networking.hostName == roleHost.${cfg.role}; + message = "modules/ax-fleet: ${config.networking.hostName} has role \"${cfg.role}\", which belongs to ${roleHost.${cfg.role}}. nas is control, coordinator is harness, worker is inference."; + } + { + assertion = inCidr cfg.lan.address cfg.lan.cidr; + message = "modules/ax-fleet: lan.address ${cfg.lan.address} is not inside lan.cidr ${cfg.lan.cidr}."; + } + { + assertion = cfg.role != "control" || cfg.lan.address == cfg.serverAddress; + message = "modules/ax-fleet: the control node's LAN address (${cfg.lan.address}) must equal serverAddress (${cfg.serverAddress})."; + } + ]; + }) + + # One switch per host: the harness role brings the ax and kubectl clients + # (modules/ax-client.nix, #454) with it. mkDefault, so a host can still + # say no. Guarded on the option existing: the NAS does not import + # ax-client.nix. + (lib.mkIf (cfg.enable && cfg.role == "harness") ( + lib.optionalAttrs (options ? myAxClient) { + myAxClient.enable = lib.mkDefault true; + } + )) + ]; +} diff --git a/modules/ax-fleet/harness.nix b/modules/ax-fleet/harness.nix new file mode 100644 index 000000000..2885bb61f --- /dev/null +++ b/modules/ax-fleet/harness.nix @@ -0,0 +1,188 @@ +{ + config, + lib, + pkgs, + ... +}: +# The harness node: the coordinator. "Agent harnesses on coordinator" (Tom, +# 2026-09-23). A k3s agent tainted with upstream Substrate's own key, so only +# atelet and the WorkerPool's gVisor worker pods land here; nothing that +# schedules or remembers does (the wifi leg is the least available link). +# DESIGN.md sections 6.2, 6.4, 6.5, 6.6, 9. +# +# What must NOT change for Tom at switch, and how this file keeps it: +# - NetworkManager is not restarted. `networking.networkmanager.unmanaged` +# rewrites NetworkManager.conf, NetworkManager's restart trigger with +# stopIfChanged = true (MEASURED nix eval), which would drop the desk's +# wifi. A conf.d drop-in plus `nmcli general reload conf` instead. +# - the tailnet: flannel and kube-proxy bind the LAN leg only, nothing is +# published, and the guard chain below keeps pods, wifi and the tailnet +# apart even though k3s turns ip_forward on (judge 1's second risk). +# - herdr and every user unit: only system units are added. +# - no containerd template, no runsc on PATH, no RuntimeClass (Substrate +# runs its own runsc in the worker pods). +let + cfg = config.myAxFleet; + on = cfg.enable && cfg.role == "harness"; + + lan = cfg.lan.interface; + podIfs = [ + "cni0" + "flannel.1" + ]; + + ipt = "${pkgs.iptables}/bin/iptables -w"; + registryPort = lib.last (lib.splitString ":" cfg.registry); + + # ── the guard chain (DESIGN 6.4), in mangle FORWARD, position 1 ── + # It runs before the filter rules kube-proxy and flannel insert. + # + # First line, a correction to the design's list: established and related + # traffic RETURNs. Without it the "wifi into pods only from the NAS" line + # also drops every REPLY to a coordinator pod's own outbound connection + # (atelet's anonymous GCS fetch, a worker pod reaching the LAN), because the + # reply arrives on the LAN leg from a source that is not the NAS. Only NEW + # flows are policed, which is the property the guard exists for. + guardRules = + [ "-m conntrack --ctstate ESTABLISHED,RELATED -j RETURN" ] + ++ lib.concatMap ( + g: + lib.concatMap (p: [ + "-i ${g} -o ${p} -j DROP" + "-i ${p} -o ${g} -j DROP" + ]) podIfs + ++ [ + "-i ${lan} -o ${g} -j DROP" + "-i ${g} -o ${lan} -j DROP" + ] + ) cfg.guardInterfaces + ++ [ + "-i ${lan} -o cni0 ! -s ${cfg.serverAddress} -j DROP" + # Pod traffic leaving on the LAN leg is masqueraded to this host's LAN + # address and would inherit every NAS rule that trusts the coordinator + # (ssh, NFS, media, paperless). From pods, the LAN gets only the + # apiserver and the registry on the NAS; the internet (atelet's GCS + # fetch) is unaffected. Everything else in-cluster rides flannel.1. + "-i cni0 -o ${lan} -d ${cfg.serverAddress} -p tcp -m multiport --dports 6443,${registryPort} -j RETURN" + "-i cni0 -o ${lan} -d ${cfg.lan.cidr} -j DROP" + ]; + + guardStart = '' + # ax-fleet guard chain (idempotent) + ${ipt} -t mangle -N ax-fleet-guard 2>/dev/null || true + ${ipt} -t mangle -F ax-fleet-guard + while ${ipt} -t mangle -D FORWARD -j ax-fleet-guard 2>/dev/null; do :; done + ${ipt} -t mangle -I FORWARD 1 -j ax-fleet-guard + ${lib.concatMapStringsSep "\n" (r: "${ipt} -t mangle -A ax-fleet-guard ${r}") guardRules} + # flannel VXLAN from the NAS only; no TCP port is opened. + ${ipt} -A nixos-fw -i ${lan} -s ${cfg.serverAddress} -p udp --dport 8472 -j nixos-fw-accept + ''; + + guardStop = '' + while ${ipt} -t mangle -D FORWARD -j ax-fleet-guard 2>/dev/null; do :; done + ${ipt} -t mangle -F ax-fleet-guard 2>/dev/null || true + ${ipt} -t mangle -X ax-fleet-guard 2>/dev/null || true + ''; + + nmDropIn = '' + # ax-fleet (modules/ax-fleet/harness.nix): k3s's own interfaces are never + # NetworkManager's. Appended (+=) to whatever NetworkManager.conf lists. + [keyfile] + unmanaged-devices+=interface-name:cni0;interface-name:flannel*;interface-name:veth* + ''; + + kubeconfigScript = pkgs.writeShellApplication { + name = "ax-fleet-kubeconfig"; + runtimeInputs = [ + pkgs.openssh + pkgs.coreutils + ]; + text = '' + # Fetch the cluster admin kubeconfig from the NAS over the existing ssh + # trust into ~/.kube/config (0600). The content is never printed. + umask 077 + mkdir -p "$HOME/.kube" + tmp=$(mktemp "$HOME/.kube/.config.XXXXXX") + trap 'rm -f "$tmp"' EXIT + ssh ''${AX_FLEET_NAS:-nas} cat /etc/ax-fleet/admin.kubeconfig > "$tmp" + [ -s "$tmp" ] || { echo "ax-fleet-kubeconfig: empty kubeconfig from the NAS" >&2; exit 1; } + mv "$tmp" "$HOME/.kube/config" + trap - EXIT + echo "wrote $HOME/.kube/config" + ''; + }; +in +{ + config = lib.mkIf on { + myAxFleet.kubelet = { + # Tom's seats, Chrome and a coordinator Halogen feel pressure after the + # sandboxes are evicted, never before. + systemReserved = lib.mkDefault "cpu=8,memory=32Gi"; + kubeReserved = lib.mkDefault "cpu=1,memory=2Gi"; + evictionHard = lib.mkDefault "memory.available<8Gi"; + }; + + services.k3s = { + role = "agent"; + # The IP, not the name `nas`. + serverAddr = "https://${cfg.serverAddress}:6443"; + nodeTaint = [ cfg.harnessTaint ]; + # Registered WITH the version label: ate-setup only labels nodes that + # exist when it runs (MEASURED version.go:113-131), and the coordinator + # joins after the NAS. Without it atelet (version-keyed) never lands. + nodeLabel = [ + "ax.mecattaf.dev/role=harness" + "ate.dev/substrate-version=${cfg.substrateVersion}" + ]; + }; + + # k3s turns ip_forward on at start anyway; the guard chain is what keeps + # it safe. IPv6 forwarding stays 0 (the wifi leg keeps accepting RAs). + boot.kernel.sysctl."net.ipv4.ip_forward" = 1; + # `default`, not `all`: only interfaces created after this applies (cni0, + # veth*, flannel.1) get proxy ARP; wlp192s0 never answers ARP for + # addresses it routes elsewhere. `all` is Tom's call (DESIGN Unknowns 11). + boot.kernel.sysctl."net.ipv4.conf.default.proxy_arp" = 1; + + networking.firewall.extraCommands = guardStart; + networking.firewall.extraStopCommands = guardStop; + + # ── NetworkManager: a drop-in and a config reload, never a restart ── + environment.etc."NetworkManager/conf.d/90-ax-fleet.conf".text = nmDropIn; + systemd.services.ax-fleet-nm-unmanaged = { + description = "ax-fleet: make NetworkManager re-read its conf.d (k3s interfaces unmanaged)"; + wantedBy = [ "multi-user.target" ]; + after = [ "NetworkManager.service" ]; + before = [ "k3s.service" ]; + restartTriggers = [ nmDropIn ]; + serviceConfig = { + Type = "oneshot"; + RemainAfterExit = true; + # Tolerated: with NetworkManager down there is nothing to reload. + ExecStart = "-${pkgs.networkmanager}/bin/nmcli general reload conf"; + }; + }; + systemd.services.k3s.wants = [ "ax-fleet-nm-unmanaged.service" ]; + + # ── ax-server on 127.0.0.1:8080, through kube-proxy's OUTPUT rules ── + # The ClusterIP is never a NodePort: ax's API has no authentication + # (upstream #376). Loopback only. + systemd.sockets.ax-server-proxy = { + description = "ax-fleet: ax-server on 127.0.0.1:8080"; + wantedBy = [ "sockets.target" ]; + listenStreams = [ "127.0.0.1:8080" ]; + }; + systemd.services.ax-server-proxy = { + description = "ax-fleet: proxy 127.0.0.1:8080 to the ax-server ClusterIP"; + requires = [ "ax-server-proxy.socket" ]; + after = [ "ax-server-proxy.socket" ]; + serviceConfig = { + ExecStart = "${config.systemd.package}/lib/systemd/systemd-socket-proxyd ${cfg.axServerClusterIP}:8080"; + DynamicUser = true; + PrivateTmp = true; + }; + }; + + environment.systemPackages = [ kubeconfigScript ]; + }; +} diff --git a/modules/ax-fleet/inference.nix b/modules/ax-fleet/inference.nix new file mode 100644 index 000000000..fd44083d7 --- /dev/null +++ b/modules/ax-fleet/inference.nix @@ -0,0 +1,31 @@ +{ + config, + lib, + ... +}: +# The inference node: the worker. "Halogen inference mainly on worker" (Tom, +# 2026-09-23). Halogen stays the worker's host service; sandboxes reach it at +# halogenEndpoint through Substrate's egress gateway on the NAS (SNAT to the +# NAS's LAN address). Nothing from this PR runs here at runtime and the worker +# is not switched in this motion: this file is one evaluation assertion, that +# Halogen's port stays open on the LAN leg. +let + cfg = config.myAxFleet; + on = cfg.enable && cfg.role == "inference"; + port = lib.toInt (lib.last (lib.splitString ":" cfg.halogenEndpoint)); + lanPorts = config.networking.firewall.interfaces.${cfg.lan.interface}.allowedTCPPorts or [ ]; +in +{ + config = lib.mkIf on { + assertions = [ + { + assertion = builtins.elem port lanPorts; + message = "modules/ax-fleet/inference.nix: Halogen's port ${toString port} must stay open on ${cfg.lan.interface}; ax sandboxes reach ${cfg.halogenEndpoint} through the egress gateway on the NAS."; + } + { + assertion = !config.services.k3s.enable; + message = "modules/ax-fleet/inference.nix: the worker is not a k3s node in this motion."; + } + ]; + }; +} diff --git a/modules/ax-fleet/interface.nix b/modules/ax-fleet/interface.nix new file mode 100644 index 000000000..0b765635f --- /dev/null +++ b/modules/ax-fleet/interface.nix @@ -0,0 +1,220 @@ +{ + lib, + inputs, + ... +}: +# ax on the fleet: the options, and nothing else. +# +# Design: ~/today/evals-2026-09-23/ax-fleet/DESIGN.md section 5. Tom's ruling +# (2026-09-23, verbatim): "this is a dotfiles task to do on my nixos fleet. the +# decisions there were already made: hypervisor on NAS, agent harnesses on +# coordinator, halogen inference mainly on worker (can also run on coordinator +# if we need redundancy or a second parallel halogen task)." +# +# `enable` is THE kill switch. Its default is false, so a host that imports +# this module and says nothing renders nothing from it. The substrate and ax +# tracks write only the three extension points at the bottom (`manifests`, +# `registrySeed`, `bootstrap`); everything above them is read by the cluster +# modules next to this file. +let + inherit (lib) mkOption mkEnableOption types; +in +{ + options.myAxFleet = { + enable = mkEnableOption "this host's part of ax on the fleet (k3s, Substrate, ax). THE kill switch"; + + role = mkOption { + type = types.enum [ + "control" + "harness" + "inference" + ]; + description = '' + control = the NAS (k3s server, Substrate and ax control planes, the + registry). harness = the coordinator (k3s agent, gVisor sandboxes). + inference = the worker (Halogen as a host service; nothing from this + PR at runtime). Asserted against the hostname in ./default.nix. + ''; + }; + + lan = { + interface = mkOption { + type = types.str; + example = "enp1s0"; + description = "The LAN leg: enp1s0 (nas), wlp192s0 (coordinator), enp191s0 (worker). Tests use eth1."; + }; + address = mkOption { + type = types.str; + example = "10.42.0.1"; + description = "This host's address on the LAN leg."; + }; + cidr = mkOption { + type = types.str; + default = "10.42.0.0/24"; + }; + }; + + guardInterfaces = mkOption { + type = types.listOf types.str; + default = [ "tailscale0" ]; + description = "Interfaces the coordinator guard chain isolates from pods and from the LAN leg (the tailnet). Tests use [ \"eth2\" ]."; + }; + + serverAddress = mkOption { + type = types.str; + default = "10.42.0.1"; + description = "The k3s server the harness agent dials: the IP, never the name."; + }; + harnessAddresses = mkOption { + type = types.listOf types.str; + default = [ "10.42.0.2" ]; + description = "LAN addresses of harness nodes: the only sources the NAS admits to 6443, the registry and VXLAN."; + }; + podCidr = mkOption { + type = types.str; + default = "10.200.0.0/16"; + }; + serviceCidr = mkOption { + type = types.str; + default = "10.201.0.0/16"; + }; + clusterDns = mkOption { + type = types.str; + default = "10.201.0.10"; + }; + axServerClusterIP = mkOption { + type = types.str; + default = "10.201.0.80"; + }; + nodePortRange = mkOption { + type = types.str; + default = "30000-30999"; + description = "Kept below 32400 (Plex on the NAS, a listener on the coordinator)."; + }; + + k3sPackage = mkOption { + type = types.package; + default = inputs.nixpkgs.legacyPackages.x86_64-linux.k3s_1_36; + defaultText = lib.literalExpression "inputs.nixpkgs.legacyPackages.x86_64-linux.k3s_1_36"; + description = "One k3s derivation for every node: the one the 2026-09-23 probe ran (1.36.2+k3s1 from the dotfiles nixpkgs pin, not stable's)."; + }; + + k3sTokenFile = mkOption { + type = types.nullOr types.str; + default = null; + description = '' + null (the default) means the agenix secret secrets/k3s-token.age, + declared by ./k3s.nix. Tests set a path to a plain file instead. + ''; + }; + + stateRoot = mkOption { + type = types.str; + default = "/mnt/fast"; + description = "The NAS's fast tier. k3s hot state (/var/lib/rancher, /var/lib/kubelet, /var/log/pods) is bind-mounted from /k3s."; + }; + localPathRoot = mkOption { + type = types.str; + default = "/mnt/nas/services/ax-fleet/local-path"; + description = "Every PersistentVolume (k3s local-path), on the data pool."; + }; + registryRoot = mkOption { + type = types.str; + default = "/mnt/nas/services/ax-fleet/registry"; + }; + registry = mkOption { + type = types.str; + default = "10.42.0.1:5000"; + description = "The NAS registry, plain HTTP, LAN only, scoped to the coordinator by nftables."; + }; + + substrateVersion = mkOption { + type = types.str; + default = "d277088b"; + description = "One string: the harness node label value, the component image tag and ate-setup's VERSION."; + }; + harnessTaint = mkOption { + type = types.str; + default = "ate.dev/sandboxClass=gvisor:NoSchedule"; + description = "Upstream Substrate's own taint key (atelet tolerates it)."; + }; + + kubelet = { + systemReserved = mkOption { + type = types.nullOr types.str; + default = null; + description = "kubelet --system-reserved. Defaults per role in control.nix / harness.nix; tests shrink it."; + }; + kubeReserved = mkOption { + type = types.nullOr types.str; + default = null; + }; + evictionHard = mkOption { + type = types.nullOr types.str; + default = null; + }; + }; + + workerPool = { + replicas = mkOption { + type = types.ints.positive; + default = 2; + }; + memoryLimit = mkOption { + type = types.str; + default = "16Gi"; + }; + unreachableTolerationSeconds = mkOption { + type = types.ints.unsigned; + default = 3600; + description = "A configuration value, not an estimate: how long worker pods tolerate an unreachable or not-ready coordinator."; + }; + }; + + halogenEndpoint = mkOption { + type = types.str; + default = "10.42.0.5:8731"; + }; + + # ── Extension points. The substrate and ax tracks write ONLY these. ── + manifests = mkOption { + type = types.attrsOf ( + types.submodule { + options.source = mkOption { + type = types.path; + description = "A YAML file; linked into k3s's auto-deploy directory on the NAS as .yaml."; + }; + } + ); + default = { }; + description = "Our own ax-system objects, auto-deployed by k3s on the NAS. Substrate itself is installed by ate-setup (bootstrap), not here."; + }; + + registrySeed = mkOption { + type = types.attrsOf ( + types.submodule { + options = { + oci = mkOption { + type = types.package; + description = "An OCI image layout at the output root (oci-layout, index.json, blobs/) plus a `digest` file holding the manifest digest (sha256:...)."; + }; + repo = mkOption { type = types.str; }; + tag = mkOption { type = types.str; }; + }; + } + ); + default = { }; + description = "Images the NAS copies into its own registry (skopeo --preserve-digests) before the bootstrap runs."; + }; + + bootstrap = mkOption { + type = types.attrsOf types.lines; + default = { }; + description = '' + "NN-name" -> idempotent bash, run by ax-fleet-bootstrap.service on the + NAS in name order, with KUBECONFIG set to the k3s admin kubeconfig. + The cluster track owns 10-api and 60-kubeconfig. + ''; + }; + }; +} diff --git a/modules/ax-fleet/k3s.nix b/modules/ax-fleet/k3s.nix new file mode 100644 index 000000000..20afb7905 --- /dev/null +++ b/modules/ax-fleet/k3s.nix @@ -0,0 +1,161 @@ +{ + config, + lib, + pkgs, + options, + ... +}: +# k3s, common to the two cluster nodes (control = nas, harness = coordinator). +# DESIGN.md sections 6.2, 6.5 and 7. What the 2026-09-23 probe MEASURED +# working on k3s 1.36.2 (probe-build/vm/flake.nix) is the core: k3s_1_36 from +# the dotfiles nixpkgs pin, the three feature gates on every component, the +# certificates.k8s.io/v1beta1 runtime-config, overlay and br_netfilter, the +# inotify raises, a registry mirror in registries.yaml. +# +# What is deliberately NOT here (DESIGN.md D2, D4, 6.2): +# - Cilium. flannel VXLAN plus kube-proxy (iptables), bound to the LAN leg +# only. No BPF socket-LB next to tailscale0. +# - a containerd template, runsc on the unit PATH, a RuntimeClass. Substrate +# runs its own runsc inside the worker pods and uses no RuntimeClass +# (MEASURED, read-substrate.md 2); containerd keeps k3s's generated config. +# - --data-dir. The NixOS module links manifests and images into fixed +# paths under /var/lib/rancher/k3s, so state moves by bind mount instead +# (./control.nix). +let + cfg = config.myAxFleet; + cluster = cfg.enable && (cfg.role == "control" || cfg.role == "harness"); + + # The feature gates Substrate's podcertcontroller needs, on all three + # components (A5a and the probe MEASURED that kubelet accepts all three). + gates = "ClusterTrustBundle=true,ClusterTrustBundleProjection=true,PodCertificateRequest=true"; + + kubeletArgs = + [ "feature-gates=${gates}" ] + ++ lib.optional (cfg.kubelet.systemReserved != null) "system-reserved=${cfg.kubelet.systemReserved}" + ++ lib.optional (cfg.kubelet.kubeReserved != null) "kube-reserved=${cfg.kubelet.kubeReserved}" + ++ lib.optional (cfg.kubelet.evictionHard != null) "eviction-hard=${cfg.kubelet.evictionHard}"; + + commonFlags = [ + "--resolv-conf=/etc/ax-fleet/resolv.conf" + "--flannel-iface=${cfg.lan.interface}" + # NodePorts (none are declared) could only ever bind the LAN /32, and + # kube-proxy never sets route_localnet. + "--kube-proxy-arg=nodeport-addresses=${cfg.lan.address}/32" + "--kube-proxy-arg=iptables-localhost-nodeports=false" + ] + ++ map (a: "--kubelet-arg=${a}") kubeletArgs; + + # containerd pulls every pod image through the NAS registry; for anything + # the mirror lacks it falls back to the upstream registry, so a missing + # seed degrades rather than breaks on the fleet. The offline VM test is what + # proves that nothing is missing. + registriesYaml = '' + mirrors: + "${cfg.registry}": + endpoint: + - "http://${cfg.registry}" + "docker.io": + endpoint: + - "http://${cfg.registry}" + "registry.k8s.io": + endpoint: + - "http://${cfg.registry}" + ''; + + # Values recorded before k3s first runs on this host, restored by + # ax-fleet-teardown. The kubelet sets the three kernel keys itself + # (INFERRED from upstream kubelet behaviour; the VM test records them). + snapshotKeys = [ + "net.ipv4.ip_forward" + "net.ipv6.conf.all.forwarding" + "net.ipv4.conf.all.proxy_arp" + "net.ipv4.conf.default.proxy_arp" + "kernel.panic" + "kernel.panic_on_oops" + "vm.overcommit_memory" + ]; +in +{ + config = lib.mkIf cluster ( + lib.mkMerge [ + { + services.k3s = { + enable = true; + package = cfg.k3sPackage; + tokenFile = + if cfg.k3sTokenFile != null then cfg.k3sTokenFile else config.age.secrets.k3s-token.path; + nodeIP = cfg.lan.address; + # k3s core images (pause, coredns, local-path and its helper) come + # from the pinned airgap tarball, never from the network. + images = [ cfg.k3sPackage.airgap-images ]; + extraFlags = commonFlags; + }; + + environment.etc."rancher/k3s/registries.yaml".text = registriesYaml; + # The kubelet's (and so CoreDNS's) upstream resolver: AdGuard on the + # NAS, never the coordinator's systemd-resolved stub. + environment.etc."ax-fleet/resolv.conf".text = "nameserver ${cfg.serverAddress}\n"; + + boot.kernelModules = [ + "overlay" + "br_netfilter" + ]; + # Both hosts define 512 / 524288 today (modules/common.nix and the + # nixpkgs default); a plain definition would be an evaluation + # conflict. mkOverride 99 raises them to the probe recipe's values. + boot.kernel.sysctl."fs.inotify.max_user_instances" = lib.mkOverride 99 8192; + boot.kernel.sysctl."fs.inotify.max_user_watches" = lib.mkOverride 99 1048576; + + # The k3s package carries k3s-killall.sh; `KillMode=process` means + # pods and shims outlive the unit, so teardown is a real step. + environment.systemPackages = [ (pkgs.callPackage ../../pkgs/ax-fleet-teardown { k3s = cfg.k3sPackage; }) ]; + + systemd.services.k3s = { + # Re-create the manifest and image links AFTER the bind mounts are + # up, whatever order a live switch ran tmpfiles and mounts in. + serviceConfig.ExecStartPre = [ + "${config.systemd.package}/bin/systemd-tmpfiles --create --prefix=/var/lib/rancher/k3s" + ]; + }; + + # ── sysctl snapshot, taken BEFORE this generation's sysctls apply ── + # An activation snippet, not a unit: at a live switch the activation + # script runs before systemd-sysctl is restarted with the new values, + # and at boot it runs before systemd starts. A unit ordered before + # k3s would record ip_forward=1 that this very generation just set. + # Written once; later generations never overwrite it. + system.activationScripts.ax-fleet-sysctl-snapshot = { + text = '' + if [ ! -e /var/lib/ax-fleet/sysctl-before.conf ]; then + mkdir -p /var/lib/ax-fleet + { + ${lib.concatMapStringsSep "\n" ( + k: "v=$(cat /proc/sys/${lib.replaceStrings [ "." ] [ "/" ] k} 2>/dev/null) && echo \"${k} = $v\" || true" + ) snapshotKeys} + } > /var/lib/ax-fleet/sysctl-before.conf.tmp + mv /var/lib/ax-fleet/sysctl-before.conf.tmp /var/lib/ax-fleet/sysctl-before.conf + fi + ''; + }; + } + + # The token: the existing agenix secret (commit 643a4196), recipients + # editors ++ delivered ++ nasOnly. Never printed. Only declared where + # agenix is imported and no test token is given. + (lib.optionalAttrs (options ? age) { + age.secrets = lib.mkIf (cfg.k3sTokenFile == null) { + k3s-token = { + file = ../../secrets/k3s-token.age; + mode = "0400"; + }; + }; + assertions = [ + { + assertion = cfg.k3sTokenFile != null || config.mySecrets.enable or false; + message = "myAxFleet needs agenix delivery (mySecrets.enable) for secrets/k3s-token.age on ${config.networking.hostName}."; + } + ]; + }) + ] + ); +} diff --git a/modules/ax-fleet/substrate.nix b/modules/ax-fleet/substrate.nix new file mode 100644 index 000000000..f56e93895 --- /dev/null +++ b/modules/ax-fleet/substrate.nix @@ -0,0 +1,4 @@ +# Track substrate fills this (DESIGN.md section 14, S3): registrySeed entries +# and bootstrap steps 20-registry-svc, 30-substrate, 40-gvisor-asset, +# 50-workerpool. It writes only the myAxFleet extension points. +_: { } diff --git a/pkgs/ax-fleet-teardown/default.nix b/pkgs/ax-fleet-teardown/default.nix new file mode 100644 index 000000000..3365a2b1b --- /dev/null +++ b/pkgs/ax-fleet-teardown/default.nix @@ -0,0 +1,68 @@ +{ + writeShellApplication, + k3s, + iptables, + iproute2, + procps, + coreutils, + gnugrep, + systemd, +}: +# ax-fleet-teardown: the second half of the kill switch (DESIGN.md section 13). +# +# The k3s unit runs with KillMode=process, so pods and containerd shims outlive +# it; switching `myAxFleet.enable = false` alone leaves them, plus cni0, +# flannel.1, the KUBE-/FLANNEL-/CNI- rules and ip_forward=1. This wraps the +# pinned package's own k3s-killall.sh, removes the coordinator guard chain +# (the firewall reload of the disabled generation does not know it), then +# restores the sysctls recorded before k3s first ran on this host +# (/var/lib/ax-fleet/sysctl-before.conf, written by the activation snippet in +# modules/ax-fleet/k3s.nix). +# +# Left on disk on purpose: /mnt/fast/k3s, the data-pool directories and +# /var/lib/ax-fleet. Deleting them is Tom's call. +writeShellApplication { + name = "ax-fleet-teardown"; + runtimeInputs = [ + k3s + iptables + iproute2 + procps + coreutils + gnugrep + systemd + ]; + text = '' + if [ "$(id -u)" -ne 0 ]; then + echo "ax-fleet-teardown: run as root (sudo)" >&2 + exit 1 + fi + + if systemctl is-enabled --quiet k3s.service 2>/dev/null; then + echo "ax-fleet-teardown: note: k3s.service is still part of this generation;" >&2 + echo " k3s-killall.sh stops it now, and it starts again at the next boot or switch." >&2 + fi + + echo "== k3s-killall.sh (${k3s.version})" + ${k3s}/bin/k3s-killall.sh || echo "k3s-killall.sh exited $?; continuing" >&2 + + echo "== guard chain" + while iptables -w -t mangle -D FORWARD -j ax-fleet-guard 2>/dev/null; do :; done + iptables -w -t mangle -F ax-fleet-guard 2>/dev/null || true + iptables -w -t mangle -X ax-fleet-guard 2>/dev/null || true + + snap=/var/lib/ax-fleet/sysctl-before.conf + if [ -s "$snap" ]; then + echo "== sysctl restore from $snap" + sysctl -p "$snap" + else + echo "ax-fleet-teardown: no $snap; sysctls left as they are" >&2 + fi + + echo "== left behind (should be empty)" + ip -br link show 2>/dev/null | grep -E '^(cni0|flannel\.1|veth)' || true + iptables-save 2>/dev/null | grep -cE 'KUBE-|FLANNEL|CNI-' || true + pgrep -a containerd-shim || true + echo "ax-fleet-teardown: done" + ''; +} From 8676f245200d29149ee45602dbc80993e6b6c092 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 08:45:54 +0200 Subject: [PATCH 13/37] hosts: switch nas, coordinator and worker to myAxFleet; drop the #446/#447 gates nas = control, coordinator = harness (myAxClient on by mkDefault), worker = inference (renders one assertion; not switched in this motion). Deletes modules/k3s-fleet.nix (with its RuntimeClass manifest) and hosts/nas/state-services.nix, superseded by modules/ax-fleet. The k3s-token comment in secrets.nix no longer claims the ciphertext is missing. Co-Authored-By: Claude Opus 5.5 --- hosts/coordinator/default.nix | 50 ++-- hosts/nas/default.nix | 43 ++- hosts/nas/state-services.nix | 362 ------------------------- hosts/worker/default.nix | 34 ++- modules/k3s-fleet.nix | 493 ---------------------------------- secrets.nix | 13 +- 6 files changed, 57 insertions(+), 938 deletions(-) delete mode 100644 hosts/nas/state-services.nix delete mode 100644 modules/k3s-fleet.nix diff --git a/hosts/coordinator/default.nix b/hosts/coordinator/default.nix index ad46a972c..070800e38 100644 --- a/hosts/coordinator/default.nix +++ b/hosts/coordinator/default.nix @@ -74,11 +74,8 @@ # this: it carries its own pins in hosts/nas/network.nix and keeps the # stock loopback mapping. ../../modules/fleet-hosts.nix - # 2026-09-20 sandbox spike: k3s AGENT. The desk box is where an - # interactive session's sandbox wants to be, because herdr and the seat - # are here; ../../modules/k3s-fleet.nix carries the numbers and the - # RuntimeClass wiring for both twins. - ../../modules/k3s-fleet.nix + # ax on the fleet (2026-09-23): the HARNESS node, see myAxFleet below. + ../../modules/ax-fleet # The REWRITE kernel (github.com/mecattaf/tally, U-B1…U-B13) as one system # service against ~/.local/state/tally-rewrite/, coexisting with the live # user-bus tally-daemon.service (U-D13). Declared here, installed by U-D19's @@ -92,26 +89,20 @@ networking.hostName = "coordinator"; - # ── k3s agent: GATE OFF (2026-09-20 sandbox spike) ───────────────────── - # Flip this, the worker's and the NAS's in the SAME commit: an agent whose - # server is not up retries forever and logs nothing useful. - # - # The labels are what the ultracode DAG schedules against, so they describe - # capability and not hardware. `fleet/desk` is the one that matters and the - # one only this box can have: herdr, the seat and the human are here, so an - # item that needs to be watched, teleported into, or answered belongs on - # this node and nowhere else. `fleet/kvm` is true on both twins (nested KVM - # MEASURED = 1 on both, 2026-09-20) and is the micro-VM sandbox class's - # precondition. There is no `fleet/gpu-proximity` here on purpose: Halogen - # is declared on this box with autoStart = false ("a resident model there - # would starve the desktop, TTS and diarization", modules/halogen.nix), so - # a GPU-adjacent item belongs on the worker. - myK3sFleet.enable = false; - myK3sFleet.role = "agent"; - myK3sFleet.nodeLabels = { - "fleet/role" = "desk"; - "fleet/desk" = "true"; - "fleet/kvm" = "true"; + # ── ax on the fleet: THE kill switch for this host ───────────────────── + # The HARNESS node (modules/ax-fleet/harness.nix): a k3s agent tainted + # ate.dev/sandboxClass=gvisor:NoSchedule, so only atelet and the gVisor + # WorkerPool land here. "agent harnesses on coordinator" (Tom, 2026-09-23). + # Switch the NAS first. `false`, switch, then + # `sudo nix run ~/dotfiles#ax-fleet-teardown` is the whole rollback. This + # also turns myAxClient (kubectl, ax) on by mkDefault. + myAxFleet = { + enable = true; + role = "harness"; + lan = { + interface = "wlp192s0"; + address = "10.42.0.2"; + }; }; # Primary physical seat again (2026-09-16); Zenbook remains a second seat. @@ -141,12 +132,9 @@ # backend; flips with the NAS's myNas.paperless.enable (2026-09-13). myNasClient.relayPaperless = true; - # OFF, and it lands OFF (modules/ax-client.nix). There is no cluster on this - # fleet to point kubectl at and no Agent Substrate for ax to delegate to, so - # flipping this today installs two binaries with nothing to talk to. The flip - # is Tom's, one host at a time, and ax-client-topology in flake.nix goes red - # on it by design. - myAxClient.enable = false; + # myAxClient (kubectl + ax) is ON here by mkDefault from myAxFleet's + # harness role (modules/ax-fleet/default.nix); ax-client-topology in + # flake.nix pins that. # The rewrite's served kernel: ONE kernel, on the coordinator (spec §2.4 Q2 — # the worker twin is a ROW this kernel serves, not a second kernel), on the diff --git a/hosts/nas/default.nix b/hosts/nas/default.nix index 4746b1c85..7e9d908d0 100644 --- a/hosts/nas/default.nix +++ b/hosts/nas/default.nix @@ -45,17 +45,12 @@ ./headscale.nix # 2026-09-01: the fleet's OWN tailnet control plane (supersedes #233) ./tailscale-personal.nix # Additional isolated SaaS ingress; never enroll the lent laptops here ./personal-https.nix # Gated, NAS-scoped DNS-01 certificates for private media - # 2026-09-20 sandbox spike: the three state services the sandbox lane - # reads -- a second database on the PostgreSQL ./media.nix already runs, - # an S3-compatible object store on the NVMe, and a container registry. - # State here, machines on the Strix boxes (Appendix J section 5). Gate OFF. - ./state-services.nix ./headscale-backup.nix # consistent identity backup before overseas handover - # 2026-09-20 sandbox spike: the fleet k3s cluster. This box is the SERVER - # and schedules nothing (disableAgent); the machines live on the Strix - # boxes. The CIDR-overlap assertions in that file evaluate whether or not - # the gate is on, which is the point of them. - ../../modules/k3s-fleet.nix + # ax on the fleet (2026-09-23): this box is the CONTROL node. k3s server + # with its kubelet, Substrate's and ax's control planes, the registry. + # "hypervisor on NAS" (Tom, 2026-09-23). The CIDR-overlap assertions in + # that module evaluate whether or not the switch below is on. + ../../modules/ax-fleet ../../modules/adguardhome.nix inputs.nixos-hardware.nixosModules.common-cpu-amd inputs.nixos-hardware.nixosModules.common-pc @@ -132,20 +127,20 @@ myNas.headscale.serverUrl = "https://nas-saas.tail8dd1.ts.net:8443"; myNas.headscale.backup.enable = true; - # ── State services for the sandbox lane: GATE OFF ─────────────────────── - # Three runbook-placed root-owned files stand between this and a flip (the - # rustfs key pair, the registry and object-store directories on /mnt/fast, - # and the Substrate role's password) -- ./state-services.nix's header has - # the commands. The appliance's no-agenix doctrine (./attic.nix) is why - # they are files placed by hand and not ciphertexts. - myNas.stateServices.enable = false; - - # ── The k3s control plane: GATE OFF ──────────────────────────────────── - # secrets/k3s-token.age does not exist in the tree; minting it needs Tom's - # admin age key. Flip all three hosts in the SAME commit -- an agent whose - # server is not up yet retries forever and logs nothing useful. - myK3sFleet.enable = false; - myK3sFleet.role = "server"; + # ── ax on the fleet: THE kill switch for this host ───────────────────── + # One line. `false`, switch, then `sudo nix run ~/dotfiles#ax-fleet-teardown` + # (k3s-killall.sh plus the sysctl restore) removes every trace but the data + # left on purpose under /mnt/fast/k3s and /mnt/nas/services/ax-fleet. The + # k3s token is the agenix secret secrets/k3s-token.age (mySecrets is on + # here). Switch order: this host first, then the coordinator. + myAxFleet = { + enable = true; + role = "control"; + lan = { + interface = "enp1s0"; + address = "10.42.0.1"; + }; + }; # Retired 2026-09-16: the Dell belongs to its owner; Tom no longer # publishes or manages Omarchy updates. Keep historical receipts only. myNas.omarchyUpdateCenter.enable = false; diff --git a/hosts/nas/state-services.nix b/hosts/nas/state-services.nix deleted file mode 100644 index 44c195460..000000000 --- a/hosts/nas/state-services.nix +++ /dev/null @@ -1,362 +0,0 @@ -{ - config, - lib, - pkgs, - unstablePkgs, - ... -}: -# ─── State services for the sandbox lane: PostgreSQL, object store, registry ─ -# -# Tom, 2026-09-20: "having local kubernetes s3 or redis or postgres or -# whatever it needs on the NAS". This is that sentence, item for item, for the -# three that Agent Substrate actually reads. It is state, not compute: the -# NAS has 8 threads and 22 GiB and runs the house's DNS, so control plane and -# state live here and the machines live on the Strix boxes. Appendix J -# section 5 is the long form of that split. -# -# ── WHY THIS IS ONE FILE AND NOT THREE ──────────────────────────────────── -# All three are LAN-only, all three are reached by the same two consumers -# (the k3s agents on the twins), all three land and flip together, and all -# three share one firewall block. Splitting them would mean three gates that -# are only ever flipped at once. -# -# ── THE APPLIANCE'S NO-AGENIX DOCTRINE APPLIES TO TWO OF THE THREE ──────── -# ./attic.nix, #130's ruling: "a root-owned env file placed by hand (runbook -# below) keeps the appliance's no-agenix doctrine." The doctrine was never -# "no agenix here" (hosts/nas/default.nix corrects that) -- it is "no STANDING -# decryption authority over ciphertext this box never reads". The test is -# whether the secret is consumed by this host alone and whether a runbook can -# place it once. -# -# rustfs access keys -> runbook-placed env file. Consumed here only. -# the atepg password -> runbook-placed file. Consumed here only. -# the tunnel creds -> agenix (./cloudflared.nix), because the ciphertext -# has to survive a reflash and be re-minted from -# Tom's key, not re-typed. -# -# ── /mnt/fast IS `nofail`, AND THAT IS LETHAL FOR STATE ─────────────────── -# ./disko.nix marks the 256G M.2 `nofail`, correct for a budget NVMe holding -# regenerable state. ./attic.nix learned the second edition of the -# signing-key trap the hard way: if that disk fails to mount, a StateDirectory -# cheerfully creates a fresh EMPTY tree and the service starts against it. For -# an object store that means Substrate's snapshots silently vanish; for a -# registry it means every image digest 404s mid-run. So every unit below -# carries RequiresMountsFor and refuses to start rather than inventing state. -# That is a Tuesday instead of an outage. Do not remove those lines. -# -# ── RUNBOOK — walk this before flipping the gate ────────────────────────── -# 1. Directories on the NVMe (the units will not create them; see above): -# install -d -m 0700 -o postgres -g postgres /mnt/fast/rustfs # no: see 2 -# 2. rustfs state and its key file: -# install -d -m 0750 -o rustfs -g rustfs /mnt/fast/rustfs -# install -d -m 0700 root:root /var/lib/rustfs-secrets -# printf 'RUSTFS_ACCESS_KEY=%s\nRUSTFS_SECRET_KEY=%s\n' \ -# > /var/lib/rustfs-secrets/env -# chmod 0400 /var/lib/rustfs-secrets/env -# Generate the pair with `openssl rand -hex 24` twice. They are NOT -# Cloudflare credentials and have nothing to do with R2. -# 3. registry storage: -# install -d -m 0750 -o docker-registry -g docker-registry \ -# /mnt/fast/registry -# 4. The Substrate database password (the role and database themselves are -# declarative below; only the password is by hand, because -# `services.postgresql.ensureUsers` deliberately cannot set one): -# install -d -m 0700 root:root /var/lib/postgresql-secrets -# openssl rand -hex 24 > /var/lib/postgresql-secrets/atepg-password -# chmod 0400 /var/lib/postgresql-secrets/atepg-password -# The oneshot below ALTERs the role from that file on every start, so -# rotating the password is "write the file, restart the unit". -# 5. Flip myNas.stateServices.enable, deploy the NAS. -# 6. Prove each one from the coordinator: -# psql "postgresql://atepg:$(cat …)@nas:5432/atepg?sslmode=disable" -c '\conninfo' -# curl -sS -o /dev/null -w '%{http_code}\n' http://nas:9000/ # rustfs -# curl -sS http://nas:5000/v2/_catalog # registry -# -# ── THE DSN SUBSTRATE WANTS ─────────────────────────────────────────────── -# MEASURED, ~/Downloads/substrate: `cmd/ateapi/main.go:347` reads -# ATE_API_POSTGRES_CONNECTION_STRING and `:348` reads ATE_API_POSTGRES_SCHEMA -# (default "public", hack/install-ate.sh:674). The installer skips its own -# bundled PostgreSQL StatefulSet entirely when the connection string is set -# (hack/install-ate.sh:274-286). The store is pgx v5 -# (cmd/ateapi/internal/store/atepg/atepg.go:36-38) and its own header comment -# says it passes "standard libpq sslmode/sslrootcert/sslcert/sslkey -# parameters" through to Connect. The upstream default DSN uses -# client-certificate auth against a projected pod certificate, which is a -# property of running PostgreSQL INSIDE the mesh; an external instance uses -# its own: -# -# ATE_API_POSTGRES_CONNECTION_STRING=postgresql://atepg:@nas:5432/atepg?sslmode=disable -# ATE_API_POSTGRES_SCHEMA=public -# -# `sslmode=disable` is honest rather than lazy: this is a LAN segment behind -# the house router, the traffic never leaves enp1s0, and a self-signed TLS -# layer here would add a certificate to rotate and no attacker it excludes. -# Revisit if the k3s agents ever stop being on the same wire. -# -# PEER AUTH OVER A UNIX SOCKET IS NOT AVAILABLE and was checked: ateapi runs -# as a pod on a Strix box, not on this host (the k3s server here sets -# disableAgent = true and schedules nothing), so the connection is necessarily -# TCP. That is what forces enableTCPIP and the pg_hba lines below. -# -# ── GATE OFF ────────────────────────────────────────────────────────────── -# Lands with `enable = false`. Nothing about this host changes until Tom walks -# the runbook and flips it. -let - cfg = config.myNas.stateServices; - - fastRoot = "/mnt/fast"; - lanInterface = "enp1s0"; - lanCidr = "10.42.0.0/24"; - - # Kept in lockstep with modules/k3s-fleet.nix. If those move, these move. - podCidr = "10.200.0.0/16"; - - atepgPasswordFile = "/var/lib/postgresql-secrets/atepg-password"; - rustfsEnvironmentFile = "/var/lib/rustfs-secrets/env"; -in -{ - options.myNas.stateServices = { - enable = lib.mkEnableOption "PostgreSQL/object-store/registry state services for the sandbox lane (2026-09-20 spike; Appendix J section 5)"; - - databaseName = lib.mkOption { - type = lib.types.str; - default = "atepg"; - description = "Substrate's database on the instance this box already runs. Upstream's own name; Paperless and Immich keep theirs, untouched."; - }; - - rustfsPort = lib.mkOption { - type = lib.types.port; - default = 9000; - description = "S3 API port for the object store. LAN address only, never 0.0.0.0."; - }; - - registryPort = lib.mkOption { - type = lib.types.port; - default = 5000; - description = "Container registry port. LAN address only, plain HTTP; see the k3s mirror note in modules/k3s-fleet.nix."; - }; - }; - - config = lib.mkIf cfg.enable { - assertions = [ - { - # Immich brings PostgreSQL up on this box (./media.nix). If that ever - # stops being true, this module is silently adding a database to - # nothing, and the failure would show up as a Substrate install that - # cannot reach its store rather than as a NixOS error. - assertion = config.services.postgresql.enable; - message = "myNas.stateServices expects the NAS's existing PostgreSQL (brought up by hosts/nas/media.nix). Enable it, or drop the Substrate database from this module."; - } - { - assertion = cfg.databaseName != "paperless" && cfg.databaseName != "immich"; - message = "myNas.stateServices.databaseName must not collide with an existing database on this instance."; - } - ]; - - # ── 1. A SECOND DATABASE ON THE INSTANCE THAT ALREADY RUNS ──────────── - # Not a second PostgreSQL. ./media.nix already put the data directory on - # the NVMe and pinned RequiresMountsFor; this rides that, so there is one - # instance, one dataDir, one backup story. - services.postgresql = { - ensureDatabases = [ cfg.databaseName ]; - ensureUsers = [ - { - name = cfg.databaseName; - # Substrate runs goose migrations at startup - # (cmd/ateapi/internal/store/atepg/schema.go) and creates its own - # tables, so it needs ownership of the database rather than grants - # on a schema someone else owns. - ensureDBOwnership = true; - } - ]; - - # Forced by the shape of the deployment, not by taste: ateapi is a pod - # on a Strix box, so there is no unix socket to peer-authenticate over. - # This flips listen_addresses to "*" -- the nftables block at the bottom - # of this file is the actual access control, exactly as ./paperless.nix - # says of its own port ("the firewall rule below is the actual access - # control"). - enableTCPIP = true; - - # mkBefore so these land ABOVE the module's own generated rules and the - # first match wins. scram-sha-256 and never trust: the LAN is not a - # trusted segment just because it is a LAN, and this database holds the - # control plane's record of every sandbox. - # - # Both source ranges are deliberate. Cilium masquerades pod traffic - # leaving the cluster to the node's own address, so in practice the - # connection arrives from 10.42.0.2 or 10.42.0.5; the pod CIDR line is - # there for the day masquerading is turned off for this destination, so - # that change is a Cilium edit and not also a pg_hba mystery. - authentication = lib.mkBefore '' - host ${cfg.databaseName} ${cfg.databaseName} ${lanCidr} scram-sha-256 - host ${cfg.databaseName} ${cfg.databaseName} ${podCidr} scram-sha-256 - ''; - }; - - # `ensureUsers` deliberately cannot set a password (a password in the Nix - # store is a password in git). This is the runbook's half: read the - # root-owned file placed by hand and ALTER the role from it. Idempotent, - # so rotation is "write the file, restart this unit". - systemd.services.substrate-postgres-password = { - description = "Set the Substrate database role's password from the runbook-placed file"; - after = [ "postgresql.service" ]; - requires = [ "postgresql.service" ]; - wantedBy = [ "multi-user.target" ]; - # No file, no unit: a fresh box before step 4 of the runbook stays quiet - # rather than failing every boot. - unitConfig.ConditionPathExists = atepgPasswordFile; - serviceConfig = { - Type = "oneshot"; - User = "postgres"; - Group = "postgres"; - RemainAfterExit = true; - # The password never reaches the command line (ps is world-readable) - # nor the journal: it goes in through psql's stdin as a bound value. - LoadCredential = "atepg-password:${atepgPasswordFile}"; - }; - script = '' - set -euo pipefail - # The password reaches psql as a bound value read from the credential - # file, so it appears in no argv (ps is world-readable) and in no - # journal line. - ${config.services.postgresql.package}/bin/psql \ - --no-psqlrc --quiet --set=ON_ERROR_STOP=1 --dbname=${cfg.databaseName} <<'SQL' - \set pw `cat "$CREDENTIALS_DIRECTORY/atepg-password"` - ALTER ROLE ${cfg.databaseName} WITH LOGIN PASSWORD :'pw'; - SQL - ''; - }; - - # ── 2. THE OBJECT STORE ─────────────────────────────────────────────── - # Substrate selects its backend by environment, not at compile time - # (MEASURED, cmd/atelet/main.go:226-246): ATE_STORAGE_BACKEND=s3 plus the - # standard AWS_* variables, with AWS_S3_USE_PATH_STYLE for a non-AWS - # endpoint. rustfs is what Substrate's own kind path uses, which keeps - # its manifests unchanged. Note that the base atelet.yaml hardcodes "gcs" - # and the kind overlay patches it, so a non-kind install has to carry - # that patch. - # - # WHY A HAND-WRITTEN UNIT AND NOT services.rustfs: this host rides - # nixpkgs-stable (nixos-26.05, flake.nix:37) and stable has NO rustfs at - # all, neither module nor package. The main pin has both -- - # nixos/modules/services/web-servers/rustfs.nix and rustfs 1.0.0-beta.9 - # -- but importing one nixpkgs's module tree into another's evaluation is - # how you get a module that references options stable does not have. So: - # the package comes across the ./unstable-pkgs.nix seam that attic-server - # and Immich already use, and the unit below is modelled line for line on - # the unstable module's own serviceConfig. When the NAS next rides a - # stable that ships the module, delete this block and use it. - # - # rustfs takes no flags worth the name; everything is environment. - users.users.rustfs = { - isSystemUser = true; - group = "rustfs"; - }; - users.groups.rustfs = { }; - - systemd.services.rustfs = { - description = "RustFS object store (Substrate snapshot backend)"; - documentation = [ "https://rustfs.com/docs/" ]; - after = [ "network-online.target" ]; - wants = [ "network-online.target" ]; - wantedBy = [ "multi-user.target" ]; - - environment = { - RUSTFS_VOLUMES = "${fastRoot}/rustfs"; - # The LAN address and nothing else. Binding 0.0.0.0 here would put an - # unauthenticated-by-default object store on every interface this box - # has, tailnet included. - RUSTFS_ADDRESS = "10.42.0.1:${toString cfg.rustfsPort}"; - # The bundled web console is a second attack surface for a store whose - # only client is a Go program. - RUSTFS_CONSOLE_ENABLE = "false"; - }; - - unitConfig = { - # The /mnt/fast lesson from ./attic.nix, second edition. Without this - # a failed NVMe mount yields an empty store and silently lost - # snapshots instead of a service that refuses to start. - RequiresMountsFor = [ "${fastRoot}/rustfs" ]; - # No keys, no start. The upstream module prints a warning and exits; - # this says the same thing before the process is spawned. - ConditionPathExists = rustfsEnvironmentFile; - }; - - serviceConfig = { - Type = "notify"; - NotifyAccess = "main"; - User = "rustfs"; - Group = "rustfs"; - EnvironmentFile = rustfsEnvironmentFile; - ExecStart = lib.getExe unstablePkgs.rustfs; - LimitNOFILE = 1048576; - LimitNPROC = 32768; - TasksMax = "infinity"; - Restart = "always"; - RestartSec = "10s"; - TimeoutStartSec = "30s"; - TimeoutStopSec = "30s"; - NoNewPrivileges = true; - ProtectHome = true; - PrivateTmp = true; - PrivateDevices = true; - ProtectClock = true; - ProtectKernelTunables = true; - ProtectKernelModules = true; - ProtectControlGroups = true; - RestrictSUIDSGID = true; - RestrictRealtime = true; - }; - }; - - systemd.tmpfiles.rules = [ - "d /var/lib/rustfs-secrets 0700 root root -" - "d /var/lib/postgresql-secrets 0700 root root -" - # NB deliberately no rule for ${fastRoot}/rustfs or ${fastRoot}/registry, - # for ./attic.nix's reason: a tmpfiles rule would race the mount and - # create an empty directory for a failed NVMe to find, which is the - # silent-data-loss path this module exists to avoid. Runbook steps 2 - # and 3 create them on the real disk, once. - ]; - - # ── 3. THE REGISTRY ─────────────────────────────────────────────────── - # The Nix-built /process image and the Substrate cmd/* images have to be - # pullable by both agents. Plain HTTP on the LAN address: see the k3s - # mirror note below, and note that k3s's containerd needs to be told this - # endpoint is not TLS -- that registries.yaml lives in - # modules/k3s-fleet.nix, not here, because it is the client's problem. - services.dockerRegistry = { - enable = true; - listenAddress = "10.42.0.1"; - port = cfg.registryPort; - storagePath = "${fastRoot}/registry"; - # A spike pushes the same tag many times. Without delete plus a - # collection pass, /mnt/fast accumulates every superseded layer forever, - # on the 118 GiB that the object store is also growing into. - enableDelete = true; - enableGarbageCollect = true; - garbageCollectDates = "weekly"; - # openFirewall is deliberately NOT used: it opens the port on every - # interface. The interface-scoped rule at the bottom of this file is the - # access control, same shape as ./attic.nix and ./paperless.nix. - }; - systemd.services.docker-registry.unitConfig.RequiresMountsFor = [ "${fastRoot}/registry" ]; - - # ── THE ONE FIREWALL BLOCK ──────────────────────────────────────────── - # Interface-scoped, not subnet-scoped: `iifname "enp1s0"` is the LAN leg - # and nothing else, so none of these three is reachable over the tailnet, - # over headscale, or through the tunnel in ./cloudflared.nix. All three - # are unauthenticated or weakly authenticated by design and all three are - # on the never-routed list in that file. - networking.firewall.extraInputRules = '' - iifname "${lanInterface}" tcp dport ${toString config.services.postgresql.settings.port} accept comment "substrate postgres, LAN leg only" - iifname "${lanInterface}" tcp dport ${toString cfg.rustfsPort} accept comment "rustfs S3 API, LAN leg only" - iifname "${lanInterface}" tcp dport ${toString cfg.registryPort} accept comment "container registry, LAN leg only" - ''; - - # The unstable seam this module takes. Named here so `grep unstablePkgs` - # finds every consumer; ./unstable-pkgs.nix's header lists the others. - warnings = lib.optional (unstablePkgs.rustfs.version or "" == "") "hosts/nas/state-services.nix: unstablePkgs.rustfs has no version attribute; the pin may have moved."; - }; -} diff --git a/hosts/worker/default.nix b/hosts/worker/default.nix index cb4eda725..3b147ff2d 100644 --- a/hosts/worker/default.nix +++ b/hosts/worker/default.nix @@ -79,9 +79,8 @@ # resolves to loopback, which every distributed library happily binds — the # rank-1-hangs-forever failure. The NAS must NOT import this. ../../modules/fleet-hosts.nix - # 2026-09-20 sandbox spike: k3s AGENT. Idle CPU and /dev/kvm while - # Halogen holds only the GPU, so this is where a long batch belongs. - ../../modules/k3s-fleet.nix + # ax on the fleet (2026-09-23): the INFERENCE role, see myAxFleet below. + ../../modules/ax-fleet # kubectl + the google/ax binaries, behind myAxClient.enable. Imported on # all three interactive hosts, OFF on all three; read that module's header # for the runbook and for what it deliberately does not declare. @@ -90,22 +89,19 @@ networking.hostName = "worker"; - # ── k3s agent: GATE OFF (2026-09-20 sandbox spike) ───────────────────── - # Flip this, the coordinator's and the NAS's in the SAME commit. - # - # `fleet/gpu-proximity=halogen` is the label that earns this box its work: - # Halogen Flash is RESIDENT here (modules/halogen.nix, http://worker:8731) - # and holds the GPU and ~68 GiB of weights for the life of the process. A - # pod scheduled here reaches it over the LAN with no hop, and -- the part - # that actually matters -- Halogen holds the GPU but NOT the CPU, which is - # idle. So this is where a long CPU batch belongs even though the box looks - # busy. No `fleet/desk`: there is no display, no seat and no herdr here. - myK3sFleet.enable = false; - myK3sFleet.role = "agent"; - myK3sFleet.nodeLabels = { - "fleet/role" = "compute"; - "fleet/kvm" = "true"; - "fleet/gpu-proximity" = "halogen"; + # ── ax on the fleet: the INFERENCE role ───────────────────────────────── + # "halogen inference mainly on worker" (Tom, 2026-09-23). Halogen stays this + # box's host service; ax sandboxes on the coordinator reach 10.42.0.5:8731 + # through Substrate's egress gateway on the NAS. This role renders nothing + # at runtime (no k3s, no unit): modules/ax-fleet/inference.nix only asserts + # that 8731 stays open on enp191s0. This host is not switched in the motion. + myAxFleet = { + enable = true; + role = "inference"; + lan = { + interface = "enp191s0"; + address = "10.42.0.5"; + }; }; # ── no display, no compositor ────────────────────────────────────────────── diff --git a/modules/k3s-fleet.nix b/modules/k3s-fleet.nix deleted file mode 100644 index 1c0e5a53e..000000000 --- a/modules/k3s-fleet.nix +++ /dev/null @@ -1,493 +0,0 @@ -{ - config, - lib, - pkgs, - ... -}: -# ─── k3s on this fleet: one module, three roles, one set of numbers ───────── -# -# Tom, 2026-09-20: "kubernetes is world-class for that ... they each stay in -# their lane. Effects ts handles the ultracode-level json dag specification -# and kubernetes schedules it on the right machines." -# -# The shape (Appendix J section 9): server on the NAS with disableAgent, so -# the appliance holds the control plane and schedules nothing; agents on the -# two Strix boxes, which have the cores, the memory and /dev/kvm. State on the -# NAS, machines on the twins. That split is the whole design and it is the -# same split hosts/nas/state-services.nix makes for PostgreSQL and the object -# store. -# -# ── THE NUMBERS, AND WHY THE ASSERTION IS NOT INSIDE THE GATE ───────────── -# k3s defaults to 10.42.0.0/16 for pods and 10.43.0.0/16 for services. THE -# HOUSE LAN IS 10.42.0.0/24 (hosts/nas/network.nix, hosts/nas/router.nix:23's -# pool, modules/fleet-hosts.nix). Taking k3s's default would put every pod on -# an address range that contains the NAS, the coordinator, the worker, the -# printer and the router, and the failure would be a same-day whole-house -# outage on the box that is also the DNS server. -# -# Tom's own 2026-09-09 research already chose the replacement: pods -# 10.200.0.0/16, services 10.201.0.0/16. Those numbers are kept exactly, and -# the overlap check below is a REAL EVALUATED ASSERTION rather than a comment, -# placed OUTSIDE the `enable` gate on purpose. It costs nothing when k3s is -# off, and it means that the day somebody edits a CIDR the flake refuses to -# evaluate rather than the house losing DNS. An assertion in the flake is the -# only thing that makes this non-forgettable. -# -# ── WHAT IS NEVER REACHABLE FROM OUTSIDE THIS HOUSE ────────────────────── -# The kube API on :6443 is the cluster's root credential surface. It is opened -# on the NAS's LAN leg (`iifname "enp1s0"`) and on nothing else, and it is on -# hosts/nas/cloudflared.nix's never-routed list, which that file enforces with -# its own assertion. A tunnel ingress bypasses every nftables rule on the -# appliance, so "not in the tunnel" is a separate guarantee from "firewalled", -# and both are needed. Same for ateapi and ax-server when they land: neither -# implements authorization at all, so reachability IS full control. -# -# ── TRACK K AND TRACK S ARE THE SAME FILE ──────────────────────────────── -# Design K is plain k3s with Cilium and a gVisor RuntimeClass, no Substrate -# and no ax. Design S adds Substrate on top of exactly this. The only thing -# Design S needs from the bottom layer that K does not is four feature-gate -# settings, and a feature gate that nothing asks for is inert: it changes no -# scheduling, admits no new controller, and costs no memory. So the gates are -# included unconditionally and this one module is both tracks. If Substrate is -# never installed, nothing here was wasted; if it is, nothing here has to -# change. -# -# ── THE FEATURE GATES, AND WHAT IS NOW MEASURED ────────────────────────── -# MEASURED from ~/Downloads/substrate, hack/create-kind-cluster.sh:104-112, -# which is upstream's own comment on why they are not optional: -# -# # cmd/podcertcontroller depends on ClusterTrustBundle & PodCertificateRequest. -# # They are not enabled by default as of Kubernetes v1.36 -# featureGates: -# ClusterTrustBundle: true -# ClusterTrustBundleProjection: true -# PodCertificateRequest: true -# runtimeConfig: -# "certificates.k8s.io/v1beta1": "true" -# -# atelet's own pod mounts a projected podCertificate volume and a -# clusterTrustBundle volume (manifests/ate-install/atelet.yaml:271-289), and -# every Substrate component's mTLS identity comes from that signer. -# -# U9 IS ANSWERED, AND THE ANSWER IS YES. The first draft of this file said the -# opposite: that nothing about these gates was measured and that the first -# switch would tell us. A5a measured it the same night, inside a k3s -# 1.35.6+k3s1 guest built from this exact pin, and the answer is that the -# pinned k3s serves the group. MEASURED there: -# -# kubectl get --raw /apis/certificates.k8s.io/v1beta1 | jq -r .resources[].name -# clustertrustbundles -# podcertificaterequests -# podcertificaterequests/status -# kubectl api-resources | grep -i "trustbundle\|podcertificate" -# clustertrustbundles certificates.k8s.io/v1beta1 false ClusterTrustBundle -# podcertificaterequests certificates.k8s.io/v1beta1 true PodCertificateRequest -# -# So Appendix J section 9's step 1 success criterion is met on the pin, its -# fallback paragraph does not have to be taken, and no newer k3s is needed for -# this reason. k3s_1_36 (1.36.2+k3s1) is in both stable and unstable if one is -# ever wanted for another. -# -# TWO THINGS THAT WOULD HAVE BEEN WRONG WITHOUT THAT MEASUREMENT, both fixed -# in this file and both worth knowing before editing it: -# 1. The three gates ALONE serve nothing. See runtimeConfigFlag below: the -# metrics read 1 while the group version stays unserved, so a check that -# read only the metric would have reported a false pass. -# 2. kubelet accepts all three gate names, including ClusterTrustBundle. -# This file used to hand kubelet a subset out of caution. It no longer -# needs to. -# -# ── GATE OFF ───────────────────────────────────────────────────────────── -# Every host lands with `enable = false`. secrets/k3s-token.age does not exist -# in the tree: minting it needs Tom's admin age key, and the overnight spike -# that opened this PR has none. Runbook in the header of each host's gate. -let - cfg = config.myK3sFleet; - - # ── The numbers. One definition, three hosts. ── - podCidr = "10.200.0.0/16"; - serviceCidr = "10.201.0.0/16"; - lanCidr = "10.42.0.0/24"; - - nasLanInterface = "enp1s0"; - nasLanAddress = "10.42.0.1"; - apiPort = 6443; - serverAddr = "https://nas:${toString apiPort}"; - - # hosts/nas/state-services.nix's registry, same box, plain HTTP on the LAN. - registryEndpoint = "${nasLanAddress}:5000"; - - # ── CIDR arithmetic, so the overlap check is arithmetic and not a wish ── - ipToInt = - s: - let - o = map lib.toInt (lib.splitString "." s); - at = builtins.elemAt o; - in - (at 0) * 16777216 + (at 1) * 65536 + (at 2) * 256 + (at 3); - - cidrRange = - c: - let - parts = lib.splitString "/" c; - base = ipToInt (builtins.head parts); - bits = lib.toInt (builtins.elemAt parts 1); - # 2 ^ (32 - bits), without a pow in lib. - size = builtins.foldl' (a: _: a * 2) 1 (lib.range 1 (32 - bits)); - in - { - lo = base; - hi = base + size - 1; - }; - - overlaps = - a: b: - let - x = cidrRange a; - y = cidrRange b; - in - x.lo <= y.hi && y.lo <= x.hi; - - # ── The feature gates Substrate needs (see the header) ── - # All three, on all three components. The earlier draft of this file gave - # kubelet only two of them, on the reasoning that ClusterTrustBundle is - # apiserver-side and an unrecognised gate name is fatal to kubelet. A5a - # MEASURED otherwise on 2026-09-20, inside a k3s 1.35.6+k3s1 guest built from - # this exact pin: "the apiserver, the controller manager and the kubelet all - # accept all three gate names. None of the three components refused an - # unknown gate, and the node reached Ready in about ten seconds." - substrateGates = [ - "ClusterTrustBundle=true" - "ClusterTrustBundleProjection=true" - "PodCertificateRequest=true" - ]; - - # ── THE FLAG THAT ACTUALLY DECIDES IT ───────────────────────────────── - # The three gates alone serve NOTHING. A5a's boot 5 held the gates on and - # dropped this line; MEASURED result: - # - # kubernetes_feature_enabled{name="ClusterTrustBundle",stage="BETA"} 1 - # kubernetes_feature_enabled{name="ClusterTrustBundleProjection"...} 1 - # kubernetes_feature_enabled{name="PodCertificateRequest",stage="BETA"} 1 - # kubectl get --raw /apis/certificates.k8s.io | jq -c .versions - # -> [{"groupVersion":"certificates.k8s.io/v1","version":"v1"}] - # kubectl api-resources | grep -i "trustbundle\|podcertificate" - # -> NO_RESOURCES - # - # So the gates flip to 1 while the group version stays unserved, and a check - # that read only the metric would have reported a false pass. The group - # version is turned on separately, by this flag, and podcertcontroller has - # nothing to talk to without it. It is quoted straight out of upstream's own - # kind config (hack/create-kind-cluster.sh:111-112, the `runtimeConfig` - # block, which is the part a reader skips). - runtimeConfigFlag = "runtime-config=certificates.k8s.io/v1beta1=true"; - - serverFlags = [ - "--cluster-cidr=${podCidr}" - "--service-cidr=${serviceCidr}" - # Cilium is the CNI (autoDeployCharts below). flannel off and k3s's own - # network policy controller off, because Cilium owns both. - "--flannel-backend=none" - "--disable-network-policy" - # The API certificate has to be valid for the name the agents dial. They - # dial `nas`, resolved by the static pins in modules/common.nix:130 and - # modules/fleet-hosts.nix, not by DNS. - "--tls-san=nas" - # A5a's MEASURED working form: the value inside a `-arg=` carries NO leading - # dashes of its own. Both flag families are required and neither is - # sufficient alone; see runtimeConfigFlag above. - "--kube-apiserver-arg=feature-gates=${lib.concatStringsSep "," substrateGates}" - "--kube-apiserver-arg=${runtimeConfigFlag}" - "--kube-controller-manager-arg=feature-gates=${lib.concatStringsSep "," substrateGates}" - "--kubelet-arg=feature-gates=${lib.concatStringsSep "," substrateGates}" - ]; - - agentFlags = [ - "--kubelet-arg=feature-gates=${lib.concatStringsSep "," substrateGates}" - ]; - - # ── containerd: add runsc WITHOUT losing the stock config ────────────── - # `{{ template "base" . }}` is the module's own documented way to keep k3s's - # generated containerd configuration and append to it. Dropping that line - # replaces the whole config and the node loses its CNI, its snapshotter and - # its registry mirrors at once. It is one line and it is load-bearing. - # - # ── THE KEY PATH BELOW IS NOT THE ONE THE NIXPKGS EXAMPLE SHOWS ─────── - # The option's example (nixos/modules/services/cluster/rancher/default.nix - # 628-646) documents - # [plugins."io.containerd.grpc.v1.cri".containerd.runtimes."custom"] - # and that path is WRONG for this k3s. A5a MEASURED, 2026-09-20, reading the - # config that k3s 1.35.6+k3s1 actually generates at - # /var/lib/rancher/k3s/agent/etc/containerd/config.toml: the file starts - # `version = 3` and its runtime table is - # [plugins.'io.containerd.cri.v1.runtime'.containerd.runtimes.runc] - # That is containerd 2.x config v3 (the node reports containerd://2.2.5-k3s2). - # A `grpc.v1.cri` block would be parsed, accepted and silently ignored, which - # is the worst of the three outcomes. Use the path below, and check it again - # the day the k3s pin moves a major version. - # - # ── AND runtime_path, NOT options.BinaryName ───────────────────────── - # The first draft of this file used runtime_type "io.containerd.runc.v2" with - # options.BinaryName pointing at an absolute runsc, which is the shape the - # nixpkgs example suggests and which looks right. A5a MEASURED it and it is a - # trap: a pod under that RuntimeClass reaches Running and STAYS "1/1 Running" - # in kubelet's view, produces NO LOGS AT ALL, and `kubectl exec` into it - # fails with `cannot execute in container ...: in state stopped`. The generic - # runc shim starts runsc but carries neither its stdio nor its state. Do not - # use BinaryName for gVisor. The nixpkgs gvisor package builds - # containerd-shim-runsc-v1 beside runsc, and that shim is what goes here. - containerdTemplate = '' - {{ template "base" . }} - - [plugins.'io.containerd.cri.v1.runtime'.containerd.runtimes.runsc] - runtime_type = "io.containerd.runsc.v1" - runtime_path = "${pkgs.gvisor}/bin/containerd-shim-runsc-v1" - ''; - - # The NAS registry is plain HTTP on the LAN (hosts/nas/state-services.nix - # explains why: this segment never leaves enp1s0 and a self-signed layer - # here adds a certificate to rotate and excludes no attacker). containerd - # will not talk to an HTTP registry unless told, and this is how it is told. - registriesYaml = '' - mirrors: - "${registryEndpoint}": - endpoint: - - "http://${registryEndpoint}" - configs: - "${registryEndpoint}": - tls: - insecure_skip_verify: true - ''; -in -{ - options.myK3sFleet = { - enable = lib.mkEnableOption "this host's membership in the fleet k3s cluster (2026-09-20 sandbox spike; Appendix J section 9)"; - - role = lib.mkOption { - type = lib.types.enum [ - "server" - "agent" - ]; - default = "agent"; - description = '' - server on the NAS (control plane only, schedules nothing); agent on - the Strix boxes (where the machines are). The default is the safe one: - an agent that dials a server it is not. The assertion below refuses a - host whose role and hostname disagree, so a forgotten `role` on the - appliance is an evaluation error and not a second control plane. - ''; - }; - - nodeLabels = lib.mkOption { - type = lib.types.attrsOf lib.types.str; - default = { }; - example = { - "fleet/role" = "desk"; - "fleet/kvm" = "true"; - }; - description = '' - Declarative node labels, applied by kubelet at registration so a node - that reboots comes back schedulable AND labelled. This is the whole - reason the labels are here and not in a `kubectl label` someone has to - remember. - ''; - }; - - ciliumVersion = lib.mkOption { - type = lib.types.str; - default = "1.18.14"; - description = "Cilium chart version. 1.18.14 is the latest of the 1.18 line; 1.19.8 exists and is the next step up."; - }; - - ciliumHash = lib.mkOption { - type = lib.types.str; - # MEASURED 2026-09-20 on the coordinator: built the chart's - # fixed-output derivation with lib.fakeHash and took the hash the - # mismatch reported, then rebuilt clean. Not guessed, and not left as - # fakeHash -- so the first switch does not fail on it. - default = "sha256-js/NLsDWeV+xlcrBc3giFaXltFTE5gey8Zaet5cqTWk="; - description = '' - Hash of the packaged Cilium chart. The module fetches the chart at - build time as a fixed-output derivation, so a wrong hash fails the - build with the right one in the message. - ''; - }; - }; - - config = lib.mkMerge [ - { - # ── OUTSIDE THE GATE, ON PURPOSE ──────────────────────────────────── - # These evaluate on every host whether or not k3s is enabled, so an edit - # to the CIDRs is caught at `nix eval` time rather than at outage time. - assertions = [ - { - assertion = !(overlaps podCidr lanCidr); - message = "modules/k3s-fleet.nix: the pod CIDR ${podCidr} overlaps the house LAN ${lanCidr}. k3s's own default (10.42.0.0/16) does exactly this and would take the NAS, the coordinator, the worker, the printer and the router with it. Pick a range outside the LAN."; - } - { - assertion = !(overlaps serviceCidr lanCidr); - message = "modules/k3s-fleet.nix: the service CIDR ${serviceCidr} overlaps the house LAN ${lanCidr}. Pick a range outside the LAN."; - } - { - assertion = !(overlaps podCidr serviceCidr); - message = "modules/k3s-fleet.nix: the pod CIDR ${podCidr} and the service CIDR ${serviceCidr} overlap each other."; - } - ]; - } - - (lib.mkIf cfg.enable { - assertions = [ - { - assertion = config.mySecrets.enable; - message = "myK3sFleet needs agenix delivery for secrets/k3s-token.age on this host."; - } - { - # The appliance is the only server, and the only server is the - # appliance. Mechanical, because `role` has a default and a - # forgotten one would otherwise be silent. - assertion = (config.networking.hostName == "nas") == (cfg.role == "server"); - message = "modules/k3s-fleet.nix: ${config.networking.hostName} has role \"${cfg.role}\". The NAS is the server and nothing else is; the Strix boxes are agents and nothing else is."; - } - ]; - - age.secrets.k3s-token = { - file = ../secrets/k3s-token.age; - mode = "0400"; - }; - - services.k3s = { - enable = true; - inherit (cfg) role; - tokenFile = config.age.secrets.k3s-token.path; - nodeLabel = lib.mapAttrsToList (k: v: "${k}=${v}") cfg.nodeLabels; - extraFlags = if cfg.role == "server" then serverFlags else agentFlags; - }; - - # Both sides of the cluster need to pull from the NAS registry. - environment.etc."rancher/k3s/registries.yaml".text = registriesYaml; - }) - - # ── THE SERVER: the NAS ─────────────────────────────────────────────── - (lib.mkIf (cfg.enable && cfg.role == "server") { - services.k3s = { - # Embedded etcd rather than the stock sqlite. One server today and one - # server for the foreseeable future, so this buys nothing operational; - # it buys the ability to add a second server later without a datastore - # migration on the box that is also the house router. - clusterInit = true; - # THE APPLIANCE SCHEDULES NOTHING. 8 threads and 22 GiB, running DNS, - # DHCP, the binary cache, Paperless, Immich and headscale. Control - # plane only; the machines are on the twins. - disableAgent = true; - disable = [ - "traefik" - "servicelb" - ]; - - # Cilium as CNI, in the generation rather than in a `cilium install` - # somebody has to remember after every reprovision. cilium-cli 0.19.6 - # is in the pin and stays in the toolbox for `cilium status` and - # `cilium connectivity test`, which are read verbs. - # - # kubeProxyReplacement with k8sServiceHost/k8sServicePort is what lets - # Cilium reach the API server before there is a CNI to reach it - # through: the chicken-and-egg that makes a half-configured cluster - # sit at "0/1 nodes ready" with no useful log line. - autoDeployCharts.cilium = { - name = "cilium"; - repo = "https://helm.cilium.io"; - version = cfg.ciliumVersion; - hash = cfg.ciliumHash; - values = { - ipam.mode = "kubernetes"; - kubeProxyReplacement = true; - k8sServiceHost = "nas"; - k8sServicePort = apiPort; - }; - }; - - manifests = { - # The gVisor class. Handler "runsc" matches the containerd runtime - # name added by containerdTemplate above on each AGENT -- a - # RuntimeClass is a cluster-scoped name for a per-node containerd - # runtime, so both halves have to agree and they are written in the - # same file for that reason. - gvisor-runtimeclass.content = { - apiVersion = "node.k8s.io/v1"; - kind = "RuntimeClass"; - metadata.name = "gvisor"; - handler = "runsc"; - }; - - # ── kata: DELIBERATELY NOT ENABLED ───────────────────────────── - # The pin has kata-runtime 3.32.0, which builds - # containerd-shim-kata-v2 with DEFAULT_HYPERVISOR=qemu and - # HYPERVISORS=qemu, so Kata on QEMU is available from the pin today - # and Kata on cloud-hypervisor is a makeFlags override rather than a - # new package (Appendix J section 9). What is NOT known is whether a - # Kata class under k3s's containerd actually runs the /process image - # as a micro-VM on Strix silicon. That is U14, measured tonight; its - # report is at - # ~/today/review/2026-09-20/sandbox-spike/ (the U14 agent's file). - # Read that before uncommenting. gVisor is the fallback class and is - # the one enabled above. - # - # kata-runtimeclass.content = { - # apiVersion = "node.k8s.io/v1"; - # kind = "RuntimeClass"; - # metadata.name = "kata"; - # handler = "kata-qemu"; - # }; - }; - }; - - # ── THE KUBE API IS LAN-ONLY, AND IS NEVER IN THE TUNNEL ─────────── - # Interface-scoped, same shape as every other door on this box. :6443 is - # the cluster's root credential surface; anything that can reach it can - # schedule a privileged pod on either Strix box. hosts/nas/cloudflared.nix - # carries the matching doctrine block and an assertion that refuses to - # route it, because a tunnel ingress bypasses this rule entirely. - networking.firewall.extraInputRules = '' - iifname "${nasLanInterface}" tcp dport ${toString apiPort} accept comment "kube API, LAN leg only, NEVER in the tunnel" - ''; - }) - - # ── THE AGENTS: coordinator and worker ─────────────────────────────── - (lib.mkIf (cfg.enable && cfg.role == "agent") { - services.k3s = { - inherit serverAddr; - containerdConfigTemplate = containerdTemplate; - }; - - # ── runsc ON THE k3s UNIT'S PATH: REQUIRED, NOT A CONVENIENCE ────── - # `runtime_path` above tells containerd where the SHIM is. The shim then - # execs `runsc` from its OWN $PATH, and the k3s unit has essentially - # none: A5a MEASURED that the nixpkgs rancher module sets - # `path = lib.optional config.boot.zfs.enabled config.boot.zfs.package` - # and nothing else (default.nix:918), so the unit PATH is empty by - # default and k3s relies on its own wrapper for iptables and friends. - # Without this line every sandbox fails at creation, MEASURED: - # - # Failed to create pod sandbox: rpc error: code = Unknown desc = - # failed to start sandbox "...": failed to create containerd task: - # failed to create shim task: OCI runtime create failed: - # exec: "runsc": executable file not found in $PATH - # - # Note also that gVisor is NOT auto-detected. A5a MEASURED that with - # `gvisor` in environment.systemPackages and runsc resolvable at - # /run/current-system/sw/bin/runsc, the generated containerd config - # still contained only `runc` and `runhcs-wcow-process`. k3s ships - # RuntimeClasses for crun, lunatic, nvidia, slight, spin, wasmedge, - # wasmer, wasmtime and wws out of the box, and none for gVisor. The - # runtime has to be declared, which is what this module does. - # - # CAUTION FOR THE NEXT EDITOR: this assignment REPLACES the unit PATH - # rather than extending a populated one. A5a MEASURED `systemctl show - # k3s -p Environment` afterwards containing only the two gvisor - # directories. It did not break kube-proxy or flannel there because the - # nixpkgs k3s package wraps its own binary with the tools it needs, but - # anyone adding a second entry should append rather than assume. - systemd.services.k3s.path = [ pkgs.gvisor ]; - }) - ]; -} diff --git a/secrets.nix b/secrets.nix index 5ab9b0af9..28b8e94d2 100644 --- a/secrets.nix +++ b/secrets.nix @@ -92,16 +92,11 @@ in # NAS private media HTTPS: zone-limited DNS-01 token, no broad Wrangler OAuth # authority. Ciphertext is provisioned before enabling personal-https.nix. "secrets/nas-cloudflare-dns.age".publicKeys = editors ++ nasOnly; - # The k3s cluster join token (2026-09-20 sandbox spike, modules/k3s-fleet.nix). - # Read by all three cluster hosts: the NAS server mints the cluster from it - # and both Strix agents present it to join, so this is the one secret whose - # tier is genuinely "the appliance AND the twins". Written out rather than - # reusing `delivered`, which deliberately excludes the nas. - # - # THE CIPHERTEXT DOES NOT EXIST YET -- this is the recipient ACL, which is - # what `agenix -e` needs to mint it. A k3s token is any sufficiently long - # opaque string; `openssl rand -hex 32` is fine. Mint with: + # The k3s cluster join token (ax on the fleet, modules/ax-fleet/k3s.nix). + # Read by the NAS (server) and the coordinator (agent). The ciphertext + # exists (commit 643a4196); rotate with: # nix develop -c agenix -e secrets/k3s-token.age + # `delivered` also reaches the worker and the client, which never read it. "secrets/k3s-token.age".publicKeys = editors ++ delivered ++ nasOnly; # --- wifi PSK tier: the coordinator, whose Freebox uplink # (wlp192s0) is now declarative too (migrated from an imperative profile on From b67ecd2c692da20398be0195809eafcf79f43400 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 08:45:54 +0200 Subject: [PATCH 14/37] checks: ax-fleet (4 VMs), ax-fleet-boot, ax-fleet-topology; app ax-fleet-teardown tests/ax-fleet: nas, coordinator, worker (Halogen stub), peer (tailnet stand-in); base configs with ax off and an ax-on specialisation switched live, NAS first. Phases 10-cluster and 90-rollback are the cluster track's; 20-substrate and 30-ax are empty for their tracks. Receipt in $out/receipt.json. ax-fleet-topology asserts the real hosts' rendered flags, firewall scope, the NetworkManager.conf no-restart property, the kill switch (mkForce false renders nothing) and parity with the VM nodes. ax-client-topology now expects myAxClient ON on the coordinator only. Co-Authored-By: Claude Opus 5.5 --- flake.nix | 43 +++- tests/ax-fleet-boot/default.nix | 45 +++++ tests/ax-fleet-topology/default.nix | 180 +++++++++++++++++ tests/ax-fleet/default.nix | 126 ++++++++++++ tests/ax-fleet/halogen_stub.py | 129 ++++++++++++ tests/ax-fleet/nodes.nix | 273 ++++++++++++++++++++++++++ tests/ax-fleet/phases/10-cluster.py | 240 ++++++++++++++++++++++ tests/ax-fleet/phases/20-substrate.py | 4 + tests/ax-fleet/phases/30-ax.py | 3 + tests/ax-fleet/phases/90-rollback.py | 52 +++++ 10 files changed, 1093 insertions(+), 2 deletions(-) create mode 100644 tests/ax-fleet-boot/default.nix create mode 100644 tests/ax-fleet-topology/default.nix create mode 100644 tests/ax-fleet/default.nix create mode 100644 tests/ax-fleet/halogen_stub.py create mode 100644 tests/ax-fleet/nodes.nix create mode 100644 tests/ax-fleet/phases/10-cluster.py create mode 100644 tests/ax-fleet/phases/20-substrate.py create mode 100644 tests/ax-fleet/phases/30-ax.py create mode 100644 tests/ax-fleet/phases/90-rollback.py diff --git a/flake.nix b/flake.nix index ff2f38e00..0d8f42681 100644 --- a/flake.nix +++ b/flake.nix @@ -706,8 +706,20 @@ ; live-iso = strixAi.live-iso; nas-installer-iso = nasInstaller.config.system.build.isoImage; + # The second half of the ax-fleet kill switch (pkgs/ax-fleet-teardown). + ax-fleet-teardown = pkgs.callPackage ./pkgs/ax-fleet-teardown { + k3s = inputs.nixpkgs.legacyPackages.${system}.k3s_1_36; + }; }; + # `sudo nix run ~/dotfiles#ax-fleet-teardown` after `myAxFleet.enable = + # false` and a switch: k3s-killall.sh, the guard chain, the sysctl restore. + apps.${system}.ax-fleet-teardown = { + type = "app"; + program = "${self.packages.${system}.ax-fleet-teardown}/bin/ax-fleet-teardown"; + meta.description = "Tear down ax-fleet's k3s leftovers and restore the pre-k3s sysctls"; + }; + # `nix build .#models.` retired with the 2026-08-21 "weights leave # nix" ruling: weights are no longer derivations, so there is nothing to # build — the NAS Library and library-fetch own materialization now. @@ -731,6 +743,22 @@ # The RAW out-of-store dotfiles are never checked at switch, so check them here. checks.${system} = { + # ax on the fleet (modules/ax-fleet, DESIGN.md 12). ax-fleet is the + # 4-VM switch/rollback proof, ax-fleet-boot the NAS-from-boot proof, + # ax-fleet-topology the evaluation-only assertions over the real hosts. + ax-fleet = import ./tests/ax-fleet { + inherit pkgs inputs; + inherit (nixpkgs) lib; + }; + ax-fleet-boot = import ./tests/ax-fleet-boot { + inherit pkgs inputs; + inherit (nixpkgs) lib; + }; + ax-fleet-topology = import ./tests/ax-fleet-topology { + inherit pkgs self; + inherit (nixpkgs) lib; + }; + qwen-speech = pkgs.runCommand "qwen-speech-tests" { @@ -769,12 +797,23 @@ ]; in assert builtins.all (host: (hostCfg host) ? myAxClient) gated; - assert builtins.all (host: (hostCfg host).myAxClient.enable == false) gated; + # ON on the coordinator only, as a mkDefault consequence of + # myAxFleet's harness role (modules/ax-fleet/default.nix); OFF on the + # worker (inference role) and the client (no fleet role). + assert (hostCfg "coordinator").myAxFleet.enable && (hostCfg "coordinator").myAxFleet.role == "harness"; + assert (hostCfg "coordinator").myAxClient.enable; + assert builtins.all (host: (hostCfg host).myAxClient.enable == false) [ + "worker" + "client" + ]; assert builtins.all ( host: !(builtins.elem pkgs.kubectl (hostCfg host).environment.systemPackages) && !(builtins.elem pkgs.ax (hostCfg host).environment.systemPackages) - ) gated; + ) [ + "worker" + "client" + ]; assert !((hostCfg "nas") ? myAxClient); pkgs.runCommand "ax-client-topology" { } '' touch "$out" diff --git a/tests/ax-fleet-boot/default.nix b/tests/ax-fleet-boot/default.nix new file mode 100644 index 000000000..01fa1ea9c --- /dev/null +++ b/tests/ax-fleet-boot/default.nix @@ -0,0 +1,45 @@ +{ + pkgs, + lib, + inputs, +}: +# checks.x86_64-linux.ax-fleet-boot (DESIGN.md 12.2): a NAS node with ax ON +# FROM BOOT. The 4-VM test proves the live-switch ordering; this proves the +# boot ordering: the bind mounts come up before k3s, the tmpfiles links land +# on the /mnt/fast side, the registry seed and the bootstrap succeed, and the +# node is Ready and untainted. The substrate track adds its Available checks +# here through the same bootstrap. +let + nodes = import ../ax-fleet/nodes.nix { inherit pkgs lib inputs; }; +in +pkgs.testers.runNixOSTest { + name = "ax-fleet-boot"; + node.specialArgs = { inherit inputs; }; + nodes.nas = { + imports = [ nodes.nas ]; + myAxFleet.enable = true; + myAxFleet.kubelet.systemReserved = "cpu=1,memory=1Gi"; + }; + globalTimeout = 3600; + testScript = '' + nas.start() + nas.wait_for_unit("k3s.service", timeout=900) + nas.wait_for_unit("ax-fleet-bootstrap.service", timeout=1800) + for p in ("/var/lib/rancher/k3s", "/var/lib/kubelet", "/var/log/pods"): + src = nas.succeed(f"findmnt -n -o SOURCE -T {p}").strip() + assert src.startswith("/dev/vdb"), f"{p} is on {src}, not /mnt/fast" + nas.succeed("ls /mnt/fast/k3s/rancher/k3s/agent/images/ | grep -q airgap") + nas.succeed("systemctl show ax-fleet-registry-seed.service -p Result --value | grep -x success") + nas.succeed("systemctl show ax-fleet-bootstrap.service -p Result --value | grep -x success") + nas.wait_until_succeeds( + "k3s kubectl get node nas -o jsonpath='{.status.conditions[?(@.type==\"Ready\")].status}' | grep -x True", + timeout=600, + ) + assert nas.succeed("k3s kubectl get node nas -o jsonpath='{.spec.taints}'").strip() == "" + nas.succeed("k3s kubectl -n kube-system wait --for=condition=Available deploy/coredns deploy/local-path-provisioner --timeout=600s") + nas.succeed("test -s /etc/ax-fleet/admin.kubeconfig") + # The snapshot was taken at first activation, before k3s ever ran. + nas.succeed("grep -q '^kernel.panic = ' /var/lib/ax-fleet/sysctl-before.conf") + nas.fail("findmnt -n -T /var/lib/rancher/k3s -o SOURCE | grep -q vda") + ''; +} diff --git a/tests/ax-fleet-topology/default.nix b/tests/ax-fleet-topology/default.nix new file mode 100644 index 000000000..4fddcf49e --- /dev/null +++ b/tests/ax-fleet-topology/default.nix @@ -0,0 +1,180 @@ +{ + pkgs, + lib, + self, +}: +# checks.x86_64-linux.ax-fleet-topology (DESIGN.md 12.3): evaluation-only +# assertions over the REAL host configurations. Every assert is eval-time and +# sits in front of the runCommand, so each runs under --no-build. +# +# Not asserted here, on purpose: "no 8731 on the coordinator's wlp192s0". That +# is #461's change; nas-topology already asserts exactly it and fails on +# origin/main until #461 is merged (inherited, not introduced). Duplicating it +# here would make this check red for a reason outside this PR. +let + hostCfg = host: self.nixosConfigurations.${host}.config; + offCfg = + host: + (self.nixosConfigurations.${host}.extendModules { + modules = [ { myAxFleet.enable = lib.mkForce false; } ]; + }).config; + + flagsOf = + cfg: + let + raw = cfg.systemd.services.k3s.serviceConfig.ExecStart; + s = if builtins.isList raw then lib.concatStringsSep " " raw else raw; + in + lib.filter (w: w != "" && w != "\\") (lib.splitString " " (lib.replaceStrings [ "\n" ] [ " " ] s)); + + has = cfg: flag: builtins.elem flag (flagsOf cfg); + hasPrefix = cfg: p: builtins.any (lib.hasPrefix p) (flagsOf cfg); + + nas = hostCfg "nas"; + coord = hostCfg "coordinator"; + worker = hostCfg "worker"; + client = hostCfg "client"; + ax = cfg: cfg.myAxFleet; + + axLines = text: lib.filter (l: lib.hasInfix "ax-fleet:" l) (lib.splitString "\n" text); + + # ── the kill switch: with enable = mkForce false nothing from this PR renders ── + killed = + host: + let + c = offCfg host; + units = lib.attrNames c.systemd.services ++ lib.attrNames c.systemd.sockets; + in + !c.services.k3s.enable + && !c.services.dockerRegistry.enable + && !(builtins.any (lib.hasPrefix "ax-fleet") units) + && !(builtins.any (lib.hasPrefix "ax-server-proxy") units) + && !(builtins.any (m: m.where == "/var/lib/rancher") c.systemd.mounts) + && !(c.environment.etc ? "NetworkManager/conf.d/90-ax-fleet.conf") + && !(c.environment.etc ? "rancher/k3s/registries.yaml") + && !(c.system.activationScripts ? ax-fleet-sysctl-snapshot) + && !(lib.hasInfix "ax-fleet-guard" c.networking.firewall.extraCommands) + && !(lib.hasInfix "ax-fleet:" (c.networking.firewall.extraInputRules or "")); + + # ── parity with the VM test: flags and firewall text, interfaces substituted ── + testNodes = self.checks.x86_64-linux.ax-fleet.nodes; + testOn = name: testNodes.${name}.specialisation.ax-on.configuration; + # Flags that legitimately differ: the token source and the VM's kubelet + # reservations (a VM has 8 GiB, the desk 128). + volatile = + f: + f == "--token-file" + || lib.hasSuffix "/k3s-token" f + || lib.hasSuffix "-ax-fleet-vm-token" f + || lib.hasInfix "reserved=" f + || lib.hasInfix "eviction-hard=" f; + normFlags = + subst: cfg: + lib.sort (a: b: a < b) ( + map (lib.replaceStrings (lib.attrNames subst) (lib.attrValues subst)) ( + lib.filter (f: !(volatile f)) (flagsOf cfg) + ) + ); + nasSubst = { + "enp1s0" = "eth1"; + }; + coordSubst = { + "wlp192s0" = "eth1"; + "tailscale0" = "eth2"; + }; + guardText = + subst: cfg: + map (lib.replaceStrings (lib.attrNames subst) (lib.attrValues subst)) ( + lib.filter (l: lib.hasInfix "ax-fleet-guard" l || lib.hasInfix "8472" l) ( + lib.splitString "\n" cfg.networking.firewall.extraCommands + ) + ); +in +# nas: the control node +assert (ax nas).enable && (ax nas).role == "control"; +assert has nas "--flannel-iface=enp1s0"; +assert has nas "--node-ip=10.42.0.1"; +assert has nas "--flannel-backend=vxlan"; +assert !(hasPrefix nas "--node-taint"); +assert has nas "--node-label=ate.dev/substrate-version=none"; +assert has nas "--node-label=ax.mecattaf.dev/role=control"; +assert has nas "--default-local-storage-path=/mnt/nas/services/ax-fleet/local-path"; +assert !(builtins.any (lib.hasInfix "local-storage") nas.services.k3s.disable); +assert has nas "--kube-apiserver-arg=runtime-config=certificates.k8s.io/v1beta1=true"; +assert has nas "--service-node-port-range=30000-30999"; +assert !(hasPrefix nas "--data-dir"); +assert !(hasPrefix nas "--cluster-init"); +assert nas.services.k3s.containerdConfigTemplate == null; +assert nas.services.k3s.package == (ax nas).k3sPackage; +assert (ax nas).k3sPackage.version == "1.36.2+k3s1"; +# every ax-fleet NAS rule is scoped to a source and to an interface, none to the tailnet +assert builtins.length (axLines nas.networking.firewall.extraInputRules) == 4; +assert builtins.all (l: lib.hasInfix "ip saddr" l && lib.hasInfix "iifname" l) ( + axLines nas.networking.firewall.extraInputRules +); +assert !(builtins.any (lib.hasInfix "tailscale0") (axLines nas.networking.firewall.extraInputRules)); +assert nas.services.dockerRegistry.listenAddress == "10.42.0.1"; +assert !nas.services.dockerRegistry.openFirewall; +assert lib.hasPrefix "/mnt/nas/" nas.services.dockerRegistry.storagePath; +assert builtins.all (m: lib.hasPrefix "/mnt/fast/" m.what) ( + lib.filter (m: m.where == "/var/lib/rancher" || m.where == "/var/lib/kubelet" || m.where == "/var/log/pods") nas.systemd.mounts +); +assert builtins.length (lib.filter (m: lib.hasPrefix "/mnt/fast/k3s" m.what) nas.systemd.mounts) == 3; +# the shared PostgreSQL is not touched: identical settings with the switch off +assert nas.services.postgresql.settings == (offCfg "nas").services.postgresql.settings; +assert nas.services.postgresql.authentication == (offCfg "nas").services.postgresql.authentication; + +# coordinator: the harness node +assert (ax coord).enable && (ax coord).role == "harness"; +assert has coord "--flannel-iface=wlp192s0"; +assert has coord "--node-ip=10.42.0.2"; +assert has coord "--server"; +assert has coord "https://10.42.0.1:6443"; +assert has coord "--node-taint=${(ax coord).harnessTaint}"; +assert (ax coord).harnessTaint == "ate.dev/sandboxClass=gvisor:NoSchedule"; +assert has coord "--node-label=ate.dev/substrate-version=${(ax coord).substrateVersion}"; +assert (ax coord).substrateVersion == (ax nas).substrateVersion; +assert coord.services.k3s.containerdConfigTemplate == null; +assert coord.services.k3s.package == nas.services.k3s.package; +assert lib.hasInfix "ax-fleet-guard" coord.networking.firewall.extraCommands; +assert lib.hasInfix "-s 10.42.0.1 -p udp --dport 8472" coord.networking.firewall.extraCommands; +assert !(coord.networking.firewall.interfaces ? cni0); +assert coord.environment.etc ? "NetworkManager/conf.d/90-ax-fleet.conf"; +# NO NetworkManager restart trigger: NetworkManager.conf renders byte-identical with the switch off +assert + coord.environment.etc."NetworkManager/NetworkManager.conf".source + == (offCfg "coordinator").environment.etc."NetworkManager/NetworkManager.conf".source; +assert coord.networking.networkmanager.unmanaged == (offCfg "coordinator").networking.networkmanager.unmanaged; +assert coord.boot.kernel.sysctl."net.ipv4.conf.default.proxy_arp" == 1; +assert !(coord.boot.kernel.sysctl ? "net.ipv4.conf.all.proxy_arp") || coord.boot.kernel.sysctl."net.ipv4.conf.all.proxy_arp" == null; +assert (coord.boot.kernel.sysctl."net.ipv6.conf.all.forwarding" or 0) == 0; +assert coord.services.tailscale.useRoutingFeatures == "none"; +assert coord.systemd.sockets.ax-server-proxy.listenStreams == [ "127.0.0.1:8080" ]; +assert coord.myAxClient.enable; + +# worker: inference, nothing at runtime +assert (ax worker).enable && (ax worker).role == "inference"; +assert !worker.services.k3s.enable; +assert builtins.elem 8731 worker.networking.firewall.interfaces.enp191s0.allowedTCPPorts; +# client: untouched +assert !(client ? myAxFleet); +assert !client.services.k3s.enable; + +# the kill switch +assert builtins.all killed [ + "nas" + "coordinator" + "worker" +]; + +# parity with the VM test +assert normFlags nasSubst nas == normFlags { } (testOn "nas"); +assert normFlags coordSubst coord == normFlags { } (testOn "coordinator"); +assert + map (lib.replaceStrings [ "enp1s0" ] [ "eth1" ]) (axLines nas.networking.firewall.extraInputRules) + == axLines (testOn "nas").networking.firewall.extraInputRules; +assert guardText coordSubst coord == guardText { } (testOn "coordinator"); + +pkgs.runCommand "ax-fleet-topology" { } '' + touch "$out" +'' diff --git a/tests/ax-fleet/default.nix b/tests/ax-fleet/default.nix new file mode 100644 index 000000000..b3e408a20 --- /dev/null +++ b/tests/ax-fleet/default.nix @@ -0,0 +1,126 @@ +{ + pkgs, + lib, + inputs, +}: +# checks.x86_64-linux.ax-fleet: the 4-VM proof before any switch (DESIGN.md +# 12.1). The script mirrors the real motion: baseline, switch the NAS, switch +# the coordinator (the worker is not switched), Tasks, resilience, rollback. +# +# The test script is phases/*.py concatenated in name order, after the +# prelude below: 10-cluster and 90-rollback (cluster track), 20-substrate +# (substrate track), 30-ax (ax track). Every subtest a phase runs through +# `step(...)` is named in $out/receipt.json with the values it recorded. +let + nodes = import ./nodes.nix { inherit pkgs lib inputs; }; + teardown = pkgs.callPackage ../../pkgs/ax-fleet-teardown { + k3s = inputs.nixpkgs.legacyPackages.x86_64-linux.k3s_1_36; + }; + phaseDir = ./phases; + phaseFiles = lib.sort (a: b: a < b) ( + lib.filter (n: lib.hasSuffix ".py" n) (lib.attrNames (builtins.readDir phaseDir)) + ); + prelude = '' + import json + import os + import time + from contextlib import contextmanager + + TEARDOWN = "${teardown}/bin/ax-fleet-teardown" + PROBE_IMAGE = "ax-fleet-probe:test" + PROBE_TARBALL = "${nodes.probeImage}" + LOCAL_PATH_ROOT = "/mnt/nas/services/ax-fleet/local-path" + SYSCTLS = [ + "net.ipv4.ip_forward", + "net.ipv6.conf.all.forwarding", + "kernel.panic", + "kernel.panic_on_oops", + "vm.overcommit_memory", + ] + + receipt = {"test": "ax-fleet", "subtests": [], "values": {}} + + + def save_receipt(): + out = os.environ.get("out", ".") + os.makedirs(out, exist_ok=True) + with open(os.path.join(out, "receipt.json"), "w") as f: + json.dump(receipt, f, indent=2, sort_keys=True) + + + def record(key, value): + receipt["values"][key] = value + save_receipt() + + + @contextmanager + def step(name): + t0 = time.monotonic() + with subtest(name): + yield + receipt["subtests"].append({"name": name, "result": "pass", "seconds": round(time.monotonic() - t0, 1)}) + save_receipt() + + + def kubectl(args): + """kubectl on the NAS, as root, with the k3s admin kubeconfig.""" + return nas.succeed(f"k3s kubectl {args}") + + + def jsonpath(obj, path): + return kubectl(f"get {obj} -o jsonpath='{path}'").strip() + + + def sysctls(machine): + return {k: machine.succeed(f"sysctl -n {k}").strip() for k in SYSCTLS} + + + def node_ready(name): + nas.wait_until_succeeds( + f"k3s kubectl get node {name} -o jsonpath='{{.status.conditions[?(@.type==\"Ready\")].status}}' | grep -x True", + timeout=600, + ) + + + def user_unit_pid(unit): + return coordinator.succeed( + "runuser -u alice -- env XDG_RUNTIME_DIR=/run/user/$(id -u alice) " + f"systemctl --user show {unit} -p MainPID --value" + ).strip() + + + def nm_invocation(): + return coordinator.succeed("systemctl show NetworkManager -p InvocationID --value").strip() + + + def unit_invocation(machine, unit): + return machine.succeed(f"systemctl show {unit} -p InvocationID --value").strip() + + ''; +in +pkgs.testers.runNixOSTest { + name = "ax-fleet"; + node.specialArgs = { inherit inputs; }; + nodes = { + inherit (nodes) + nas + coordinator + worker + peer + ; + }; + # Long waits are test parameters (k3s start, image import), not estimates. + globalTimeout = 3 * 3600; + testScript = + prelude + + lib.concatMapStrings (f: '' + + # ───────────── phases/${f} ───────────── + ${builtins.readFile (phaseDir + "/${f}")} + '') phaseFiles + + '' + + save_receipt() + ''; + passthru.axFleet = { inherit nodes teardown; }; +} diff --git a/tests/ax-fleet/halogen_stub.py b/tests/ax-fleet/halogen_stub.py new file mode 100644 index 000000000..ee14ea1a7 --- /dev/null +++ b/tests/ax-fleet/halogen_stub.py @@ -0,0 +1,129 @@ +#!/usr/bin/env python3 +"""Halogen stand-in for the ax-fleet VM test (DESIGN.md 12.1, node `worker`). + +OpenAI-compatible enough for the ax task image's two model paths: + - `halogen-smoke`: one non-streaming POST /v1/chat/completions (curl); + - `pi`: streaming POST /v1/chat/completions (SSE, `data: {...}` chunks, then + `data: [DONE]`). +Also GET /v1/models and GET /health, like the real server. + +Every request is appended to --log as one JSON line with the source address +and a per-request id, so the test can prove that sandbox traffic arrived +SNAT'd from the NAS (10.42.0.1) and count requests per Task. +""" + +import argparse +import json +import threading +import time +import uuid +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +MODEL = "halogen-qwen3.8-flash-next" +REPLY = "halogen-stub-ok" +LOCK = threading.Lock() + + +def make_handler(log_path): + class Handler(BaseHTTPRequestHandler): + protocol_version = "HTTP/1.1" + + def log_message(self, fmt, *args): # quiet stderr; the jsonl is the log + pass + + def _record(self, body): + entry = { + "id": uuid.uuid4().hex, + "ts": time.time(), + "src": self.client_address[0], + "method": self.command, + "path": self.path, + "bytes": len(body), + } + with LOCK, open(log_path, "a", encoding="utf-8") as f: + f.write(json.dumps(entry) + "\n") + return entry["id"] + + def _json(self, code, obj): + data = json.dumps(obj).encode() + self.send_response(code) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(data))) + self.end_headers() + self.wfile.write(data) + + def do_GET(self): + rid = self._record(b"") + if self.path.rstrip("/") == "/health": + self._json(200, {"status": "ok", "in_flight": 0, "queued": 0, "request_id": rid}) + elif self.path.rstrip("/") == "/v1/models": + self._json(200, {"object": "list", "data": [{"id": MODEL, "object": "model", "owned_by": "halogen"}]}) + else: + self._json(404, {"error": "not found", "request_id": rid}) + + def do_POST(self): + length = int(self.headers.get("Content-Length") or 0) + body = self.rfile.read(length) if length else b"" + rid = self._record(body) + if self.path.rstrip("/") != "/v1/chat/completions": + self._json(404, {"error": "not found", "request_id": rid}) + return + try: + req = json.loads(body or b"{}") + except json.JSONDecodeError: + self._json(400, {"error": "bad json", "request_id": rid}) + return + created = int(time.time()) + cid = "chatcmpl-" + rid + if not req.get("stream"): + self._json( + 200, + { + "id": cid, + "object": "chat.completion", + "created": created, + "model": MODEL, + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": REPLY}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + }, + ) + return + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Cache-Control", "no-cache") + self.send_header("Connection", "close") + self.end_headers() + chunks = [ + {"role": "assistant", "content": ""}, + {"content": REPLY}, + ] + for i, delta in enumerate(chunks + [None]): + choice = {"index": 0, "delta": delta or {}, "finish_reason": None if delta else "stop"} + obj = {"id": cid, "object": "chat.completion.chunk", "created": created, "model": MODEL, "choices": [choice]} + if delta is None: + obj["usage"] = {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2} + self.wfile.write(b"data: " + json.dumps(obj).encode() + b"\n\n") + self.wfile.flush() + self.wfile.write(b"data: [DONE]\n\n") + self.wfile.flush() + self.close_connection = True + + return Handler + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--port", type=int, default=8731) + ap.add_argument("--log", default="/tmp/halogen-stub.jsonl") + args = ap.parse_args() + ThreadingHTTPServer(("0.0.0.0", args.port), make_handler(args.log)).serve_forever() + + +if __name__ == "__main__": + main() diff --git a/tests/ax-fleet/nodes.nix b/tests/ax-fleet/nodes.nix new file mode 100644 index 000000000..b9454a17b --- /dev/null +++ b/tests/ax-fleet/nodes.nix @@ -0,0 +1,273 @@ +{ + pkgs, + lib, + inputs, +}: +# The four VMs of checks.x86_64-linux.ax-fleet (DESIGN.md 12.1), shared with +# checks.x86_64-linux.ax-fleet-boot. They carry the fleet's real hostnames and +# LAN addresses, so every manifest and firewall rule the modules render is the +# production one; only interface names (eth1, eth2), the token and the kubelet +# reservations differ, and ax-fleet-topology pins that parity. +# +# Each base config is "today", with ax OFF. `specialisation.ax-on` sets +# myAxFleet.enable = true, and the test script switches to it live, NAS first, +# exactly as Tom will. +let + # A plain file, test-only, not a secret: the fleet reads agenix instead. + token = pkgs.writeText "ax-fleet-vm-token" "ax-fleet-vm-test-token-0123456789abcdef"; + + # busybox (httpd, nslookup) plus curl, imported by k3s from the images + # directory on both nodes, so probe pods need no registry and no network. + probeImage = pkgs.dockerTools.buildImage { + name = "ax-fleet-probe"; + tag = "test"; + copyToRoot = pkgs.buildEnv { + name = "ax-fleet-probe-root"; + paths = [ + pkgs.busybox + pkgs.curl + pkgs.cacert + ]; + pathsToLink = [ + "/bin" + "/etc" + ]; + }; + config.Cmd = [ + "/bin/sh" + "-c" + "sleep 1000000" + ]; + }; + + lanAddr = address: { + interface = "eth1"; + inherit address; + }; + + setAddr = iface: address: prefixLength: { + networking.interfaces.${iface}.ipv4.addresses = lib.mkForce [ { inherit address prefixLength; } ]; + }; + + # Everything that makes a node a fleet node in the test. + fleetNode = + { role, address }: + { + imports = [ + ../../modules/ax-fleet + inputs.agenix.nixosModules.default + ]; + system.switch.enable = true; + myAxFleet = { + inherit role; + lan = lanAddr address; + k3sTokenFile = "${token}"; + guardInterfaces = [ "eth2" ]; + }; + environment.systemPackages = [ + pkgs.jq + pkgs.curl + pkgs.iptables + pkgs.nftables + pkgs.iproute2 + pkgs.dnsutils + ]; + }; + + axOn = extra: { + specialisation.ax-on.configuration = lib.mkMerge [ + { + myAxFleet.enable = true; + services.k3s.images = [ probeImage ]; + } + extra + ]; + }; + + vmReservations = { + systemReserved = "cpu=500m,memory=512Mi"; + kubeReserved = "cpu=250m,memory=256Mi"; + evictionHard = "memory.available<256Mi"; + }; +in +{ + inherit probeImage token; + + nas = + { ... }: + { + imports = [ + (fleetNode { + role = "control"; + address = "10.42.0.1"; + }) + (axOn { + myAxFleet.kubelet = { + systemReserved = "cpu=1,memory=1Gi"; + }; + }) + (setAddr "eth1" "10.42.0.1" 24) + ]; + networking.hostName = "nas"; + virtualisation = { + vlans = [ 1 ]; + memorySize = 10240; + cores = 6; + diskSize = 8192; + emptyDiskImages = [ + 8192 + 16384 + ]; + fileSystems = { + "/mnt/fast" = { + device = "/dev/vdb"; + fsType = "ext4"; + autoFormat = true; + }; + "/mnt/nas" = { + device = "/dev/vdc"; + fsType = "btrfs"; + autoFormat = true; + }; + }; + }; + boot.supportedFilesystems = [ "btrfs" ]; + + # The NAS firewall as the real one is shaped: nftables, interface-scoped + # extraInputRules, filterForward off, strict rpfilter, and a reload that + # keeps other tables (flushRuleset = false). + networking.nftables.enable = true; + networking.nftables.flushRuleset = false; + networking.firewall = { + enable = true; + filterForward = false; + checkReversePath = "strict"; + extraInputRules = '' + iifname "eth1" udp dport 53 accept comment "stand-in AdGuard" + iifname "eth1" tcp dport 53 accept comment "stand-in AdGuard" + ''; + }; + + # AdGuard stand-in: answers one record nobody else serves. + services.dnsmasq = { + enable = true; + resolveLocalQueries = false; + settings = { + listen-address = [ "10.42.0.1" ]; + bind-interfaces = true; + no-resolv = true; + address = [ "/only-nas.test/10.42.0.77" ]; + }; + }; + + # The bystander: the NAS's shared PostgreSQL (Paperless, Immich) must + # not restart and must not change. + services.postgresql = { + enable = true; + ensureDatabases = [ "paperless" ]; + }; + }; + + coordinator = + { ... }: + { + imports = [ + (fleetNode { + role = "harness"; + address = "10.42.0.2"; + }) + (axOn { myAxFleet.kubelet = vmReservations; }) + (setAddr "eth1" "10.42.0.2" 24) + (setAddr "eth2" "100.105.121.73" 10) + ]; + networking.hostName = "coordinator"; + virtualisation = { + vlans = [ + 1 + 2 + ]; + memorySize = 8192; + cores = 4; + diskSize = 8192; + podman.enable = true; + }; + # The desk's shape: iptables firewall, strict rpfilter, NetworkManager + # running (its InvocationID must not change at switch), zram swap. + networking.nftables.enable = false; + networking.firewall = { + enable = true; + checkReversePath = "strict"; + allowedTCPPorts = [ 80 ]; + }; + networking.networkmanager.enable = true; + # eth1/eth2 are the test driver's static addresses; identical in base + # and specialisation, so NetworkManager.conf does not change at switch. + networking.networkmanager.unmanaged = [ + "eth1" + "eth2" + ]; + zramSwap.enable = true; + + services.caddy = { + enable = true; + virtualHosts.":80".extraConfig = '' + respond "caddy-ok" + ''; + }; + + # herdr stand-in: a lingering user's long-lived user unit with the same + # X-SwitchMethod herdr has. Its MainPID must survive switch and rollback. + users.users.alice = { + isNormalUser = true; + linger = true; + }; + systemd.user.services.herdr-standin = { + wantedBy = [ "default.target" ]; + unitConfig."X-SwitchMethod" = "keep-old"; + serviceConfig.ExecStart = "${pkgs.coreutils}/bin/sleep 1000000"; + }; + }; + + worker = + { ... }: + { + imports = [ + (fleetNode { + role = "inference"; + address = "10.42.0.5"; + }) + (setAddr "eth1" "10.42.0.5" 24) + ]; + networking.hostName = "worker"; + virtualisation.vlans = [ 1 ]; + # The worker is not switched in the motion; its fleet role is ON from + # boot and renders only the 8731 assertion, as on the real host. + myAxFleet.enable = true; + networking.firewall = { + enable = true; + interfaces.eth1.allowedTCPPorts = [ 8731 ]; + }; + systemd.services.halogen-stub = { + wantedBy = [ "multi-user.target" ]; + serviceConfig = { + ExecStart = "${pkgs.python3}/bin/python3 ${./halogen_stub.py} --port 8731 --log /var/lib/halogen-stub/requests.jsonl"; + StateDirectory = "halogen-stub"; + }; + }; + }; + + peer = + { ... }: + { + imports = [ (setAddr "eth1" "100.64.0.9" 10) ]; + networking.hostName = "peer"; + virtualisation.vlans = [ 2 ]; + networking.firewall.enable = false; + environment.systemPackages = [ pkgs.curl ]; + # Something to reach: the guard must stop pods from getting here. + systemd.services.peer-http = { + wantedBy = [ "multi-user.target" ]; + serviceConfig.ExecStart = "${pkgs.busybox}/bin/httpd -f -p 8000 -h /etc"; + }; + }; +} diff --git a/tests/ax-fleet/phases/10-cluster.py b/tests/ax-fleet/phases/10-cluster.py new file mode 100644 index 000000000..b8dcfaf84 --- /dev/null +++ b/tests/ax-fleet/phases/10-cluster.py @@ -0,0 +1,240 @@ +# Phase 1 (baseline), 2 and 3 (switch nas, then coordinator): the cluster +# assertions. Track cluster. DESIGN.md 12.1 and 6.6. The substrate and ax +# phases (20, 30) run after this one, on the cluster it leaves behind. + +AX_ON = "/run/current-system/specialisation/ax-on/bin/switch-to-configuration test" + +PROBE_POD = """ +apiVersion: v1 +kind: Pod +metadata: + name: {name} + namespace: default + labels: {{app: ax-fleet-probe}} +spec: + nodeSelector: {{ax.mecattaf.dev/role: {role}}} + tolerations: + - {{key: ate.dev/sandboxClass, operator: Exists, effect: NoSchedule}} + containers: + - name: probe + image: ax-fleet-probe:test + imagePullPolicy: Never + command: ["/bin/sh", "-c", "echo probe-started-{name}; mkdir -p /www; echo pod-ok > /www/index.html; exec httpd -f -p 8000 -h /www"] + ports: + - {{containerPort: 8000, hostPort: {host_port}}} + {volume_mounts} + {volumes} +""" + + +def apply_probe(name, role, host_port, pvc=None): + vm = "volumeMounts: [{name: data, mountPath: /data}]" if pvc else "" + vols = f"volumes: [{{name: data, persistentVolumeClaim: {{claimName: {pvc}}}}}]" if pvc else "" + doc = PROBE_POD.format(name=name, role=role, host_port=host_port, volume_mounts=vm, volumes=vols) + nas.succeed(f"cat > /tmp/{name}.yaml <<'EOF'\n{doc}\nEOF") + kubectl(f"apply -f /tmp/{name}.yaml") + kubectl(f"wait --for=condition=Ready pod/{name} --timeout=300s") + + +with step("baseline"): + start_all() + for m in (nas, coordinator, worker, peer): + m.wait_for_unit("multi-user.target") + coordinator.wait_for_unit("caddy.service") + coordinator.wait_for_unit("NetworkManager.service") + nas.wait_for_unit("postgresql.service") + nas.wait_for_unit("dnsmasq.service") + worker.wait_for_unit("halogen-stub.service") + worker.wait_for_open_port(8731) + coordinator.wait_until_succeeds( + "runuser -u alice -- env XDG_RUNTIME_DIR=/run/user/$(id -u alice) systemctl --user is-active herdr-standin" + ) + worker.succeed("curl -sf --max-time 10 http://10.42.0.2/ | grep -x caddy-ok") + peer.succeed("curl -sf --max-time 10 http://100.105.121.73/ | grep -x caddy-ok") + # Not a k3s node, never switched: nothing from the fleet module runs here. + worker.fail("systemctl cat k3s.service") + nas.fail("systemctl cat k3s.service") + coordinator.fail("systemctl cat k3s.service") + + base = { + "herdr_pid": user_unit_pid("herdr-standin"), + "nm_invocation": nm_invocation(), + "postgres_invocation": unit_invocation(nas, "postgresql.service"), + "nas_databases": sorted(nas.succeed("runuser -u postgres -- psql -Atc 'select datname from pg_database'").split()), + "nas_nft": nas.succeed("nft -s list ruleset"), + "nas_nixos_fw": nas.succeed("nft -s list table inet nixos-fw"), + "sysctl_nas": sysctls(nas), + "sysctl_coordinator": sysctls(coordinator), + } + record("baseline", {k: v for k, v in base.items() if k not in ("nas_nft", "nas_nixos_fw")}) + + +with step("switch nas"): + nas.succeed(f"{AX_ON} >&2") + nas.succeed("systemctl is-active k3s.service docker-registry.service") + for u in ("ax-fleet-registry-seed.service", "ax-fleet-bootstrap.service"): + nas.succeed(f"systemctl is-active {u}") + nas.succeed(f"systemctl show {u} -p Result --value | grep -x success") + + +with step("nas: k3s state on the fast tier, links on that disk"): + src = nas.succeed("findmnt -n -o SOURCE -T /var/lib/rancher/k3s").strip() + record("nas_rancher_source", src) + assert src.startswith("/dev/vdb"), f"/var/lib/rancher/k3s is on {src}, not the /mnt/fast disk" + for p in ("/var/lib/kubelet", "/var/log/pods"): + s = nas.succeed(f"findmnt -n -o SOURCE -T {p}").strip() + assert s.startswith("/dev/vdb"), f"{p} is on {s}" + # The airgap images link lives under the bind mount, i.e. on /mnt/fast. + nas.succeed("ls -l /var/lib/rancher/k3s/agent/images/ | grep -q airgap") + nas.succeed("ls -l /mnt/fast/k3s/rancher/k3s/agent/images/ | grep -q airgap") + nas.succeed("test -s /etc/ax-fleet/admin.kubeconfig") + nas.succeed("stat -c '%a %U %G' /etc/ax-fleet/admin.kubeconfig | grep -x '640 root wheel'") + nas.succeed("grep -q 'server: https://10.42.0.1:6443' /etc/ax-fleet/admin.kubeconfig") + nas.succeed("test -s /var/lib/ax-fleet/sysctl-before.conf") + + +with step("nas: node Ready, untainted, control labels"): + node_ready("nas") + assert jsonpath("node nas", "{.spec.taints}") == "", "the NAS must be untainted" + labels = json.loads(kubectl("get node nas -o jsonpath='{.metadata.labels}'")) + assert labels.get("ax.mecattaf.dev/role") == "control", labels + assert labels.get("ate.dev/substrate-version") == "none", labels + ip = jsonpath("node nas", "{.status.addresses[?(@.type==\"InternalIP\")].address}") + assert ip == "10.42.0.1", ip + kubectl("-n kube-system wait --for=condition=Available deploy/coredns deploy/local-path-provisioner --timeout=600s") + + +with step("nas: certificates.k8s.io/v1beta1 serves the Substrate resources"): + res = json.loads(kubectl("get --raw /apis/certificates.k8s.io/v1beta1")) + names = sorted(r["name"] for r in res["resources"]) + record("certificates_v1beta1", names) + assert "clustertrustbundles" in names and "podcertificaterequests" in names, names + + +with step("nas: PersistentVolumes land on the data pool"): + nas.succeed( + "cat > /tmp/pvc.yaml <<'EOF'\n" + "apiVersion: v1\nkind: PersistentVolumeClaim\nmetadata: {name: ax-fleet-probe-pvc, namespace: default}\n" + "spec: {accessModes: [ReadWriteOnce], storageClassName: local-path, resources: {requests: {storage: 16Mi}}}\n" + "EOF" + ) + kubectl("apply -f /tmp/pvc.yaml") + apply_probe("probe-nas", "control", 18086, pvc="ax-fleet-probe-pvc") + kubectl("exec probe-nas -- sh -c 'echo data-pool > /data/marker'") + paths = kubectl( + "get pv -o jsonpath='{range .items[*]}{.spec.hostPath.path}{.spec.local.path}{\"\\n\"}{end}'" + ).split() + record("pv_paths", paths) + assert paths and all(p.startswith(LOCAL_PATH_ROOT) for p in paths), paths + nas.succeed(f"grep -rqx data-pool {LOCAL_PATH_ROOT}") + src = nas.succeed(f"findmnt -n -o SOURCE -T {LOCAL_PATH_ROOT}").strip() + assert src.startswith("/dev/vdc"), f"local-path root is on {src}, not the data pool" + + +with step("nas: bystanders untouched"): + assert unit_invocation(nas, "postgresql.service") == base["postgres_invocation"], "the shared PostgreSQL restarted" + dbs = sorted(nas.succeed("runuser -u postgres -- psql -Atc 'select datname from pg_database'").split()) + assert dbs == base["nas_databases"], (dbs, base["nas_databases"]) + nas.succeed("dig +short @10.42.0.1 only-nas.test | grep -x 10.42.0.77") + ruleset = nas.succeed("nft list ruleset") + record("nas_kube_services_in_nft", "KUBE-SERVICES" in ruleset) + record("sysctl_nas_after_switch", sysctls(nas)) + + +with step("switch coordinator"): + coordinator.succeed(f"{AX_ON} >&2") + coordinator.succeed("systemctl is-active k3s.service ax-fleet-nm-unmanaged.service ax-server-proxy.socket") + node_ready("coordinator") + + +with step("coordinator: Ready with the harness taint and labels"): + taints = json.loads(kubectl("get node coordinator -o jsonpath='{.spec.taints}'")) + assert {"key": "ate.dev/sandboxClass", "value": "gvisor", "effect": "NoSchedule"} in taints, taints + labels = json.loads(kubectl("get node coordinator -o jsonpath='{.metadata.labels}'")) + assert labels.get("ax.mecattaf.dev/role") == "harness", labels + assert labels.get("ate.dev/substrate-version") == "d277088b", labels + ip = jsonpath("node coordinator", "{.status.addresses[?(@.type==\"InternalIP\")].address}") + assert ip == "10.42.0.2", ip + coordinator.succeed("ip -d link show flannel.1 | grep -q 'dev eth1'") + record("sysctl_coordinator_after_switch", sysctls(coordinator)) + + +with step("coordinator: only tolerating pods land here; control stays on the NAS"): + apply_probe("probe-coord", "harness", 18085) + placed = kubectl( + "get pods -A -o jsonpath='{range .items[*]}{.metadata.namespace}/{.metadata.name} {.spec.nodeName}{\"\\n\"}{end}'" + ) + record("pod_placement", placed.strip().splitlines()) + for line in placed.strip().splitlines(): + name, node = line.split() + if node == "coordinator": + assert ( + name == "default/probe-coord" or "/atelet" in name or "/ateom-gvisor" in name + ), f"{name} landed on the coordinator" + # kubectl logs of a coordinator pod rides the agent tunnel (egress-selector agent). + kubectl("logs probe-coord | grep -q probe-started-probe-coord") + + +with step("coordinator: nothing changed for Tom"): + assert user_unit_pid("herdr-standin") == base["herdr_pid"], "herdr stand-in restarted" + assert nm_invocation() == base["nm_invocation"], "NetworkManager restarted" + coordinator.wait_until_succeeds("nmcli -t -f DEVICE,STATE d | grep -x 'cni0:unmanaged'", timeout=60) + coordinator.succeed("nmcli -t -f DEVICE,STATE d | grep -x 'flannel.1:unmanaged'") + record("nmcli_devices", coordinator.succeed("nmcli -t -f DEVICE,STATE d").strip().splitlines()) + worker.succeed("curl -sf --max-time 10 http://10.42.0.2/ | grep -x caddy-ok") + peer.succeed("curl -sf --max-time 10 http://100.105.121.73/ | grep -x caddy-ok") + # A podman container on the default bridge still reaches the host's Caddy + # (br_netfilter is loaded now, rpfilter is strict). + loaded = coordinator.succeed(f"podman load -i {PROBE_TARBALL}") + ref = [l.split("Loaded image:")[1].strip() for l in loaded.splitlines() if "Loaded image" in l][0] + coordinator.succeed(f"podman run --rm --network bridge {ref} curl -sf --max-time 10 http://10.88.0.1/ | grep -x caddy-ok") + coordinator.succeed("lsmod | grep -q br_netfilter") + + +with step("coordinator: the guards hold"): + pod_ip = jsonpath("pod probe-coord", "{.status.podIP}") + record("probe_coord_ip", pod_ip) + # Make the negative tests discriminating: the peer routes the LAN and the + # pod network through the coordinator, and the worker routes the tailnet + # back through it. Without the guard chain, ip_forward=1 would carry these. + peer.succeed("ip route replace 10.42.0.0/24 via 100.105.121.73") + peer.succeed("ip route replace 10.200.0.0/16 via 100.105.121.73") + worker.succeed("ip route replace 100.64.0.0/10 via 10.42.0.2") + peer.fail("curl -sf --max-time 5 http://10.42.0.5:8731/health") + peer.fail(f"curl -sf --max-time 5 http://{pod_ip}:8000/") + for port in (8085, 9090, 10250, 6443, 18085): + peer.fail(f"curl -sk --max-time 5 -o /dev/null https://100.105.121.73:{port}/") + peer.fail(f"curl -s --max-time 5 -o /dev/null http://100.105.121.73:{port}/") + # hostPorts: from the NAS yes (atelet's peers), from the worker no. + nas.succeed("curl -sf --max-time 10 http://10.42.0.2:18085/ | grep -x pod-ok") + worker.fail("curl -sf --max-time 5 http://10.42.0.2:18085/") + # Pods never reach the tailnet; the coordinator host itself still does. + coordinator.succeed("curl -sf --max-time 5 http://100.64.0.9:8000/hostname") + kubectl("exec probe-coord -- sh -c '! curl -sf --max-time 5 http://100.64.0.9:8000/hostname'") + # Pods reach the apiserver Service (any HTTP answer is reachability). + code = kubectl("exec probe-coord -- curl -sk -o /dev/null -w '%{http_code}' --max-time 10 https://10.201.0.1/readyz").strip() + assert code != "000", "a coordinator pod cannot reach the apiserver Service" + # Pods never reach the LAN except the NAS's apiserver and registry. + kubectl("exec probe-coord -- sh -c '! curl -sf --max-time 5 http://10.42.0.5:8731/health'") + worker.succeed("ip route del 100.64.0.0/10 via 10.42.0.2") + assert coordinator.succeed("sysctl -n net.ipv6.conf.all.forwarding").strip() == "0" + coordinator.succeed("iptables -t mangle -S nixos-fw-rpfilter | grep -q rpfilter") + coordinator.succeed("iptables -t mangle -S FORWARD 1 | grep -q ax-fleet-guard") + coordinator.fail("ip -br link | grep -qi cilium") + record("guard_chain", coordinator.succeed("iptables -t mangle -S ax-fleet-guard").strip().splitlines()) + + +with step("cluster plumbing: DNS through the NAS stand-in"): + kubectl("exec probe-coord -- nslookup only-nas.test | grep -q 10.42.0.77") + kubectl("exec probe-nas -- nslookup only-nas.test | grep -q 10.42.0.77") + + +with step("resilience: coordinator link flap"): + uid = jsonpath("pod probe-coord", "{.metadata.uid}") + coordinator.succeed("ip link set eth1 down") + time.sleep(20) + coordinator.succeed("ip link set eth1 up") + node_ready("coordinator") + kubectl("wait --for=condition=Ready pod/probe-coord --timeout=300s") + assert jsonpath("pod probe-coord", "{.metadata.uid}") == uid, "the probe pod was replaced by the flap" + nas.wait_until_succeeds("curl -sf --max-time 5 http://10.42.0.2:18085/ | grep -x pod-ok", timeout=120) diff --git a/tests/ax-fleet/phases/20-substrate.py b/tests/ax-fleet/phases/20-substrate.py new file mode 100644 index 000000000..06783242b --- /dev/null +++ b/tests/ax-fleet/phases/20-substrate.py @@ -0,0 +1,4 @@ +# Track substrate fills this (DESIGN.md 12.1 phases 2 and 3, Substrate +# subtests): every Substrate workload Available, atelet Ready on the +# coordinator only, the WorkerPool Ready, gVisor through the RustFS fallback. +# Left empty by the cluster track on purpose; it runs after 10-cluster. diff --git a/tests/ax-fleet/phases/30-ax.py b/tests/ax-fleet/phases/30-ax.py new file mode 100644 index 000000000..3708733f8 --- /dev/null +++ b/tests/ax-fleet/phases/30-ax.py @@ -0,0 +1,3 @@ +# Track ax fills this (DESIGN.md 12.1 phases 4 and 5): T1 to T5 through +# ax-fleet-smoke, and the ax resilience checks. Left empty by the cluster +# track on purpose; it runs after 20-substrate. diff --git a/tests/ax-fleet/phases/90-rollback.py b/tests/ax-fleet/phases/90-rollback.py new file mode 100644 index 000000000..987ad662e --- /dev/null +++ b/tests/ax-fleet/phases/90-rollback.py @@ -0,0 +1,52 @@ +# Phase 6: rollback, the kill switch proven (DESIGN.md 12.1, 13). Track +# cluster. Reverse order: coordinator, then nas. Each host goes back to its +# base toplevel (ax off) and then ax-fleet-teardown runs, as Tom would. + +BASE = "/run/booted-system/bin/switch-to-configuration test" +LEFTOVER_RULES = "iptables-save 2>/dev/null | grep -E 'KUBE-|FLANNEL|CNI-'" + + +with step("rollback coordinator"): + coordinator.succeed(f"{BASE} >&2") + coordinator.fail("systemctl is-active k3s.service") + coordinator.succeed(f"{TEARDOWN} >&2") + coordinator.fail("ip link show cni0") + coordinator.fail("ip link show flannel.1") + coordinator.fail(LEFTOVER_RULES) + coordinator.fail("pgrep -f containerd-shim") + coordinator.fail("iptables -t mangle -S ax-fleet-guard") + after = sysctls(coordinator) + record("sysctl_coordinator_after_rollback", after) + assert after == base["sysctl_coordinator"], (after, base["sysctl_coordinator"]) + assert user_unit_pid("herdr-standin") == base["herdr_pid"], "herdr stand-in restarted" + assert nm_invocation() == base["nm_invocation"], "NetworkManager restarted" + coordinator.fail("test -e /etc/NetworkManager/conf.d/90-ax-fleet.conf") + worker.succeed("curl -sf --max-time 10 http://10.42.0.2/ | grep -x caddy-ok") + peer.succeed("curl -sf --max-time 10 http://100.105.121.73/ | grep -x caddy-ok") + + +with step("rollback nas"): + nas.succeed(f"{BASE} >&2") + nas.fail("systemctl is-active k3s.service") + nas.fail("systemctl is-active docker-registry.service") + nas.succeed(f"{TEARDOWN} >&2") + nas.fail("ip link show cni0") + nas.fail("ip link show flannel.1") + nas.fail("pgrep -f containerd-shim") + ruleset = nas.succeed("nft -s list ruleset") + for marker in ("KUBE-", "FLANNEL", "CNI-", "ax-fleet"): + assert marker not in ruleset, f"{marker} left in the NAS ruleset" + nixos_fw = nas.succeed("nft -s list table inet nixos-fw") + assert nixos_fw == base["nas_nixos_fw"], "the NAS firewall table differs from the baseline" + record("nas_ruleset_equal_baseline", ruleset == base["nas_nft"]) + if ruleset != base["nas_nft"]: + record("nas_ruleset_after_rollback", ruleset) + after = sysctls(nas) + record("sysctl_nas_after_rollback", after) + assert after == base["sysctl_nas"], (after, base["sysctl_nas"]) + assert unit_invocation(nas, "postgresql.service") == base["postgres_invocation"], "the shared PostgreSQL restarted" + dbs = sorted(nas.succeed("runuser -u postgres -- psql -Atc 'select datname from pg_database'").split()) + assert dbs == base["nas_databases"], dbs + # Left on disk on purpose; deleting it is Tom's call. + nas.succeed("test -d /mnt/fast/k3s/rancher/k3s") + nas.fail("findmnt /var/lib/rancher") From 71ce913b5d112a8ff873fbd3f2e68a7604e5f6f7 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 08:46:02 +0200 Subject: [PATCH 15/37] ax-fleet: the ax module, ax-fleet-smoke, and the Task phase of the VM test modules/ax-fleet/ax.nix writes only the myAxFleet extension points and the coordinator's ax scripts: - manifests.ax-fleet-40-ax: Namespace ax-system; ax-redis (AOF on a local-path PVC, no password, nothing on argv); ax-server on ClusterIP 10.201.0.80, never a NodePort; ax-controller with upstream's projected token and ClusterTrustBundle, --running-resync (P1), ATENET_ROUTER_ADDR and AX_SNAPSHOTS_BUCKET=gs://ate-snapshots/ax/. Every image by digest, the digests substituted at build time and checked (no IFD). - registrySeed: ax-server, ax-controller, ax-redis, ax-agent. Images come from the flake's own nixpkgs and `ax`, never the host's pkgs, so the NAS seeds the digest the coordinator names. - harness: ax-fleet-image-ref, ax-fleet-smoke (halogen, pi, exit N, egress-deny, floor N; JSON receipts) and AX_SERVER=http://127.0.0.1:8080. - options myAxFleet.ax.{runningResyncSeconds,atespace}. tests/ax-fleet/phases/30-ax.py: T1 to T5 and the resilience checks. Co-Authored-By: Claude Opus 5.5 --- modules/ax-fleet/ax-fleet-smoke.sh | 151 +++++++++ modules/ax-fleet/ax.nix | 521 +++++++++++++++++++++++++++++ tests/ax-fleet/phases/30-ax.py | 118 +++++++ 3 files changed, 790 insertions(+) create mode 100644 modules/ax-fleet/ax-fleet-smoke.sh create mode 100644 modules/ax-fleet/ax.nix create mode 100644 tests/ax-fleet/phases/30-ax.py diff --git a/modules/ax-fleet/ax-fleet-smoke.sh b/modules/ax-fleet/ax-fleet-smoke.sh new file mode 100644 index 000000000..c7ff574bf --- /dev/null +++ b/modules/ax-fleet/ax-fleet-smoke.sh @@ -0,0 +1,151 @@ +# ax-fleet-smoke: run one proof case as a real ax Task and print a JSON receipt. +# (Wrapped by writeShellApplication in ./ax.nix: strict mode, pinned PATH, and +# AX_FLEET_{ATESPACE,HALOGEN,HALOGEN_CIDR,RESYNC_SECONDS} from the module.) +# +# ax-fleet-smoke halogen [--hold N] one chat completion against Halogen +# ax-fleet-smoke pi [--hold N] pi against Halogen, schema-valid result +# ax-fleet-smoke exit N [--hold N] the command exits N (Failed ExitCode=N) +# ax-fleet-smoke egress-deny [URL] GET a non-allowlisted URL (default the +# coordinator's LAN address); must fail +# ax-fleet-smoke floor N N Tasks in a row; none ResourceExhausted +# +# Options: --hold N (re-read the phase after N seconds), --timeout N (per Task, +# default 900), --keep (do not delete the Tasks). Exit 0 only when the case +# passes. The receipt carries no secret: Tasks carry none (DESIGN.md 11). + +export AX_SERVER="${AX_SERVER:-http://127.0.0.1:8080}" +ns="$AX_FLEET_ATESPACE" +hold=0 +timeout=900 +keep=0 +args=() +while [ $# -gt 0 ]; do + case "$1" in + --hold) hold="$2"; shift 2 ;; + --timeout) timeout="$2"; shift 2 ;; + --keep) keep=1; shift ;; + *) args+=("$1"); shift ;; + esac +done +[ "${#args[@]}" -ge 1 ] || { sed -n '2,14p' "$0" >&2; exit 64; } +case_="${args[0]}" +image="$(ax-fleet-image-ref)" +run_id="$(date +%s)-$$" + +ensure_gateway() { + ax -a "$ns" apply -f - >/dev/null </dev/null | yq -o json '.' 2>/dev/null || echo '{}' +} + +# run_task NAME CMD...: apply, wait for a terminal phase, hold, print a receipt. +run_task() { + local name="$1" + shift + local cmd_json + cmd_json="$(jq -cn '$ARGS.positional' --args -- "$@")" + ax -a "$ns" apply -f - >/dev/null </dev/null | jq -c '.' 2>/dev/null || echo null)" + fi + jq -cn --arg name "$name" --arg phase "$phase" --arg after "$after" \ + --argjson t "$t" --argjson result "$result" --argjson secs "$((settled - t0))" --argjson hold "$hold" \ + '{task:$name, phase:$phase, phase_after_hold:$after, hold_seconds:$hold, settle_seconds:$secs, + ready:(($t.status.conditions // []) | map(select(.type=="Ready")) | .[0] // null | if . then {reason, message} else null end), + exit_code:($t.status.command.exitCode // null), result_bytes:($t.status.command.resultBytes // null), + result_sha256:($t.status.command.resultSha256 // null), result:$result}' + if [ "$keep" -eq 0 ]; then ax -a "$ns" delete task "$name" >/dev/null 2>&1 || true; fi +} + +run_case() { +case "$case_" in +halogen) + r="$(run_task "smoke-halogen-$run_id" ax-agent halogen-smoke)" + jq -c --arg case "$case_" '. + {case:$case, pass:(.phase=="Completed" and .phase_after_hold=="Completed" and .exit_code==0)}' <<<"$r" + ;; +pi) + r="$(run_task "smoke-pi-$run_id" ax-agent pi)" + jq -c --arg case "$case_" '. + {case:$case, pass:(.phase=="Completed" and .phase_after_hold=="Completed" and (.result.valid == true))}' <<<"$r" + ;; +exit) + n="${args[1]:?usage: ax-fleet-smoke exit N}" + r="$(run_task "smoke-exit-$run_id" ax-agent exit "$n")" + if [ "$n" -eq 0 ]; then want=Completed; else want=Failed; fi + jq -c --arg case "exit $n" --arg want "$want" --argjson n "$n" \ + '. + {case:$case, pass:(.phase==$want and .phase_after_hold==$want and .exit_code==$n and .ready.reason=="CommandExited")}' <<<"$r" + ;; +egress-deny) + url="${args[1]:-http://10.42.0.2/}" + r="$(run_task "smoke-egress-$run_id" ax-agent fetch "$url")" + jq -c --arg case "$case_" --arg url "$url" \ + '. + {case:$case, url:$url, pass:(.phase=="Failed" and .ready.reason=="CommandExited" and (.exit_code // 0) != 0)}' <<<"$r" + ;; +floor) + n="${args[1]:?usage: ax-fleet-smoke floor N}" + all='[]' + for i in $(seq 1 "$n"); do + r="$(run_task "smoke-floor-$run_id-$i" ax-agent exit 0)" + all="$(jq -c --argjson r "$r" '. + [$r]' <<<"$all")" + done + jq -c --arg case "floor $n" --argjson n "$n" \ + '{case:$case, tasks:., pass:((length == $n) and all(.[]; .phase=="Completed" and ((.ready.message // "") | test("ResourceExhausted") | not)))}' <<<"$all" + ;; +*) + echo "unknown case: $case_" >&2 + exit 64 + ;; +esac +} + +ensure_gateway +receipt="$(run_case)" +printf '%s\n' "$receipt" +jq -e '.pass' <<<"$receipt" >/dev/null diff --git a/modules/ax-fleet/ax.nix b/modules/ax-fleet/ax.nix new file mode 100644 index 000000000..efaa19dfc --- /dev/null +++ b/modules/ax-fleet/ax.nix @@ -0,0 +1,521 @@ +{ + config, + lib, + pkgs, + inputs, + ... +}: +# ax on the fleet, track ax (DESIGN.md sections 9, 10 and 14 A3). This module +# writes only the myAxFleet extension points (manifests, registrySeed) and, on +# the harness, the coordinator's ax scripts and AX_SERVER. Everything else +# (k3s, the registry, the bootstrap runner, ax-server-proxy.socket) belongs to +# the cluster modules next to this file. +# +# ONE image set for every host. The images are built from the flake's own +# nixpkgs pin and the flake's `ax` package, never from the host's `pkgs`: the +# NAS evaluates from nixpkgs-stable, and the digest the NAS seeds must be the +# digest the coordinator's `ax-fleet-image-ref` prints. +# +# Day one: pi against Halogen only. No Claude credential, no claude-code in the +# task image, no secret in any Task or manifest (DESIGN.md section 11). +let + cfg = config.myAxFleet; + system = "x86_64-linux"; + + fleetPkgs = inputs.nixpkgs.legacyPackages.${system}; + ax = inputs.self.packages.${system}.ax; + pi = inputs.llm-agents.packages.${system}.pi; + + images = fleetPkgs.callPackage ../../pkgs/ax/images.nix { inherit ax; }; + agentImage = fleetPkgs.callPackage ../../pkgs/ax-agent-image { inherit ax pi; }; + + seeds = { + ax-server = { + oci = images.ax-server; + repo = "ax/ax-server"; + tag = images.tag; + }; + ax-controller = { + oci = images.ax-controller; + repo = "ax/ax-controller"; + tag = images.tag; + }; + ax-redis = { + oci = images.ax-redis; + repo = "ax/ax-redis"; + tag = images.ax-redis.passthru.tag; + }; + ax-agent = { + oci = agentImage; + repo = "ax/ax-agent"; + tag = agentImage.passthru.tag; + }; + }; + + # Pod image references through the containerd mirror for cfg.registry. + # @digest:@ is replaced at build time from "/digest". + podImage = name: "${cfg.registry}/${seeds.${name}.repo}@@digest:${name}@"; + + controlSelector = { + "ax.mecattaf.dev/role" = "control"; + }; + + labels = name: { + "app.kubernetes.io/name" = name; + "app.kubernetes.io/part-of" = "ax"; + "app.kubernetes.io/managed-by" = "ax-fleet"; + }; + + # The ax-system objects, as data. Rendered to YAML below, then the digests + # are substituted in a build step (no import-from-derivation). + objects = [ + { + apiVersion = "v1"; + kind = "Namespace"; + metadata = { + name = "ax-system"; + labels = labels "ax-system"; + }; + } + + # ── ax-redis: AOF on a local-path volume (the data pool on the NAS) ── + { + apiVersion = "v1"; + kind = "PersistentVolumeClaim"; + metadata = { + name = "ax-redis-data"; + namespace = "ax-system"; + labels = labels "ax-redis"; + }; + spec = { + accessModes = [ "ReadWriteOnce" ]; + storageClassName = "local-path"; + resources.requests.storage = "5Gi"; + }; + } + { + apiVersion = "apps/v1"; + kind = "Deployment"; + metadata = { + name = "ax-redis"; + namespace = "ax-system"; + labels = labels "ax-redis"; + }; + spec = { + replicas = 1; + # One RWO volume: never two pods at once. + strategy.type = "Recreate"; + selector.matchLabels."app.kubernetes.io/name" = "ax-redis"; + template = { + metadata.labels = labels "ax-redis"; + spec = { + nodeSelector = controlSelector; + containers = [ + { + name = "redis"; + image = podImage "ax-redis"; + imagePullPolicy = "IfNotPresent"; + # No password: reachable only as a ClusterIP, and a sandbox + # reaches only its Gateway allowlist. If one is ever added, both + # daemons read REDIS_PASSWORD from the environment; never argv. + args = [ + "--appendonly" + "yes" + "--dir" + "/data" + "--protected-mode" + "no" + ]; + ports = [ + { + containerPort = 6379; + name = "redis"; + } + ]; + readinessProbe = { + tcpSocket.port = 6379; + periodSeconds = 5; + }; + resources = { + requests = { + cpu = "100m"; + memory = "128Mi"; + }; + limits.memory = "1Gi"; + }; + volumeMounts = [ + { + name = "data"; + mountPath = "/data"; + } + ]; + } + ]; + volumes = [ + { + name = "data"; + persistentVolumeClaim.claimName = "ax-redis-data"; + } + ]; + }; + }; + }; + } + { + apiVersion = "v1"; + kind = "Service"; + metadata = { + name = "ax-redis"; + namespace = "ax-system"; + labels = labels "ax-redis"; + }; + spec = { + type = "ClusterIP"; + selector."app.kubernetes.io/name" = "ax-redis"; + ports = [ + { + name = "redis"; + port = 6379; + targetPort = 6379; + } + ]; + }; + } + + # ── ax-server: ClusterIP pinned, never a NodePort (the API has no auth) ── + { + apiVersion = "apps/v1"; + kind = "Deployment"; + metadata = { + name = "ax-server"; + namespace = "ax-system"; + labels = labels "ax-server"; + }; + spec = { + replicas = 1; + selector.matchLabels."app.kubernetes.io/name" = "ax-server"; + template = { + metadata.labels = labels "ax-server"; + spec = { + nodeSelector = controlSelector; + containers = [ + { + name = "ax-server"; + image = podImage "ax-server"; + imagePullPolicy = "IfNotPresent"; + args = [ + "--addr=:8080" + "--redis-addr=ax-redis.ax-system.svc.cluster.local:6379" + ]; + ports = [ + { + containerPort = 8080; + name = "http"; + } + ]; + readinessProbe = { + httpGet = { + path = "/healthz"; + port = 8080; + }; + initialDelaySeconds = 2; + periodSeconds = 5; + }; + livenessProbe = { + httpGet = { + path = "/healthz"; + port = 8080; + }; + initialDelaySeconds = 5; + periodSeconds = 10; + }; + resources = { + requests = { + cpu = "100m"; + memory = "128Mi"; + }; + limits.memory = "1Gi"; + }; + securityContext = { + readOnlyRootFilesystem = true; + allowPrivilegeEscalation = false; + runAsNonRoot = true; + }; + } + ]; + }; + }; + }; + } + { + apiVersion = "v1"; + kind = "Service"; + metadata = { + name = "ax-server"; + namespace = "ax-system"; + labels = labels "ax-server"; + }; + spec = { + type = "ClusterIP"; + clusterIP = cfg.axServerClusterIP; + selector."app.kubernetes.io/name" = "ax-server"; + ports = [ + { + name = "http"; + port = 8080; + targetPort = 8080; + } + ]; + }; + } + + # ── ax-controller: upstream deploy/ax-controller.yaml, plus P1's resync ── + { + apiVersion = "v1"; + kind = "ServiceAccount"; + metadata = { + name = "ax-controller"; + namespace = "ax-system"; + labels = labels "ax-controller"; + }; + } + { + apiVersion = "rbac.authorization.k8s.io/v1"; + kind = "ClusterRole"; + metadata = { + name = "ax-controller"; + labels = labels "ax-controller"; + }; + rules = [ + { + apiGroups = [ "" ]; + resources = [ "secrets" ]; + verbs = [ + "get" + "list" + "watch" + ]; + } + ]; + } + { + apiVersion = "rbac.authorization.k8s.io/v1"; + kind = "ClusterRoleBinding"; + metadata = { + name = "ax-controller"; + labels = labels "ax-controller"; + }; + subjects = [ + { + kind = "ServiceAccount"; + name = "ax-controller"; + namespace = "ax-system"; + } + ]; + roleRef = { + apiGroup = "rbac.authorization.k8s.io"; + kind = "ClusterRole"; + name = "ax-controller"; + }; + } + { + apiVersion = "apps/v1"; + kind = "Deployment"; + metadata = { + name = "ax-controller"; + namespace = "ax-system"; + labels = labels "ax-controller"; + }; + spec = { + # One consumer: P1's resync assumes one controller per Redis group. + replicas = 1; + strategy.type = "Recreate"; + selector.matchLabels."app.kubernetes.io/name" = "ax-controller"; + template = { + metadata.labels = labels "ax-controller"; + spec = { + nodeSelector = controlSelector; + serviceAccountName = "ax-controller"; + containers = [ + { + name = "controller"; + image = podImage "ax-controller"; + imagePullPolicy = "IfNotPresent"; + args = [ + "--redis-addr=ax-redis.ax-system.svc.cluster.local:6379" + "--substrate-endpoint=api.ate-system.svc.cluster.local:443" + "--substrate-authority=api.ate-system.svc" + "--substrate-token-file=/var/run/secrets/ateapi/token" + "--substrate-ca-file=/run/servicedns-ca/trust-bundle.pem" + "--template=default-template" + "--template-atespace=ax-system" + "--running-resync=${toString cfg.ax.runningResyncSeconds}s" + ]; + env = [ + { + name = "ATENET_ROUTER_ADDR"; + value = "atenet-router.ate-system.svc.cluster.local:80"; + } + { + # MEASURED working on RustFS by the 2026-09-23 probe. + name = "AX_SNAPSHOTS_BUCKET"; + value = "gs://ate-snapshots/ax/"; + } + ]; + resources = { + requests = { + cpu = "100m"; + memory = "128Mi"; + }; + limits.memory = "512Mi"; + }; + securityContext = { + readOnlyRootFilesystem = true; + allowPrivilegeEscalation = false; + runAsNonRoot = true; + }; + volumeMounts = [ + { + mountPath = "/var/run/secrets/ateapi"; + name = "ate-token"; + readOnly = true; + } + { + mountPath = "/run/servicedns-ca"; + name = "servicedns-ca"; + readOnly = true; + } + ]; + } + ]; + volumes = [ + { + name = "ate-token"; + projected = { + defaultMode = 292; # 0444: the controller runs as uid 65532 + sources = [ + { + serviceAccountToken = { + audience = "api.ate-system.svc"; + expirationSeconds = 7200; + path = "token"; + }; + } + ]; + }; + } + { + name = "servicedns-ca"; + projected = { + defaultMode = 420; + sources = [ + { + clusterTrustBundle = { + signerName = "servicedns.podcert.ate.dev/identity"; + labelSelector.matchLabels."podcert.ate.dev/canarying" = "live"; + path = "trust-bundle.pem"; + }; + } + ]; + }; + } + ]; + }; + }; + }; + } + ]; + + manifestTemplate = fleetPkgs.writeText "ax-fleet-40-ax.yaml.in" ( + lib.concatMapStringsSep "\n---\n" builtins.toJSON objects + "\n" + ); + + # JSON documents are valid YAML; k3s's deploy controller reads multi-doc + # YAML. Digests are substituted from each image's `digest` file here, at + # build time, then every document is checked for a leftover placeholder. + manifest = + fleetPkgs.runCommand "ax-fleet-40-ax.yaml" + { + nativeBuildInputs = [ fleetPkgs.yq-go ]; + } + '' + cp ${manifestTemplate} $out + chmod u+w $out + ${lib.concatMapStrings (n: '' + substituteInPlace $out --replace-quiet '@digest:${n}@' "$(cat ${seeds.${n}.oci}/digest)" + '') (builtins.attrNames seeds)} + if grep -q '@digest:' $out; then echo "unsubstituted digest in $out" >&2; exit 1; fi + # Every document parses, and every image is pinned by digest. + yq -e 'select(.kind == "Deployment") | .spec.template.spec.containers[].image | test("@sha256:[0-9a-f]{64}$")' $out >/dev/null + test "$(yq -N 'select(.kind != null) | .kind' $out | wc -l)" -eq ${toString (builtins.length objects)} + ''; + + agentRef = fleetPkgs.writeShellScriptBin "ax-fleet-image-ref" '' + # The ax-agent Task image by digest. atelet rewrites localhost:5000. + printf 'localhost:5000/ax/ax-agent@%s\n' "$(cat ${agentImage}/digest)" + ''; + + smoke = fleetPkgs.writeShellApplication { + name = "ax-fleet-smoke"; + runtimeInputs = [ + ax + agentRef + fleetPkgs.coreutils + fleetPkgs.jq + fleetPkgs.yq-go + ]; + text = builtins.readFile ./ax-fleet-smoke.sh; + runtimeEnv = { + AX_FLEET_ATESPACE = cfg.ax.atespace; + AX_FLEET_HALOGEN = cfg.halogenEndpoint; + AX_FLEET_HALOGEN_CIDR = "${builtins.head (lib.splitString ":" cfg.halogenEndpoint)}/32"; + AX_FLEET_RESYNC_SECONDS = toString cfg.ax.runningResyncSeconds; + }; + }; +in +{ + options.myAxFleet.ax = { + runningResyncSeconds = lib.mkOption { + type = lib.types.ints.positive; + default = 15; + description = "P1's --running-resync: how often ax-controller re-checks Running Tasks for a command exit. A trade between detection lag and controller load, not an estimate."; + }; + atespace = lib.mkOption { + type = lib.types.str; + default = "fleet"; + description = "The ax atespace ax-fleet-smoke runs Tasks and the halogen Gateway in."; + }; + images = lib.mkOption { + type = lib.types.attrsOf lib.types.package; + readOnly = true; + internal = true; + default = { + inherit (images) ax-server ax-controller ax-redis; + ax-agent = agentImage; + manifest = manifest; + }; + description = "The ax images and the rendered manifest, for tests and the flake's packages."; + }; + }; + + config = lib.mkIf cfg.enable ( + lib.mkMerge [ + { + # Consumed on the NAS (control role) by the cluster modules; declared + # on every role so the digests any host names are the digests seeded. + myAxFleet.manifests.ax-fleet-40-ax.source = manifest; + myAxFleet.registrySeed = lib.mapAttrs (_: s: s) seeds; + } + + (lib.mkIf (cfg.role == "harness") { + environment.systemPackages = [ + agentRef + smoke + ]; + # ax-server-proxy.socket (harness.nix) listens here and forwards to + # the ax-server ClusterIP. The ax CLI and ax-conwip both honour it. + environment.sessionVariables.AX_SERVER = "http://127.0.0.1:8080"; + }) + ] + ); +} diff --git a/tests/ax-fleet/phases/30-ax.py b/tests/ax-fleet/phases/30-ax.py new file mode 100644 index 000000000..04ec7681f --- /dev/null +++ b/tests/ax-fleet/phases/30-ax.py @@ -0,0 +1,118 @@ +# Phase 4 (Tasks) and phase 5 (resilience) of checks.x86_64-linux.ax-fleet. +# Track ax; DESIGN.md section 12.1. Concatenated after 10-cluster and +# 20-substrate by tests/ax-fleet/default.nix, so `nas`, `coordinator` and +# `worker` are the test nodes and the cluster, Substrate and the WorkerPool are +# already up. Every Task runs through ax-fleet-smoke, the script Tom runs on the +# real coordinator after the switch. +import json +import re + + +def ax_smoke(case, *args, expect_pass=True): + cmd = "ax-fleet-smoke " + " ".join([case, *map(str, args)]) + rc, out = coordinator.execute(cmd + " 2>/dev/null") + lines = [l for l in out.strip().splitlines() if l.startswith("{")] + assert lines, f"{cmd}: no receipt (rc={rc}): {out!r}" + receipt = json.loads(lines[-1]) + print(f"RECEIPT {cmd}: {json.dumps(receipt)}") + if expect_pass: + assert rc == 0 and receipt.get("pass") is True, f"{cmd} failed: {receipt}" + return receipt + + +def kubectl(args): + return nas.succeed(f"k3s kubectl {args}") + + +AX = "AX_SERVER=http://127.0.0.1:8080 ax -a fleet" + + +def task_phase(name): + # `ax get task NAME` prints the Task as YAML; status.phase is its only + # `phase:` key. + out = coordinator.succeed(f"{AX} get task {name}") + m = re.search(r"^\s+phase:\s*(\S+)", out, re.M) + assert m, f"no phase for {name}: {out!r}" + return m.group(1).strip("\"'") + + +with subtest("ax control plane is Available on the NAS, nowhere else"): + for d in ["ax-redis", "ax-server", "ax-controller"]: + kubectl(f"-n ax-system rollout status deploy/{d} --timeout=600s") + node = kubectl( + f"-n ax-system get pods -l app.kubernetes.io/name={d} -o jsonpath='{{.items[0].spec.nodeName}}'" + ).strip() + assert node == "nas", f"{d} runs on {node}" + pv = kubectl( + "get pv -o jsonpath='{range .items[?(@.spec.claimRef.name==\"ax-redis-data\")]}{.spec.local.path}{.spec.hostPath.path}{end}'" + ) + assert pv.startswith("/mnt/nas/services/ax-fleet/local-path"), f"ax-redis volume at {pv!r}" + coordinator.wait_until_succeeds("curl -sf http://127.0.0.1:8080/healthz", timeout=300) + # No Claude credential, and no secret, in any ax object. + kubectl("-n ax-system get secrets -o name | (! grep -q .)") + +with subtest("T1 halogen: Completed, and still Completed after the hold"): + t1 = ax_smoke("halogen", "--hold", 90, "--keep") + assert t1["exit_code"] == 0 and t1["phase_after_hold"] == "Completed", t1 + # The golden-snapshot double execution judge 2 inferred: record, do not fail. + rc, count = worker.execute("curl -sf http://127.0.0.1:8731/stub/requests") + print(f"RECEIPT halogen stub requests after T1: rc={rc} {count.strip()!r}") + +with subtest("T2 pi: Completed with a schema-valid result read back through P1"): + t2 = ax_smoke("pi", "--hold", 30) + assert t2["result"]["valid"] is True, t2 + assert t2["result_bytes"] and t2["result_sha256"], t2 + +with subtest("T3 exit 3: Failed with ExitCode=3"): + t3 = ax_smoke("exit", 3, "--hold", 30) + assert t3["phase"] == "Failed" and t3["ready"]["message"].startswith("ExitCode=3"), t3 + +with subtest("T4 egress-deny: a non-allowlisted target is refused by the Gateway"): + # The coordinator's LAN address answers the worker in this run (checked + # first), so a failure inside the sandbox is the Gateway's refusal. + worker.succeed("curl -s -o /dev/null --max-time 10 http://10.42.0.2/") + t4 = ax_smoke("egress-deny", "http://10.42.0.2/") + assert t4["phase"] == "Failed", t4 + +with subtest("T5 floor 4: four Tasks in a row on the 2-worker pool, none ResourceExhausted"): + t5 = ax_smoke("floor", 4) + for t in t5["tasks"]: + assert t["phase"] == "Completed", t + occupancy = kubectl( + "-n ate-system get pods --field-selector spec.nodeName=coordinator -o name | grep -c ateom || true" + ).strip() + print(f"RECEIPT worker pods on coordinator after {t['task']}: {occupancy}") + +t1_name = t1["task"] + +with subtest("resilience: a restarted ax-controller leaves a finished Task Completed"): + kubectl("-n ax-system delete pod -l app.kubernetes.io/name=ax-controller --wait=true") + kubectl("-n ax-system rollout status deploy/ax-controller --timeout=300s") + coordinator.sleep(45) # three resync periods + assert task_phase(t1_name) == "Completed" + +with subtest("resilience: a restarted ax-redis keeps every Task (AOF on the volume)"): + before = coordinator.succeed(f"{AX} get tasks | tail -n +2 | wc -l").strip() + kubectl("-n ax-system delete pod -l app.kubernetes.io/name=ax-redis --wait=true") + kubectl("-n ax-system rollout status deploy/ax-redis --timeout=300s") + coordinator.wait_until_succeeds(f"{AX} get tasks >/dev/null", timeout=120) + after = coordinator.succeed(f"{AX} get tasks | tail -n +2 | wc -l").strip() + assert before == after and int(after) >= 1, f"tasks before={before} after={after}" + assert task_phase(t1_name) == "Completed" + +with subtest("resilience: the LAN leg flaps, worker pods keep their names, a new T1 completes"): + pods = lambda: kubectl( + "-n ate-system get pods --field-selector spec.nodeName=coordinator -o name | grep ateom | sort" + ) + before = pods() + coordinator.succeed("ip link set eth1 down") + coordinator.sleep(20) + coordinator.succeed("ip link set eth1 up") + nas.wait_until_succeeds( + "k3s kubectl get node coordinator -o jsonpath='{.status.conditions[?(@.type==\"Ready\")].status}' | grep -qx True", + timeout=300, + ) + assert pods() == before, f"worker pods changed: {before!r}" + ax_smoke("halogen") + +coordinator.succeed(f"{AX} delete task {t1_name}") From dc78340c0268856c0824adfff2cd89a625d168d1 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 08:47:37 +0200 Subject: [PATCH 16/37] modules/ax-fleet/substrate: seed the NAS registry, install Substrate with ate-setup Fills only the myAxFleet extension points, on the control role: - registrySeed: six component images (substrate/:d277088b) and the eight third-party images under their upstream repositories; - bootstrap 20-registry-svc: Namespace ate-system and the kind-registry Service plus EndpointSlice to the NAS registry, which atelet's --localhost-registry-replacement names; - 30-substrate: upstream ate-setup --kind deploy ate-system with --image-repo /substrate --image-tag d277088b, once per (installer, images) closure, stamped in kube-system/ax-fleet-substrate; - 40-gvisor-asset: bucket gvisor in the in-cluster RustFS holds the pinned runsc tarball at atelet's fallback key, sha256-verified; the upstream kind credential is read from the Deployment and kept off argv; - 50-workerpool: WorkerPool ateom-gvisor, replicas and memory limit from the options, harness nodeSelector and taint toleration, by digest. tests/ax-fleet/phases/20-substrate.py: the Substrate subtests of the 4-VM check. images.nix: drop an unused argument (deadnix). Co-Authored-By: Claude Opus 5.5 --- modules/ax-fleet/substrate.nix | 290 ++++++++++++++++++++++++++ pkgs/substrate/images.nix | 3 +- tests/ax-fleet/phases/20-substrate.py | 150 +++++++++++++ 3 files changed, 441 insertions(+), 2 deletions(-) create mode 100644 modules/ax-fleet/substrate.nix create mode 100644 tests/ax-fleet/phases/20-substrate.py diff --git a/modules/ax-fleet/substrate.nix b/modules/ax-fleet/substrate.nix new file mode 100644 index 000000000..286db878c --- /dev/null +++ b/modules/ax-fleet/substrate.nix @@ -0,0 +1,290 @@ +{ + config, + lib, + pkgs, + inputs, + ... +}: +# Track substrate (DESIGN.md sections 8, 9 and 14 S3): upstream Agent Substrate +# d277088b on the fleet's k3s, installed by its own `ate-setup --kind` exactly +# as the 2026-09-23 probe ran it (MEASURED rc 0, probe-build.md section 4), with +# --image-repo/--image-tag pointing at nix-built images in the NAS registry +# instead of ko. +# +# This file writes ONLY the myAxFleet extension points: +# registrySeed the six component images, the eight third-party images +# bootstrap 20-registry-svc, 30-substrate, 40-gvisor-asset, 50-workerpool +# and an assertion that the version is one string everywhere. It renders +# nothing on a host where myAxFleet.enable is false, and the extension points +# are consumed only on the control role (the NAS), so the coordinator and the +# worker see an unchanged closure from this file. +let + cfg = config.myAxFleet; + on = cfg.enable && cfg.role == "control"; + + inherit (pkgs.stdenv.hostPlatform) system; + + # The NAS evaluates on nixpkgs-stable; the Go toolchain still comes from the + # nixpkgs-go input, the same derivation pkgs/ax uses. + substrate = pkgs.callPackage ../../pkgs/substrate { + go_1_27 = inputs.nixpkgs-go.legacyPackages.${system}.go_1_27; + }; + images = pkgs.callPackage ../../pkgs/substrate/images.nix { inherit substrate; }; + + kubectl = "${cfg.k3sPackage}/bin/kubectl"; + jq = "${pkgs.jq}/bin/jq"; + curl = "${pkgs.curl}/bin/curl"; + + ns = "ate-system"; + + # 10.42.0.1:5000 -> host and port, for the kind-registry EndpointSlice. + registryParts = lib.splitString ":" cfg.registry; + registryHost = lib.head registryParts; + registryPort = lib.toInt (lib.last registryParts); + + # "ate.dev/sandboxClass=gvisor:NoSchedule" -> key, value, effect. + taint = + let + m = builtins.match "([^=]+)=([^:]*):(.+)" cfg.harnessTaint; + in + { + key = lib.elemAt m 0; + value = lib.elemAt m 1; + effect = lib.elemAt m 2; + }; + + # ── 20: the Service atelet's --localhost-registry-replacement names ── + # The kind overlay runs atelet with --localhost-registry-replacement= + # kind-registry:5000 (manifests/ate-install/kind/atelet/kustomization.yaml), + # so every localhost:5000/... image atelet pulls itself (the pause image, the + # Task images ax names) resolves to this Service. The probe needed exactly + # this object pointing at its node's registry (MEASURED probe-build.md 4). + registrySvc = pkgs.writeText "ax-fleet-20-registry-svc.yaml" ( + builtins.toJSON { + apiVersion = "v1"; + kind = "List"; + items = [ + { + apiVersion = "v1"; + kind = "Namespace"; + metadata.name = ns; + } + { + apiVersion = "v1"; + kind = "Service"; + metadata = { + name = "kind-registry"; + namespace = ns; + labels."app.kubernetes.io/managed-by" = "ax-fleet"; + }; + spec.ports = [ + { + name = "registry"; + port = 5000; + targetPort = registryPort; + protocol = "TCP"; + } + ]; + } + { + apiVersion = "discovery.k8s.io/v1"; + kind = "EndpointSlice"; + metadata = { + name = "kind-registry-nas"; + namespace = ns; + labels = { + "kubernetes.io/service-name" = "kind-registry"; + "endpointslice.kubernetes.io/managed-by" = "ax-fleet"; + }; + }; + addressType = "IPv4"; + ports = [ + { + name = "registry"; + port = registryPort; + protocol = "TCP"; + } + ]; + endpoints = [ + { + addresses = [ registryHost ]; + conditions.ready = true; + } + ]; + } + ]; + } + ); + + # The install is re-run only when what it installs changes: the patched + # manifests, the installer and every image are all in these two paths. + stamp = "${substrate}|${images}"; + + # ── 50: the gVisor WorkerPool, on the harness only ── + # nodeSelector and tolerations go through spec.template (MEASURED + # pkg/api/v1alpha1/workerpool_types.go). Resources: limits.memory caps each + # worker pod's cgroup, which holds the gVisor sandbox (DESIGN.md section 2, + # replacing ax patch P3). Capacity is read from limits only and an unreported + # dimension is unconstrained (cmd/ateapi/internal/scheduling HasRoom), so no + # CPU limit. The requests are set explicitly: with limits alone Kubernetes + # defaults requests to limits, and 2 x 16Gi would be reserved up front. + workerPoolSpec = { + replicas = cfg.workerPool.replicas; + sandboxClass = "gvisor"; + workerImage = "@WORKER_IMAGE@"; + template = { + labels."ax.mecattaf.dev/pool" = "ateom-gvisor"; + nodeSelector = { + "ax.mecattaf.dev/role" = "harness"; + "ate.dev/substrate-version" = cfg.substrateVersion; + }; + tolerations = [ + { + inherit (taint) key value effect; + operator = "Equal"; + } + { + key = "node.kubernetes.io/unreachable"; + operator = "Exists"; + effect = "NoExecute"; + tolerationSeconds = cfg.workerPool.unreachableTolerationSeconds; + } + { + key = "node.kubernetes.io/not-ready"; + operator = "Exists"; + effect = "NoExecute"; + tolerationSeconds = cfg.workerPool.unreachableTolerationSeconds; + } + ]; + resources = { + limits.memory = cfg.workerPool.memoryLimit; + requests = { + cpu = "250m"; + memory = "1Gi"; + }; + }; + }; + }; + workerPoolTemplate = pkgs.writeText "ax-fleet-50-workerpool.json" ( + builtins.toJSON { + apiVersion = "ate.dev/v1alpha1"; + kind = "WorkerPool"; + metadata = { + name = "ateom-gvisor"; + namespace = ns; + labels."app.kubernetes.io/managed-by" = "ax-fleet"; + }; + spec = workerPoolSpec; + } + ); +in +{ + config = lib.mkIf on { + assertions = [ + { + assertion = cfg.substrateVersion == substrate.version; + message = "myAxFleet.substrateVersion (${cfg.substrateVersion}) must equal pkgs/substrate's version (${substrate.version}): it is the node label, the image tag and ate-setup's VERSION at once."; + } + { + assertion = builtins.match "([^=]+)=([^:]*):(.+)" cfg.harnessTaint != null; + message = "myAxFleet.harnessTaint must read key=value:Effect."; + } + { + assertion = builtins.length registryParts == 2; + message = "myAxFleet.registry must read host:port."; + } + ]; + + # Components under substrate/:; third-party images under + # their upstream repositories, which the docker.io and registry.k8s.io + # mirrors in registries.yaml resolve to. The pause image under `pause`, + # which atelet reaches as kind-registry:5000/pause. + myAxFleet.registrySeed = images.seed; + + myAxFleet.bootstrap = { + "20-registry-svc" = '' + # 20-registry-svc: Namespace ${ns} and the kind-registry Service -> + # ${cfg.registry}, before ate-setup starts atelet. + ${kubectl} apply -f ${registrySvc} + ''; + + "30-substrate" = '' + # 30-substrate: upstream's own installer, once per (installer, images) + # pair. The stamp lives in the cluster, so a wiped cluster re-installs. + want='${stamp}' + have=$(${kubectl} -n kube-system get configmap ax-fleet-substrate \ + -o jsonpath='{.data.stamp}' 2>/dev/null || true) + if [ "$have" = "$want" ]; then + echo "substrate ${substrate.version} already installed from this closure" + else + # ate-setup walks up from its working directory to go.mod and reads + # manifests/ from there; it writes nothing under that root. + cd ${substrate.source} + env -u GCE_REGION -u CLUSTER_LOCATION -u NETWORK -u SUBNETWORK \ + -u MEMORYSTORE_INSTANCE -u PROJECT_ID \ + VERSION=${cfg.substrateVersion} BUCKET_NAME=ate-snapshots \ + HOME="''${HOME:-/var/lib/ax-fleet}" \ + ${substrate}/bin/ate-setup --kind --no-dev-env \ + --kubeconfig "$KUBECONFIG" --context default \ + --rollout-timeout 10m \ + --image-repo ${cfg.registry}/substrate \ + --image-tag ${cfg.substrateVersion} \ + deploy ate-system + cd / + ${kubectl} -n kube-system create configmap ax-fleet-substrate \ + --from-literal=stamp="$want" \ + --from-literal=version=${cfg.substrateVersion} \ + --dry-run=client -o yaml | ${kubectl} apply -f - + echo "substrate ${substrate.version} installed" + fi + ''; + + "40-gvisor-asset" = '' + # 40-gvisor-asset: put the pinned runsc tarball where atelet's S3 + # fallback looks for gs://${images.gvisor.bucket}/${images.gvisor.key} + # (same bucket and key; the scheme is ignored). The upstream kind + # credential is read from the RustFS Deployment and kept off argv. + ${kubectl} -n ${ns} rollout status deploy/rustfs --timeout=10m + ip=$(${kubectl} -n ${ns} get svc rustfs -o jsonpath='{.spec.clusterIP}') + base="http://$ip:9000" + creds=$(mktemp) + trap 'rm -f "$creds"' EXIT + chmod 600 "$creds" + ${kubectl} -n ${ns} get deploy rustfs -o json | ${jq} -r ' + .spec.template.spec.containers[0].env + | map({(.name): .value}) | add + | "user = \"\(.RUSTFS_ACCESS_KEY):\(.RUSTFS_SECRET_KEY)\""' > "$creds" + s3() { ${curl} -sS --aws-sigv4 "aws:amz:us-east-1:s3" -K "$creds" "$@"; } + code=$(s3 -o /dev/null -w '%{http_code}' -I "$base/${images.gvisor.bucket}") + if [ "$code" != 200 ]; then + s3 -f -X PUT "$base/${images.gvisor.bucket}" -o /dev/null + echo "created bucket ${images.gvisor.bucket}" + fi + obj="$base/${images.gvisor.bucket}/${images.gvisor.key}" + code=$(s3 -o /dev/null -w '%{http_code}' -I "$obj") + if [ "$code" != 200 ]; then + s3 -f -T ${images.gvisor} \ + -H "x-amz-content-sha256: ${images.gvisor.sha256}" \ + -H "Content-Type: application/zstd" "$obj" -o /dev/null + echo "uploaded ${images.gvisor.key}" + fi + got=$(s3 -f "$obj" | sha256sum | cut -d' ' -f1) + if [ "$got" != "${images.gvisor.sha256}" ]; then + echo "gvisor asset sha256 mismatch in RustFS: $got" >&2 + exit 1 + fi + echo "gvisor asset verified (sha256 ${images.gvisor.sha256})" + ''; + + "50-workerpool" = '' + # 50-workerpool: WorkerPool ateom-gvisor on the harness node(s), by + # digest (the CRD's CEL rule wants an @). The digest is read from the + # store at run time, so evaluation needs no IFD. + digest=$(tr -d '[:space:]' < ${images.components.ateom-gvisor}/digest) + ref="${cfg.registry}/substrate/ateom-gvisor:${cfg.substrateVersion}@$digest" + sed "s|@WORKER_IMAGE@|$ref|" ${workerPoolTemplate} | ${kubectl} apply -f - + echo "workerpool ateom-gvisor: ${toString cfg.workerPool.replicas} replica(s), $ref" + ''; + }; + }; +} diff --git a/pkgs/substrate/images.nix b/pkgs/substrate/images.nix index 6b4b70a0e..83da1c65c 100644 --- a/pkgs/substrate/images.nix +++ b/pkgs/substrate/images.nix @@ -56,7 +56,6 @@ let # ko push. toOci = { - name, image, tag, repo, @@ -113,7 +112,7 @@ let }; in toOci { - inherit name image; + inherit image; tag = version; repo = "substrate/${name}"; }; diff --git a/tests/ax-fleet/phases/20-substrate.py b/tests/ax-fleet/phases/20-substrate.py new file mode 100644 index 000000000..8a6d742e1 --- /dev/null +++ b/tests/ax-fleet/phases/20-substrate.py @@ -0,0 +1,150 @@ +# Phase 20: Substrate (track substrate; DESIGN.md sections 12.1 phases 2 and 3, +# and section 14 S4). Concatenated after 10-cluster.py, so both switches have +# happened: the NAS runs k3s, the registry seed and the bootstrap, and the +# coordinator has joined as the tainted harness node. +# +# Uses only the machine objects `nas` and `coordinator`. Every kubectl runs on +# the NAS as root against /etc/rancher/k3s/k3s.yaml. +import json + +SUB_NS = "ate-system" +SUB_LOCAL_PATH = "/mnt/nas/services/ax-fleet/local-path/" +SUB_CONTROL_DEPLOYMENTS = [ + "ate-api-server", + "ate-controller", + "atenet-router", + "podcertificate-controller", +] + + +def sub_k(args, timeout=120): + return nas.succeed(f"KUBECONFIG=/etc/rancher/k3s/k3s.yaml kubectl {args}", timeout=timeout) + + +def sub_json(args): + return json.loads(sub_k(f"{args} -o json")) + + +with subtest("substrate: bootstrap steps 20-50 ran"): + nas.wait_for_unit("ax-fleet-bootstrap.service", timeout=3600) + log = nas.succeed("journalctl -b -u ax-fleet-bootstrap.service --no-pager") + for step in ["20-registry-svc", "30-substrate", "40-gvisor-asset", "50-workerpool"]: + assert f"step {step}" in log, f"bootstrap step {step} did not run" + assert "gvisor asset verified" in log, "40-gvisor-asset did not verify the tarball" + stamp = sub_k( + "-n kube-system get configmap ax-fleet-substrate -o jsonpath='{.data.version}'" + ).strip() + assert stamp == "d277088b", f"install stamp version {stamp!r}" + +with subtest("substrate: every control workload Available, on the NAS"): + nas.wait_until_succeeds( + f"KUBECONFIG=/etc/rancher/k3s/k3s.yaml kubectl -n {SUB_NS} wait " + "--for=condition=Available deploy --all --timeout=10s", + timeout=900, + ) + names = {d["metadata"]["name"] for d in sub_json(f"-n {SUB_NS} get deploy")["items"]} + for want in SUB_CONTROL_DEPLOYMENTS: + assert want in names, f"deployment {want} missing: {sorted(names)}" + # atenet-egress is the actor egress gateway; its kind differs across + # dataplanes, so it is found by name among Deployments and StatefulSets. + sts = {s["metadata"]["name"] for s in sub_json(f"-n {SUB_NS} get statefulset")["items"]} + assert any(n.startswith("atenet-egress") for n in names | sts), "no atenet-egress" + assert "postgres" in sts, f"postgres StatefulSet missing: {sorted(sts)}" + sub_k(f"-n {SUB_NS} rollout status statefulset/postgres --timeout=600s", timeout=660) + for pod in sub_json(f"-n {SUB_NS} get pods")["items"]: + owner = (pod["metadata"].get("ownerReferences") or [{}])[0].get("kind", "") + name = pod["metadata"]["name"] + node = pod["spec"].get("nodeName", "") + if name.startswith("atelet") or pod["metadata"].get("labels", {}).get( + "ax.mecattaf.dev/pool" + ): + continue + if owner == "Job" and pod["status"].get("phase") == "Succeeded": + continue + assert node == "nas", f"control pod {name} landed on {node!r}" + +with subtest("substrate: nas keeps substrate-version=none, coordinator carries the version"): + nodes = {n["metadata"]["name"]: n for n in sub_json("get nodes")["items"]} + assert nodes["nas"]["metadata"]["labels"].get("ate.dev/substrate-version") == "none" + assert ( + nodes["coordinator"]["metadata"]["labels"].get("ate.dev/substrate-version") + == "d277088b" + ) + +with subtest("substrate: atelet runs on the coordinator only"): + # The atelet DaemonSet is version-keyed by ate-setup; find it by label. + ds = sub_k(f"-n {SUB_NS} get ds -l app=atelet -o jsonpath='{{.items[0].metadata.name}}'").strip() + sub_k(f"-n {SUB_NS} rollout status ds/{ds} --timeout=600s", timeout=660) + atelet = [ + p + for p in sub_json(f"-n {SUB_NS} get pods")["items"] + if p["metadata"]["name"].startswith("atelet") + ] + assert atelet, "no atelet pod" + for p in atelet: + assert p["spec"]["nodeName"] == "coordinator", ( + f"atelet on {p['spec']['nodeName']}" + ) + +with subtest("substrate: ClusterTrustBundles and pod certificates served"): + res = sub_k("get --raw /apis/certificates.k8s.io/v1beta1") + assert "clustertrustbundles" in res and "podcertificaterequests" in res, res + assert sub_json("get clustertrustbundles")["items"], "no ClusterTrustBundle objects" + +with subtest("substrate: WorkerPool ateom-gvisor Ready 2 on the coordinator"): + nas.wait_until_succeeds( + "test \"$(KUBECONFIG=/etc/rancher/k3s/k3s.yaml kubectl -n ate-system get " + "workerpool ateom-gvisor -o jsonpath='{.status.readyReplicas}')\" = 2", + timeout=900, + ) + wp = sub_json(f"-n {SUB_NS} get workerpool ateom-gvisor") + assert "@sha256:" in wp["spec"]["workerImage"], wp["spec"]["workerImage"] + workers = [ + p + for p in sub_json(f"-n {SUB_NS} get pods -l ax.mecattaf.dev/pool=ateom-gvisor")["items"] + ] + assert len(workers) == 2, f"{len(workers)} worker pods" + for p in workers: + assert p["spec"]["nodeName"] == "coordinator", p["spec"]["nodeName"] + lim = p["spec"]["containers"][0]["resources"]["limits"]["memory"] + assert lim, "worker pod has no memory limit" + +with subtest("substrate: gVisor fetched through the RustFS fallback"): + # No internet in the VM: atelet's anonymous GCS open of gs://gvisor/... + # fails, then its S3 client reads the same bucket and key from the + # in-cluster RustFS (40-gvisor-asset). The prewarmer logs "Sandbox assets + # prewarmed" for gvisor-default only once the pause image and the gVisor + # tarball (sha256-checked by atelet) are both on the node + # (cmd/atelet/sandbox_prewarm.go). + nas.wait_until_succeeds( + "KUBECONFIG=/etc/rancher/k3s/k3s.yaml kubectl -n ate-system logs -l app=atelet " + "--all-containers --tail=-1 | grep 'Sandbox assets prewarmed' | grep -q gvisor-default", + timeout=900, + ) + logs = sub_k(f"-n {SUB_NS} logs -l app=atelet --all-containers --tail=-1", timeout=300) + print("\n".join(l for l in logs.splitlines() if "gvisor" in l.lower())[-4000:]) + assert "gVisor release download complete" in logs or "gvisor" in logs.lower() + +with subtest("substrate: every PersistentVolume on the data pool"): + pvs = sub_json("get pv")["items"] + assert pvs, "no PersistentVolumes" + for pv in pvs: + spec = pv["spec"] + path = (spec.get("hostPath") or spec.get("local") or {}).get("path", "") + assert path.startswith(SUB_LOCAL_PATH), (pv["metadata"]["name"], path) + node_terms = ( + spec.get("nodeAffinity", {}).get("required", {}).get("nodeSelectorTerms", []) + ) + values = [ + v + for t in node_terms + for e in t.get("matchExpressions", []) + for v in e.get("values", []) + ] + assert "coordinator" not in values, f"PV {pv['metadata']['name']} on the coordinator" + +with subtest("substrate: every image came from the NAS registry by digest"): + pods = sub_json(f"-n {SUB_NS} get pods")["items"] + for p in pods: + for c in p["spec"].get("containers", []) + p["spec"].get("initContainers", []): + assert "@sha256:" in c["image"], f"{p['metadata']['name']}: {c['image']}" From 1165567be060aad7696cfcdfff7e1b557d52a394 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 08:51:23 +0200 Subject: [PATCH 17/37] pkgs/substrate: keep only ate-setup and the install tree in the NAS closure The bootstrap referenced the full substrate output (seven binaries, 377 MB) and the patched source (172 MB) at run time. It now runs passthru.ate-setup (53 MB) from passthru.installTree (go.mod, manifests/, hack/: 1.4 MB), which is everything deploy ate-system reads. The component binaries reach the NAS only inside the images. Saves about 0.5 GB on the NAS's eMMC root. Co-Authored-By: Claude Opus 5.5 --- modules/ax-fleet/substrate.nix | 11 ++-- pkgs/substrate/default.nix | 111 +++++++++++++++++++++------------ 2 files changed, 76 insertions(+), 46 deletions(-) diff --git a/modules/ax-fleet/substrate.nix b/modules/ax-fleet/substrate.nix index 286db878c..a91e779a1 100644 --- a/modules/ax-fleet/substrate.nix +++ b/modules/ax-fleet/substrate.nix @@ -116,9 +116,10 @@ let } ); - # The install is re-run only when what it installs changes: the patched - # manifests, the installer and every image are all in these two paths. - stamp = "${substrate}|${images}"; + # The install is re-run only when what it installs changes: the installer, + # the patched manifests and every image are all in these three paths. The + # component binaries themselves are not in the NAS closure, only the images. + stamp = "${substrate.ate-setup}|${substrate.installTree}|${images}"; # ── 50: the gVisor WorkerPool, on the harness only ── # nodeSelector and tolerations go through spec.template (MEASURED @@ -219,12 +220,12 @@ in else # ate-setup walks up from its working directory to go.mod and reads # manifests/ from there; it writes nothing under that root. - cd ${substrate.source} + cd ${substrate.installTree} env -u GCE_REGION -u CLUSTER_LOCATION -u NETWORK -u SUBNETWORK \ -u MEMORYSTORE_INSTANCE -u PROJECT_ID \ VERSION=${cfg.substrateVersion} BUCKET_NAME=ate-snapshots \ HOME="''${HOME:-/var/lib/ax-fleet}" \ - ${substrate}/bin/ate-setup --kind --no-dev-env \ + ${substrate.ate-setup}/bin/ate-setup --kind --no-dev-env \ --kubeconfig "$KUBECONFIG" --context default \ --rollout-timeout 10m \ --image-repo ${cfg.registry}/substrate \ diff --git a/pkgs/substrate/default.nix b/pkgs/substrate/default.nix index ebc063978..153f62a47 100644 --- a/pkgs/substrate/default.nix +++ b/pkgs/substrate/default.nix @@ -3,6 +3,7 @@ buildGoModule, fetchFromGitHub, applyPatches, + runCommand, # Passed in by the caller from the nixpkgs-go input, exactly as pkgs/ax gets # it (see overlays/default.nix and the nixpkgs-go comment in flake.nix). # Substrate's go.mod says `go 1.27.0`; the probe built it with 1.27.1 @@ -26,8 +27,13 @@ # 0002 third-party images -> their linux/amd64 child digests, so the NAS # seeds ~0.7 GB instead of every platform (see the patch header). # ate-setup reads the manifests from its working directory's repository root -# (it walks up to go.mod), so the patched tree is passthru.source and the -# bootstrap runs ate-setup from there. +# (it walks up to go.mod). What it reads for `deploy ate-system` is go.mod, +# manifests/ and hack/ (kustomize overlays, CSI manifests; MEASURED grep of +# cmd/ate-setup/internal/steps), so passthru.installTree is those three, +# about 1.3 MB instead of the 172 MB tree, and passthru.ate-setup is the +# installer alone (55 MB instead of all seven binaries). Only those two sit in +# the NAS closure at run time; the component binaries reach the NAS inside +# the images. let version = "d277088b"; rev = "d277088bc1d081ef716d81dd7986d05d0a36ad3a"; @@ -51,6 +57,45 @@ let buildGo127Module = buildGoModule.override { go = go_1_27; }; + common = { + inherit version; + src = source; + + # The tree is vendored (vendor/modules.txt), so no module download. + vendorHash = null; + + env.CGO_ENABLED = "0"; + + # The Makefile's LDFLAGS (Makefile:44-45), with the version the node label, + # the image tag and ate-setup's VERSION all share. + ldflags = [ + "-s" + "-w" + "-X=github.com/agent-substrate/substrate/internal/version.Version=${version}" + ]; + + # Upstream's suite needs Docker (225 PostgreSQL testcontainer tests), root + # (19 tests) and an FHS /bin/sleep; the probe ran it outside nix (1422 + # pass, 2 environment-only failures, MEASURED probe-build.md section 3). + # Inside the nix sandbox it can only fail for the same reasons. + doCheck = false; + }; + + ate-setup = buildGo127Module ( + common + // { + pname = "ate-setup"; + subPackages = [ "cmd/ate-setup" ]; + meta.mainProgram = "ate-setup"; + } + ); + + installTree = runCommand "substrate-install-tree-${version}" { } '' + mkdir -p $out + cp ${source}/go.mod $out/ + cp -r ${source}/manifests ${source}/hack $out/ + ''; + # The six components the kind install and the WorkerPool run, plus the # installer. Image names are the import paths' last element, which is what # ate-setup's --image-repo mode looks up (cmd/ate-setup/internal/images). @@ -63,45 +108,29 @@ let "ateom-gvisor" ]; in -buildGo127Module { - pname = "substrate"; - inherit version; - src = source; +buildGo127Module ( + common + // { + pname = "substrate"; - # The tree is vendored (vendor/modules.txt), so no module download. - vendorHash = null; + subPackages = [ "cmd/ate-setup" ] ++ map (c: "cmd/${c}") components; - subPackages = [ "cmd/ate-setup" ] ++ map (c: "cmd/${c}") components; - - env.CGO_ENABLED = "0"; - - # The Makefile's LDFLAGS (Makefile:44-45), with the version the node label, - # the image tag and ate-setup's VERSION all share. - ldflags = [ - "-s" - "-w" - "-X=github.com/agent-substrate/substrate/internal/version.Version=${version}" - ]; - - # Upstream's suite needs Docker (225 PostgreSQL testcontainer tests), root - # (19 tests) and an FHS /bin/sleep; the probe ran it outside nix (1422 pass, - # 2 environment-only failures, MEASURED probe-build.md section 3). Inside the - # nix sandbox it can only fail for the same environmental reasons. - doCheck = false; - - passthru = { - inherit - source - rev - components - ; - }; + passthru = { + inherit + source + installTree + ate-setup + rev + components + ; + }; - meta = { - description = "Agent Substrate (upstream main ${version}): ate-setup and the gVisor control plane"; - homepage = "https://github.com/agent-substrate/substrate"; - license = lib.licenses.asl20; - platforms = [ "x86_64-linux" ]; - mainProgram = "ate-setup"; - }; -} + meta = { + description = "Agent Substrate (upstream main ${version}): ate-setup and the gVisor control plane"; + homepage = "https://github.com/agent-substrate/substrate"; + license = lib.licenses.asl20; + platforms = [ "x86_64-linux" ]; + mainProgram = "ate-setup"; + }; + } +) From cc39eac44e74c26c65c18e4a262a214c0805d98d Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 09:00:31 +0200 Subject: [PATCH 18/37] ax-fleet: fixes from the VM runs; the 4-VM and boot checks pass - ax-fleet-kubeconfig: quote the ssh host (shellcheck SC2086 failed the coordinator toplevel build). - tests: wait out k3s's transient node taints and the flannel.1 link instead of asserting once; A-record DNS query (the stand-in has no upstream, AAAA is REFUSED); a cross-node pod-to-pod subtest over VXLAN; typed receipt (test-driver type check); records also printed to the log. Co-Authored-By: Claude Opus 5.5 --- modules/ax-fleet/harness.nix | 2 +- tests/ax-fleet-boot/default.nix | 4 ++- tests/ax-fleet/default.nix | 12 ++++++-- tests/ax-fleet/phases/10-cluster.py | 44 ++++++++++++++++++++++++++--- 4 files changed, 53 insertions(+), 9 deletions(-) diff --git a/modules/ax-fleet/harness.nix b/modules/ax-fleet/harness.nix index 2885bb61f..96f412b65 100644 --- a/modules/ax-fleet/harness.nix +++ b/modules/ax-fleet/harness.nix @@ -104,7 +104,7 @@ let mkdir -p "$HOME/.kube" tmp=$(mktemp "$HOME/.kube/.config.XXXXXX") trap 'rm -f "$tmp"' EXIT - ssh ''${AX_FLEET_NAS:-nas} cat /etc/ax-fleet/admin.kubeconfig > "$tmp" + ssh "''${AX_FLEET_NAS:-nas}" cat /etc/ax-fleet/admin.kubeconfig > "$tmp" [ -s "$tmp" ] || { echo "ax-fleet-kubeconfig: empty kubeconfig from the NAS" >&2; exit 1; } mv "$tmp" "$HOME/.kube/config" trap - EXIT diff --git a/tests/ax-fleet-boot/default.nix b/tests/ax-fleet-boot/default.nix index 01fa1ea9c..77dfb9902 100644 --- a/tests/ax-fleet-boot/default.nix +++ b/tests/ax-fleet-boot/default.nix @@ -35,7 +35,9 @@ pkgs.testers.runNixOSTest { "k3s kubectl get node nas -o jsonpath='{.status.conditions[?(@.type==\"Ready\")].status}' | grep -x True", timeout=600, ) - assert nas.succeed("k3s kubectl get node nas -o jsonpath='{.spec.taints}'").strip() == "" + # k3s's own transient taints (uninitialized, not-ready) clear on their own; + # the NAS carries no taint of ours. + nas.wait_until_succeeds("test -z \"$(k3s kubectl get node nas -o jsonpath='{.spec.taints}')\"", timeout=300) nas.succeed("k3s kubectl -n kube-system wait --for=condition=Available deploy/coredns deploy/local-path-provisioner --timeout=600s") nas.succeed("test -s /etc/ax-fleet/admin.kubeconfig") # The snapshot was taken at first activation, before k3s ever ran. diff --git a/tests/ax-fleet/default.nix b/tests/ax-fleet/default.nix index b3e408a20..1288feb2b 100644 --- a/tests/ax-fleet/default.nix +++ b/tests/ax-fleet/default.nix @@ -38,7 +38,11 @@ let "vm.overcommit_memory", ] - receipt = {"test": "ax-fleet", "subtests": [], "values": {}} + from typing import Any + + receipt_subtests: list[dict[str, Any]] = [] + receipt_values: dict[str, Any] = {} + receipt: dict[str, Any] = {"test": "ax-fleet", "subtests": receipt_subtests, "values": receipt_values} def save_receipt(): @@ -49,7 +53,9 @@ let def record(key, value): - receipt["values"][key] = value + receipt_values[key] = value + # Also into the build log: a failed build keeps no $out. + print(f"AXFLEET-RECORD {key} = {json.dumps(value, sort_keys=True)}") save_receipt() @@ -58,7 +64,7 @@ let t0 = time.monotonic() with subtest(name): yield - receipt["subtests"].append({"name": name, "result": "pass", "seconds": round(time.monotonic() - t0, 1)}) + receipt_subtests.append({"name": name, "result": "pass", "seconds": round(time.monotonic() - t0, 1)}) save_receipt() diff --git a/tests/ax-fleet/phases/10-cluster.py b/tests/ax-fleet/phases/10-cluster.py index b8dcfaf84..6262b4fa1 100644 --- a/tests/ax-fleet/phases/10-cluster.py +++ b/tests/ax-fleet/phases/10-cluster.py @@ -95,7 +95,8 @@ def apply_probe(name, role, host_port, pvc=None): with step("nas: node Ready, untainted, control labels"): node_ready("nas") - assert jsonpath("node nas", "{.spec.taints}") == "", "the NAS must be untainted" + # k3s's own transient taints (uninitialized, not-ready) clear on their own. + nas.wait_until_succeeds("test -z \"$(k3s kubectl get node nas -o jsonpath='{.spec.taints}')\"", timeout=300) labels = json.loads(kubectl("get node nas -o jsonpath='{.metadata.labels}'")) assert labels.get("ax.mecattaf.dev/role") == "control", labels assert labels.get("ate.dev/substrate-version") == "none", labels @@ -155,7 +156,8 @@ def apply_probe(name, role, host_port, pvc=None): assert labels.get("ate.dev/substrate-version") == "d277088b", labels ip = jsonpath("node coordinator", "{.status.addresses[?(@.type==\"InternalIP\")].address}") assert ip == "10.42.0.2", ip - coordinator.succeed("ip -d link show flannel.1 | grep -q 'dev eth1'") + # flannel.1 appears shortly after the node registers; wait for it. + coordinator.wait_until_succeeds("ip -d link show flannel.1 | grep -q 'dev eth1'", timeout=180) record("sysctl_coordinator_after_switch", sysctls(coordinator)) @@ -224,9 +226,43 @@ def apply_probe(name, role, host_port, pvc=None): record("guard_chain", coordinator.succeed("iptables -t mangle -S ax-fleet-guard").strip().splitlines()) +def diag(cmds): + out = {} + for name, (machine, cmd) in cmds.items(): + _, text = machine.execute(cmd + " 2>&1") + out[name] = text[-3000:] + return out + + +with step("cluster plumbing: pod to pod across nodes (flannel VXLAN)"): + nas_pod_ip = jsonpath("pod probe-nas", "{.status.podIP}") + status, _ = nas.execute(f"k3s kubectl exec probe-coord -- curl -sf --max-time 10 http://{nas_pod_ip}:8000/") + if status != 0: + record("diag_flannel", diag({ + "coord_routes": (coordinator, "ip route; ip -d link show flannel.1"), + "nas_routes": (nas, "ip route; ip -d link show flannel.1"), + "coord_fw": (coordinator, "iptables -S nixos-fw; iptables -t mangle -S ax-fleet-guard"), + "nas_nft_input": (nas, "nft list chain inet nixos-fw input-allow"), + })) + nas.succeed(f"k3s kubectl exec probe-coord -- curl -sf --max-time 10 http://{nas_pod_ip}:8000/ | grep -x pod-ok") + + with step("cluster plumbing: DNS through the NAS stand-in"): - kubectl("exec probe-coord -- nslookup only-nas.test | grep -q 10.42.0.77") - kubectl("exec probe-nas -- nslookup only-nas.test | grep -q 10.42.0.77") + # A records only: the stand-in has no upstream, so AAAA answers REFUSED. + status, _ = nas.execute("k3s kubectl exec probe-coord -- nslookup -type=a only-nas.test 2>&1 | grep -q 10.42.0.77") + if status != 0: + record("diag_dns", diag({ + "coord_nslookup": (nas, "k3s kubectl exec probe-coord -- nslookup only-nas.test"), + "coord_nslookup_svc": (nas, "k3s kubectl exec probe-coord -- nslookup kubernetes.default.svc.cluster.local"), + "nas_pod_direct": (nas, "k3s kubectl exec probe-nas -- nslookup only-nas.test 10.42.0.1"), + "nas_pod_coredns": (nas, "k3s kubectl exec probe-nas -- nslookup only-nas.test"), + "coredns_logs": (nas, "k3s kubectl -n kube-system logs deploy/coredns --tail=50"), + "coredns_pod_resolv": (nas, "k3s kubectl -n kube-system exec deploy/coredns -- cat /etc/resolv.conf"), + "dnsmasq": (nas, "journalctl -u dnsmasq --no-pager | tail -30; ss -lunp | grep :53"), + "nas_nft": (nas, "nft list table inet nixos-fw"), + })) + kubectl("exec probe-coord -- nslookup -type=a only-nas.test | grep -q 10.42.0.77") + kubectl("exec probe-nas -- nslookup -type=a only-nas.test | grep -q 10.42.0.77") with step("resilience: coordinator link flap"): From c539dbfb10ec6d57ad514c999d282e843d141272 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 09:53:39 +0200 Subject: [PATCH 19/37] ax-fleet: integrate the three tracks, run the Task image under gVisor, prove it in the VM test Merges ax/fleet-cluster (cc39eac4), ax/fleet-substrate (1165567b) and ax/fleet-ax (71ce913b); the four add/add placeholder conflicts take each track's own file. The gates stay ON in hosts/{nas,coordinator,worker} (control, harness, inference). - pkgs/ax-agent-image: pi's bun binary lists PT_LOAD out of p_vaddr order. Linux runs it; gVisor refuses it (exit 126 inside the sandbox, MEASURED in the VM test). elf-sort-load.py reorders the program header table only, so the Task image's pi runs in gVisor (T2 schema-valid, MEASURED). - pkgs/ax-agent-image: a `probe` mode (claude --version when the image has claude-code, a GET of Halogen's /v1/models, /proc/version); pi mode records its stderr tail; test-only `extraPaths`/`variant` arguments. - ax-fleet-smoke: --image, --sandbox-class and the `probe` case; an exited command with protobuf-omitted exitCode reads as 0. - tests/ax-fleet: phase 25-harness-probe (one sandboxClass gvisor Task on the coordinator in a test-only image variant with claude-code and no credential: Completed, sampled Completed for 60 s); the NAS test node seeds that variant; test nodes import modules/ax-client.nix as the hosts do; the Halogen stub answers a JSON Schema prompt with {"answer": 42}; 20-substrate checks the pod-certificate controller in its own namespace and stops shadowing the driver's `step` and `log`. - flake: packages ax-agent-image and substrate-images. checks.x86_64-linux.ax-fleet (4 VMs): rc 0, 37 subtests. Co-Authored-By: Claude Opus 5.5 --- flake.nix | 14 +++ modules/ax-fleet/ax-fleet-smoke.sh | 34 +++++- pkgs/ax-agent-image/ax-agent.sh | 31 +++++- pkgs/ax-agent-image/default.nix | 35 +++++- pkgs/ax-agent-image/elf-sort-load.py | 51 +++++++++ tests/ax-fleet/default.nix | 3 + tests/ax-fleet/halogen_stub.py | 18 ++- tests/ax-fleet/nodes.nix | 25 ++++- tests/ax-fleet/phases/20-substrate.py | 24 ++-- tests/ax-fleet/phases/25-harness-probe.py | 128 ++++++++++++++++++++++ 10 files changed, 340 insertions(+), 23 deletions(-) create mode 100644 pkgs/ax-agent-image/elf-sort-load.py create mode 100644 tests/ax-fleet/phases/25-harness-probe.py diff --git a/flake.nix b/flake.nix index 0d8f42681..0bc55c47e 100644 --- a/flake.nix +++ b/flake.nix @@ -710,6 +710,20 @@ ax-fleet-teardown = pkgs.callPackage ./pkgs/ax-fleet-teardown { k3s = inputs.nixpkgs.legacyPackages.${system}.k3s_1_36; }; + # The fleet's ax Task image (OCI layout plus `digest`), built exactly + # as modules/ax-fleet/ax.nix builds the one the NAS seeds: the flake's + # nixpkgs, the patched `ax`, pi from llm-agents. pi only, no claude-code. + ax-agent-image = inputs.nixpkgs.legacyPackages.${system}.callPackage ./pkgs/ax-agent-image { + inherit (self.packages.${system}) ax; + inherit (inputs.llm-agents.packages.${system}) pi; + }; + # Upstream Agent Substrate d277088b: the six component images, the + # third-party images by digest and the gVisor asset (pkgs/substrate). + substrate-images = pkgs.callPackage ./pkgs/substrate/images.nix { + substrate = pkgs.callPackage ./pkgs/substrate { + go_1_27 = inputs.nixpkgs-go.legacyPackages.${system}.go_1_27; + }; + }; }; # `sudo nix run ~/dotfiles#ax-fleet-teardown` after `myAxFleet.enable = diff --git a/modules/ax-fleet/ax-fleet-smoke.sh b/modules/ax-fleet/ax-fleet-smoke.sh index c7ff574bf..082f260fa 100644 --- a/modules/ax-fleet/ax-fleet-smoke.sh +++ b/modules/ax-fleet/ax-fleet-smoke.sh @@ -8,28 +8,38 @@ # ax-fleet-smoke egress-deny [URL] GET a non-allowlisted URL (default the # coordinator's LAN address); must fail # ax-fleet-smoke floor N N Tasks in a row; none ResourceExhausted +# ax-fleet-smoke probe [--hold N] `claude --version` and a GET of Halogen's +# /v1/models, sandboxClass gvisor; needs an +# --image that carries claude-code (the +# fleet image does not, DESIGN.md 11) # # Options: --hold N (re-read the phase after N seconds), --timeout N (per Task, -# default 900), --keep (do not delete the Tasks). Exit 0 only when the case -# passes. The receipt carries no secret: Tasks carry none (DESIGN.md 11). +# default 900), --keep (do not delete the Tasks), --image REF (default: the +# fleet image, ax-fleet-image-ref), --sandbox-class C (spec.sandboxClass; empty +# means gVisor). Exit 0 only when the case passes. The receipt carries no secret: Tasks carry none (DESIGN.md 11). export AX_SERVER="${AX_SERVER:-http://127.0.0.1:8080}" ns="$AX_FLEET_ATESPACE" hold=0 timeout=900 keep=0 +image="" +sandbox_class="" args=() while [ $# -gt 0 ]; do case "$1" in --hold) hold="$2"; shift 2 ;; --timeout) timeout="$2"; shift 2 ;; --keep) keep=1; shift ;; + --image) image="$2"; shift 2 ;; + --sandbox-class) sandbox_class="$2"; shift 2 ;; *) args+=("$1"); shift ;; esac done -[ "${#args[@]}" -ge 1 ] || { sed -n '2,14p' "$0" >&2; exit 64; } +[ "${#args[@]}" -ge 1 ] || { sed -n '2,21p' "$0" >&2; exit 64; } case_="${args[0]}" -image="$(ax-fleet-image-ref)" +[ -n "$image" ] || image="$(ax-fleet-image-ref)" +[ "$case_" != probe ] || sandbox_class="${sandbox_class:-gvisor}" run_id="$(date +%s)-$$" ensure_gateway() { @@ -48,7 +58,9 @@ spec: YAML } -# task_json NAME: the Task as JSON (empty object when absent). +# task_json NAME: the Task as JSON (empty object when absent). The status is +# protobuf JSON, which omits zero values: an exited command with exit code 0 +# carries `exited: true` and no `exitCode` (MEASURED in the VM test). task_json() { ax -a "$ns" get task "$1" 2>/dev/null | yq -o json '.' 2>/dev/null || echo '{}' } @@ -59,6 +71,8 @@ run_task() { shift local cmd_json cmd_json="$(jq -cn '$ARGS.positional' --args -- "$@")" + local class_line="" + [ -z "$sandbox_class" ] || class_line=" sandboxClass: \"$sandbox_class\"" ax -a "$ns" apply -f - >/dev/null </dev/null 2>&1 || true; fi } @@ -138,6 +153,13 @@ floor) jq -c --arg case "floor $n" --argjson n "$n" \ '{case:$case, tasks:., pass:((length == $n) and all(.[]; .phase=="Completed" and ((.ready.message // "") | test("ResourceExhausted") | not)))}' <<<"$all" ;; +probe) + r="$(run_task "smoke-probe-$run_id" ax-agent probe)" + jq -c --arg case "$case_" --arg class "$sandbox_class" --arg image "$image" \ + '. + {case:$case, sandbox_class:$class, image:$image, + pass:(.phase=="Completed" and .phase_after_hold=="Completed" and .exit_code==0 + and .result.claude_rc==0 and .result.http_code=="200")}' <<<"$r" + ;; *) echo "unknown case: $case_" >&2 exit 64 diff --git a/pkgs/ax-agent-image/ax-agent.sh b/pkgs/ax-agent-image/ax-agent.sh index 77c25c0ed..0959c2707 100644 --- a/pkgs/ax-agent-image/ax-agent.sh +++ b/pkgs/ax-agent-image/ax-agent.sh @@ -5,6 +5,11 @@ # ax-agent pi pi against Halogen, result validated against a # JSON Schema (the probe's adapter-pi.sh, e1) # ax-agent fetch URL GET URL; exits with curl's code (egress checks) +# ax-agent probe what a sandbox can do without a credential: +# `claude --version` (only in an image that carries +# claude-code; the fleet image does not) and one GET +# of Halogen's /v1/models; records /proc/version, +# which names gVisor's kernel inside a sandbox # ax-agent exit N exit with code N # # Every mode writes its result to $AX_RESULT_PATH (default .ax/result.json @@ -56,8 +61,10 @@ pi) vrc=$? jq -n --arg label "${AX_CONWIP_LABEL:-}" --arg model "$model" --arg effort "${AX_CONWIP_EFFORT:-low}" \ --argjson rc "$rc" --argjson v "$verdict" --arg secs "$(python3 -c "print($t1-$t0)")" --arg cwd "$PWD" \ + --arg err "$(tail -c 2000 "$out.stderr" 2>/dev/null)" \ '{label:$label, model:$model, effort:$effort, harness:"pi", harness_rc:$rc, - valid:$v.valid, errors:$v.errors, result:$v.value, seconds:($secs|tonumber), cwd:$cwd}' >"$out" + valid:$v.valid, errors:$v.errors, result:$v.value, seconds:($secs|tonumber), cwd:$cwd, + stderr_tail:$err}' >"$out" echo "ax-agent pi: valid=$(jq .valid "$out") rc=$rc vrc=$vrc" [ "$rc" -eq 0 ] && [ "$vrc" -eq 0 ] ;; @@ -71,6 +78,26 @@ fetch) exit "$rc" ;; +probe) + kernel=$(cat /proc/version 2>/dev/null) + if command -v claude >/dev/null 2>&1; then + cv=$(claude --version 2>&1) + crc=$? + else + cv="claude: not in this image" + crc=127 + fi + code=$(curl -sS -o "$out.models" -w '%{http_code}' --max-time 60 "$halogen/v1/models") + hrc=$? + seen=$(jq -r '.data[0].id // empty' "$out.models" 2>/dev/null) + jq -n --arg cv "$cv" --argjson crc "$crc" --arg url "$halogen" --argjson hrc "$hrc" --arg code "$code" \ + --arg seen "$seen" --arg kernel "$kernel" \ + '{mode:"probe", claude_version:$cv, claude_rc:$crc, halogen:$url, curl_rc:$hrc, http_code:$code, + model:$seen, proc_version:$kernel, ok:($crc == 0 and $hrc == 0 and $code == "200")}' >"$out" + echo "ax-agent probe: claude_rc=$crc curl_rc=$hrc http=$code" + [ "$crc" -eq 0 ] && [ "$hrc" -eq 0 ] && [ "$code" = 200 ] + ;; + exit) n="${1:-0}" jq -n --argjson n "$n" '{mode:"exit", code:$n}' >"$out" @@ -78,7 +105,7 @@ exit) ;; *) - echo "usage: ax-agent halogen-smoke | pi | fetch URL | exit N" >&2 + echo "usage: ax-agent halogen-smoke | pi | fetch URL | probe | exit N" >&2 exit 64 ;; esac diff --git a/pkgs/ax-agent-image/default.nix b/pkgs/ax-agent-image/default.nix index d264c2de8..c93ad2886 100644 --- a/pkgs/ax-agent-image/default.nix +++ b/pkgs/ax-agent-image/default.nix @@ -24,6 +24,10 @@ # pi from llm-agents. Halogen only on day one; no claude-code in this image # until Tom rules on Claude credentials in sandboxes (DESIGN.md section 11). pi, + # Test-only variants (tests/ax-fleet): extra store paths linked into /bin and + # a name suffix. The fleet image passes neither: pi only, no claude-code. + extraPaths ? [ ], + variant ? "", }: # The ax Task image for the fleet (DESIGN.md section 10.2): `ax-agent`. # @@ -55,6 +59,26 @@ let ''; # The runner at the path Substrate's template names. + # pi with its ELF program headers in the order the ELF spec asks for. pi + # 0.85.1's bun binary lists PT_LOAD out of p_vaddr order; Linux runs it, but + # gVisor's loader refuses it (ENOEXEC, exit 126 inside the sandbox, MEASURED + # in the ax-fleet VM test, INTEGRATE.md). Only table entries move; segments, + # addresses and bytes stay. The name keeps pi's, so the store path length is + # unchanged; only the text wrapper in bin/ is repointed. + piSandbox = runCommand pi.name { nativeBuildInputs = [ python3 ]; } '' + cp -a ${pi} $out + chmod -R u+w $out + find $out -type f -print0 | while IFS= read -r -d "" f; do + if [ "$(head -c 4 "$f" | od -An -c | tr -d ' ')" = '177ELF' ]; then + python3 ${./elf-sort-load.py} "$f" + fi + done + for f in $out/bin/*; do + substituteInPlace "$f" --replace-quiet ${pi} $out + done + chmod -R a-w $out + ''; + runner = runCommand "ax-task-runner-usr-local" { } '' mkdir -p $out/usr/local/bin ln -s ${ax}/bin/ax-task-runner $out/usr/local/bin/ax-task-runner @@ -90,9 +114,10 @@ let gnutar gzip git - pi + piSandbox ax-agent - ]; + ] + ++ extraPaths; pathsToLink = [ "/bin" "/etc/ssl" @@ -101,7 +126,7 @@ let }; image = dockerTools.buildLayeredImage { - name = "ax/ax-agent"; + name = "ax/ax-agent${variant}"; tag = "v${ax.version}-p1"; contents = [ env @@ -128,13 +153,13 @@ let }; in (ociLayout { - name = "ax/ax-agent"; + name = "ax/ax-agent${variant}"; tag = "v${ax.version}-p1"; inherit image; }).overrideAttrs (old: { passthru = old.passthru // { - inherit ax-agent image; + inherit ax-agent image piSandbox; }; meta = { description = "ax Task image for the fleet: ax-task-runner, pi, and the ax-agent adapter (OCI layout)"; diff --git a/pkgs/ax-agent-image/elf-sort-load.py b/pkgs/ax-agent-image/elf-sort-load.py new file mode 100644 index 000000000..0af001f78 --- /dev/null +++ b/pkgs/ax-agent-image/elf-sort-load.py @@ -0,0 +1,51 @@ +#!/usr/bin/env python3 +"""Reorder an ELF64 program header table so gVisor's loader accepts it. + +The ELF spec says loadable segments appear in ascending p_vaddr order, and +PT_PHDR and PT_INTERP precede every PT_LOAD. Linux tolerates a table that +breaks this; gVisor's loader refuses it with ENOEXEC, which bash reports as +exit 126. pi 0.85.1 from llm-agents ships such a table: its first PT_LOAD maps +vaddr 0x6431000 and a later one 0x1ff000 (MEASURED 2026-09-23, ax-fleet +INTEGRATE.md). The fix moves table entries only: every segment keeps its file +offset, address, size and flags, and the table keeps its place and length. + +Usage: elf-sort-load.py FILE... (files that are not ELF64 LE, or are already +in order, are left alone; prints one line per file it changed) +""" +import struct +import sys + +PT_LOAD, PT_INTERP, PT_PHDR = 1, 3, 6 + + +def fix(path): + with open(path, "r+b") as f: + ident = f.read(64) + if len(ident) < 64 or ident[:4] != b"\x7fELF" or ident[4] != 2 or ident[5] != 1: + return False + (phoff,) = struct.unpack_from(" {order}") + return True + + +if __name__ == "__main__": + for p in sys.argv[1:]: + fix(p) diff --git a/tests/ax-fleet/default.nix b/tests/ax-fleet/default.nix index 1288feb2b..bff1722cf 100644 --- a/tests/ax-fleet/default.nix +++ b/tests/ax-fleet/default.nix @@ -29,6 +29,9 @@ let TEARDOWN = "${teardown}/bin/ax-fleet-teardown" PROBE_IMAGE = "ax-fleet-probe:test" PROBE_TARBALL = "${nodes.probeImage}" + # Test-only task image variant with claude-code (nodes.nix); its OCI layout + # carries the manifest digest the Task names. + CLAUDE_PROBE_OCI = "${nodes.claudeProbeImage}" LOCAL_PATH_ROOT = "/mnt/nas/services/ax-fleet/local-path" SYSCTLS = [ "net.ipv4.ip_forward", diff --git a/tests/ax-fleet/halogen_stub.py b/tests/ax-fleet/halogen_stub.py index ee14ea1a7..e5efe6f37 100644 --- a/tests/ax-fleet/halogen_stub.py +++ b/tests/ax-fleet/halogen_stub.py @@ -21,6 +21,20 @@ MODEL = "halogen-qwen3.8-flash-next" REPLY = "halogen-stub-ok" +# The pi mode asks for one JSON object against a JSON Schema in its system +# prompt; the adapter's default schema wants {"answer": }. +JSON_REPLY = '{"answer": 42}' + + +def reply_for(req): + """REPLY, or JSON_REPLY when any message mentions a JSON Schema.""" + for msg in req.get("messages") or []: + content = msg.get("content") if isinstance(msg, dict) else None + if isinstance(content, list): + content = " ".join(str(p.get("text", "")) for p in content if isinstance(p, dict)) + if isinstance(content, str) and "JSON Schema" in content: + return JSON_REPLY + return REPLY LOCK = threading.Lock() @@ -86,7 +100,7 @@ def do_POST(self): "choices": [ { "index": 0, - "message": {"role": "assistant", "content": REPLY}, + "message": {"role": "assistant", "content": reply_for(req)}, "finish_reason": "stop", } ], @@ -101,7 +115,7 @@ def do_POST(self): self.end_headers() chunks = [ {"role": "assistant", "content": ""}, - {"content": REPLY}, + {"content": reply_for(req)}, ] for i, delta in enumerate(chunks + [None]): choice = {"index": 0, "delta": delta or {}, "finish_reason": None if delta else "stop"} diff --git a/tests/ax-fleet/nodes.nix b/tests/ax-fleet/nodes.nix index b9454a17b..a9d890763 100644 --- a/tests/ax-fleet/nodes.nix +++ b/tests/ax-fleet/nodes.nix @@ -40,6 +40,20 @@ let ]; }; + # Test-only: the fleet task image (pkgs/ax-agent-image, the same ax and pi + # the fleet seeds) plus claude-code, for 25-harness-probe's `claude + # --version`. No credential exists in it or in any Task. The fleet image + # itself stays pi-only on day one (DESIGN.md 10.2 and 11); only the NAS test + # node seeds this variant, from its ax-on specialisation. + claudeProbeImage = + inputs.nixpkgs.legacyPackages.x86_64-linux.callPackage ../../pkgs/ax-agent-image + { + ax = inputs.self.packages.x86_64-linux.ax; + pi = inputs.llm-agents.packages.x86_64-linux.pi; + extraPaths = [ inputs.llm-agents.packages.x86_64-linux.claude-code ]; + variant = "-claude-probe"; + }; + lanAddr = address: { interface = "eth1"; inherit address; @@ -55,6 +69,10 @@ let { imports = [ ../../modules/ax-fleet + # As on the real hosts (hosts/{coordinator,worker,client}): the harness + # role turns myAxClient on by mkDefault, which puts `ax` and `kubectl` + # on the coordinator's PATH for the Task phases. + ../../modules/ax-client.nix inputs.agenix.nixosModules.default ]; system.switch.enable = true; @@ -91,7 +109,7 @@ let }; in { - inherit probeImage token; + inherit probeImage token claudeProbeImage; nas = { ... }: @@ -105,6 +123,11 @@ in myAxFleet.kubelet = { systemReserved = "cpu=1,memory=1Gi"; }; + myAxFleet.registrySeed.ax-agent-claude-probe = { + oci = claudeProbeImage; + repo = "ax/ax-agent-claude-probe"; + inherit (claudeProbeImage.passthru) tag; + }; }) (setAddr "eth1" "10.42.0.1" 24) ]; diff --git a/tests/ax-fleet/phases/20-substrate.py b/tests/ax-fleet/phases/20-substrate.py index 8a6d742e1..f95a18724 100644 --- a/tests/ax-fleet/phases/20-substrate.py +++ b/tests/ax-fleet/phases/20-substrate.py @@ -13,8 +13,10 @@ "ate-api-server", "ate-controller", "atenet-router", - "podcertificate-controller", ] +# ate-setup installs the pod-certificate controller in its own namespace +# (manifests/ate-install/pod-certificate-controller.yaml), not in ate-system. +SUB_PODCERT_NS = "podcertificate-controller-system" def sub_k(args, timeout=120): @@ -27,10 +29,12 @@ def sub_json(args): with subtest("substrate: bootstrap steps 20-50 ran"): nas.wait_for_unit("ax-fleet-bootstrap.service", timeout=3600) - log = nas.succeed("journalctl -b -u ax-fleet-bootstrap.service --no-pager") - for step in ["20-registry-svc", "30-substrate", "40-gvisor-asset", "50-workerpool"]: - assert f"step {step}" in log, f"bootstrap step {step} did not run" - assert "gvisor asset verified" in log, "40-gvisor-asset did not verify the tarball" + boot_log = nas.succeed("journalctl -b -u ax-fleet-bootstrap.service --no-pager") + # Not `step`: that name is the prelude's receipt context manager, which + # 90-rollback still needs. + for step_name in ["20-registry-svc", "30-substrate", "40-gvisor-asset", "50-workerpool"]: + assert f"step {step_name}" in boot_log, f"bootstrap step {step_name} did not run" + assert "gvisor asset verified" in boot_log, "40-gvisor-asset did not verify the tarball" stamp = sub_k( "-n kube-system get configmap ax-fleet-substrate -o jsonpath='{.data.version}'" ).strip() @@ -51,6 +55,12 @@ def sub_json(args): assert any(n.startswith("atenet-egress") for n in names | sts), "no atenet-egress" assert "postgres" in sts, f"postgres StatefulSet missing: {sorted(sts)}" sub_k(f"-n {SUB_NS} rollout status statefulset/postgres --timeout=600s", timeout=660) + sub_k( + f"-n {SUB_PODCERT_NS} wait --for=condition=Available deploy/podcertificate-controller --timeout=600s", + timeout=660, + ) + for pod in sub_json(f"-n {SUB_PODCERT_NS} get pods")["items"]: + assert pod["spec"].get("nodeName") == "nas", f"{pod['metadata']['name']} on {pod['spec'].get('nodeName')!r}" for pod in sub_json(f"-n {SUB_NS} get pods")["items"]: owner = (pod["metadata"].get("ownerReferences") or [{}])[0].get("kind", "") name = pod["metadata"]["name"] @@ -106,8 +116,8 @@ def sub_json(args): assert len(workers) == 2, f"{len(workers)} worker pods" for p in workers: assert p["spec"]["nodeName"] == "coordinator", p["spec"]["nodeName"] - lim = p["spec"]["containers"][0]["resources"]["limits"]["memory"] - assert lim, "worker pod has no memory limit" + lims = [c.get("resources", {}).get("limits", {}).get("memory") for c in p["spec"]["containers"]] + assert any(lims), f"worker pod has no memory limit: {lims}" with subtest("substrate: gVisor fetched through the RustFS fallback"): # No internet in the VM: atelet's anonymous GCS open of gs://gvisor/... diff --git a/tests/ax-fleet/phases/25-harness-probe.py b/tests/ax-fleet/phases/25-harness-probe.py new file mode 100644 index 000000000..1659e95d6 --- /dev/null +++ b/tests/ax-fleet/phases/25-harness-probe.py @@ -0,0 +1,128 @@ +# Phase 25: the integration's harness probe (ax-fleet INTEGRATE.md). It runs +# after 10-cluster (both switches) and 20-substrate, before the T-cases of +# 30-ax, so its result is in the log even if a later case fails. +# +# One ax Task with spec.sandboxClass gvisor, run by `ax-fleet-smoke probe` +# (the script Tom runs on the real coordinator) in a test-only variant of the +# fleet task image that also carries claude-code. No credential exists in the +# image, the Task or the cluster. The Task runs `claude --version` and one GET +# of the worker's Halogen stand-in, must reach Completed with exit 0, and must +# stay Completed for PROBE_HOLD seconds, sampled every PROBE_EVERY seconds +# (four and more of P1's 15 s resync periods). +# +# There is no containerd RuntimeClass on this path: Substrate's ateom-gvisor +# worker pods run runsc themselves (DESIGN.md D2, 6.2). "gVisor" is proven from +# inside the Task: its /proc/version is not the coordinator VM's kernel. +import json +import re + +PROBE_HOLD = 60 +PROBE_EVERY = 5 +PROBE_AX = "AX_SERVER=http://127.0.0.1:8080 ax -a fleet" +PROBE_STUB_LOG = "/var/lib/halogen-stub/requests.jsonl" + + +def probe_task_phase(name): + # `ax get task NAME` prints YAML; status.phase is its only `phase:` key. + out = coordinator.succeed(f"{PROBE_AX} get task {name}") + m = re.search(r"^\s+phase:\s*(\S+)", out, re.M) + assert m, f"no phase for {name}: {out!r}" + return m.group(1).strip("\"'") + + +def probe_stub_lines(): + return worker.succeed(f"cat {PROBE_STUB_LOG} 2>/dev/null || true").splitlines() + + +with step("probe: k3s nodes Ready, Substrate healthy, ax-server and ax-controller up"): + node_ready("nas") + node_ready("coordinator") + kubectl("-n ate-system wait --for=condition=Available deploy --all --timeout=600s") + kubectl("-n ate-system rollout status statefulset/postgres --timeout=600s") + nas.wait_until_succeeds( + "test \"$(k3s kubectl -n ate-system get workerpool ateom-gvisor -o jsonpath='{.status.readyReplicas}')\" = 2", + timeout=900, + ) + for d in ("ax-redis", "ax-server", "ax-controller"): + kubectl(f"-n ax-system rollout status deploy/{d} --timeout=600s") + coordinator.wait_until_succeeds("curl -sf http://127.0.0.1:8080/healthz", timeout=300) + record("probe_nodes", kubectl("get nodes -o wide").strip().splitlines()) + record("probe_ax_pods", kubectl("-n ax-system get pods -o wide").strip().splitlines()) + workers_wide = kubectl("-n ate-system get pods -l ax.mecattaf.dev/pool=ateom-gvisor -o wide") + record("probe_worker_pods", workers_wide.strip().splitlines()) + worker_nodes = kubectl( + "-n ate-system get pods -l ax.mecattaf.dev/pool=ateom-gvisor " + "-o jsonpath='{range .items[*]}{.spec.nodeName}{\"\\n\"}{end}'" + ).split() + assert worker_nodes and all(n == "coordinator" for n in worker_nodes), worker_nodes + + +with step("probe: a gVisor ax Task on the coordinator runs claude --version and reaches Halogen, Completed and still Completed 60 s later"): + digest = coordinator.succeed(f"cat {CLAUDE_PROBE_OCI}/digest").strip() + assert digest.startswith("sha256:"), digest + image = f"localhost:5000/ax/ax-agent-claude-probe@{digest}" + stub_before = len(probe_stub_lines()) + host_kernel = coordinator.succeed("cat /proc/version").strip() + + rc, out = coordinator.execute( + f"ax-fleet-smoke probe --image {image} --keep --timeout 1500 2>/tmp/probe-smoke.stderr", + timeout=1800, + ) + lines = [l for l in out.strip().splitlines() if l.startswith("{")] + if not lines or rc != 0: + _, err = coordinator.execute("tail -c 4000 /tmp/probe-smoke.stderr") + _, tasks = coordinator.execute(f"{PROBE_AX} get tasks 2>&1 | tail -20") + _, atelet = nas.execute("k3s kubectl -n ate-system logs -l app=atelet --all-containers --tail=80 2>&1") + _, wp = nas.execute("k3s kubectl -n ate-system logs -l ax.mecattaf.dev/pool=ateom-gvisor --all-containers --tail=60 2>&1") + _, ctl = nas.execute("k3s kubectl -n ax-system logs deploy/ax-controller --tail=80 2>&1") + record("probe_diag", {"stderr": err[-4000:], "tasks": tasks[-3000:], "atelet": atelet[-6000:], + "workers": wp[-6000:], "controller": ctl[-6000:]}) + assert lines, f"ax-fleet-smoke probe printed no receipt (rc={rc}): {out!r}" + r = json.loads(lines[-1]) + record("probe_receipt", r) + assert rc == 0 and r.get("pass") is True, r + name = r["task"] + + # Completed, and STAYS Completed: sample for PROBE_HOLD seconds. + samples = [] + t_end = time.monotonic() + PROBE_HOLD + while True: + samples.append(probe_task_phase(name)) + if time.monotonic() >= t_end: + break + time.sleep(PROBE_EVERY) + record("probe_phase_samples", {"hold_seconds": PROBE_HOLD, "every_seconds": PROBE_EVERY, "phases": samples}) + assert len(samples) >= PROBE_HOLD // PROBE_EVERY and all(p == "Completed" for p in samples), samples + record("probe_task_yaml", coordinator.succeed(f"{PROBE_AX} get task {name}").strip().splitlines()) + + res = r["result"] + # claude-code answered inside the sandbox, with no credential anywhere. + assert res["claude_rc"] == 0 and re.search(r"\d+\.\d+\.\d+", res["claude_version"]), res + # The Halogen stand-in on the worker VM answered through the Gateway. + assert res["curl_rc"] == 0 and res["http_code"] == "200", res + assert res["model"] == "halogen-qwen3.8-flash-next", res + # gVisor, not runc: a runc container would report the VM's own kernel. + record("probe_kernels", {"sandbox": res["proc_version"], "coordinator_vm": host_kernel}) + assert res["proc_version"] and res["proc_version"] != host_kernel, (res["proc_version"], host_kernel) + + new = [json.loads(l) for l in probe_stub_lines()[stub_before:] if l.strip()] + models = [q for q in new if q.get("path", "").rstrip("/") == "/v1/models"] + record("probe_stub_requests", models) + assert models, f"the Halogen stand-in logged no /v1/models request: {new!r}" + + # The image was pulled by atelet, which runs on the coordinator only. + atelet_logs = kubectl("-n ate-system logs -l app=atelet --all-containers --tail=-1") + hits = [ + l for l in atelet_logs.splitlines() if "ax-agent-claude-probe" in l or digest.split(":", 1)[1][:16] in l + ] + # Recorded, not asserted: the placement proof is the pool (every worker pod + # on the coordinator, asserted in the first step) and the sandbox kernel above. + record("probe_atelet_image_lines", hits[-5:]) + # '[r]unsc' so the pattern does not match the shell that runs pgrep. + record("probe_runsc_processes", { + "coordinator": coordinator.succeed("pgrep -c -f '[r]unsc' || true").strip(), + "nas": nas.succeed("pgrep -c -f '[r]unsc' || true").strip(), + }) + nas.fail("pgrep -f '[r]unsc'") + + coordinator.succeed(f"{PROBE_AX} delete task {name}") From 806589781dd18b7413d368163d8c487946f03de0 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 07:41:14 +0200 Subject: [PATCH 20/37] flake: drop #448's parakeet socket asserts that rode in without #448 (DF-6) a775d58f (remove the SessionEnd harvest hook) also carried the home-profiles asserts from f1d689b2 (#448, parakeet socket activation), which this branch does not contain. The coordinator home here still has parakeet-service with Install.WantedBy = [ "default.target" ] and no socket, so checks.home-profiles failed with `attribute 'parakeet-service' missing` at flake.nix:2036. Restore main's asserts for this check. The socket asserts belong in #448 and return when it merges. No module or host change; only the check. Co-Authored-By: Claude Opus 5.5 --- flake.nix | 10 +++------- 1 file changed, 3 insertions(+), 7 deletions(-) diff --git a/flake.nix b/flake.nix index 7df7b4def..17e9e127e 100644 --- a/flake.nix +++ b/flake.nix @@ -2248,12 +2248,10 @@ assert coordinatorHome.services.tally.enable; # Direct Parakeet is coordinator-only; capture has no virtual mic or # service-start download. Native Herdr owns client key handling. - # #448: the socket is the ONLY activation path — a service with its - # own Install would go resident again behind the config's back. + # The #448 socket-activation asserts rode in on the harvest-hook + # commit without #448 itself; they return with #448 (DF-6). assert - coordinatorHome.systemd.user.sockets.parakeet-service.Install.WantedBy == [ "sockets.target" ]; - assert !(coordinatorHome.systemd.user.services.parakeet-service ? Install); - assert coordinatorHome.systemd.user.services.parakeet-service.Service.Restart == "no"; + coordinatorHome.systemd.user.services.parakeet-service.Install.WantedBy == [ "default.target" ]; assert !(coordinatorHome.systemd.user.services ? voxtype); assert !(coordinatorHome.xdg.configFile ? "pipewire/pipewire.conf.d/60-client-mic.conf"); assert !(coordinatorHome.systemd.user.services.parakeet-service.Service ? ExecStartPre); @@ -2339,7 +2337,6 @@ assert workerHome.programs.atuin.settings.auto_sync; assert !workerHome.services.tally.enable; assert !(workerHome.systemd.user.services ? parakeet-service); - assert !(workerHome.systemd.user.sockets ? parakeet-service); assert !(workerHome.systemd.user.services ? voxtype); # …and the herdr SERVER. The worker still gets the herdr binary (it is # how `herdr --remote coordinator` works at all), just no unit. @@ -2420,7 +2417,6 @@ assert !(clientHome.xdg.configFile ? "voxtype/config.toml"); assert !(clientHome.systemd.user.services ? voxtype); assert !(clientHome.systemd.user.services ? parakeet-service); - assert !(clientHome.systemd.user.sockets ? parakeet-service); assert clientHome.systemd.user.services.speech-wake.Install.WantedBy == [ "default.target" ]; assert !(builtins.any (p: nixpkgs.lib.getName p == "dictate-hold") clientHome.home.packages); assert nixpkgs.lib.hasInfix "HERDR_DICTATION_COMMAND=speech-dictate" ( From 95b065df70b122c39540cb028f716951fa3e6b8d Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 10:51:53 +0200 Subject: [PATCH 21/37] probe(ax-fleet): no-P1 variant, completion to a stand-in floor, driver deletes the Task Probe branch, not for merge. pkgs/ax builds v0.3.0 with sandbox-class.patch only (p1-completion.patch removed) and ax-controller runs without --running-resync. The VM test keeps phases 10 and 20 and replaces the rest with 30-nop1: each Task's command POSTs its result to a floor stand-in on the worker stub, and the driver (standing in for the link) deletes the Task through the stock ax client. Test-only kubectl-ate on the nas node for actor, template and worker listings. Co-Authored-By: Claude Opus 5.5 --- modules/ax-fleet/ax.nix | 2 +- pkgs/ax/default.nix | 3 +- pkgs/ax/patches/p1-completion.patch | 3545 --------------------- tests/ax-fleet/halogen_stub.py | 22 + tests/ax-fleet/nodes.nix | 18 +- tests/ax-fleet/phases/25-harness-probe.py | 128 - tests/ax-fleet/phases/30-ax.py | 118 - tests/ax-fleet/phases/30-nop1.py | 358 +++ tests/ax-fleet/phases/90-rollback.py | 52 - 9 files changed, 400 insertions(+), 3846 deletions(-) delete mode 100644 pkgs/ax/patches/p1-completion.patch delete mode 100644 tests/ax-fleet/phases/25-harness-probe.py delete mode 100644 tests/ax-fleet/phases/30-ax.py create mode 100644 tests/ax-fleet/phases/30-nop1.py delete mode 100644 tests/ax-fleet/phases/90-rollback.py diff --git a/modules/ax-fleet/ax.nix b/modules/ax-fleet/ax.nix index efaa19dfc..ca9f02854 100644 --- a/modules/ax-fleet/ax.nix +++ b/modules/ax-fleet/ax.nix @@ -349,7 +349,7 @@ let "--substrate-ca-file=/run/servicedns-ca/trust-bundle.pem" "--template=default-template" "--template-atespace=ax-system" - "--running-resync=${toString cfg.ax.runningResyncSeconds}s" + # probe/ax-fleet-nop1: stock v0.3.0 has no --running-resync (P1 removed). ]; env = [ { diff --git a/pkgs/ax/default.nix b/pkgs/ax/default.nix index 244a1ffea..c4d0864ee 100644 --- a/pkgs/ax/default.nix +++ b/pkgs/ax/default.nix @@ -98,7 +98,8 @@ buildGo127Module { # that without an exit report the third is refused ResourceExhausted. patches = [ ./patches/sandbox-class.patch - ./patches/p1-completion.patch + # probe/ax-fleet-nop1: p1-completion.patch removed; completion is reported + # by the Task to the floor and the link deletes the Task. ]; # subPackages left unset so all four commands build, matching upstream's diff --git a/pkgs/ax/patches/p1-completion.patch b/pkgs/ax/patches/p1-completion.patch deleted file mode 100644 index 9176e8464..000000000 --- a/pkgs/ax/patches/p1-completion.patch +++ /dev/null @@ -1,3545 +0,0 @@ -diff --git a/cmd/ax-controller/main.go b/cmd/ax-controller/main.go -index 9d1020e..f882e2d 100644 ---- a/cmd/ax-controller/main.go -+++ b/cmd/ax-controller/main.go -@@ -21,6 +21,7 @@ import ( - "os" - "os/signal" - "syscall" -+ "time" - - "github.com/google/ax/internal/controller" - "github.com/google/ax/internal/store/redis" -@@ -42,6 +43,7 @@ func main() { - redisPassword string - redisGroup string - redisConsumer string -+ runningResync time.Duration - ) - - flag.StringVar(&redisAddr, "redis-addr", "localhost:6379", "Redis server address (e.g. localhost:6379)") -@@ -56,6 +58,7 @@ func main() { - flag.BoolVar(&substratePlaintext, "substrate-plaintext", false, "Use insecure plaintext gRPC connection to Substrate") - flag.StringVar(&defaultTemplate, "template", "default-template", "Default Substrate ActorTemplate name") - flag.StringVar(&defaultTemplateAtespace, "template-atespace", "ax-system", "Default Substrate ActorTemplate atespace") -+ flag.DurationVar(&runningResync, "running-resync", controller.DefaultRunningResync, "How often Running tasks are re-checked for a command exit (0 disables)") - flag.Parse() - - if envRedis := os.Getenv("REDIS_ADDR"); envRedis != "" { -@@ -104,6 +107,7 @@ func main() { - - rStore := redis.NewStore(rClient, redis.Options{}) - worker := controller.NewWorker(rStore, reconciler, redisGroup, redisConsumer) -+ worker.RunningResync = runningResync - if err := worker.Run(ctx); err != nil && err != context.Canceled { - slog.Error("redis worker stopped with error", "error", err) - os.Exit(1) -diff --git a/cmd/ax/main.go b/cmd/ax/main.go -index 7b2922e..fff6379 100644 ---- a/cmd/ax/main.go -+++ b/cmd/ax/main.go -@@ -147,6 +147,8 @@ func main() { - err = runDelete(serverURL, atespace, cleanArgs) - case "ssh": - err = runSSH(serverURL, atespace, kubeContext, cleanArgs) -+ case "result": -+ err = runResult(serverURL, atespace, cleanArgs) - default: - fmt.Fprintf(os.Stderr, "unknown command: %s\n", cmd) - printUsage() -@@ -183,6 +185,7 @@ Available Commands: - ssh [-- cmd] Run a command or shell inside the running task container - suspend task Suspend execution of a task and checkpoint state - resume task Resume execution of a suspended task -+ result task Print the result file a finished task's command wrote - delete task Delete a task - delete gateway Delete a gateway - delete workspace Delete a workspace -@@ -1228,3 +1231,33 @@ func runSSH(serverURL, atespace, kubeContext string, args []string) error { - - return nil - } -+ -+// runResult prints the result content the controller copied from a finished -+// task's sandbox, byte for byte, to stdout. -+func runResult(serverURL, atespace string, args []string) error { -+ name := "" -+ switch { -+ case len(args) == 1: -+ name = args[0] -+ case len(args) >= 2 && (args[0] == "task" || args[0] == "tasks"): -+ name = args[1] -+ default: -+ return fmt.Errorf("usage: ax result task ") -+ } -+ -+ client, conn, err := getAXClient(serverURL) -+ if err != nil { -+ return err -+ } -+ defer conn.Close() -+ -+ ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second) -+ defer cancel() -+ -+ res, err := client.GetTaskResult(ctx, &v1alpha1.GetTaskResultRequest{Atespace: atespace, Name: name}) -+ if err != nil { -+ return fmt.Errorf("getting task result: %w", err) -+ } -+ _, err = os.Stdout.Write(res.GetContent()) -+ return err -+} -diff --git a/internal/controller/completion_test.go b/internal/controller/completion_test.go -new file mode 100644 -index 0000000..d286b42 ---- /dev/null -+++ b/internal/controller/completion_test.go -@@ -0,0 +1,396 @@ -+// Copyright 2026 Google LLC -+// -+// Licensed under the Apache License, Version 2.0 (the "License"); -+// you may not use this file except in compliance with the License. -+// You may obtain a copy of the License at -+// -+// http://www.apache.org/licenses/LICENSE-2.0 -+// -+// Unless required by applicable law or agreed to in writing, software -+// distributed under the License is distributed on an "AS IS" BASIS, -+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -+// See the License for the specific language governing permissions and -+// limitations under the License. -+ -+package controller_test -+ -+import ( -+ "context" -+ "encoding/json" -+ "fmt" -+ "net" -+ "net/http" -+ "net/http/httptest" -+ "sync" -+ "sync/atomic" -+ "testing" -+ "time" -+ -+ "github.com/agent-substrate/substrate/pkg/proto/ateapipb" -+ "github.com/google/ax/internal/controller" -+ "github.com/google/ax/internal/metadata" -+ "github.com/google/ax/internal/store" -+ "github.com/google/ax/internal/store/memory" -+ "github.com/google/ax/internal/substrate" -+ "github.com/google/ax/pkg/apis/v1alpha1" -+ "google.golang.org/grpc" -+ "google.golang.org/grpc/codes" -+ "google.golang.org/grpc/credentials/insecure" -+ "google.golang.org/grpc/status" -+) -+ -+// poolControl is a Substrate mock with a fixed number of workers. A resumed -+// actor holds a worker until it is suspended or deleted; with every worker -+// held, ResumeActor fails ResourceExhausted, as the real scheduler does. -+type poolControl struct { -+ ateapipb.UnimplementedControlServer -+ mu sync.Mutex -+ workers int -+ workerIP string -+ running map[string]bool -+ crashed map[string]bool -+ resumes map[string]int -+ suspends map[string]int -+} -+ -+func newPoolControl(workers int, workerIP string) *poolControl { -+ return &poolControl{workers: workers, workerIP: workerIP, running: map[string]bool{}, crashed: map[string]bool{}, resumes: map[string]int{}, suspends: map[string]int{}} -+} -+ -+func (m *poolControl) CreateAtespace(ctx context.Context, req *ateapipb.CreateAtespaceRequest) (*ateapipb.Atespace, error) { -+ return &ateapipb.Atespace{Metadata: req.GetAtespace().GetMetadata()}, nil -+} -+ -+func (m *poolControl) CreateActorTemplate(ctx context.Context, req *ateapipb.CreateActorTemplateRequest) (*ateapipb.ActorTemplate, error) { -+ return req.GetActorTemplate(), nil -+} -+ -+func (m *poolControl) GetActorTemplate(ctx context.Context, req *ateapipb.GetActorTemplateRequest) (*ateapipb.ActorTemplate, error) { -+ return &ateapipb.ActorTemplate{Metadata: &ateapipb.ResourceMetadata{Atespace: req.GetActorTemplate().GetAtespace(), Name: req.GetActorTemplate().GetName()}}, nil -+} -+ -+func (m *poolControl) CreateActor(ctx context.Context, req *ateapipb.CreateActorRequest) (*ateapipb.Actor, error) { -+ return &ateapipb.Actor{Metadata: req.GetActor().GetMetadata(), Status: &ateapipb.ActorStatus{State: ateapipb.ActorState_ACTOR_STATE_SUSPENDED}}, nil -+} -+ -+func (m *poolControl) CreateActorEgressPolicy(ctx context.Context, req *ateapipb.CreateActorEgressPolicyRequest) (*ateapipb.EgressPolicy, error) { -+ return &ateapipb.EgressPolicy{}, nil -+} -+ -+func (m *poolControl) GetActor(ctx context.Context, req *ateapipb.GetActorRequest) (*ateapipb.Actor, error) { -+ m.mu.Lock() -+ defer m.mu.Unlock() -+ name := req.GetActor().GetName() -+ state := ateapipb.ActorState_ACTOR_STATE_SUSPENDED -+ switch { -+ case m.crashed[name]: -+ state = ateapipb.ActorState_ACTOR_STATE_CRASHED -+ case m.running[name]: -+ state = ateapipb.ActorState_ACTOR_STATE_RUNNING -+ } -+ return &ateapipb.Actor{Metadata: &ateapipb.ResourceMetadata{Name: name}, Status: &ateapipb.ActorStatus{State: state}}, nil -+} -+ -+func (m *poolControl) ResumeActor(ctx context.Context, req *ateapipb.ResumeActorRequest) (*ateapipb.ResumeActorResponse, error) { -+ m.mu.Lock() -+ defer m.mu.Unlock() -+ name := req.GetActor().GetName() -+ m.resumes[name]++ -+ if !m.running[name] { -+ if len(m.running) >= m.workers { -+ return nil, status.Error(codes.ResourceExhausted, "no free workers available") -+ } -+ m.running[name] = true -+ } -+ return &ateapipb.ResumeActorResponse{Actor: &ateapipb.Actor{ -+ Metadata: &ateapipb.ResourceMetadata{Name: name}, -+ Status: &ateapipb.ActorStatus{ -+ State: ateapipb.ActorState_ACTOR_STATE_RUNNING, -+ WorkerAssignment: &ateapipb.WorkerAssignment{WorkerPod: "w", WorkerPodIp: m.workerIP}, -+ }, -+ }, Resumed: true}, nil -+} -+ -+func (m *poolControl) SuspendActor(ctx context.Context, req *ateapipb.SuspendActorRequest) (*ateapipb.SuspendActorResponse, error) { -+ m.mu.Lock() -+ defer m.mu.Unlock() -+ name := req.GetActor().GetName() -+ m.suspends[name]++ -+ delete(m.running, name) -+ return &ateapipb.SuspendActorResponse{}, nil -+} -+ -+func (m *poolControl) resumeCount(name string) int { -+ m.mu.Lock() -+ defer m.mu.Unlock() -+ return m.resumes[name] -+} -+ -+// fakeRunner serves the runner's metadata endpoints. exited toggles whether -+// the command has exited; exitCode is what it reports. -+type fakeRunner struct { -+ exited atomic.Bool -+ exitCode atomic.Int32 -+ result string -+} -+ -+func (f *fakeRunner) handler() http.Handler { -+ mux := http.NewServeMux() -+ mux.HandleFunc("/readyz", func(w http.ResponseWriter, r *http.Request) { w.WriteHeader(http.StatusOK) }) -+ mux.HandleFunc("/metadata/v1alpha1/ax/exit", func(w http.ResponseWriter, r *http.Request) { -+ _ = json.NewEncoder(w).Encode(metadata.CommandExitStatus{Exited: f.exited.Load(), ExitCode: int(f.exitCode.Load()), FinishedAt: time.Now()}) -+ }) -+ mux.HandleFunc("/metadata/v1alpha1/ax/result", func(w http.ResponseWriter, r *http.Request) { -+ if !f.exited.Load() { -+ http.Error(w, "not exited", http.StatusConflict) -+ return -+ } -+ if f.result == "" { -+ http.NotFound(w, r) -+ return -+ } -+ _, _ = w.Write([]byte(f.result)) -+ }) -+ mux.HandleFunc("/metadata/v1alpha1/ax/usage", func(w http.ResponseWriter, r *http.Request) { -+ _, _ = w.Write([]byte(`{"prompt_tokens":11,"completion_tokens":7,"tool_calls":2}`)) -+ }) -+ return mux -+} -+ -+type harness struct { -+ pool *poolControl -+ runner *fakeRunner -+ reconciler *controller.TaskReconciler -+ store *memory.MemoryStore -+} -+ -+func newHarness(t *testing.T, workers int) *harness { -+ t.Helper() -+ fr := &fakeRunner{result: `{"answer":42}`} -+ hs := httptest.NewServer(fr.handler()) -+ t.Cleanup(hs.Close) -+ pool := newPoolControl(workers, hs.Listener.Addr().String()) -+ -+ lis, err := net.Listen("tcp", "127.0.0.1:0") -+ if err != nil { -+ t.Fatal(err) -+ } -+ gs := grpc.NewServer() -+ ateapipb.RegisterControlServer(gs, pool) -+ go gs.Serve(lis) -+ t.Cleanup(gs.Stop) -+ client, err := substrate.NewClient(lis.Addr().String(), grpc.WithTransportCredentials(insecure.NewCredentials())) -+ if err != nil { -+ t.Fatal(err) -+ } -+ t.Cleanup(func() { client.Close() }) -+ r := controller.NewTaskReconciler(client, "default-template", "ax-system") -+ r.SecretResolver = noSecrets -+ r.WorkspaceReadyTimeout = 100 * time.Millisecond -+ return &harness{pool: pool, runner: fr, reconciler: r, store: memory.NewStore()} -+} -+ -+func newTask(name string) *v1alpha1.Task { -+ return &v1alpha1.Task{ -+ Metadata: &v1alpha1.ObjectMeta{Name: name, Atespace: "fleet"}, -+ Spec: &v1alpha1.TaskSpec{Image: "localhost:5000/ax/ax-agent@sha256:00", Command: []string{"true"}}, -+ } -+} -+ -+func readyReason(task *v1alpha1.Task) string { -+ for _, c := range task.GetStatus().GetConditions() { -+ if c.GetType() == "Ready" { -+ return c.GetReason() -+ } -+ } -+ return "" -+} -+ -+func TestReconcile_ExitWriteBack(t *testing.T) { -+ ctx := context.Background() -+ for _, tc := range []struct { -+ code int32 -+ wantPhase string -+ }{{0, "Completed"}, {3, "Failed"}} { -+ t.Run(tc.wantPhase, func(t *testing.T) { -+ h := newHarness(t, 2) -+ h.runner.exited.Store(true) -+ h.runner.exitCode.Store(tc.code) -+ var saved []byte -+ h.reconciler.SaveResult = func(ctx context.Context, atespace, name string, content []byte) error { -+ saved = content -+ return nil -+ } -+ task, err := h.reconciler.Reconcile(ctx, newTask("t"), nil) -+ if err != nil { -+ t.Fatal(err) -+ } -+ if task.Status.Phase != tc.wantPhase { -+ t.Fatalf("phase = %q, want %q", task.Status.Phase, tc.wantPhase) -+ } -+ if readyReason(task) != "CommandExited" { -+ t.Fatalf("Ready reason = %q", readyReason(task)) -+ } -+ if want := fmt.Sprintf("ExitCode=%d", tc.code); task.Status.Conditions[len(task.Status.Conditions)-1].GetMessage() != want && !hasMessage(task, want) { -+ t.Fatalf("no condition message %q", want) -+ } -+ if c := task.Status.GetCommand(); !c.GetExited() || c.GetExitCode() != tc.code || c.GetResultBytes() != int64(len(`{"answer":42}`)) || c.GetResultSha256() == "" { -+ t.Fatalf("command status = %v", c) -+ } -+ if string(saved) != `{"answer":42}` { -+ t.Fatalf("saved result = %q", saved) -+ } -+ if u := task.Status.GetUsage(); u.GetPromptTokens() != 11 || u.GetToolCalls() != 2 { -+ t.Fatalf("usage = %v", u) -+ } -+ if h.pool.suspends["t"] != 1 || task.Status.WorkerIp != "" { -+ t.Fatalf("worker not freed: suspends=%d workerIP=%q", h.pool.suspends["t"], task.Status.WorkerIp) -+ } -+ if !controller.IsTerminal(task) { -+ t.Fatal("finished task is not terminal") -+ } -+ }) -+ } -+} -+ -+func hasMessage(task *v1alpha1.Task, msg string) bool { -+ for _, c := range task.GetStatus().GetConditions() { -+ if c.GetMessage() == msg { -+ return true -+ } -+ } -+ return false -+} -+ -+// The probe measured that stock v0.3.0 turns a Completed write-back into -+// Running and resumes the actor again. With the terminal guard it stays put. -+func TestWorker_CompletedWriteBackIsNotOverwritten(t *testing.T) { -+ ctx, cancel := context.WithCancel(context.Background()) -+ defer cancel() -+ h := newHarness(t, 2) -+ task := newTask("done") -+ task.Status = &v1alpha1.TaskStatus{Phase: "Completed"} -+ if err := h.store.SaveTask(ctx, task); err != nil { -+ t.Fatal(err) -+ } -+ w := controller.NewWorker(h.store, h.reconciler, "g", "c") -+ go func() { _ = w.Run(ctx) }() -+ time.Sleep(300 * time.Millisecond) -+ got, err := h.store.GetTask(ctx, "fleet", "done") -+ if err != nil { -+ t.Fatal(err) -+ } -+ if got.Status.Phase != "Completed" { -+ t.Fatalf("phase = %q, want Completed", got.Status.Phase) -+ } -+ if n := h.pool.resumeCount("done"); n != 0 { -+ t.Fatalf("terminal task resumed %d times", n) -+ } -+} -+ -+// runSequence applies n tasks one after another through a worker, each command -+// exiting after its first resume, and returns the final phase of each. -+func runSequence(t *testing.T, h *harness, n int) []*v1alpha1.Task { -+ t.Helper() -+ ctx, cancel := context.WithCancel(context.Background()) -+ defer cancel() -+ w := controller.NewWorker(h.store, h.reconciler, "g", "c") -+ w.RunningResync = 50 * time.Millisecond -+ go func() { _ = w.Run(ctx) }() -+ var out []*v1alpha1.Task -+ for i := 0; i < n; i++ { -+ name := fmt.Sprintf("floor-%d", i) -+ if err := h.store.SaveTask(ctx, newTask(name)); err != nil { -+ t.Fatal(err) -+ } -+ var got *v1alpha1.Task -+ deadline := time.Now().Add(5 * time.Second) -+ for time.Now().Before(deadline) { -+ tk, err := h.store.GetTask(ctx, "fleet", name) -+ if err == nil && tk.Status.GetPhase() != "Pending" && tk.Status.GetPhase() != "Running" { -+ got = tk -+ break -+ } -+ if err == nil && tk.Status.GetPhase() == "Running" && !h.runner.exited.Load() { -+ // Without an exit report a Running task never finishes. -+ got = tk -+ break -+ } -+ time.Sleep(20 * time.Millisecond) -+ } -+ if got == nil { -+ t.Fatalf("%s did not settle", name) -+ } -+ out = append(out, got) -+ } -+ return out -+} -+ -+// Four Tasks in a row on a 2-worker pool: every one reaches a terminal phase -+// and none fails ResourceExhausted, because each finished Task frees its worker. -+func TestWorker_FloorFourOnTwoWorkers(t *testing.T) { -+ h := newHarness(t, 2) -+ h.runner.exited.Store(true) -+ for _, tk := range runSequence(t, h, 4) { -+ if tk.Status.Phase != "Completed" { -+ t.Fatalf("%s: phase %q reason %q, want Completed", tk.Metadata.Name, tk.Status.Phase, readyReason(tk)) -+ } -+ } -+} -+ -+// The contrast the probe measured on stock v0.3.0: when no exit is reported, -+// finished work keeps its worker and the third Task on 2 workers is refused. -+func TestWorker_NoExitReportExhaustsPool(t *testing.T) { -+ h := newHarness(t, 2) -+ tasks := runSequence(t, h, 3) -+ if tasks[2].Status.Phase != "Failed" || readyReason(tasks[2]) != "ActorResumeFailed" { -+ t.Fatalf("third task: phase %q reason %q, want Failed ActorResumeFailed", tasks[2].Status.Phase, readyReason(tasks[2])) -+ } -+} -+ -+func TestWorker_ResyncFinishesRunningTask(t *testing.T) { -+ ctx := context.Background() -+ h := newHarness(t, 2) -+ w := controller.NewWorker(h.store, h.reconciler, "g", "c") -+ task, err := h.reconciler.Reconcile(ctx, newTask("slow"), nil) -+ if err != nil || task.Status.Phase != "Running" { -+ t.Fatalf("reconcile: %v phase %q", err, task.GetStatus().GetPhase()) -+ } -+ if err := h.store.SaveTask(ctx, task); err != nil { -+ t.Fatal(err) -+ } -+ if n := w.ResyncOnce(ctx); n != 0 { -+ t.Fatalf("resync changed %d tasks before exit", n) -+ } -+ h.runner.exited.Store(true) -+ if n := w.ResyncOnce(ctx); n != 1 { -+ t.Fatalf("resync changed %d tasks after exit, want 1", n) -+ } -+ got, _ := h.store.GetTask(ctx, "fleet", "slow") -+ if got.Status.Phase != "Completed" { -+ t.Fatalf("phase = %q", got.Status.Phase) -+ } -+ if b, err := h.store.GetTaskResult(ctx, "fleet", "slow"); err != nil || string(b) != `{"answer":42}` { -+ t.Fatalf("stored result = %q, %v", b, err) -+ } -+ if _, err := h.store.GetTaskResult(ctx, "fleet", "nope"); err != store.ErrNotFound { -+ t.Fatalf("missing result err = %v", err) -+ } -+} -+ -+func TestResync_CrashedActorIsFailed(t *testing.T) { -+ ctx := context.Background() -+ h := newHarness(t, 2) -+ task, err := h.reconciler.Reconcile(ctx, newTask("boom"), nil) -+ if err != nil { -+ t.Fatal(err) -+ } -+ h.pool.mu.Lock() -+ h.pool.crashed["boom"] = true -+ h.pool.mu.Unlock() -+ task.Status.WorkerIp = "127.0.0.1:1" // runner unreachable -+ changed, _ := h.reconciler.ResyncRunning(ctx, task) -+ if !changed || task.Status.Phase != "Failed" || readyReason(task) != "ActorCrashed" { -+ t.Fatalf("changed=%v phase=%q reason=%q", changed, task.Status.Phase, readyReason(task)) -+ } -+} -diff --git a/internal/controller/reconciler.go b/internal/controller/reconciler.go -index c4aebda..a4692ca 100644 ---- a/internal/controller/reconciler.go -+++ b/internal/controller/reconciler.go -@@ -17,8 +17,11 @@ package controller - import ( - "context" - "crypto/sha256" -+ "encoding/hex" -+ "encoding/json" - "errors" - "fmt" -+ "io" - "log/slog" - "net" - "net/http" -@@ -28,6 +31,8 @@ import ( - "strings" - "time" - -+ "github.com/agent-substrate/substrate/pkg/proto/ateapipb" -+ "github.com/google/ax/internal/metadata" - "github.com/google/ax/internal/model" - "github.com/google/ax/internal/substrate" - "github.com/google/ax/pkg/apis/v1alpha1" -@@ -70,6 +75,10 @@ type TaskReconciler struct { - // WorkspaceReadyTimeout bounds how long Reconcile waits for the actor's workspace - // to report ready before recording it as still initializing. - WorkspaceReadyTimeout time.Duration -+ -+ // SaveResult stores the result content copied from a finished task's -+ // sandbox. Nil drops the content; the digest and size still reach status. -+ SaveResult func(ctx context.Context, atespace, name string, content []byte) error - } - - // NewTaskReconciler creates a new TaskReconciler. -@@ -111,6 +120,14 @@ func (r *TaskReconciler) Reconcile(ctx context.Context, task *v1alpha1.Task, gat - task.Spec.Image = v1alpha1.DefaultTaskImage - } - -+ // Terminal guard: a task whose command already exited is never resumed -+ // again. Without it every later event (an apply, a status write-back that -+ // republishes) would resume the actor and flip the phase back to Running. -+ if IsTerminal(task) { -+ slog.Info("task is terminal, not reconciling", "name", task.Metadata.Name, "atespace", atespace, "phase", task.Status.Phase) -+ return task, nil -+ } -+ - slog.Info("reconciling task", "name", task.Metadata.Name, "atespace", atespace, "image", task.Spec.Image) - - now := time.Now() -@@ -293,6 +310,15 @@ func (r *TaskReconciler) Reconcile(ctx context.Context, task *v1alpha1.Task, gat - } - DonePolling: - -+ // A short command may already have finished; if so, record it now. -+ if workerIP != "" { -+ if done, err := r.finishIfExited(ctx, task, host, port); err != nil { -+ slog.Warn("could not read command exit", "name", task.Metadata.Name, "error", err) -+ } else if done { -+ return task, nil -+ } -+ } -+ - // The task is Ready only once its actor is running and the workspace inside it is set up. - if workspaceReady { - if !r.conditionTrue(task, condWorkspaceReady) { -@@ -317,6 +343,11 @@ DonePolling: - - // Condition types reported on Task status. - const ( -+ // reasonCommandExited is the Ready reason on a task whose command exited. -+ reasonCommandExited = "CommandExited" -+ // reasonActorCrashed is the Ready reason on a task whose actor crashed. -+ reasonActorCrashed = "ActorCrashed" -+ - // condReady reports whether the task as a whole is ready to do work: its actor is - // running and the workspace inside it has finished setting up. - condReady = "Ready" -@@ -483,3 +514,172 @@ func marshalWorkspaces(workspaces []*v1alpha1.Workspace) (string, error) { - } - return sb.String(), nil - } -+ -+// IsTerminal reports whether the task's command has finished: phase Completed, -+// or Failed with the Ready reason CommandExited. Such a task is never resumed. -+func IsTerminal(task *v1alpha1.Task) bool { -+ st := task.GetStatus() -+ switch st.GetPhase() { -+ case "Completed": -+ return true -+ case "Failed": -+ for _, c := range st.GetConditions() { -+ if c.GetType() == condReady && c.GetReason() == reasonCommandExited { -+ return true -+ } -+ } -+ } -+ return false -+} -+ -+// ResyncRunning re-checks a Running task, because no event fires when its -+// command exits. It returns true when it changed the task's status: the command -+// exited (Completed or Failed, worker freed) or the actor crashed (Failed). -+func (r *TaskReconciler) ResyncRunning(ctx context.Context, task *v1alpha1.Task) (bool, error) { -+ if task.GetStatus().GetPhase() != "Running" || task.GetMetadata() == nil || IsTerminal(task) { -+ return false, nil -+ } -+ atespace := task.Metadata.Atespace -+ if atespace == "" { -+ atespace = "default" -+ } -+ host, port := task.Status.WorkerIp, "80" -+ if h, p, err := net.SplitHostPort(host); err == nil { -+ host, port = h, p -+ } -+ done, exitErr := r.finishIfExited(ctx, task, host, port) -+ if done { -+ return true, nil -+ } -+ // No exit report: a crashed actor is Failed, not restarted behind the -+ // caller's back. -+ state, err := r.client.ActorState(ctx, atespace, task.Metadata.Name) -+ if err == nil && state == ateapipb.ActorState_ACTOR_STATE_CRASHED { -+ task.Status.Phase = "Failed" -+ task.Status.WorkerIp = "" -+ r.setNotReady(task, reasonActorCrashed, "Substrate reports the actor CRASHED", time.Now()) -+ return true, nil -+ } -+ if exitErr != nil { -+ return false, exitErr -+ } -+ return false, err -+} -+ -+// finishIfExited asks the runner whether the command exited. If it did, it -+// copies the result and usage, writes the outcome into status, and suspends -+// the actor so its worker is free for the next task. -+func (r *TaskReconciler) finishIfExited(ctx context.Context, task *v1alpha1.Task, host, port string) (bool, error) { -+ atespace := task.Metadata.Atespace -+ actorName := task.Metadata.Name -+ body, err := r.metadataGet(ctx, atespace, actorName, host, port, "/metadata/v1alpha1/ax/exit") -+ if err != nil { -+ return false, err -+ } -+ var exit metadata.CommandExitStatus -+ if err := json.Unmarshal(body, &exit); err != nil { -+ return false, fmt.Errorf("decoding exit status: %w", err) -+ } -+ if !exit.Exited { -+ return false, nil -+ } -+ -+ now := time.Now() -+ cmdStatus := &v1alpha1.CommandStatus{Exited: true, ExitCode: int32(exit.ExitCode)} -+ if !exit.FinishedAt.IsZero() { -+ cmdStatus.FinishedAt = timestamppb.New(exit.FinishedAt) -+ } else { -+ cmdStatus.FinishedAt = timestamppb.New(now) -+ } -+ if result, err := r.metadataGet(ctx, atespace, actorName, host, port, "/metadata/v1alpha1/ax/result"); err == nil { -+ sum := sha256.Sum256(result) -+ cmdStatus.ResultBytes = int64(len(result)) -+ cmdStatus.ResultSha256 = hex.EncodeToString(sum[:]) -+ if r.SaveResult != nil { -+ if err := r.SaveResult(ctx, atespace, actorName, result); err != nil { -+ slog.Warn("could not store task result", "name", actorName, "error", err) -+ } -+ } -+ } else { -+ slog.Info("task produced no result", "name", actorName, "reason", err) -+ } -+ if usageBody, err := r.metadataGet(ctx, atespace, actorName, host, port, "/metadata/v1alpha1/ax/usage"); err == nil { -+ var u struct { -+ PromptTokens int32 `json:"prompt_tokens"` -+ CompletionTokens int32 `json:"completion_tokens"` -+ ToolCalls int32 `json:"tool_calls"` -+ } -+ if json.Unmarshal(usageBody, &u) == nil { -+ task.Status.Usage = &v1alpha1.UsageStats{PromptTokens: u.PromptTokens, CompletionTokens: u.CompletionTokens, ToolCalls: u.ToolCalls} -+ } -+ } -+ task.Status.Command = cmdStatus -+ -+ msg := fmt.Sprintf("ExitCode=%d", exit.ExitCode) -+ if exit.ExitCode == 0 { -+ task.Status.Phase = "Completed" -+ } else { -+ task.Status.Phase = "Failed" -+ } -+ r.setCondition(task, condReady, "False", reasonCommandExited, msg, now) -+ -+ // Free the worker. The DATA snapshot keeps /workspace for inspection. -+ if err := r.client.SuspendActor(ctx, atespace, actorName); err != nil { -+ slog.Warn("could not suspend finished task's actor", "name", actorName, "error", err) -+ r.setCondition(task, condReady, "False", reasonCommandExited, msg+"; suspend failed: "+err.Error(), now) -+ } else { -+ task.Status.WorkerIp = "" -+ } -+ slog.Info("task command exited", "name", actorName, "atespace", atespace, "exitCode", exit.ExitCode, "phase", task.Status.Phase) -+ return true, nil -+} -+ -+// metadataGet reads a runner metadata path, directly from the worker first and -+// then through the atenet router when ATENET_ROUTER_ADDR is set. -+func (r *TaskReconciler) metadataGet(ctx context.Context, atespace, actorName, host, port, path string) ([]byte, error) { -+ var errs []error -+ if host != "" { -+ b, err := r.fetch(ctx, fmt.Sprintf("http://%s%s", net.JoinHostPort(host, port), path), "") -+ if err == nil { -+ return b, nil -+ } -+ errs = append(errs, err) -+ } -+ if routerAddr := os.Getenv("ATENET_ROUTER_ADDR"); routerAddr != "" { -+ b, err := r.fetch(ctx, fmt.Sprintf("http://%s%s", routerAddr, path), fmt.Sprintf("%s/%s", atespace, actorName)) -+ if err == nil { -+ return b, nil -+ } -+ errs = append(errs, err) -+ } -+ if len(errs) == 0 { -+ return nil, errors.New("no route to the task's metadata server") -+ } -+ return nil, errors.Join(errs...) -+} -+ -+func (r *TaskReconciler) fetch(ctx context.Context, url, targetActor string) ([]byte, error) { -+ req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil) -+ if err != nil { -+ return nil, err -+ } -+ if targetActor != "" { -+ req.Header.Set("ate-target-actor", targetActor) -+ } -+ resp, err := r.httpClient.Do(req) -+ if err != nil { -+ return nil, err -+ } -+ defer resp.Body.Close() -+ body, err := io.ReadAll(io.LimitReader(resp.Body, metadata.MaxResultBytes+1)) -+ if err != nil { -+ return nil, err -+ } -+ if resp.StatusCode != http.StatusOK { -+ return nil, fmt.Errorf("GET %s: %s", url, resp.Status) -+ } -+ if int64(len(body)) > metadata.MaxResultBytes { -+ return nil, fmt.Errorf("GET %s: body over %d bytes", url, metadata.MaxResultBytes) -+ } -+ return body, nil -+} -diff --git a/internal/controller/worker.go b/internal/controller/worker.go -index 398b090..c783e8a 100644 ---- a/internal/controller/worker.go -+++ b/internal/controller/worker.go -@@ -20,6 +20,7 @@ import ( - "fmt" - "log/slog" - "os" -+ "sync" - "time" - - "github.com/google/ax/internal/store" -@@ -31,6 +32,9 @@ const ( - // readRetryDelay is how long the worker waits after a transient error from the - // event queue before trying again. - readRetryDelay = time.Second -+ // DefaultRunningResync is how often Running tasks are re-checked for a -+ // command exit, since nothing publishes an event when a command exits. -+ DefaultRunningResync = 15 * time.Second - ) - - // Worker consumes task events from the store's event queue and reconciles each -@@ -41,6 +45,14 @@ type Worker struct { - reconciler *TaskReconciler - group string - consumer string -+ -+ // RunningResync is the period of the Running-task re-check. Zero or less -+ // disables it. -+ RunningResync time.Duration -+ -+ // mu serializes event handling and the resync, so one task is never -+ // reconciled twice at once by this worker. -+ mu sync.Mutex - } - - // NewWorker creates a worker that joins group as consumer. An empty group uses the -@@ -53,11 +65,15 @@ func NewWorker(s store.Store, reconciler *TaskReconciler, group, consumer string - hostname, _ := os.Hostname() - consumer = fmt.Sprintf("%s-%d", hostname, time.Now().UnixNano()%10000) - } -+ if reconciler != nil && reconciler.SaveResult == nil { -+ reconciler.SaveResult = s.SaveTaskResult -+ } - return &Worker{ -- store: s, -- reconciler: reconciler, -- group: group, -- consumer: consumer, -+ store: s, -+ reconciler: reconciler, -+ group: group, -+ consumer: consumer, -+ RunningResync: DefaultRunningResync, - } - } - -@@ -73,6 +89,10 @@ func (w *Worker) Run(ctx context.Context) error { - } - defer sub.Close() - -+ if w.RunningResync > 0 { -+ go w.resyncLoop(ctx) -+ } -+ - for { - ev, err := sub.Next(ctx) - if err != nil { -@@ -89,7 +109,10 @@ func (w *Worker) Run(ctx context.Context) error { - continue - } - -- if err := w.processEvent(ctx, ev); err != nil { -+ w.mu.Lock() -+ err = w.processEvent(ctx, ev) -+ w.mu.Unlock() -+ if err != nil { - slog.Error("error processing task event", - "id", ev.ID, - "atespace", ev.Atespace, -@@ -165,3 +188,53 @@ func (w *Worker) processEvent(ctx context.Context, ev store.TaskEvent) error { - - return nil - } -+ -+// resyncLoop re-checks every Running task each RunningResync until ctx is done. -+func (w *Worker) resyncLoop(ctx context.Context) { -+ ticker := time.NewTicker(w.RunningResync) -+ defer ticker.Stop() -+ for { -+ select { -+ case <-ctx.Done(): -+ return -+ case <-ticker.C: -+ w.ResyncOnce(ctx) -+ } -+ } -+} -+ -+// ResyncOnce re-checks every Running task once and writes back any that -+// finished. It returns how many tasks changed. -+func (w *Worker) ResyncOnce(ctx context.Context) int { -+ w.mu.Lock() -+ defer w.mu.Unlock() -+ tasks, err := w.store.ListTasks(ctx, "", 0, 0) -+ if err != nil { -+ slog.Warn("resync: listing tasks", "error", err) -+ return 0 -+ } -+ changed := 0 -+ for _, t := range tasks { -+ if t.GetStatus().GetPhase() != "Running" { -+ continue -+ } -+ // Re-read so a status written since the list is not overwritten. -+ task, err := w.store.GetTask(ctx, t.Metadata.Atespace, t.Metadata.Name) -+ if err != nil || task.GetStatus().GetPhase() != "Running" { -+ continue -+ } -+ ok, err := w.reconciler.ResyncRunning(ctx, task) -+ if err != nil { -+ slog.Debug("resync: task not finished or unreachable", "atespace", task.Metadata.Atespace, "name", task.Metadata.Name, "error", err) -+ } -+ if !ok { -+ continue -+ } -+ if err := w.store.UpdateTaskStatus(ctx, task.Metadata.Atespace, task.Metadata.Name, task.Status); err != nil { -+ slog.Error("resync: writing task status", "name", task.Metadata.Name, "error", err) -+ continue -+ } -+ changed++ -+ } -+ return changed -+} -diff --git a/internal/metadata/exit_test.go b/internal/metadata/exit_test.go -new file mode 100644 -index 0000000..85da1c1 ---- /dev/null -+++ b/internal/metadata/exit_test.go -@@ -0,0 +1,118 @@ -+// Copyright 2026 Google LLC -+// -+// Licensed under the Apache License, Version 2.0 (the "License"); -+// you may not use this file except in compliance with the License. -+// You may obtain a copy of the License at -+// -+// http://www.apache.org/licenses/LICENSE-2.0 -+// -+// Unless required by applicable law or agreed to in writing, software -+// distributed under the License is distributed on an "AS IS" BASIS, -+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -+// See the License for the specific language governing permissions and -+// limitations under the License. -+ -+package metadata_test -+ -+import ( -+ "context" -+ "encoding/json" -+ "fmt" -+ "io" -+ "net" -+ "net/http" -+ "os" -+ "path/filepath" -+ "strings" -+ "testing" -+ "time" -+ -+ "github.com/google/ax/internal/metadata" -+ "github.com/google/ax/pkg/apis/v1alpha1" -+) -+ -+func freePort(t *testing.T) int { -+ t.Helper() -+ l, err := net.Listen("tcp", "127.0.0.1:0") -+ if err != nil { -+ t.Fatal(err) -+ } -+ defer l.Close() -+ return l.Addr().(*net.TCPAddr).Port -+} -+ -+func get(t *testing.T, url string) (int, string) { -+ t.Helper() -+ var resp *http.Response -+ var err error -+ for i := 0; i < 50; i++ { -+ resp, err = http.Get(url) -+ if err == nil { -+ break -+ } -+ time.Sleep(20 * time.Millisecond) -+ } -+ if err != nil { -+ t.Fatalf("GET %s: %v", url, err) -+ } -+ defer resp.Body.Close() -+ b, _ := io.ReadAll(resp.Body) -+ return resp.StatusCode, string(b) -+} -+ -+func TestMetadataServer_ExitAndResult(t *testing.T) { -+ dir := t.TempDir() -+ resultPath := filepath.Join(dir, "result.json") -+ usagePath := filepath.Join(dir, "usage.json") -+ port := freePort(t) -+ task := &v1alpha1.Task{Metadata: &v1alpha1.ObjectMeta{Name: "t1"}, Spec: &v1alpha1.TaskSpec{}} -+ s := metadata.NewServer(port, task, nil, metadata.ServerOptions{ResultPath: resultPath, UsagePath: usagePath}) -+ if err := s.Start(); err != nil { -+ t.Fatal(err) -+ } -+ defer s.Stop(context.Background()) -+ base := fmt.Sprintf("http://127.0.0.1:%d/metadata/v1alpha1/ax", port) -+ -+ // Before exit: exit says not exited, result is refused. -+ code, body := get(t, base+"/exit") -+ var st metadata.CommandExitStatus -+ if code != http.StatusOK || json.Unmarshal([]byte(body), &st) != nil || st.Exited { -+ t.Fatalf("exit before command exit = %d %q", code, body) -+ } -+ if code, _ := get(t, base+"/result"); code != http.StatusConflict { -+ t.Fatalf("result before exit = %d, want 409", code) -+ } -+ -+ if err := os.WriteFile(resultPath, []byte(`{"ok":true}`), 0o644); err != nil { -+ t.Fatal(err) -+ } -+ if err := os.WriteFile(usagePath, []byte(`{"prompt_tokens":3}`), 0o644); err != nil { -+ t.Fatal(err) -+ } -+ s.SetCommandExit(3, time.Now()) -+ s.SetCommandExit(0, time.Now()) // only the first report counts -+ -+ code, body = get(t, base+"/exit") -+ if err := json.Unmarshal([]byte(body), &st); err != nil || code != http.StatusOK || !st.Exited || st.ExitCode != 3 { -+ t.Fatalf("exit after command exit = %d %q", code, body) -+ } -+ if code, body := get(t, base+"/result"); code != http.StatusOK || body != `{"ok":true}` { -+ t.Fatalf("result = %d %q", code, body) -+ } -+ if code, body := get(t, base+"/usage"); code != http.StatusOK || !strings.Contains(body, "prompt_tokens") { -+ t.Fatalf("usage = %d %q", code, body) -+ } -+ -+ // Over the cap: refused, not truncated. -+ if err := os.WriteFile(resultPath, make([]byte, metadata.MaxResultBytes+1), 0o644); err != nil { -+ t.Fatal(err) -+ } -+ if code, _ := get(t, base+"/result"); code != http.StatusRequestEntityTooLarge { -+ t.Fatalf("oversized result = %d, want 413", code) -+ } -+ // Missing: 404. -+ _ = os.Remove(resultPath) -+ if code, _ := get(t, base+"/result"); code != http.StatusNotFound { -+ t.Fatalf("missing result = %d, want 404", code) -+ } -+} -diff --git a/internal/metadata/server.go b/internal/metadata/server.go -index fb4feba..909e879 100644 ---- a/internal/metadata/server.go -+++ b/internal/metadata/server.go -@@ -16,13 +16,17 @@ package metadata - - import ( - "context" -+ "encoding/json" - "errors" - "fmt" -+ "io" - "log/slog" - "net" - "net/http" -+ "os" - "strings" - "sync" -+ "time" - - "github.com/agent-substrate/env/guest" - "github.com/google/ax/pkg/apis/v1alpha1" -@@ -41,12 +45,31 @@ type Server struct { - task *v1alpha1.Task - workspaces []*v1alpha1.Workspace - workspaceReady bool -+ resultPath string -+ usagePath string -+ exit *CommandExitStatus -+} -+ -+// MaxResultBytes caps the result file served at /metadata/v1alpha1/ax/result. -+// A larger file is refused with 413 rather than truncated. -+const MaxResultBytes = 1 << 20 -+ -+// CommandExitStatus is the JSON body of /metadata/v1alpha1/ax/exit. -+type CommandExitStatus struct { -+ Exited bool `json:"exited"` -+ ExitCode int `json:"exitCode"` -+ FinishedAt time.Time `json:"finishedAt,omitempty"` - } - - // ServerOptions configures optional settings for the metadata and guest server. - type ServerOptions struct { - WorkspacePath string - LogDir string -+ // ResultPath is the file served at /metadata/v1alpha1/ax/result once the -+ // command has exited. Empty disables the endpoint. -+ ResultPath string -+ // UsagePath is the JSON usage file served at /metadata/v1alpha1/ax/usage. -+ UsagePath string - } - - // NewServer creates a new metadata and guest server serving the task and its -@@ -65,6 +88,8 @@ func NewServer(port int, task *v1alpha1.Task, workspaces []*v1alpha1.Workspace, - if len(opts) > 0 { - opt = opts[0] - } -+ s.resultPath = opt.ResultPath -+ s.usagePath = opt.UsagePath - - // Guest services expose process execution and file access inside the container, - // so they are only served when the task opts in via spec.debug. -@@ -98,6 +123,12 @@ func NewServer(port int, task *v1alpha1.Task, workspaces []*v1alpha1.Workspace, - // /metadata/v1alpha1/ax/workspaces every bound Workspace, as a YAML stream - mux.HandleFunc("/metadata/v1alpha1/ax/task", s.handleTask) - mux.HandleFunc("/metadata/v1alpha1/ax/workspaces", s.handleWorkspaces) -+ // /metadata/v1alpha1/ax/exit whether the command exited, and its code -+ // /metadata/v1alpha1/ax/result the command's result file, after exit -+ // /metadata/v1alpha1/ax/usage the command's usage file, after exit -+ mux.HandleFunc("/metadata/v1alpha1/ax/exit", s.handleExit) -+ mux.HandleFunc("/metadata/v1alpha1/ax/result", s.handleResult) -+ mux.HandleFunc("/metadata/v1alpha1/ax/usage", s.handleUsage) - - handler := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if s.grpcServer != nil && r.ProtoMajor == 2 && strings.HasPrefix(r.Header.Get("Content-Type"), "application/grpc") { -@@ -236,3 +267,72 @@ func (s *Server) handleWorkspaces(w http.ResponseWriter, r *http.Request) { - http.Error(w, err.Error(), http.StatusInternalServerError) - } - } -+ -+// SetCommandExit records that the task command exited on its own with code. -+// Only the first call counts. -+func (s *Server) SetCommandExit(code int, at time.Time) { -+ s.mu.Lock() -+ defer s.mu.Unlock() -+ if s.exit == nil { -+ s.exit = &CommandExitStatus{Exited: true, ExitCode: code, FinishedAt: at.UTC()} -+ } -+} -+ -+func (s *Server) commandExit() *CommandExitStatus { -+ s.mu.RLock() -+ defer s.mu.RUnlock() -+ if s.exit == nil { -+ return nil -+ } -+ cp := *s.exit -+ return &cp -+} -+ -+// handleExit reports whether the command has exited. It always answers 200 -+// with a JSON body, so a caller can tell "still running" from "unreachable". -+func (s *Server) handleExit(w http.ResponseWriter, r *http.Request) { -+ st := s.commandExit() -+ if st == nil { -+ st = &CommandExitStatus{} -+ } -+ w.Header().Set("Content-Type", "application/json") -+ _ = json.NewEncoder(w).Encode(st) -+} -+ -+func (s *Server) handleResult(w http.ResponseWriter, r *http.Request) { -+ s.serveAfterExit(w, s.resultPath, MaxResultBytes) -+} -+ -+func (s *Server) handleUsage(w http.ResponseWriter, r *http.Request) { -+ s.serveAfterExit(w, s.usagePath, 64<<10) -+} -+ -+// serveAfterExit serves the file at path once the command has exited: 409 -+// before that, 404 when there is no file, 413 when it exceeds limit. -+func (s *Server) serveAfterExit(w http.ResponseWriter, path string, limit int64) { -+ if s.commandExit() == nil { -+ http.Error(w, "command has not exited", http.StatusConflict) -+ return -+ } -+ if path == "" { -+ http.Error(w, "not configured", http.StatusNotFound) -+ return -+ } -+ f, err := os.Open(path) -+ if err != nil { -+ http.Error(w, "no file", http.StatusNotFound) -+ return -+ } -+ defer f.Close() -+ fi, err := f.Stat() -+ if err != nil || !fi.Mode().IsRegular() { -+ http.Error(w, "no file", http.StatusNotFound) -+ return -+ } -+ if fi.Size() > limit { -+ http.Error(w, fmt.Sprintf("file is %d bytes, over the %d byte cap", fi.Size(), limit), http.StatusRequestEntityTooLarge) -+ return -+ } -+ w.Header().Set("Content-Type", "application/octet-stream") -+ _, _ = io.Copy(w, io.LimitReader(f, limit)) -+} -diff --git a/internal/server/server.go b/internal/server/server.go -index 27e89ab..669cbac 100644 ---- a/internal/server/server.go -+++ b/internal/server/server.go -@@ -16,6 +16,8 @@ package server - - import ( - "context" -+ "crypto/sha256" -+ "encoding/hex" - "errors" - "net/http" - "strings" -@@ -88,6 +90,27 @@ func (s *Server) GetTask(ctx context.Context, req *v1alpha1.GetTaskRequest) (*v1 - return task, nil - } - -+// GetTaskResult returns the result content the controller copied from the -+// task's sandbox when its command exited. -+func (s *Server) GetTaskResult(ctx context.Context, req *v1alpha1.GetTaskResultRequest) (*v1alpha1.TaskResult, error) { -+ if req == nil || req.Name == "" { -+ return nil, status.Error(codes.InvalidArgument, "missing task name") -+ } -+ atespace := req.Atespace -+ if atespace == "" { -+ atespace = "default" -+ } -+ content, err := s.store.GetTaskResult(ctx, atespace, req.Name) -+ if err != nil { -+ if errors.Is(err, store.ErrNotFound) { -+ return nil, status.Errorf(codes.NotFound, "no result for task %q in atespace %q", req.Name, atespace) -+ } -+ return nil, status.Errorf(codes.Internal, "getting task result: %v", err) -+ } -+ sum := sha256.Sum256(content) -+ return &v1alpha1.TaskResult{Content: content, Sha256: hex.EncodeToString(sum[:])}, nil -+} -+ - func (s *Server) ListTasks(ctx context.Context, req *v1alpha1.ListTasksRequest) (*v1alpha1.ListTasksResponse, error) { - atespace := "" - limit := int64(50) -diff --git a/internal/store/memory/store.go b/internal/store/memory/store.go -index 8a0b13d..215f9a4 100644 ---- a/internal/store/memory/store.go -+++ b/internal/store/memory/store.go -@@ -47,6 +47,7 @@ type MemoryStore struct { - workspaces map[string]*v1alpha1.Workspace - events chan store.TaskEvent - watchers map[string][]chan *v1alpha1.Task -+ results map[string][]byte - } - - // NewStore creates a new in-memory Store. -@@ -58,9 +59,29 @@ func NewStore() *MemoryStore { - workspaces: make(map[string]*v1alpha1.Workspace), - events: make(chan store.TaskEvent, 1000), - watchers: make(map[string][]chan *v1alpha1.Task), -+ results: make(map[string][]byte), - } - } - -+// SaveTaskResult stores a copy of the task's result content. -+func (s *MemoryStore) SaveTaskResult(ctx context.Context, atespace, name string, content []byte) error { -+ s.mu.Lock() -+ defer s.mu.Unlock() -+ s.results[taskKey(atespace, name)] = append([]byte(nil), content...) -+ return nil -+} -+ -+// GetTaskResult returns a copy of the stored result content. -+func (s *MemoryStore) GetTaskResult(ctx context.Context, atespace, name string) ([]byte, error) { -+ s.mu.RLock() -+ defer s.mu.RUnlock() -+ content, ok := s.results[taskKey(atespace, name)] -+ if !ok { -+ return nil, store.ErrNotFound -+ } -+ return append([]byte(nil), content...), nil -+} -+ - func taskKey(atespace, name string) string { - if atespace == "" { - atespace = "default" -@@ -216,6 +237,7 @@ func (s *MemoryStore) DeleteTask(ctx context.Context, atespace, name string) err - s.mu.Lock() - defer s.mu.Unlock() - delete(s.tasks, taskKey(atespace, name)) -+ delete(s.results, taskKey(atespace, name)) - return nil - } - -diff --git a/internal/store/redis/store.go b/internal/store/redis/store.go -index 0bbf7aa..34da35b 100644 ---- a/internal/store/redis/store.go -+++ b/internal/store/redis/store.go -@@ -82,6 +82,37 @@ func (s *Store) taskKey(atespace, name string) string { - return fmt.Sprintf("%s:task:%s:%s", s.opts.KeyPrefix, atespace, name) - } - -+func (s *Store) taskResultKey(atespace, name string) string { -+ return fmt.Sprintf("%s:task-result:%s:%s", s.opts.KeyPrefix, atespace, name) -+} -+ -+// SaveTaskResult stores the task's result content under its own key, so task -+// records, lists and watches stay small. It publishes no event. -+func (s *Store) SaveTaskResult(ctx context.Context, atespace, name string, content []byte) error { -+ if atespace == "" { -+ atespace = "default" -+ } -+ if err := s.client.Set(ctx, s.taskResultKey(atespace, name), content, s.opts.TTL).Err(); err != nil { -+ return fmt.Errorf("saving task result in redis: %w", err) -+ } -+ return nil -+} -+ -+// GetTaskResult returns the stored result content, or store.ErrNotFound. -+func (s *Store) GetTaskResult(ctx context.Context, atespace, name string) ([]byte, error) { -+ if atespace == "" { -+ atespace = "default" -+ } -+ content, err := s.client.Get(ctx, s.taskResultKey(atespace, name)).Bytes() -+ if err != nil { -+ if errors.Is(err, redis.Nil) { -+ return nil, store.ErrNotFound -+ } -+ return nil, fmt.Errorf("getting task result from redis: %w", err) -+ } -+ return content, nil -+} -+ - func (s *Store) gwKey(atespace, name string) string { - return fmt.Sprintf("%s:gw:%s:%s", s.opts.KeyPrefix, atespace, name) - } -@@ -335,6 +366,7 @@ func (s *Store) DeleteTask(ctx context.Context, atespace, name string) error { - - pipe := s.client.TxPipeline() - pipe.Del(ctx, s.taskKey(atespace, name)) -+ pipe.Del(ctx, s.taskResultKey(atespace, name)) - pipe.ZRem(ctx, s.taskIndexKey(), member) - pipe.ZRem(ctx, s.taskAtespaceIndexKey(atespace), name) - if _, err := pipe.Exec(ctx); err != nil { -diff --git a/internal/store/store.go b/internal/store/store.go -index 1e84a0f..684f140 100644 ---- a/internal/store/store.go -+++ b/internal/store/store.go -@@ -86,5 +86,11 @@ type Store interface { - DeleteModel(ctx context.Context, atespace, name string) error - - WatchTask(ctx context.Context, atespace, name string) (<-chan *v1alpha1.Task, io.Closer, error) -+ -+ // SaveTaskResult stores the result file a task's command wrote, as copied -+ // by the controller once the command exited. It publishes no event. -+ SaveTaskResult(ctx context.Context, atespace, name string, content []byte) error -+ // GetTaskResult returns the stored result. ErrNotFound when none was saved. -+ GetTaskResult(ctx context.Context, atespace, name string) ([]byte, error) - Close() error - } -diff --git a/internal/substrate/client.go b/internal/substrate/client.go -index 2840ce2..3693e99 100644 ---- a/internal/substrate/client.go -+++ b/internal/substrate/client.go -@@ -389,6 +389,17 @@ func (c *Client) EnsureActor(ctx context.Context, atespace, actorName, templateA - return actor, nil - } - -+// ActorState returns the actor's current state as Substrate reports it. -+func (c *Client) ActorState(ctx context.Context, atespace, actorName string) (ateapipb.ActorState, error) { -+ actor, err := c.control.GetActor(ctx, &ateapipb.GetActorRequest{ -+ Actor: &ateapipb.ObjectRef{Atespace: atespace, Name: actorName}, -+ }) -+ if err != nil { -+ return ateapipb.ActorState_ACTOR_STATE_UNSPECIFIED, fmt.Errorf("getting actor %s/%s: %w", atespace, actorName, err) -+ } -+ return actor.GetStatus().GetState(), nil -+} -+ - // ResumeActor resumes the specified actor onto a worker and returns the worker details. - func (c *Client) ResumeActor(ctx context.Context, atespace, actorName string) (*ateapipb.Actor, string, error) { - req := &ateapipb.ResumeActorRequest{ -diff --git a/pkg/apis/v1alpha1/ax.pb.go b/pkg/apis/v1alpha1/ax.pb.go -index 9a71dc0..0ce2bf0 100644 ---- a/pkg/apis/v1alpha1/ax.pb.go -+++ b/pkg/apis/v1alpha1/ax.pb.go -@@ -571,8 +571,10 @@ type TaskStatus struct { - PendingApproval *PendingApproval `protobuf:"bytes,5,opt,name=pending_approval,json=pendingApproval,proto3" json:"pending_approval,omitempty"` - Usage *UsageStats `protobuf:"bytes,6,opt,name=usage,proto3" json:"usage,omitempty"` - Conditions []*Condition `protobuf:"bytes,7,rep,name=conditions,proto3" json:"conditions,omitempty"` -- unknownFields protoimpl.UnknownFields -- sizeCache protoimpl.SizeCache -+ // command reports how the task's command finished. Unset while it runs. -+ Command *CommandStatus `protobuf:"bytes,8,opt,name=command,proto3" json:"command,omitempty"` -+ unknownFields protoimpl.UnknownFields -+ sizeCache protoimpl.SizeCache - } - - func (x *TaskStatus) Reset() { -@@ -654,6 +656,92 @@ func (x *TaskStatus) GetConditions() []*Condition { - return nil - } - -+func (x *TaskStatus) GetCommand() *CommandStatus { -+ if x != nil { -+ return x.Command -+ } -+ return nil -+} -+ -+// CommandStatus is written by the controller once the runner reports that the -+// task's command exited. The result content itself is served by GetTaskResult. -+type CommandStatus struct { -+ state protoimpl.MessageState `protogen:"open.v1"` -+ Exited bool `protobuf:"varint,1,opt,name=exited,proto3" json:"exited,omitempty"` -+ ExitCode int32 `protobuf:"varint,2,opt,name=exit_code,json=exitCode,proto3" json:"exit_code,omitempty"` -+ FinishedAt *timestamppb.Timestamp `protobuf:"bytes,3,opt,name=finished_at,json=finishedAt,proto3" json:"finished_at,omitempty"` -+ // result_bytes is the size of the copied result, 0 when there was none. -+ ResultBytes int64 `protobuf:"varint,4,opt,name=result_bytes,json=resultBytes,proto3" json:"result_bytes,omitempty"` -+ ResultSha256 string `protobuf:"bytes,5,opt,name=result_sha256,json=resultSha256,proto3" json:"result_sha256,omitempty"` -+ unknownFields protoimpl.UnknownFields -+ sizeCache protoimpl.SizeCache -+} -+ -+func (x *CommandStatus) Reset() { -+ *x = CommandStatus{} -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[9] -+ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) -+ ms.StoreMessageInfo(mi) -+} -+ -+func (x *CommandStatus) String() string { -+ return protoimpl.X.MessageStringOf(x) -+} -+ -+func (*CommandStatus) ProtoMessage() {} -+ -+func (x *CommandStatus) ProtoReflect() protoreflect.Message { -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[9] -+ if x != nil { -+ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) -+ if ms.LoadMessageInfo() == nil { -+ ms.StoreMessageInfo(mi) -+ } -+ return ms -+ } -+ return mi.MessageOf(x) -+} -+ -+// Deprecated: Use CommandStatus.ProtoReflect.Descriptor instead. -+func (*CommandStatus) Descriptor() ([]byte, []int) { -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{9} -+} -+ -+func (x *CommandStatus) GetExited() bool { -+ if x != nil { -+ return x.Exited -+ } -+ return false -+} -+ -+func (x *CommandStatus) GetExitCode() int32 { -+ if x != nil { -+ return x.ExitCode -+ } -+ return 0 -+} -+ -+func (x *CommandStatus) GetFinishedAt() *timestamppb.Timestamp { -+ if x != nil { -+ return x.FinishedAt -+ } -+ return nil -+} -+ -+func (x *CommandStatus) GetResultBytes() int64 { -+ if x != nil { -+ return x.ResultBytes -+ } -+ return 0 -+} -+ -+func (x *CommandStatus) GetResultSha256() string { -+ if x != nil { -+ return x.ResultSha256 -+ } -+ return "" -+} -+ - type PendingApproval struct { - state protoimpl.MessageState `protogen:"open.v1"` - Id string `protobuf:"bytes,1,opt,name=id,proto3" json:"id,omitempty"` -@@ -665,7 +753,7 @@ type PendingApproval struct { - - func (x *PendingApproval) Reset() { - *x = PendingApproval{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[9] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[10] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -677,7 +765,7 @@ func (x *PendingApproval) String() string { - func (*PendingApproval) ProtoMessage() {} - - func (x *PendingApproval) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[9] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[10] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -690,7 +778,7 @@ func (x *PendingApproval) ProtoReflect() protoreflect.Message { - - // Deprecated: Use PendingApproval.ProtoReflect.Descriptor instead. - func (*PendingApproval) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{9} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{10} - } - - func (x *PendingApproval) GetId() string { -@@ -718,13 +806,14 @@ type UsageStats struct { - state protoimpl.MessageState `protogen:"open.v1"` - PromptTokens int32 `protobuf:"varint,1,opt,name=prompt_tokens,json=promptTokens,proto3" json:"prompt_tokens,omitempty"` - CompletionTokens int32 `protobuf:"varint,2,opt,name=completion_tokens,json=completionTokens,proto3" json:"completion_tokens,omitempty"` -+ ToolCalls int32 `protobuf:"varint,3,opt,name=tool_calls,json=toolCalls,proto3" json:"tool_calls,omitempty"` - unknownFields protoimpl.UnknownFields - sizeCache protoimpl.SizeCache - } - - func (x *UsageStats) Reset() { - *x = UsageStats{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[10] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[11] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -736,7 +825,7 @@ func (x *UsageStats) String() string { - func (*UsageStats) ProtoMessage() {} - - func (x *UsageStats) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[10] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[11] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -749,7 +838,7 @@ func (x *UsageStats) ProtoReflect() protoreflect.Message { - - // Deprecated: Use UsageStats.ProtoReflect.Descriptor instead. - func (*UsageStats) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{10} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{11} - } - - func (x *UsageStats) GetPromptTokens() int32 { -@@ -766,6 +855,13 @@ func (x *UsageStats) GetCompletionTokens() int32 { - return 0 - } - -+func (x *UsageStats) GetToolCalls() int32 { -+ if x != nil { -+ return x.ToolCalls -+ } -+ return 0 -+} -+ - type Condition struct { - state protoimpl.MessageState `protogen:"open.v1"` - Type string `protobuf:"bytes,1,opt,name=type,proto3" json:"type,omitempty"` -@@ -779,7 +875,7 @@ type Condition struct { - - func (x *Condition) Reset() { - *x = Condition{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[11] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[12] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -791,7 +887,7 @@ func (x *Condition) String() string { - func (*Condition) ProtoMessage() {} - - func (x *Condition) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[11] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[12] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -804,7 +900,7 @@ func (x *Condition) ProtoReflect() protoreflect.Message { - - // Deprecated: Use Condition.ProtoReflect.Descriptor instead. - func (*Condition) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{11} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{12} - } - - func (x *Condition) GetType() string { -@@ -854,7 +950,7 @@ type Gateway struct { - - func (x *Gateway) Reset() { - *x = Gateway{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[12] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[13] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -866,7 +962,7 @@ func (x *Gateway) String() string { - func (*Gateway) ProtoMessage() {} - - func (x *Gateway) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[12] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[13] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -879,7 +975,7 @@ func (x *Gateway) ProtoReflect() protoreflect.Message { - - // Deprecated: Use Gateway.ProtoReflect.Descriptor instead. - func (*Gateway) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{12} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{13} - } - - func (x *Gateway) GetApiVersion() string { -@@ -920,7 +1016,7 @@ type GatewaySpec struct { - - func (x *GatewaySpec) Reset() { - *x = GatewaySpec{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[13] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[14] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -932,7 +1028,7 @@ func (x *GatewaySpec) String() string { - func (*GatewaySpec) ProtoMessage() {} - - func (x *GatewaySpec) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[13] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[14] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -945,7 +1041,7 @@ func (x *GatewaySpec) ProtoReflect() protoreflect.Message { - - // Deprecated: Use GatewaySpec.ProtoReflect.Descriptor instead. - func (*GatewaySpec) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{13} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{14} - } - - func (x *GatewaySpec) GetListeners() []*Listener { -@@ -973,7 +1069,7 @@ type Listener struct { - - func (x *Listener) Reset() { - *x = Listener{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[14] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[15] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -985,7 +1081,7 @@ func (x *Listener) String() string { - func (*Listener) ProtoMessage() {} - - func (x *Listener) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[14] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[15] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -998,7 +1094,7 @@ func (x *Listener) ProtoReflect() protoreflect.Message { - - // Deprecated: Use Listener.ProtoReflect.Descriptor instead. - func (*Listener) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{14} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{15} - } - - func (x *Listener) GetName() string { -@@ -1031,7 +1127,7 @@ type EgressConfig struct { - - func (x *EgressConfig) Reset() { - *x = EgressConfig{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[15] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[16] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1043,7 +1139,7 @@ func (x *EgressConfig) String() string { - func (*EgressConfig) ProtoMessage() {} - - func (x *EgressConfig) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[15] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[16] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1056,7 +1152,7 @@ func (x *EgressConfig) ProtoReflect() protoreflect.Message { - - // Deprecated: Use EgressConfig.ProtoReflect.Descriptor instead. - func (*EgressConfig) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{15} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{16} - } - - func (x *EgressConfig) GetAllowlist() *EgressAllowlist { -@@ -1075,7 +1171,7 @@ type EgressAllowlist struct { - - func (x *EgressAllowlist) Reset() { - *x = EgressAllowlist{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[16] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[17] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1087,7 +1183,7 @@ func (x *EgressAllowlist) String() string { - func (*EgressAllowlist) ProtoMessage() {} - - func (x *EgressAllowlist) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[16] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[17] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1100,7 +1196,7 @@ func (x *EgressAllowlist) ProtoReflect() protoreflect.Message { - - // Deprecated: Use EgressAllowlist.ProtoReflect.Descriptor instead. - func (*EgressAllowlist) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{16} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{17} - } - - func (x *EgressAllowlist) GetHosts() []*HostRule { -@@ -1120,7 +1216,7 @@ type HostRule struct { - - func (x *HostRule) Reset() { - *x = HostRule{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[17] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[18] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1132,7 +1228,7 @@ func (x *HostRule) String() string { - func (*HostRule) ProtoMessage() {} - - func (x *HostRule) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[17] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[18] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1145,7 +1241,7 @@ func (x *HostRule) ProtoReflect() protoreflect.Message { - - // Deprecated: Use HostRule.ProtoReflect.Descriptor instead. - func (*HostRule) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{17} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{18} - } - - func (x *HostRule) GetHost() string { -@@ -1174,7 +1270,7 @@ type Workspace struct { - - func (x *Workspace) Reset() { - *x = Workspace{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[18] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[19] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1186,7 +1282,7 @@ func (x *Workspace) String() string { - func (*Workspace) ProtoMessage() {} - - func (x *Workspace) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[18] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[19] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1199,7 +1295,7 @@ func (x *Workspace) ProtoReflect() protoreflect.Message { - - // Deprecated: Use Workspace.ProtoReflect.Descriptor instead. - func (*Workspace) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{18} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{19} - } - - func (x *Workspace) GetApiVersion() string { -@@ -1241,7 +1337,7 @@ type WorkspaceSpec struct { - - func (x *WorkspaceSpec) Reset() { - *x = WorkspaceSpec{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[19] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[20] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1253,7 +1349,7 @@ func (x *WorkspaceSpec) String() string { - func (*WorkspaceSpec) ProtoMessage() {} - - func (x *WorkspaceSpec) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[19] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[20] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1266,7 +1362,7 @@ func (x *WorkspaceSpec) ProtoReflect() protoreflect.Message { - - // Deprecated: Use WorkspaceSpec.ProtoReflect.Descriptor instead. - func (*WorkspaceSpec) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{19} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{20} - } - - func (x *WorkspaceSpec) GetGit() []*GitRepo { -@@ -1303,7 +1399,7 @@ type GitRepo struct { - - func (x *GitRepo) Reset() { - *x = GitRepo{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[20] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[21] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1315,7 +1411,7 @@ func (x *GitRepo) String() string { - func (*GitRepo) ProtoMessage() {} - - func (x *GitRepo) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[20] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[21] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1328,7 +1424,7 @@ func (x *GitRepo) ProtoReflect() protoreflect.Message { - - // Deprecated: Use GitRepo.ProtoReflect.Descriptor instead. - func (*GitRepo) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{20} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{21} - } - - func (x *GitRepo) GetName() string { -@@ -1376,7 +1472,7 @@ type MCPConfig struct { - - func (x *MCPConfig) Reset() { - *x = MCPConfig{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[21] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[22] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1388,7 +1484,7 @@ func (x *MCPConfig) String() string { - func (*MCPConfig) ProtoMessage() {} - - func (x *MCPConfig) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[21] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[22] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1401,7 +1497,7 @@ func (x *MCPConfig) ProtoReflect() protoreflect.Message { - - // Deprecated: Use MCPConfig.ProtoReflect.Descriptor instead. - func (*MCPConfig) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{21} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{22} - } - - func (x *MCPConfig) GetRegistries() []*MCPRegistry { -@@ -1430,7 +1526,7 @@ type MCPRegistry struct { - - func (x *MCPRegistry) Reset() { - *x = MCPRegistry{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[22] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[23] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1442,7 +1538,7 @@ func (x *MCPRegistry) String() string { - func (*MCPRegistry) ProtoMessage() {} - - func (x *MCPRegistry) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[22] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[23] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1455,7 +1551,7 @@ func (x *MCPRegistry) ProtoReflect() protoreflect.Message { - - // Deprecated: Use MCPRegistry.ProtoReflect.Descriptor instead. - func (*MCPRegistry) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{22} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{23} - } - - func (x *MCPRegistry) GetProvider() string { -@@ -1498,7 +1594,7 @@ type MCPServer struct { - - func (x *MCPServer) Reset() { - *x = MCPServer{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[23] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[24] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1510,7 +1606,7 @@ func (x *MCPServer) String() string { - func (*MCPServer) ProtoMessage() {} - - func (x *MCPServer) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[23] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[24] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1523,7 +1619,7 @@ func (x *MCPServer) ProtoReflect() protoreflect.Message { - - // Deprecated: Use MCPServer.ProtoReflect.Descriptor instead. - func (*MCPServer) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{23} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{24} - } - - func (x *MCPServer) GetName() string { -@@ -1564,7 +1660,7 @@ type SkillsConfig struct { - - func (x *SkillsConfig) Reset() { - *x = SkillsConfig{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[24] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[25] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1576,7 +1672,7 @@ func (x *SkillsConfig) String() string { - func (*SkillsConfig) ProtoMessage() {} - - func (x *SkillsConfig) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[24] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[25] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1589,7 +1685,7 @@ func (x *SkillsConfig) ProtoReflect() protoreflect.Message { - - // Deprecated: Use SkillsConfig.ProtoReflect.Descriptor instead. - func (*SkillsConfig) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{24} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{25} - } - - func (x *SkillsConfig) GetRegistries() []*SkillRegistry { -@@ -1617,7 +1713,7 @@ type SkillRegistry struct { - - func (x *SkillRegistry) Reset() { - *x = SkillRegistry{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[25] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[26] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1629,7 +1725,7 @@ func (x *SkillRegistry) String() string { - func (*SkillRegistry) ProtoMessage() {} - - func (x *SkillRegistry) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[25] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[26] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1642,7 +1738,7 @@ func (x *SkillRegistry) ProtoReflect() protoreflect.Message { - - // Deprecated: Use SkillRegistry.ProtoReflect.Descriptor instead. - func (*SkillRegistry) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{25} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{26} - } - - func (x *SkillRegistry) GetProvider() string { -@@ -1678,7 +1774,7 @@ type Model struct { - - func (x *Model) Reset() { - *x = Model{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[26] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[27] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1690,7 +1786,7 @@ func (x *Model) String() string { - func (*Model) ProtoMessage() {} - - func (x *Model) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[26] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[27] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1703,7 +1799,7 @@ func (x *Model) ProtoReflect() protoreflect.Message { - - // Deprecated: Use Model.ProtoReflect.Descriptor instead. - func (*Model) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{26} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{27} - } - - func (x *Model) GetApiVersion() string { -@@ -1748,7 +1844,7 @@ type ModelSpec struct { - - func (x *ModelSpec) Reset() { - *x = ModelSpec{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[27] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[28] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1760,7 +1856,7 @@ func (x *ModelSpec) String() string { - func (*ModelSpec) ProtoMessage() {} - - func (x *ModelSpec) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[27] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[28] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1773,7 +1869,7 @@ func (x *ModelSpec) ProtoReflect() protoreflect.Message { - - // Deprecated: Use ModelSpec.ProtoReflect.Descriptor instead. - func (*ModelSpec) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{27} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{28} - } - - func (x *ModelSpec) GetProvider() string { -@@ -1814,7 +1910,7 @@ type SecretKeyRef struct { - - func (x *SecretKeyRef) Reset() { - *x = SecretKeyRef{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[28] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[29] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1826,7 +1922,7 @@ func (x *SecretKeyRef) String() string { - func (*SecretKeyRef) ProtoMessage() {} - - func (x *SecretKeyRef) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[28] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[29] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1839,7 +1935,7 @@ func (x *SecretKeyRef) ProtoReflect() protoreflect.Message { - - // Deprecated: Use SecretKeyRef.ProtoReflect.Descriptor instead. - func (*SecretKeyRef) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{28} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{29} - } - - func (x *SecretKeyRef) GetName() string { -@@ -1867,7 +1963,7 @@ type GetTaskRequest struct { - - func (x *GetTaskRequest) Reset() { - *x = GetTaskRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[29] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[30] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1879,7 +1975,7 @@ func (x *GetTaskRequest) String() string { - func (*GetTaskRequest) ProtoMessage() {} - - func (x *GetTaskRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[29] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[30] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1892,7 +1988,7 @@ func (x *GetTaskRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use GetTaskRequest.ProtoReflect.Descriptor instead. - func (*GetTaskRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{29} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{30} - } - - func (x *GetTaskRequest) GetAtespace() string { -@@ -1909,6 +2005,110 @@ func (x *GetTaskRequest) GetName() string { - return "" - } - -+type GetTaskResultRequest struct { -+ state protoimpl.MessageState `protogen:"open.v1"` -+ Atespace string `protobuf:"bytes,1,opt,name=atespace,proto3" json:"atespace,omitempty"` -+ Name string `protobuf:"bytes,2,opt,name=name,proto3" json:"name,omitempty"` -+ unknownFields protoimpl.UnknownFields -+ sizeCache protoimpl.SizeCache -+} -+ -+func (x *GetTaskResultRequest) Reset() { -+ *x = GetTaskResultRequest{} -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[31] -+ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) -+ ms.StoreMessageInfo(mi) -+} -+ -+func (x *GetTaskResultRequest) String() string { -+ return protoimpl.X.MessageStringOf(x) -+} -+ -+func (*GetTaskResultRequest) ProtoMessage() {} -+ -+func (x *GetTaskResultRequest) ProtoReflect() protoreflect.Message { -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[31] -+ if x != nil { -+ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) -+ if ms.LoadMessageInfo() == nil { -+ ms.StoreMessageInfo(mi) -+ } -+ return ms -+ } -+ return mi.MessageOf(x) -+} -+ -+// Deprecated: Use GetTaskResultRequest.ProtoReflect.Descriptor instead. -+func (*GetTaskResultRequest) Descriptor() ([]byte, []int) { -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{31} -+} -+ -+func (x *GetTaskResultRequest) GetAtespace() string { -+ if x != nil { -+ return x.Atespace -+ } -+ return "" -+} -+ -+func (x *GetTaskResultRequest) GetName() string { -+ if x != nil { -+ return x.Name -+ } -+ return "" -+} -+ -+type TaskResult struct { -+ state protoimpl.MessageState `protogen:"open.v1"` -+ Content []byte `protobuf:"bytes,1,opt,name=content,proto3" json:"content,omitempty"` -+ Sha256 string `protobuf:"bytes,2,opt,name=sha256,proto3" json:"sha256,omitempty"` -+ unknownFields protoimpl.UnknownFields -+ sizeCache protoimpl.SizeCache -+} -+ -+func (x *TaskResult) Reset() { -+ *x = TaskResult{} -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[32] -+ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) -+ ms.StoreMessageInfo(mi) -+} -+ -+func (x *TaskResult) String() string { -+ return protoimpl.X.MessageStringOf(x) -+} -+ -+func (*TaskResult) ProtoMessage() {} -+ -+func (x *TaskResult) ProtoReflect() protoreflect.Message { -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[32] -+ if x != nil { -+ ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) -+ if ms.LoadMessageInfo() == nil { -+ ms.StoreMessageInfo(mi) -+ } -+ return ms -+ } -+ return mi.MessageOf(x) -+} -+ -+// Deprecated: Use TaskResult.ProtoReflect.Descriptor instead. -+func (*TaskResult) Descriptor() ([]byte, []int) { -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{32} -+} -+ -+func (x *TaskResult) GetContent() []byte { -+ if x != nil { -+ return x.Content -+ } -+ return nil -+} -+ -+func (x *TaskResult) GetSha256() string { -+ if x != nil { -+ return x.Sha256 -+ } -+ return "" -+} -+ - type ListTasksRequest struct { - state protoimpl.MessageState `protogen:"open.v1"` - Atespace string `protobuf:"bytes,1,opt,name=atespace,proto3" json:"atespace,omitempty"` -@@ -1920,7 +2120,7 @@ type ListTasksRequest struct { - - func (x *ListTasksRequest) Reset() { - *x = ListTasksRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[30] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[33] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1932,7 +2132,7 @@ func (x *ListTasksRequest) String() string { - func (*ListTasksRequest) ProtoMessage() {} - - func (x *ListTasksRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[30] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[33] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -1945,7 +2145,7 @@ func (x *ListTasksRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use ListTasksRequest.ProtoReflect.Descriptor instead. - func (*ListTasksRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{30} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{33} - } - - func (x *ListTasksRequest) GetAtespace() string { -@@ -1978,7 +2178,7 @@ type ListTasksResponse struct { - - func (x *ListTasksResponse) Reset() { - *x = ListTasksResponse{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[31] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[34] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -1990,7 +2190,7 @@ func (x *ListTasksResponse) String() string { - func (*ListTasksResponse) ProtoMessage() {} - - func (x *ListTasksResponse) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[31] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[34] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2003,7 +2203,7 @@ func (x *ListTasksResponse) ProtoReflect() protoreflect.Message { - - // Deprecated: Use ListTasksResponse.ProtoReflect.Descriptor instead. - func (*ListTasksResponse) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{31} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{34} - } - - func (x *ListTasksResponse) GetTasks() []*Task { -@@ -2022,7 +2222,7 @@ type UpdateTaskRequest struct { - - func (x *UpdateTaskRequest) Reset() { - *x = UpdateTaskRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[32] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[35] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2034,7 +2234,7 @@ func (x *UpdateTaskRequest) String() string { - func (*UpdateTaskRequest) ProtoMessage() {} - - func (x *UpdateTaskRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[32] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[35] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2047,7 +2247,7 @@ func (x *UpdateTaskRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use UpdateTaskRequest.ProtoReflect.Descriptor instead. - func (*UpdateTaskRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{32} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{35} - } - - func (x *UpdateTaskRequest) GetTask() *Task { -@@ -2067,7 +2267,7 @@ type DeleteTaskRequest struct { - - func (x *DeleteTaskRequest) Reset() { - *x = DeleteTaskRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[33] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[36] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2079,7 +2279,7 @@ func (x *DeleteTaskRequest) String() string { - func (*DeleteTaskRequest) ProtoMessage() {} - - func (x *DeleteTaskRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[33] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[36] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2092,7 +2292,7 @@ func (x *DeleteTaskRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use DeleteTaskRequest.ProtoReflect.Descriptor instead. - func (*DeleteTaskRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{33} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{36} - } - - func (x *DeleteTaskRequest) GetAtespace() string { -@@ -2117,7 +2317,7 @@ type DeleteTaskResponse struct { - - func (x *DeleteTaskResponse) Reset() { - *x = DeleteTaskResponse{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[34] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[37] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2129,7 +2329,7 @@ func (x *DeleteTaskResponse) String() string { - func (*DeleteTaskResponse) ProtoMessage() {} - - func (x *DeleteTaskResponse) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[34] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[37] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2142,7 +2342,7 @@ func (x *DeleteTaskResponse) ProtoReflect() protoreflect.Message { - - // Deprecated: Use DeleteTaskResponse.ProtoReflect.Descriptor instead. - func (*DeleteTaskResponse) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{34} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{37} - } - - type SuspendTaskRequest struct { -@@ -2155,7 +2355,7 @@ type SuspendTaskRequest struct { - - func (x *SuspendTaskRequest) Reset() { - *x = SuspendTaskRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[35] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[38] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2167,7 +2367,7 @@ func (x *SuspendTaskRequest) String() string { - func (*SuspendTaskRequest) ProtoMessage() {} - - func (x *SuspendTaskRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[35] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[38] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2180,7 +2380,7 @@ func (x *SuspendTaskRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use SuspendTaskRequest.ProtoReflect.Descriptor instead. - func (*SuspendTaskRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{35} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{38} - } - - func (x *SuspendTaskRequest) GetAtespace() string { -@@ -2207,7 +2407,7 @@ type ResumeTaskRequest struct { - - func (x *ResumeTaskRequest) Reset() { - *x = ResumeTaskRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[36] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[39] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2219,7 +2419,7 @@ func (x *ResumeTaskRequest) String() string { - func (*ResumeTaskRequest) ProtoMessage() {} - - func (x *ResumeTaskRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[36] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[39] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2232,7 +2432,7 @@ func (x *ResumeTaskRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use ResumeTaskRequest.ProtoReflect.Descriptor instead. - func (*ResumeTaskRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{36} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{39} - } - - func (x *ResumeTaskRequest) GetAtespace() string { -@@ -2259,7 +2459,7 @@ type WatchTaskRequest struct { - - func (x *WatchTaskRequest) Reset() { - *x = WatchTaskRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[37] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[40] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2271,7 +2471,7 @@ func (x *WatchTaskRequest) String() string { - func (*WatchTaskRequest) ProtoMessage() {} - - func (x *WatchTaskRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[37] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[40] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2284,7 +2484,7 @@ func (x *WatchTaskRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use WatchTaskRequest.ProtoReflect.Descriptor instead. - func (*WatchTaskRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{37} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{40} - } - - func (x *WatchTaskRequest) GetAtespace() string { -@@ -2311,7 +2511,7 @@ type WatchTaskResponse struct { - - func (x *WatchTaskResponse) Reset() { - *x = WatchTaskResponse{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[38] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[41] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2323,7 +2523,7 @@ func (x *WatchTaskResponse) String() string { - func (*WatchTaskResponse) ProtoMessage() {} - - func (x *WatchTaskResponse) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[38] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[41] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2336,7 +2536,7 @@ func (x *WatchTaskResponse) ProtoReflect() protoreflect.Message { - - // Deprecated: Use WatchTaskResponse.ProtoReflect.Descriptor instead. - func (*WatchTaskResponse) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{38} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{41} - } - - func (x *WatchTaskResponse) GetTask() *Task { -@@ -2364,7 +2564,7 @@ type GetGatewayRequest struct { - - func (x *GetGatewayRequest) Reset() { - *x = GetGatewayRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[39] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[42] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2376,7 +2576,7 @@ func (x *GetGatewayRequest) String() string { - func (*GetGatewayRequest) ProtoMessage() {} - - func (x *GetGatewayRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[39] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[42] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2389,7 +2589,7 @@ func (x *GetGatewayRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use GetGatewayRequest.ProtoReflect.Descriptor instead. - func (*GetGatewayRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{39} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{42} - } - - func (x *GetGatewayRequest) GetAtespace() string { -@@ -2415,7 +2615,7 @@ type ListGatewaysRequest struct { - - func (x *ListGatewaysRequest) Reset() { - *x = ListGatewaysRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[40] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[43] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2427,7 +2627,7 @@ func (x *ListGatewaysRequest) String() string { - func (*ListGatewaysRequest) ProtoMessage() {} - - func (x *ListGatewaysRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[40] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[43] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2440,7 +2640,7 @@ func (x *ListGatewaysRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use ListGatewaysRequest.ProtoReflect.Descriptor instead. - func (*ListGatewaysRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{40} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{43} - } - - func (x *ListGatewaysRequest) GetAtespace() string { -@@ -2459,7 +2659,7 @@ type ListGatewaysResponse struct { - - func (x *ListGatewaysResponse) Reset() { - *x = ListGatewaysResponse{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[41] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[44] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2471,7 +2671,7 @@ func (x *ListGatewaysResponse) String() string { - func (*ListGatewaysResponse) ProtoMessage() {} - - func (x *ListGatewaysResponse) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[41] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[44] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2484,7 +2684,7 @@ func (x *ListGatewaysResponse) ProtoReflect() protoreflect.Message { - - // Deprecated: Use ListGatewaysResponse.ProtoReflect.Descriptor instead. - func (*ListGatewaysResponse) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{41} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{44} - } - - func (x *ListGatewaysResponse) GetGateways() []*Gateway { -@@ -2503,7 +2703,7 @@ type UpdateGatewayRequest struct { - - func (x *UpdateGatewayRequest) Reset() { - *x = UpdateGatewayRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[42] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[45] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2515,7 +2715,7 @@ func (x *UpdateGatewayRequest) String() string { - func (*UpdateGatewayRequest) ProtoMessage() {} - - func (x *UpdateGatewayRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[42] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[45] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2528,7 +2728,7 @@ func (x *UpdateGatewayRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use UpdateGatewayRequest.ProtoReflect.Descriptor instead. - func (*UpdateGatewayRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{42} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{45} - } - - func (x *UpdateGatewayRequest) GetGateway() *Gateway { -@@ -2548,7 +2748,7 @@ type DeleteGatewayRequest struct { - - func (x *DeleteGatewayRequest) Reset() { - *x = DeleteGatewayRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[43] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[46] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2560,7 +2760,7 @@ func (x *DeleteGatewayRequest) String() string { - func (*DeleteGatewayRequest) ProtoMessage() {} - - func (x *DeleteGatewayRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[43] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[46] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2573,7 +2773,7 @@ func (x *DeleteGatewayRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use DeleteGatewayRequest.ProtoReflect.Descriptor instead. - func (*DeleteGatewayRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{43} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{46} - } - - func (x *DeleteGatewayRequest) GetAtespace() string { -@@ -2598,7 +2798,7 @@ type DeleteGatewayResponse struct { - - func (x *DeleteGatewayResponse) Reset() { - *x = DeleteGatewayResponse{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[44] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[47] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2610,7 +2810,7 @@ func (x *DeleteGatewayResponse) String() string { - func (*DeleteGatewayResponse) ProtoMessage() {} - - func (x *DeleteGatewayResponse) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[44] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[47] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2623,7 +2823,7 @@ func (x *DeleteGatewayResponse) ProtoReflect() protoreflect.Message { - - // Deprecated: Use DeleteGatewayResponse.ProtoReflect.Descriptor instead. - func (*DeleteGatewayResponse) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{44} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{47} - } - - // Workspaces -@@ -2637,7 +2837,7 @@ type GetWorkspaceRequest struct { - - func (x *GetWorkspaceRequest) Reset() { - *x = GetWorkspaceRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[45] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[48] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2649,7 +2849,7 @@ func (x *GetWorkspaceRequest) String() string { - func (*GetWorkspaceRequest) ProtoMessage() {} - - func (x *GetWorkspaceRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[45] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[48] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2662,7 +2862,7 @@ func (x *GetWorkspaceRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use GetWorkspaceRequest.ProtoReflect.Descriptor instead. - func (*GetWorkspaceRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{45} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{48} - } - - func (x *GetWorkspaceRequest) GetAtespace() string { -@@ -2688,7 +2888,7 @@ type ListWorkspacesRequest struct { - - func (x *ListWorkspacesRequest) Reset() { - *x = ListWorkspacesRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[46] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[49] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2700,7 +2900,7 @@ func (x *ListWorkspacesRequest) String() string { - func (*ListWorkspacesRequest) ProtoMessage() {} - - func (x *ListWorkspacesRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[46] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[49] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2713,7 +2913,7 @@ func (x *ListWorkspacesRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use ListWorkspacesRequest.ProtoReflect.Descriptor instead. - func (*ListWorkspacesRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{46} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{49} - } - - func (x *ListWorkspacesRequest) GetAtespace() string { -@@ -2732,7 +2932,7 @@ type ListWorkspacesResponse struct { - - func (x *ListWorkspacesResponse) Reset() { - *x = ListWorkspacesResponse{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[47] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[50] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2744,7 +2944,7 @@ func (x *ListWorkspacesResponse) String() string { - func (*ListWorkspacesResponse) ProtoMessage() {} - - func (x *ListWorkspacesResponse) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[47] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[50] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2757,7 +2957,7 @@ func (x *ListWorkspacesResponse) ProtoReflect() protoreflect.Message { - - // Deprecated: Use ListWorkspacesResponse.ProtoReflect.Descriptor instead. - func (*ListWorkspacesResponse) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{47} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{50} - } - - func (x *ListWorkspacesResponse) GetWorkspaces() []*Workspace { -@@ -2776,7 +2976,7 @@ type UpdateWorkspaceRequest struct { - - func (x *UpdateWorkspaceRequest) Reset() { - *x = UpdateWorkspaceRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[48] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[51] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2788,7 +2988,7 @@ func (x *UpdateWorkspaceRequest) String() string { - func (*UpdateWorkspaceRequest) ProtoMessage() {} - - func (x *UpdateWorkspaceRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[48] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[51] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2801,7 +3001,7 @@ func (x *UpdateWorkspaceRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use UpdateWorkspaceRequest.ProtoReflect.Descriptor instead. - func (*UpdateWorkspaceRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{48} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{51} - } - - func (x *UpdateWorkspaceRequest) GetWorkspace() *Workspace { -@@ -2821,7 +3021,7 @@ type DeleteWorkspaceRequest struct { - - func (x *DeleteWorkspaceRequest) Reset() { - *x = DeleteWorkspaceRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[49] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[52] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2833,7 +3033,7 @@ func (x *DeleteWorkspaceRequest) String() string { - func (*DeleteWorkspaceRequest) ProtoMessage() {} - - func (x *DeleteWorkspaceRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[49] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[52] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2846,7 +3046,7 @@ func (x *DeleteWorkspaceRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use DeleteWorkspaceRequest.ProtoReflect.Descriptor instead. - func (*DeleteWorkspaceRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{49} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{52} - } - - func (x *DeleteWorkspaceRequest) GetAtespace() string { -@@ -2871,7 +3071,7 @@ type DeleteWorkspaceResponse struct { - - func (x *DeleteWorkspaceResponse) Reset() { - *x = DeleteWorkspaceResponse{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[50] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[53] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2883,7 +3083,7 @@ func (x *DeleteWorkspaceResponse) String() string { - func (*DeleteWorkspaceResponse) ProtoMessage() {} - - func (x *DeleteWorkspaceResponse) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[50] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[53] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2896,7 +3096,7 @@ func (x *DeleteWorkspaceResponse) ProtoReflect() protoreflect.Message { - - // Deprecated: Use DeleteWorkspaceResponse.ProtoReflect.Descriptor instead. - func (*DeleteWorkspaceResponse) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{50} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{53} - } - - // Models -@@ -2910,7 +3110,7 @@ type GetModelRequest struct { - - func (x *GetModelRequest) Reset() { - *x = GetModelRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[51] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[54] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2922,7 +3122,7 @@ func (x *GetModelRequest) String() string { - func (*GetModelRequest) ProtoMessage() {} - - func (x *GetModelRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[51] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[54] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2935,7 +3135,7 @@ func (x *GetModelRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use GetModelRequest.ProtoReflect.Descriptor instead. - func (*GetModelRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{51} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{54} - } - - func (x *GetModelRequest) GetAtespace() string { -@@ -2961,7 +3161,7 @@ type ListModelsRequest struct { - - func (x *ListModelsRequest) Reset() { - *x = ListModelsRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[52] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[55] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -2973,7 +3173,7 @@ func (x *ListModelsRequest) String() string { - func (*ListModelsRequest) ProtoMessage() {} - - func (x *ListModelsRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[52] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[55] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -2986,7 +3186,7 @@ func (x *ListModelsRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use ListModelsRequest.ProtoReflect.Descriptor instead. - func (*ListModelsRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{52} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{55} - } - - func (x *ListModelsRequest) GetAtespace() string { -@@ -3005,7 +3205,7 @@ type ListModelsResponse struct { - - func (x *ListModelsResponse) Reset() { - *x = ListModelsResponse{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[53] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[56] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -3017,7 +3217,7 @@ func (x *ListModelsResponse) String() string { - func (*ListModelsResponse) ProtoMessage() {} - - func (x *ListModelsResponse) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[53] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[56] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -3030,7 +3230,7 @@ func (x *ListModelsResponse) ProtoReflect() protoreflect.Message { - - // Deprecated: Use ListModelsResponse.ProtoReflect.Descriptor instead. - func (*ListModelsResponse) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{53} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{56} - } - - func (x *ListModelsResponse) GetModels() []*Model { -@@ -3049,7 +3249,7 @@ type UpdateModelRequest struct { - - func (x *UpdateModelRequest) Reset() { - *x = UpdateModelRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[54] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[57] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -3061,7 +3261,7 @@ func (x *UpdateModelRequest) String() string { - func (*UpdateModelRequest) ProtoMessage() {} - - func (x *UpdateModelRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[54] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[57] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -3074,7 +3274,7 @@ func (x *UpdateModelRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use UpdateModelRequest.ProtoReflect.Descriptor instead. - func (*UpdateModelRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{54} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{57} - } - - func (x *UpdateModelRequest) GetModel() *Model { -@@ -3094,7 +3294,7 @@ type DeleteModelRequest struct { - - func (x *DeleteModelRequest) Reset() { - *x = DeleteModelRequest{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[55] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[58] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -3106,7 +3306,7 @@ func (x *DeleteModelRequest) String() string { - func (*DeleteModelRequest) ProtoMessage() {} - - func (x *DeleteModelRequest) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[55] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[58] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -3119,7 +3319,7 @@ func (x *DeleteModelRequest) ProtoReflect() protoreflect.Message { - - // Deprecated: Use DeleteModelRequest.ProtoReflect.Descriptor instead. - func (*DeleteModelRequest) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{55} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{58} - } - - func (x *DeleteModelRequest) GetAtespace() string { -@@ -3144,7 +3344,7 @@ type DeleteModelResponse struct { - - func (x *DeleteModelResponse) Reset() { - *x = DeleteModelResponse{} -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[56] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[59] - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - ms.StoreMessageInfo(mi) - } -@@ -3156,7 +3356,7 @@ func (x *DeleteModelResponse) String() string { - func (*DeleteModelResponse) ProtoMessage() {} - - func (x *DeleteModelResponse) ProtoReflect() protoreflect.Message { -- mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[56] -+ mi := &file_pkg_apis_v1alpha1_ax_proto_msgTypes[59] - if x != nil { - ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) - if ms.LoadMessageInfo() == nil { -@@ -3169,7 +3369,7 @@ func (x *DeleteModelResponse) ProtoReflect() protoreflect.Message { - - // Deprecated: Use DeleteModelResponse.ProtoReflect.Descriptor instead. - func (*DeleteModelResponse) Descriptor() ([]byte, []int) { -- return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{56} -+ return file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP(), []int{59} - } - - var File_pkg_apis_v1alpha1_ax_proto protoreflect.FileDescriptor -@@ -3218,7 +3418,7 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + - "\x04goal\x18\x03 \x01(\tR\x04goal\" \n" + - "\n" + - "GatewayRef\x12\x12\n" + -- "\x04name\x18\x01 \x01(\tR\x04name\"\x95\x02\n" + -+ "\x04name\x18\x01 \x01(\tR\x04name\"\xcb\x02\n" + - "\n" + - "TaskStatus\x12\x14\n" + - "\x05phase\x18\x01 \x01(\tR\x05phase\x12\x0e\n" + -@@ -3229,15 +3429,25 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + - "\x05usage\x18\x06 \x01(\v2\x17.ax.v1alpha1.UsageStatsR\x05usage\x126\n" + - "\n" + - "conditions\x18\a \x03(\v2\x16.ax.v1alpha1.ConditionR\n" + -- "conditions\"x\n" + -+ "conditions\x124\n" + -+ "\acommand\x18\b \x01(\v2\x1a.ax.v1alpha1.CommandStatusR\acommand\"\xc9\x01\n" + -+ "\rCommandStatus\x12\x16\n" + -+ "\x06exited\x18\x01 \x01(\bR\x06exited\x12\x1b\n" + -+ "\texit_code\x18\x02 \x01(\x05R\bexitCode\x12;\n" + -+ "\vfinished_at\x18\x03 \x01(\v2\x1a.google.protobuf.TimestampR\n" + -+ "finishedAt\x12!\n" + -+ "\fresult_bytes\x18\x04 \x01(\x03R\vresultBytes\x12#\n" + -+ "\rresult_sha256\x18\x05 \x01(\tR\fresultSha256\"x\n" + - "\x0fPendingApproval\x12\x0e\n" + - "\x02id\x18\x01 \x01(\tR\x02id\x12\x16\n" + - "\x06action\x18\x02 \x01(\tR\x06action\x12=\n" + -- "\frequested_at\x18\x03 \x01(\v2\x1a.google.protobuf.TimestampR\vrequestedAt\"^\n" + -+ "\frequested_at\x18\x03 \x01(\v2\x1a.google.protobuf.TimestampR\vrequestedAt\"}\n" + - "\n" + - "UsageStats\x12#\n" + - "\rprompt_tokens\x18\x01 \x01(\x05R\fpromptTokens\x12+\n" + -- "\x11completion_tokens\x18\x02 \x01(\x05R\x10completionTokens\"\xb7\x01\n" + -+ "\x11completion_tokens\x18\x02 \x01(\x05R\x10completionTokens\x12\x1d\n" + -+ "\n" + -+ "tool_calls\x18\x03 \x01(\x05R\ttoolCalls\"\xb7\x01\n" + - "\tCondition\x12\x12\n" + - "\x04type\x18\x01 \x01(\tR\x04type\x12\x16\n" + - "\x06status\x18\x02 \x01(\tR\x06status\x12L\n" + -@@ -3324,7 +3534,14 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + - "\x03key\x18\x02 \x01(\tR\x03key\"@\n" + - "\x0eGetTaskRequest\x12\x1a\n" + - "\batespace\x18\x01 \x01(\tR\batespace\x12\x12\n" + -- "\x04name\x18\x02 \x01(\tR\x04name\"\\\n" + -+ "\x04name\x18\x02 \x01(\tR\x04name\"F\n" + -+ "\x14GetTaskResultRequest\x12\x1a\n" + -+ "\batespace\x18\x01 \x01(\tR\batespace\x12\x12\n" + -+ "\x04name\x18\x02 \x01(\tR\x04name\">\n" + -+ "\n" + -+ "TaskResult\x12\x18\n" + -+ "\acontent\x18\x01 \x01(\fR\acontent\x12\x16\n" + -+ "\x06sha256\x18\x02 \x01(\tR\x06sha256\"\\\n" + - "\x10ListTasksRequest\x12\x1a\n" + - "\batespace\x18\x01 \x01(\tR\batespace\x12\x14\n" + - "\x05limit\x18\x02 \x01(\x03R\x05limit\x12\x16\n" + -@@ -3389,7 +3606,7 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + - "\x12DeleteModelRequest\x12\x1a\n" + - "\batespace\x18\x01 \x01(\tR\batespace\x12\x12\n" + - "\x04name\x18\x02 \x01(\tR\x04name\"\x15\n" + -- "\x13DeleteModelResponse2\x9e\v\n" + -+ "\x13DeleteModelResponse2\xeb\v\n" + - "\x02AX\x129\n" + - "\aGetTask\x12\x1b.ax.v1alpha1.GetTaskRequest\x1a\x11.ax.v1alpha1.Task\x12J\n" + - "\tListTasks\x12\x1d.ax.v1alpha1.ListTasksRequest\x1a\x1e.ax.v1alpha1.ListTasksResponse\x12?\n" + -@@ -3400,7 +3617,8 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + - "\vSuspendTask\x12\x1f.ax.v1alpha1.SuspendTaskRequest\x1a\x11.ax.v1alpha1.Task\x12?\n" + - "\n" + - "ResumeTask\x12\x1e.ax.v1alpha1.ResumeTaskRequest\x1a\x11.ax.v1alpha1.Task\x12L\n" + -- "\tWatchTask\x12\x1d.ax.v1alpha1.WatchTaskRequest\x1a\x1e.ax.v1alpha1.WatchTaskResponse0\x01\x12B\n" + -+ "\tWatchTask\x12\x1d.ax.v1alpha1.WatchTaskRequest\x1a\x1e.ax.v1alpha1.WatchTaskResponse0\x01\x12K\n" + -+ "\rGetTaskResult\x12!.ax.v1alpha1.GetTaskResultRequest\x1a\x17.ax.v1alpha1.TaskResult\x12B\n" + - "\n" + - "GetGateway\x12\x1e.ax.v1alpha1.GetGatewayRequest\x1a\x14.ax.v1alpha1.Gateway\x12S\n" + - "\fListGateways\x12 .ax.v1alpha1.ListGatewaysRequest\x1a!.ax.v1alpha1.ListGatewaysResponse\x12H\n" + -@@ -3428,7 +3646,7 @@ func file_pkg_apis_v1alpha1_ax_proto_rawDescGZIP() []byte { - return file_pkg_apis_v1alpha1_ax_proto_rawDescData - } - --var file_pkg_apis_v1alpha1_ax_proto_msgTypes = make([]protoimpl.MessageInfo, 57) -+var file_pkg_apis_v1alpha1_ax_proto_msgTypes = make([]protoimpl.MessageInfo, 60) - var file_pkg_apis_v1alpha1_ax_proto_goTypes = []any{ - (*ObjectMeta)(nil), // 0: ax.v1alpha1.ObjectMeta - (*Task)(nil), // 1: ax.v1alpha1.Task -@@ -3439,59 +3657,62 @@ var file_pkg_apis_v1alpha1_ax_proto_goTypes = []any{ - (*WorkspaceRef)(nil), // 6: ax.v1alpha1.WorkspaceRef - (*GatewayRef)(nil), // 7: ax.v1alpha1.GatewayRef - (*TaskStatus)(nil), // 8: ax.v1alpha1.TaskStatus -- (*PendingApproval)(nil), // 9: ax.v1alpha1.PendingApproval -- (*UsageStats)(nil), // 10: ax.v1alpha1.UsageStats -- (*Condition)(nil), // 11: ax.v1alpha1.Condition -- (*Gateway)(nil), // 12: ax.v1alpha1.Gateway -- (*GatewaySpec)(nil), // 13: ax.v1alpha1.GatewaySpec -- (*Listener)(nil), // 14: ax.v1alpha1.Listener -- (*EgressConfig)(nil), // 15: ax.v1alpha1.EgressConfig -- (*EgressAllowlist)(nil), // 16: ax.v1alpha1.EgressAllowlist -- (*HostRule)(nil), // 17: ax.v1alpha1.HostRule -- (*Workspace)(nil), // 18: ax.v1alpha1.Workspace -- (*WorkspaceSpec)(nil), // 19: ax.v1alpha1.WorkspaceSpec -- (*GitRepo)(nil), // 20: ax.v1alpha1.GitRepo -- (*MCPConfig)(nil), // 21: ax.v1alpha1.MCPConfig -- (*MCPRegistry)(nil), // 22: ax.v1alpha1.MCPRegistry -- (*MCPServer)(nil), // 23: ax.v1alpha1.MCPServer -- (*SkillsConfig)(nil), // 24: ax.v1alpha1.SkillsConfig -- (*SkillRegistry)(nil), // 25: ax.v1alpha1.SkillRegistry -- (*Model)(nil), // 26: ax.v1alpha1.Model -- (*ModelSpec)(nil), // 27: ax.v1alpha1.ModelSpec -- (*SecretKeyRef)(nil), // 28: ax.v1alpha1.SecretKeyRef -- (*GetTaskRequest)(nil), // 29: ax.v1alpha1.GetTaskRequest -- (*ListTasksRequest)(nil), // 30: ax.v1alpha1.ListTasksRequest -- (*ListTasksResponse)(nil), // 31: ax.v1alpha1.ListTasksResponse -- (*UpdateTaskRequest)(nil), // 32: ax.v1alpha1.UpdateTaskRequest -- (*DeleteTaskRequest)(nil), // 33: ax.v1alpha1.DeleteTaskRequest -- (*DeleteTaskResponse)(nil), // 34: ax.v1alpha1.DeleteTaskResponse -- (*SuspendTaskRequest)(nil), // 35: ax.v1alpha1.SuspendTaskRequest -- (*ResumeTaskRequest)(nil), // 36: ax.v1alpha1.ResumeTaskRequest -- (*WatchTaskRequest)(nil), // 37: ax.v1alpha1.WatchTaskRequest -- (*WatchTaskResponse)(nil), // 38: ax.v1alpha1.WatchTaskResponse -- (*GetGatewayRequest)(nil), // 39: ax.v1alpha1.GetGatewayRequest -- (*ListGatewaysRequest)(nil), // 40: ax.v1alpha1.ListGatewaysRequest -- (*ListGatewaysResponse)(nil), // 41: ax.v1alpha1.ListGatewaysResponse -- (*UpdateGatewayRequest)(nil), // 42: ax.v1alpha1.UpdateGatewayRequest -- (*DeleteGatewayRequest)(nil), // 43: ax.v1alpha1.DeleteGatewayRequest -- (*DeleteGatewayResponse)(nil), // 44: ax.v1alpha1.DeleteGatewayResponse -- (*GetWorkspaceRequest)(nil), // 45: ax.v1alpha1.GetWorkspaceRequest -- (*ListWorkspacesRequest)(nil), // 46: ax.v1alpha1.ListWorkspacesRequest -- (*ListWorkspacesResponse)(nil), // 47: ax.v1alpha1.ListWorkspacesResponse -- (*UpdateWorkspaceRequest)(nil), // 48: ax.v1alpha1.UpdateWorkspaceRequest -- (*DeleteWorkspaceRequest)(nil), // 49: ax.v1alpha1.DeleteWorkspaceRequest -- (*DeleteWorkspaceResponse)(nil), // 50: ax.v1alpha1.DeleteWorkspaceResponse -- (*GetModelRequest)(nil), // 51: ax.v1alpha1.GetModelRequest -- (*ListModelsRequest)(nil), // 52: ax.v1alpha1.ListModelsRequest -- (*ListModelsResponse)(nil), // 53: ax.v1alpha1.ListModelsResponse -- (*UpdateModelRequest)(nil), // 54: ax.v1alpha1.UpdateModelRequest -- (*DeleteModelRequest)(nil), // 55: ax.v1alpha1.DeleteModelRequest -- (*DeleteModelResponse)(nil), // 56: ax.v1alpha1.DeleteModelResponse -- (*timestamppb.Timestamp)(nil), // 57: google.protobuf.Timestamp -- (*structpb.Struct)(nil), // 58: google.protobuf.Struct -+ (*CommandStatus)(nil), // 9: ax.v1alpha1.CommandStatus -+ (*PendingApproval)(nil), // 10: ax.v1alpha1.PendingApproval -+ (*UsageStats)(nil), // 11: ax.v1alpha1.UsageStats -+ (*Condition)(nil), // 12: ax.v1alpha1.Condition -+ (*Gateway)(nil), // 13: ax.v1alpha1.Gateway -+ (*GatewaySpec)(nil), // 14: ax.v1alpha1.GatewaySpec -+ (*Listener)(nil), // 15: ax.v1alpha1.Listener -+ (*EgressConfig)(nil), // 16: ax.v1alpha1.EgressConfig -+ (*EgressAllowlist)(nil), // 17: ax.v1alpha1.EgressAllowlist -+ (*HostRule)(nil), // 18: ax.v1alpha1.HostRule -+ (*Workspace)(nil), // 19: ax.v1alpha1.Workspace -+ (*WorkspaceSpec)(nil), // 20: ax.v1alpha1.WorkspaceSpec -+ (*GitRepo)(nil), // 21: ax.v1alpha1.GitRepo -+ (*MCPConfig)(nil), // 22: ax.v1alpha1.MCPConfig -+ (*MCPRegistry)(nil), // 23: ax.v1alpha1.MCPRegistry -+ (*MCPServer)(nil), // 24: ax.v1alpha1.MCPServer -+ (*SkillsConfig)(nil), // 25: ax.v1alpha1.SkillsConfig -+ (*SkillRegistry)(nil), // 26: ax.v1alpha1.SkillRegistry -+ (*Model)(nil), // 27: ax.v1alpha1.Model -+ (*ModelSpec)(nil), // 28: ax.v1alpha1.ModelSpec -+ (*SecretKeyRef)(nil), // 29: ax.v1alpha1.SecretKeyRef -+ (*GetTaskRequest)(nil), // 30: ax.v1alpha1.GetTaskRequest -+ (*GetTaskResultRequest)(nil), // 31: ax.v1alpha1.GetTaskResultRequest -+ (*TaskResult)(nil), // 32: ax.v1alpha1.TaskResult -+ (*ListTasksRequest)(nil), // 33: ax.v1alpha1.ListTasksRequest -+ (*ListTasksResponse)(nil), // 34: ax.v1alpha1.ListTasksResponse -+ (*UpdateTaskRequest)(nil), // 35: ax.v1alpha1.UpdateTaskRequest -+ (*DeleteTaskRequest)(nil), // 36: ax.v1alpha1.DeleteTaskRequest -+ (*DeleteTaskResponse)(nil), // 37: ax.v1alpha1.DeleteTaskResponse -+ (*SuspendTaskRequest)(nil), // 38: ax.v1alpha1.SuspendTaskRequest -+ (*ResumeTaskRequest)(nil), // 39: ax.v1alpha1.ResumeTaskRequest -+ (*WatchTaskRequest)(nil), // 40: ax.v1alpha1.WatchTaskRequest -+ (*WatchTaskResponse)(nil), // 41: ax.v1alpha1.WatchTaskResponse -+ (*GetGatewayRequest)(nil), // 42: ax.v1alpha1.GetGatewayRequest -+ (*ListGatewaysRequest)(nil), // 43: ax.v1alpha1.ListGatewaysRequest -+ (*ListGatewaysResponse)(nil), // 44: ax.v1alpha1.ListGatewaysResponse -+ (*UpdateGatewayRequest)(nil), // 45: ax.v1alpha1.UpdateGatewayRequest -+ (*DeleteGatewayRequest)(nil), // 46: ax.v1alpha1.DeleteGatewayRequest -+ (*DeleteGatewayResponse)(nil), // 47: ax.v1alpha1.DeleteGatewayResponse -+ (*GetWorkspaceRequest)(nil), // 48: ax.v1alpha1.GetWorkspaceRequest -+ (*ListWorkspacesRequest)(nil), // 49: ax.v1alpha1.ListWorkspacesRequest -+ (*ListWorkspacesResponse)(nil), // 50: ax.v1alpha1.ListWorkspacesResponse -+ (*UpdateWorkspaceRequest)(nil), // 51: ax.v1alpha1.UpdateWorkspaceRequest -+ (*DeleteWorkspaceRequest)(nil), // 52: ax.v1alpha1.DeleteWorkspaceRequest -+ (*DeleteWorkspaceResponse)(nil), // 53: ax.v1alpha1.DeleteWorkspaceResponse -+ (*GetModelRequest)(nil), // 54: ax.v1alpha1.GetModelRequest -+ (*ListModelsRequest)(nil), // 55: ax.v1alpha1.ListModelsRequest -+ (*ListModelsResponse)(nil), // 56: ax.v1alpha1.ListModelsResponse -+ (*UpdateModelRequest)(nil), // 57: ax.v1alpha1.UpdateModelRequest -+ (*DeleteModelRequest)(nil), // 58: ax.v1alpha1.DeleteModelRequest -+ (*DeleteModelResponse)(nil), // 59: ax.v1alpha1.DeleteModelResponse -+ (*timestamppb.Timestamp)(nil), // 60: google.protobuf.Timestamp -+ (*structpb.Struct)(nil), // 61: google.protobuf.Struct - } - var file_pkg_apis_v1alpha1_ax_proto_depIdxs = []int32{ -- 57, // 0: ax.v1alpha1.ObjectMeta.creation_timestamp:type_name -> google.protobuf.Timestamp -+ 60, // 0: ax.v1alpha1.ObjectMeta.creation_timestamp:type_name -> google.protobuf.Timestamp - 0, // 1: ax.v1alpha1.Task.metadata:type_name -> ax.v1alpha1.ObjectMeta - 2, // 2: ax.v1alpha1.Task.spec:type_name -> ax.v1alpha1.TaskSpec - 8, // 3: ax.v1alpha1.Task.status:type_name -> ax.v1alpha1.TaskStatus -@@ -3501,82 +3722,86 @@ var file_pkg_apis_v1alpha1_ax_proto_depIdxs = []int32{ - 7, // 7: ax.v1alpha1.TaskSpec.gateway:type_name -> ax.v1alpha1.GatewayRef - 5, // 8: ax.v1alpha1.ResourceReqs.requests:type_name -> ax.v1alpha1.ResourceList - 5, // 9: ax.v1alpha1.ResourceReqs.limits:type_name -> ax.v1alpha1.ResourceList -- 9, // 10: ax.v1alpha1.TaskStatus.pending_approval:type_name -> ax.v1alpha1.PendingApproval -- 10, // 11: ax.v1alpha1.TaskStatus.usage:type_name -> ax.v1alpha1.UsageStats -- 11, // 12: ax.v1alpha1.TaskStatus.conditions:type_name -> ax.v1alpha1.Condition -- 57, // 13: ax.v1alpha1.PendingApproval.requested_at:type_name -> google.protobuf.Timestamp -- 57, // 14: ax.v1alpha1.Condition.last_transition_time:type_name -> google.protobuf.Timestamp -- 0, // 15: ax.v1alpha1.Gateway.metadata:type_name -> ax.v1alpha1.ObjectMeta -- 13, // 16: ax.v1alpha1.Gateway.spec:type_name -> ax.v1alpha1.GatewaySpec -- 14, // 17: ax.v1alpha1.GatewaySpec.listeners:type_name -> ax.v1alpha1.Listener -- 15, // 18: ax.v1alpha1.GatewaySpec.egress:type_name -> ax.v1alpha1.EgressConfig -- 16, // 19: ax.v1alpha1.EgressConfig.allowlist:type_name -> ax.v1alpha1.EgressAllowlist -- 17, // 20: ax.v1alpha1.EgressAllowlist.hosts:type_name -> ax.v1alpha1.HostRule -- 0, // 21: ax.v1alpha1.Workspace.metadata:type_name -> ax.v1alpha1.ObjectMeta -- 19, // 22: ax.v1alpha1.Workspace.spec:type_name -> ax.v1alpha1.WorkspaceSpec -- 20, // 23: ax.v1alpha1.WorkspaceSpec.git:type_name -> ax.v1alpha1.GitRepo -- 21, // 24: ax.v1alpha1.WorkspaceSpec.mcp:type_name -> ax.v1alpha1.MCPConfig -- 24, // 25: ax.v1alpha1.WorkspaceSpec.skills:type_name -> ax.v1alpha1.SkillsConfig -- 22, // 26: ax.v1alpha1.MCPConfig.registries:type_name -> ax.v1alpha1.MCPRegistry -- 23, // 27: ax.v1alpha1.MCPConfig.servers:type_name -> ax.v1alpha1.MCPServer -- 23, // 28: ax.v1alpha1.MCPRegistry.servers:type_name -> ax.v1alpha1.MCPServer -- 25, // 29: ax.v1alpha1.SkillsConfig.registries:type_name -> ax.v1alpha1.SkillRegistry -- 0, // 30: ax.v1alpha1.Model.metadata:type_name -> ax.v1alpha1.ObjectMeta -- 27, // 31: ax.v1alpha1.Model.spec:type_name -> ax.v1alpha1.ModelSpec -- 28, // 32: ax.v1alpha1.ModelSpec.secret_key:type_name -> ax.v1alpha1.SecretKeyRef -- 58, // 33: ax.v1alpha1.ModelSpec.parameters:type_name -> google.protobuf.Struct -- 1, // 34: ax.v1alpha1.ListTasksResponse.tasks:type_name -> ax.v1alpha1.Task -- 1, // 35: ax.v1alpha1.UpdateTaskRequest.task:type_name -> ax.v1alpha1.Task -- 1, // 36: ax.v1alpha1.WatchTaskResponse.task:type_name -> ax.v1alpha1.Task -- 12, // 37: ax.v1alpha1.ListGatewaysResponse.gateways:type_name -> ax.v1alpha1.Gateway -- 12, // 38: ax.v1alpha1.UpdateGatewayRequest.gateway:type_name -> ax.v1alpha1.Gateway -- 18, // 39: ax.v1alpha1.ListWorkspacesResponse.workspaces:type_name -> ax.v1alpha1.Workspace -- 18, // 40: ax.v1alpha1.UpdateWorkspaceRequest.workspace:type_name -> ax.v1alpha1.Workspace -- 26, // 41: ax.v1alpha1.ListModelsResponse.models:type_name -> ax.v1alpha1.Model -- 26, // 42: ax.v1alpha1.UpdateModelRequest.model:type_name -> ax.v1alpha1.Model -- 29, // 43: ax.v1alpha1.AX.GetTask:input_type -> ax.v1alpha1.GetTaskRequest -- 30, // 44: ax.v1alpha1.AX.ListTasks:input_type -> ax.v1alpha1.ListTasksRequest -- 32, // 45: ax.v1alpha1.AX.UpdateTask:input_type -> ax.v1alpha1.UpdateTaskRequest -- 33, // 46: ax.v1alpha1.AX.DeleteTask:input_type -> ax.v1alpha1.DeleteTaskRequest -- 35, // 47: ax.v1alpha1.AX.SuspendTask:input_type -> ax.v1alpha1.SuspendTaskRequest -- 36, // 48: ax.v1alpha1.AX.ResumeTask:input_type -> ax.v1alpha1.ResumeTaskRequest -- 37, // 49: ax.v1alpha1.AX.WatchTask:input_type -> ax.v1alpha1.WatchTaskRequest -- 39, // 50: ax.v1alpha1.AX.GetGateway:input_type -> ax.v1alpha1.GetGatewayRequest -- 40, // 51: ax.v1alpha1.AX.ListGateways:input_type -> ax.v1alpha1.ListGatewaysRequest -- 42, // 52: ax.v1alpha1.AX.UpdateGateway:input_type -> ax.v1alpha1.UpdateGatewayRequest -- 43, // 53: ax.v1alpha1.AX.DeleteGateway:input_type -> ax.v1alpha1.DeleteGatewayRequest -- 45, // 54: ax.v1alpha1.AX.GetWorkspace:input_type -> ax.v1alpha1.GetWorkspaceRequest -- 46, // 55: ax.v1alpha1.AX.ListWorkspaces:input_type -> ax.v1alpha1.ListWorkspacesRequest -- 48, // 56: ax.v1alpha1.AX.UpdateWorkspace:input_type -> ax.v1alpha1.UpdateWorkspaceRequest -- 49, // 57: ax.v1alpha1.AX.DeleteWorkspace:input_type -> ax.v1alpha1.DeleteWorkspaceRequest -- 51, // 58: ax.v1alpha1.AX.GetModel:input_type -> ax.v1alpha1.GetModelRequest -- 52, // 59: ax.v1alpha1.AX.ListModels:input_type -> ax.v1alpha1.ListModelsRequest -- 54, // 60: ax.v1alpha1.AX.UpdateModel:input_type -> ax.v1alpha1.UpdateModelRequest -- 55, // 61: ax.v1alpha1.AX.DeleteModel:input_type -> ax.v1alpha1.DeleteModelRequest -- 1, // 62: ax.v1alpha1.AX.GetTask:output_type -> ax.v1alpha1.Task -- 31, // 63: ax.v1alpha1.AX.ListTasks:output_type -> ax.v1alpha1.ListTasksResponse -- 1, // 64: ax.v1alpha1.AX.UpdateTask:output_type -> ax.v1alpha1.Task -- 34, // 65: ax.v1alpha1.AX.DeleteTask:output_type -> ax.v1alpha1.DeleteTaskResponse -- 1, // 66: ax.v1alpha1.AX.SuspendTask:output_type -> ax.v1alpha1.Task -- 1, // 67: ax.v1alpha1.AX.ResumeTask:output_type -> ax.v1alpha1.Task -- 38, // 68: ax.v1alpha1.AX.WatchTask:output_type -> ax.v1alpha1.WatchTaskResponse -- 12, // 69: ax.v1alpha1.AX.GetGateway:output_type -> ax.v1alpha1.Gateway -- 41, // 70: ax.v1alpha1.AX.ListGateways:output_type -> ax.v1alpha1.ListGatewaysResponse -- 12, // 71: ax.v1alpha1.AX.UpdateGateway:output_type -> ax.v1alpha1.Gateway -- 44, // 72: ax.v1alpha1.AX.DeleteGateway:output_type -> ax.v1alpha1.DeleteGatewayResponse -- 18, // 73: ax.v1alpha1.AX.GetWorkspace:output_type -> ax.v1alpha1.Workspace -- 47, // 74: ax.v1alpha1.AX.ListWorkspaces:output_type -> ax.v1alpha1.ListWorkspacesResponse -- 18, // 75: ax.v1alpha1.AX.UpdateWorkspace:output_type -> ax.v1alpha1.Workspace -- 50, // 76: ax.v1alpha1.AX.DeleteWorkspace:output_type -> ax.v1alpha1.DeleteWorkspaceResponse -- 26, // 77: ax.v1alpha1.AX.GetModel:output_type -> ax.v1alpha1.Model -- 53, // 78: ax.v1alpha1.AX.ListModels:output_type -> ax.v1alpha1.ListModelsResponse -- 26, // 79: ax.v1alpha1.AX.UpdateModel:output_type -> ax.v1alpha1.Model -- 56, // 80: ax.v1alpha1.AX.DeleteModel:output_type -> ax.v1alpha1.DeleteModelResponse -- 62, // [62:81] is the sub-list for method output_type -- 43, // [43:62] is the sub-list for method input_type -- 43, // [43:43] is the sub-list for extension type_name -- 43, // [43:43] is the sub-list for extension extendee -- 0, // [0:43] is the sub-list for field type_name -+ 10, // 10: ax.v1alpha1.TaskStatus.pending_approval:type_name -> ax.v1alpha1.PendingApproval -+ 11, // 11: ax.v1alpha1.TaskStatus.usage:type_name -> ax.v1alpha1.UsageStats -+ 12, // 12: ax.v1alpha1.TaskStatus.conditions:type_name -> ax.v1alpha1.Condition -+ 9, // 13: ax.v1alpha1.TaskStatus.command:type_name -> ax.v1alpha1.CommandStatus -+ 60, // 14: ax.v1alpha1.CommandStatus.finished_at:type_name -> google.protobuf.Timestamp -+ 60, // 15: ax.v1alpha1.PendingApproval.requested_at:type_name -> google.protobuf.Timestamp -+ 60, // 16: ax.v1alpha1.Condition.last_transition_time:type_name -> google.protobuf.Timestamp -+ 0, // 17: ax.v1alpha1.Gateway.metadata:type_name -> ax.v1alpha1.ObjectMeta -+ 14, // 18: ax.v1alpha1.Gateway.spec:type_name -> ax.v1alpha1.GatewaySpec -+ 15, // 19: ax.v1alpha1.GatewaySpec.listeners:type_name -> ax.v1alpha1.Listener -+ 16, // 20: ax.v1alpha1.GatewaySpec.egress:type_name -> ax.v1alpha1.EgressConfig -+ 17, // 21: ax.v1alpha1.EgressConfig.allowlist:type_name -> ax.v1alpha1.EgressAllowlist -+ 18, // 22: ax.v1alpha1.EgressAllowlist.hosts:type_name -> ax.v1alpha1.HostRule -+ 0, // 23: ax.v1alpha1.Workspace.metadata:type_name -> ax.v1alpha1.ObjectMeta -+ 20, // 24: ax.v1alpha1.Workspace.spec:type_name -> ax.v1alpha1.WorkspaceSpec -+ 21, // 25: ax.v1alpha1.WorkspaceSpec.git:type_name -> ax.v1alpha1.GitRepo -+ 22, // 26: ax.v1alpha1.WorkspaceSpec.mcp:type_name -> ax.v1alpha1.MCPConfig -+ 25, // 27: ax.v1alpha1.WorkspaceSpec.skills:type_name -> ax.v1alpha1.SkillsConfig -+ 23, // 28: ax.v1alpha1.MCPConfig.registries:type_name -> ax.v1alpha1.MCPRegistry -+ 24, // 29: ax.v1alpha1.MCPConfig.servers:type_name -> ax.v1alpha1.MCPServer -+ 24, // 30: ax.v1alpha1.MCPRegistry.servers:type_name -> ax.v1alpha1.MCPServer -+ 26, // 31: ax.v1alpha1.SkillsConfig.registries:type_name -> ax.v1alpha1.SkillRegistry -+ 0, // 32: ax.v1alpha1.Model.metadata:type_name -> ax.v1alpha1.ObjectMeta -+ 28, // 33: ax.v1alpha1.Model.spec:type_name -> ax.v1alpha1.ModelSpec -+ 29, // 34: ax.v1alpha1.ModelSpec.secret_key:type_name -> ax.v1alpha1.SecretKeyRef -+ 61, // 35: ax.v1alpha1.ModelSpec.parameters:type_name -> google.protobuf.Struct -+ 1, // 36: ax.v1alpha1.ListTasksResponse.tasks:type_name -> ax.v1alpha1.Task -+ 1, // 37: ax.v1alpha1.UpdateTaskRequest.task:type_name -> ax.v1alpha1.Task -+ 1, // 38: ax.v1alpha1.WatchTaskResponse.task:type_name -> ax.v1alpha1.Task -+ 13, // 39: ax.v1alpha1.ListGatewaysResponse.gateways:type_name -> ax.v1alpha1.Gateway -+ 13, // 40: ax.v1alpha1.UpdateGatewayRequest.gateway:type_name -> ax.v1alpha1.Gateway -+ 19, // 41: ax.v1alpha1.ListWorkspacesResponse.workspaces:type_name -> ax.v1alpha1.Workspace -+ 19, // 42: ax.v1alpha1.UpdateWorkspaceRequest.workspace:type_name -> ax.v1alpha1.Workspace -+ 27, // 43: ax.v1alpha1.ListModelsResponse.models:type_name -> ax.v1alpha1.Model -+ 27, // 44: ax.v1alpha1.UpdateModelRequest.model:type_name -> ax.v1alpha1.Model -+ 30, // 45: ax.v1alpha1.AX.GetTask:input_type -> ax.v1alpha1.GetTaskRequest -+ 33, // 46: ax.v1alpha1.AX.ListTasks:input_type -> ax.v1alpha1.ListTasksRequest -+ 35, // 47: ax.v1alpha1.AX.UpdateTask:input_type -> ax.v1alpha1.UpdateTaskRequest -+ 36, // 48: ax.v1alpha1.AX.DeleteTask:input_type -> ax.v1alpha1.DeleteTaskRequest -+ 38, // 49: ax.v1alpha1.AX.SuspendTask:input_type -> ax.v1alpha1.SuspendTaskRequest -+ 39, // 50: ax.v1alpha1.AX.ResumeTask:input_type -> ax.v1alpha1.ResumeTaskRequest -+ 40, // 51: ax.v1alpha1.AX.WatchTask:input_type -> ax.v1alpha1.WatchTaskRequest -+ 31, // 52: ax.v1alpha1.AX.GetTaskResult:input_type -> ax.v1alpha1.GetTaskResultRequest -+ 42, // 53: ax.v1alpha1.AX.GetGateway:input_type -> ax.v1alpha1.GetGatewayRequest -+ 43, // 54: ax.v1alpha1.AX.ListGateways:input_type -> ax.v1alpha1.ListGatewaysRequest -+ 45, // 55: ax.v1alpha1.AX.UpdateGateway:input_type -> ax.v1alpha1.UpdateGatewayRequest -+ 46, // 56: ax.v1alpha1.AX.DeleteGateway:input_type -> ax.v1alpha1.DeleteGatewayRequest -+ 48, // 57: ax.v1alpha1.AX.GetWorkspace:input_type -> ax.v1alpha1.GetWorkspaceRequest -+ 49, // 58: ax.v1alpha1.AX.ListWorkspaces:input_type -> ax.v1alpha1.ListWorkspacesRequest -+ 51, // 59: ax.v1alpha1.AX.UpdateWorkspace:input_type -> ax.v1alpha1.UpdateWorkspaceRequest -+ 52, // 60: ax.v1alpha1.AX.DeleteWorkspace:input_type -> ax.v1alpha1.DeleteWorkspaceRequest -+ 54, // 61: ax.v1alpha1.AX.GetModel:input_type -> ax.v1alpha1.GetModelRequest -+ 55, // 62: ax.v1alpha1.AX.ListModels:input_type -> ax.v1alpha1.ListModelsRequest -+ 57, // 63: ax.v1alpha1.AX.UpdateModel:input_type -> ax.v1alpha1.UpdateModelRequest -+ 58, // 64: ax.v1alpha1.AX.DeleteModel:input_type -> ax.v1alpha1.DeleteModelRequest -+ 1, // 65: ax.v1alpha1.AX.GetTask:output_type -> ax.v1alpha1.Task -+ 34, // 66: ax.v1alpha1.AX.ListTasks:output_type -> ax.v1alpha1.ListTasksResponse -+ 1, // 67: ax.v1alpha1.AX.UpdateTask:output_type -> ax.v1alpha1.Task -+ 37, // 68: ax.v1alpha1.AX.DeleteTask:output_type -> ax.v1alpha1.DeleteTaskResponse -+ 1, // 69: ax.v1alpha1.AX.SuspendTask:output_type -> ax.v1alpha1.Task -+ 1, // 70: ax.v1alpha1.AX.ResumeTask:output_type -> ax.v1alpha1.Task -+ 41, // 71: ax.v1alpha1.AX.WatchTask:output_type -> ax.v1alpha1.WatchTaskResponse -+ 32, // 72: ax.v1alpha1.AX.GetTaskResult:output_type -> ax.v1alpha1.TaskResult -+ 13, // 73: ax.v1alpha1.AX.GetGateway:output_type -> ax.v1alpha1.Gateway -+ 44, // 74: ax.v1alpha1.AX.ListGateways:output_type -> ax.v1alpha1.ListGatewaysResponse -+ 13, // 75: ax.v1alpha1.AX.UpdateGateway:output_type -> ax.v1alpha1.Gateway -+ 47, // 76: ax.v1alpha1.AX.DeleteGateway:output_type -> ax.v1alpha1.DeleteGatewayResponse -+ 19, // 77: ax.v1alpha1.AX.GetWorkspace:output_type -> ax.v1alpha1.Workspace -+ 50, // 78: ax.v1alpha1.AX.ListWorkspaces:output_type -> ax.v1alpha1.ListWorkspacesResponse -+ 19, // 79: ax.v1alpha1.AX.UpdateWorkspace:output_type -> ax.v1alpha1.Workspace -+ 53, // 80: ax.v1alpha1.AX.DeleteWorkspace:output_type -> ax.v1alpha1.DeleteWorkspaceResponse -+ 27, // 81: ax.v1alpha1.AX.GetModel:output_type -> ax.v1alpha1.Model -+ 56, // 82: ax.v1alpha1.AX.ListModels:output_type -> ax.v1alpha1.ListModelsResponse -+ 27, // 83: ax.v1alpha1.AX.UpdateModel:output_type -> ax.v1alpha1.Model -+ 59, // 84: ax.v1alpha1.AX.DeleteModel:output_type -> ax.v1alpha1.DeleteModelResponse -+ 65, // [65:85] is the sub-list for method output_type -+ 45, // [45:65] is the sub-list for method input_type -+ 45, // [45:45] is the sub-list for extension type_name -+ 45, // [45:45] is the sub-list for extension extendee -+ 0, // [0:45] is the sub-list for field type_name - } - - func init() { file_pkg_apis_v1alpha1_ax_proto_init() } -@@ -3590,7 +3815,7 @@ func file_pkg_apis_v1alpha1_ax_proto_init() { - GoPackagePath: reflect.TypeOf(x{}).PkgPath(), - RawDescriptor: unsafe.Slice(unsafe.StringData(file_pkg_apis_v1alpha1_ax_proto_rawDesc), len(file_pkg_apis_v1alpha1_ax_proto_rawDesc)), - NumEnums: 0, -- NumMessages: 57, -+ NumMessages: 60, - NumExtensions: 0, - NumServices: 1, - }, -diff --git a/pkg/apis/v1alpha1/ax.proto b/pkg/apis/v1alpha1/ax.proto -index cf35316..072305b 100644 ---- a/pkg/apis/v1alpha1/ax.proto -+++ b/pkg/apis/v1alpha1/ax.proto -@@ -34,6 +34,9 @@ service AX { - rpc SuspendTask(SuspendTaskRequest) returns (Task); - rpc ResumeTask(ResumeTaskRequest) returns (Task); - rpc WatchTask(WatchTaskRequest) returns (stream WatchTaskResponse); -+ // GetTaskResult returns the result file the task's command wrote, as copied -+ // by the controller when the command exited. NotFound until then. -+ rpc GetTaskResult(GetTaskResultRequest) returns (TaskResult); - - // Gateways - rpc GetGateway(GetGatewayRequest) returns (Gateway); -@@ -143,6 +146,19 @@ message TaskStatus { - PendingApproval pending_approval = 5; - UsageStats usage = 6; - repeated Condition conditions = 7; -+ // command reports how the task's command finished. Unset while it runs. -+ CommandStatus command = 8; -+} -+ -+// CommandStatus is written by the controller once the runner reports that the -+// task's command exited. The result content itself is served by GetTaskResult. -+message CommandStatus { -+ bool exited = 1; -+ int32 exit_code = 2; -+ google.protobuf.Timestamp finished_at = 3; -+ // result_bytes is the size of the copied result, 0 when there was none. -+ int64 result_bytes = 4; -+ string result_sha256 = 5; - } - - message PendingApproval { -@@ -154,6 +170,7 @@ message PendingApproval { - message UsageStats { - int32 prompt_tokens = 1; - int32 completion_tokens = 2; -+ int32 tool_calls = 3; - } - - message Condition { -@@ -285,6 +302,16 @@ message GetTaskRequest { - string name = 2; - } - -+message GetTaskResultRequest { -+ string atespace = 1; -+ string name = 2; -+} -+ -+message TaskResult { -+ bytes content = 1; -+ string sha256 = 2; -+} -+ - message ListTasksRequest { - string atespace = 1; - int64 limit = 2; -diff --git a/pkg/apis/v1alpha1/ax_grpc.pb.go b/pkg/apis/v1alpha1/ax_grpc.pb.go -index d106157..a8b2ca8 100644 ---- a/pkg/apis/v1alpha1/ax_grpc.pb.go -+++ b/pkg/apis/v1alpha1/ax_grpc.pb.go -@@ -14,8 +14,8 @@ - - // Code generated by protoc-gen-go-grpc. DO NOT EDIT. - // versions: --// - protoc-gen-go-grpc v1.6.0 --// - protoc v7.34.1 -+// - protoc-gen-go-grpc v1.6.2 -+// - protoc v7.35.1 - // source: pkg/apis/v1alpha1/ax.proto - - package v1alpha1 -@@ -40,6 +40,7 @@ const ( - AX_SuspendTask_FullMethodName = "/ax.v1alpha1.AX/SuspendTask" - AX_ResumeTask_FullMethodName = "/ax.v1alpha1.AX/ResumeTask" - AX_WatchTask_FullMethodName = "/ax.v1alpha1.AX/WatchTask" -+ AX_GetTaskResult_FullMethodName = "/ax.v1alpha1.AX/GetTaskResult" - AX_GetGateway_FullMethodName = "/ax.v1alpha1.AX/GetGateway" - AX_ListGateways_FullMethodName = "/ax.v1alpha1.AX/ListGateways" - AX_UpdateGateway_FullMethodName = "/ax.v1alpha1.AX/UpdateGateway" -@@ -68,6 +69,9 @@ type AXClient interface { - SuspendTask(ctx context.Context, in *SuspendTaskRequest, opts ...grpc.CallOption) (*Task, error) - ResumeTask(ctx context.Context, in *ResumeTaskRequest, opts ...grpc.CallOption) (*Task, error) - WatchTask(ctx context.Context, in *WatchTaskRequest, opts ...grpc.CallOption) (grpc.ServerStreamingClient[WatchTaskResponse], error) -+ // GetTaskResult returns the result file the task's command wrote, as copied -+ // by the controller when the command exited. NotFound until then. -+ GetTaskResult(ctx context.Context, in *GetTaskResultRequest, opts ...grpc.CallOption) (*TaskResult, error) - // Gateways - GetGateway(ctx context.Context, in *GetGatewayRequest, opts ...grpc.CallOption) (*Gateway, error) - ListGateways(ctx context.Context, in *ListGatewaysRequest, opts ...grpc.CallOption) (*ListGatewaysResponse, error) -@@ -172,6 +176,16 @@ func (c *aXClient) WatchTask(ctx context.Context, in *WatchTaskRequest, opts ... - // This type alias is provided for backwards compatibility with existing code that references the prior non-generic stream type by name. - type AX_WatchTaskClient = grpc.ServerStreamingClient[WatchTaskResponse] - -+func (c *aXClient) GetTaskResult(ctx context.Context, in *GetTaskResultRequest, opts ...grpc.CallOption) (*TaskResult, error) { -+ cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...) -+ out := new(TaskResult) -+ err := c.cc.Invoke(ctx, AX_GetTaskResult_FullMethodName, in, out, cOpts...) -+ if err != nil { -+ return nil, err -+ } -+ return out, nil -+} -+ - func (c *aXClient) GetGateway(ctx context.Context, in *GetGatewayRequest, opts ...grpc.CallOption) (*Gateway, error) { - cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...) - out := new(Gateway) -@@ -306,6 +320,9 @@ type AXServer interface { - SuspendTask(context.Context, *SuspendTaskRequest) (*Task, error) - ResumeTask(context.Context, *ResumeTaskRequest) (*Task, error) - WatchTask(*WatchTaskRequest, grpc.ServerStreamingServer[WatchTaskResponse]) error -+ // GetTaskResult returns the result file the task's command wrote, as copied -+ // by the controller when the command exited. NotFound until then. -+ GetTaskResult(context.Context, *GetTaskResultRequest) (*TaskResult, error) - // Gateways - GetGateway(context.Context, *GetGatewayRequest) (*Gateway, error) - ListGateways(context.Context, *ListGatewaysRequest) (*ListGatewaysResponse, error) -@@ -352,6 +369,9 @@ func (UnimplementedAXServer) ResumeTask(context.Context, *ResumeTaskRequest) (*T - func (UnimplementedAXServer) WatchTask(*WatchTaskRequest, grpc.ServerStreamingServer[WatchTaskResponse]) error { - return status.Error(codes.Unimplemented, "method WatchTask not implemented") - } -+func (UnimplementedAXServer) GetTaskResult(context.Context, *GetTaskResultRequest) (*TaskResult, error) { -+ return nil, status.Error(codes.Unimplemented, "method GetTaskResult not implemented") -+} - func (UnimplementedAXServer) GetGateway(context.Context, *GetGatewayRequest) (*Gateway, error) { - return nil, status.Error(codes.Unimplemented, "method GetGateway not implemented") - } -@@ -528,6 +548,24 @@ func _AX_WatchTask_Handler(srv interface{}, stream grpc.ServerStream) error { - // This type alias is provided for backwards compatibility with existing code that references the prior non-generic stream type by name. - type AX_WatchTaskServer = grpc.ServerStreamingServer[WatchTaskResponse] - -+func _AX_GetTaskResult_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { -+ in := new(GetTaskResultRequest) -+ if err := dec(in); err != nil { -+ return nil, err -+ } -+ if interceptor == nil { -+ return srv.(AXServer).GetTaskResult(ctx, in) -+ } -+ info := &grpc.UnaryServerInfo{ -+ Server: srv, -+ FullMethod: AX_GetTaskResult_FullMethodName, -+ } -+ handler := func(ctx context.Context, req interface{}) (interface{}, error) { -+ return srv.(AXServer).GetTaskResult(ctx, req.(*GetTaskResultRequest)) -+ } -+ return interceptor(ctx, in, info, handler) -+} -+ - func _AX_GetGateway_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { - in := new(GetGatewayRequest) - if err := dec(in); err != nil { -@@ -775,6 +813,10 @@ var AX_ServiceDesc = grpc.ServiceDesc{ - MethodName: "ResumeTask", - Handler: _AX_ResumeTask_Handler, - }, -+ { -+ MethodName: "GetTaskResult", -+ Handler: _AX_GetTaskResult_Handler, -+ }, - { - MethodName: "GetGateway", - Handler: _AX_GetGateway_Handler, -diff --git a/runner/exit_test.go b/runner/exit_test.go -new file mode 100644 -index 0000000..a340da9 ---- /dev/null -+++ b/runner/exit_test.go -@@ -0,0 +1,65 @@ -+// Copyright 2026 Google LLC -+// -+// Licensed under the Apache License, Version 2.0 (the "License"); -+// you may not use this file except in compliance with the License. -+// You may obtain a copy of the License at -+// -+// http://www.apache.org/licenses/LICENSE-2.0 -+// -+// Unless required by applicable law or agreed to in writing, software -+// distributed under the License is distributed on an "AS IS" BASIS, -+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -+// See the License for the specific language governing permissions and -+// limitations under the License. -+ -+package runner_test -+ -+import ( -+ "encoding/json" -+ "fmt" -+ "io" -+ "net/http" -+ "testing" -+ "time" -+ -+ "github.com/google/ax/internal/metadata" -+) -+ -+// The runner reports the command's exit and serves its result file from the -+// default path under the workspace, which is what the controller reads back. -+func TestRun_ServesExitAndResult(t *testing.T) { -+ h := newHarness(t) -+ h.task.Spec.Command = []string{"/bin/sh", "-c", `mkdir -p .ax && printf '{"ok":1}' > .ax/result.json && exit 5`} -+ h.start(t) -+ -+ select { -+ case e := <-h.exited: -+ if e.ExitCode != 5 { -+ t.Fatalf("exit code = %d, want 5", e.ExitCode) -+ } -+ case <-time.After(10 * time.Second): -+ t.Fatal("command did not exit") -+ } -+ -+ base := fmt.Sprintf("http://127.0.0.1:%d/metadata/v1alpha1/ax", h.cfg.Port) -+ resp, err := http.Get(base + "/exit") -+ if err != nil { -+ t.Fatal(err) -+ } -+ var st metadata.CommandExitStatus -+ err = json.NewDecoder(resp.Body).Decode(&st) -+ resp.Body.Close() -+ if err != nil || !st.Exited || st.ExitCode != 5 { -+ t.Fatalf("exit status = %+v, %v", st, err) -+ } -+ -+ resp, err = http.Get(base + "/result") -+ if err != nil { -+ t.Fatal(err) -+ } -+ body, _ := io.ReadAll(resp.Body) -+ resp.Body.Close() -+ if resp.StatusCode != http.StatusOK || string(body) != `{"ok":1}` { -+ t.Fatalf("result = %d %q", resp.StatusCode, body) -+ } -+} -diff --git a/runner/runner.go b/runner/runner.go -index 4eaadb0..7ec8da9 100644 ---- a/runner/runner.go -+++ b/runner/runner.go -@@ -28,6 +28,7 @@ import ( - "log/slog" - "os" - "os/exec" -+ "path/filepath" - "syscall" - "time" - -@@ -44,6 +45,12 @@ const ( - // binds no workspaces. - DefaultWorkspacePath = v1alpha1.DefaultWorkspacePath - -+ // DefaultResultPath is where the task command writes its result, relative -+ // to the first workspace, unless AX_RESULT_PATH says otherwise. -+ DefaultResultPath = ".ax/result.json" -+ // DefaultUsagePath is where the task command writes its usage JSON. -+ DefaultUsagePath = ".ax/usage.json" -+ - // stopGracePeriod is how long a running task command gets to exit after - // SIGTERM before it is killed during shutdown. - stopGracePeriod = 10 * time.Second -@@ -143,7 +150,11 @@ func Run(ctx context.Context, cfg Config) error { - _ = os.Setenv(e.GetName(), e.GetValue()) - } - -- metaServer := metadata.NewServer(port, cfg.Task, workspaces, metadata.ServerOptions{WorkspacePath: wsPath}) -+ metaServer := metadata.NewServer(port, cfg.Task, workspaces, metadata.ServerOptions{ -+ WorkspacePath: wsPath, -+ ResultPath: outputPath(wsPath, DefaultResultPath, "AX_RESULT_PATH", "AX_CONWIP_RESULT_PATH"), -+ UsagePath: outputPath(wsPath, DefaultUsagePath, "AX_USAGE_PATH", "AX_CONWIP_USAGE_PATH"), -+ }) - if err := metaServer.Start(); err != nil { - return fmt.Errorf("starting metadata server: %w", err) - } -@@ -191,6 +202,13 @@ func Run(ctx context.Context, cfg Config) error { - - select { - case err := <-exited: -+ // Only a command that exited on its own is reported as finished; a -+ // command stopped at shutdown is not, so a suspend never reads as exit. -+ code := -1 -+ if cmd.ProcessState != nil { -+ code = cmd.ProcessState.ExitCode() -+ } -+ metaServer.SetCommandExit(code, time.Now()) - reportExit(cfg, cmd, err) - // Keep the sandbox up and inspectable until told to stop. - <-ctx.Done() -@@ -244,3 +262,19 @@ func commandEnv(task *v1alpha1.Task, port int) []string { - } - return env - } -+ -+// outputPath resolves a command output file: the first of envVars that is set, -+// else def; a relative path is taken relative to the workspace. -+func outputPath(wsPath, def string, envVars ...string) string { -+ p := def -+ for _, v := range envVars { -+ if val := os.Getenv(v); val != "" { -+ p = val -+ break -+ } -+ } -+ if !filepath.IsAbs(p) { -+ p = filepath.Join(wsPath, p) -+ } -+ return p -+} diff --git a/tests/ax-fleet/halogen_stub.py b/tests/ax-fleet/halogen_stub.py index e5efe6f37..190c9b84c 100644 --- a/tests/ax-fleet/halogen_stub.py +++ b/tests/ax-fleet/halogen_stub.py @@ -37,6 +37,13 @@ def reply_for(req): return REPLY LOCK = threading.Lock() +# probe/ax-fleet-nop1: a stand-in floor. A Task's command POSTs its own +# completion to /floor/complete (the in-sandbox adapter reporting straight to +# the floor, LINK-DESIGN L7); the test driver, standing in for the link, reads +# GET /floor/results and then deletes the Task. Every POST is kept, so a +# golden-snapshot pre-run of the command would show as a second report. +FLOOR = [] + def make_handler(log_path): class Handler(BaseHTTPRequestHandler): @@ -68,6 +75,10 @@ def _json(self, code, obj): def do_GET(self): rid = self._record(b"") + if self.path.rstrip("/") == "/floor/results": + with LOCK: + self._json(200, {"reports": list(FLOOR)}) + return if self.path.rstrip("/") == "/health": self._json(200, {"status": "ok", "in_flight": 0, "queued": 0, "request_id": rid}) elif self.path.rstrip("/") == "/v1/models": @@ -79,6 +90,17 @@ def do_POST(self): length = int(self.headers.get("Content-Length") or 0) body = self.rfile.read(length) if length else b"" rid = self._record(body) + if self.path.rstrip("/") == "/floor/complete": + try: + report = json.loads(body or b"{}") + except json.JSONDecodeError: + self._json(400, {"error": "bad json", "request_id": rid}) + return + with LOCK: + FLOOR.append({"received": time.time(), "src": self.client_address[0], "request_id": rid, "report": report}) + n = len(FLOOR) + self._json(200, {"ok": True, "request_id": rid, "n": n}) + return if self.path.rstrip("/") != "/v1/chat/completions": self._json(404, {"error": "not found", "request_id": rid}) return diff --git a/tests/ax-fleet/nodes.nix b/tests/ax-fleet/nodes.nix index a9d890763..6c3f5989b 100644 --- a/tests/ax-fleet/nodes.nix +++ b/tests/ax-fleet/nodes.nix @@ -54,6 +54,21 @@ let variant = "-claude-probe"; }; + # probe/ax-fleet-nop1, test-only: Substrate's own kubectl plugin from the + # same pinned source, so the no-P1 phase can list actors, templates and + # workers (leak and worker-freed checks). Not in any host closure. + kubectlAte = + (pkgs.callPackage ../../pkgs/substrate { + go_1_27 = inputs.nixpkgs-go.legacyPackages.x86_64-linux.go_1_27; + }).ate-setup.overrideAttrs + (o: { + pname = "kubectl-ate"; + subPackages = [ "cmd/kubectl-ate" ]; + meta = o.meta // { + mainProgram = "kubectl-ate"; + }; + }); + lanAddr = address: { interface = "eth1"; inherit address; @@ -109,7 +124,7 @@ let }; in { - inherit probeImage token claudeProbeImage; + inherit probeImage token claudeProbeImage kubectlAte; nas = { ... }: @@ -132,6 +147,7 @@ in (setAddr "eth1" "10.42.0.1" 24) ]; networking.hostName = "nas"; + environment.systemPackages = [ kubectlAte ]; virtualisation = { vlans = [ 1 ]; memorySize = 10240; diff --git a/tests/ax-fleet/phases/25-harness-probe.py b/tests/ax-fleet/phases/25-harness-probe.py deleted file mode 100644 index 1659e95d6..000000000 --- a/tests/ax-fleet/phases/25-harness-probe.py +++ /dev/null @@ -1,128 +0,0 @@ -# Phase 25: the integration's harness probe (ax-fleet INTEGRATE.md). It runs -# after 10-cluster (both switches) and 20-substrate, before the T-cases of -# 30-ax, so its result is in the log even if a later case fails. -# -# One ax Task with spec.sandboxClass gvisor, run by `ax-fleet-smoke probe` -# (the script Tom runs on the real coordinator) in a test-only variant of the -# fleet task image that also carries claude-code. No credential exists in the -# image, the Task or the cluster. The Task runs `claude --version` and one GET -# of the worker's Halogen stand-in, must reach Completed with exit 0, and must -# stay Completed for PROBE_HOLD seconds, sampled every PROBE_EVERY seconds -# (four and more of P1's 15 s resync periods). -# -# There is no containerd RuntimeClass on this path: Substrate's ateom-gvisor -# worker pods run runsc themselves (DESIGN.md D2, 6.2). "gVisor" is proven from -# inside the Task: its /proc/version is not the coordinator VM's kernel. -import json -import re - -PROBE_HOLD = 60 -PROBE_EVERY = 5 -PROBE_AX = "AX_SERVER=http://127.0.0.1:8080 ax -a fleet" -PROBE_STUB_LOG = "/var/lib/halogen-stub/requests.jsonl" - - -def probe_task_phase(name): - # `ax get task NAME` prints YAML; status.phase is its only `phase:` key. - out = coordinator.succeed(f"{PROBE_AX} get task {name}") - m = re.search(r"^\s+phase:\s*(\S+)", out, re.M) - assert m, f"no phase for {name}: {out!r}" - return m.group(1).strip("\"'") - - -def probe_stub_lines(): - return worker.succeed(f"cat {PROBE_STUB_LOG} 2>/dev/null || true").splitlines() - - -with step("probe: k3s nodes Ready, Substrate healthy, ax-server and ax-controller up"): - node_ready("nas") - node_ready("coordinator") - kubectl("-n ate-system wait --for=condition=Available deploy --all --timeout=600s") - kubectl("-n ate-system rollout status statefulset/postgres --timeout=600s") - nas.wait_until_succeeds( - "test \"$(k3s kubectl -n ate-system get workerpool ateom-gvisor -o jsonpath='{.status.readyReplicas}')\" = 2", - timeout=900, - ) - for d in ("ax-redis", "ax-server", "ax-controller"): - kubectl(f"-n ax-system rollout status deploy/{d} --timeout=600s") - coordinator.wait_until_succeeds("curl -sf http://127.0.0.1:8080/healthz", timeout=300) - record("probe_nodes", kubectl("get nodes -o wide").strip().splitlines()) - record("probe_ax_pods", kubectl("-n ax-system get pods -o wide").strip().splitlines()) - workers_wide = kubectl("-n ate-system get pods -l ax.mecattaf.dev/pool=ateom-gvisor -o wide") - record("probe_worker_pods", workers_wide.strip().splitlines()) - worker_nodes = kubectl( - "-n ate-system get pods -l ax.mecattaf.dev/pool=ateom-gvisor " - "-o jsonpath='{range .items[*]}{.spec.nodeName}{\"\\n\"}{end}'" - ).split() - assert worker_nodes and all(n == "coordinator" for n in worker_nodes), worker_nodes - - -with step("probe: a gVisor ax Task on the coordinator runs claude --version and reaches Halogen, Completed and still Completed 60 s later"): - digest = coordinator.succeed(f"cat {CLAUDE_PROBE_OCI}/digest").strip() - assert digest.startswith("sha256:"), digest - image = f"localhost:5000/ax/ax-agent-claude-probe@{digest}" - stub_before = len(probe_stub_lines()) - host_kernel = coordinator.succeed("cat /proc/version").strip() - - rc, out = coordinator.execute( - f"ax-fleet-smoke probe --image {image} --keep --timeout 1500 2>/tmp/probe-smoke.stderr", - timeout=1800, - ) - lines = [l for l in out.strip().splitlines() if l.startswith("{")] - if not lines or rc != 0: - _, err = coordinator.execute("tail -c 4000 /tmp/probe-smoke.stderr") - _, tasks = coordinator.execute(f"{PROBE_AX} get tasks 2>&1 | tail -20") - _, atelet = nas.execute("k3s kubectl -n ate-system logs -l app=atelet --all-containers --tail=80 2>&1") - _, wp = nas.execute("k3s kubectl -n ate-system logs -l ax.mecattaf.dev/pool=ateom-gvisor --all-containers --tail=60 2>&1") - _, ctl = nas.execute("k3s kubectl -n ax-system logs deploy/ax-controller --tail=80 2>&1") - record("probe_diag", {"stderr": err[-4000:], "tasks": tasks[-3000:], "atelet": atelet[-6000:], - "workers": wp[-6000:], "controller": ctl[-6000:]}) - assert lines, f"ax-fleet-smoke probe printed no receipt (rc={rc}): {out!r}" - r = json.loads(lines[-1]) - record("probe_receipt", r) - assert rc == 0 and r.get("pass") is True, r - name = r["task"] - - # Completed, and STAYS Completed: sample for PROBE_HOLD seconds. - samples = [] - t_end = time.monotonic() + PROBE_HOLD - while True: - samples.append(probe_task_phase(name)) - if time.monotonic() >= t_end: - break - time.sleep(PROBE_EVERY) - record("probe_phase_samples", {"hold_seconds": PROBE_HOLD, "every_seconds": PROBE_EVERY, "phases": samples}) - assert len(samples) >= PROBE_HOLD // PROBE_EVERY and all(p == "Completed" for p in samples), samples - record("probe_task_yaml", coordinator.succeed(f"{PROBE_AX} get task {name}").strip().splitlines()) - - res = r["result"] - # claude-code answered inside the sandbox, with no credential anywhere. - assert res["claude_rc"] == 0 and re.search(r"\d+\.\d+\.\d+", res["claude_version"]), res - # The Halogen stand-in on the worker VM answered through the Gateway. - assert res["curl_rc"] == 0 and res["http_code"] == "200", res - assert res["model"] == "halogen-qwen3.8-flash-next", res - # gVisor, not runc: a runc container would report the VM's own kernel. - record("probe_kernels", {"sandbox": res["proc_version"], "coordinator_vm": host_kernel}) - assert res["proc_version"] and res["proc_version"] != host_kernel, (res["proc_version"], host_kernel) - - new = [json.loads(l) for l in probe_stub_lines()[stub_before:] if l.strip()] - models = [q for q in new if q.get("path", "").rstrip("/") == "/v1/models"] - record("probe_stub_requests", models) - assert models, f"the Halogen stand-in logged no /v1/models request: {new!r}" - - # The image was pulled by atelet, which runs on the coordinator only. - atelet_logs = kubectl("-n ate-system logs -l app=atelet --all-containers --tail=-1") - hits = [ - l for l in atelet_logs.splitlines() if "ax-agent-claude-probe" in l or digest.split(":", 1)[1][:16] in l - ] - # Recorded, not asserted: the placement proof is the pool (every worker pod - # on the coordinator, asserted in the first step) and the sandbox kernel above. - record("probe_atelet_image_lines", hits[-5:]) - # '[r]unsc' so the pattern does not match the shell that runs pgrep. - record("probe_runsc_processes", { - "coordinator": coordinator.succeed("pgrep -c -f '[r]unsc' || true").strip(), - "nas": nas.succeed("pgrep -c -f '[r]unsc' || true").strip(), - }) - nas.fail("pgrep -f '[r]unsc'") - - coordinator.succeed(f"{PROBE_AX} delete task {name}") diff --git a/tests/ax-fleet/phases/30-ax.py b/tests/ax-fleet/phases/30-ax.py deleted file mode 100644 index 04ec7681f..000000000 --- a/tests/ax-fleet/phases/30-ax.py +++ /dev/null @@ -1,118 +0,0 @@ -# Phase 4 (Tasks) and phase 5 (resilience) of checks.x86_64-linux.ax-fleet. -# Track ax; DESIGN.md section 12.1. Concatenated after 10-cluster and -# 20-substrate by tests/ax-fleet/default.nix, so `nas`, `coordinator` and -# `worker` are the test nodes and the cluster, Substrate and the WorkerPool are -# already up. Every Task runs through ax-fleet-smoke, the script Tom runs on the -# real coordinator after the switch. -import json -import re - - -def ax_smoke(case, *args, expect_pass=True): - cmd = "ax-fleet-smoke " + " ".join([case, *map(str, args)]) - rc, out = coordinator.execute(cmd + " 2>/dev/null") - lines = [l for l in out.strip().splitlines() if l.startswith("{")] - assert lines, f"{cmd}: no receipt (rc={rc}): {out!r}" - receipt = json.loads(lines[-1]) - print(f"RECEIPT {cmd}: {json.dumps(receipt)}") - if expect_pass: - assert rc == 0 and receipt.get("pass") is True, f"{cmd} failed: {receipt}" - return receipt - - -def kubectl(args): - return nas.succeed(f"k3s kubectl {args}") - - -AX = "AX_SERVER=http://127.0.0.1:8080 ax -a fleet" - - -def task_phase(name): - # `ax get task NAME` prints the Task as YAML; status.phase is its only - # `phase:` key. - out = coordinator.succeed(f"{AX} get task {name}") - m = re.search(r"^\s+phase:\s*(\S+)", out, re.M) - assert m, f"no phase for {name}: {out!r}" - return m.group(1).strip("\"'") - - -with subtest("ax control plane is Available on the NAS, nowhere else"): - for d in ["ax-redis", "ax-server", "ax-controller"]: - kubectl(f"-n ax-system rollout status deploy/{d} --timeout=600s") - node = kubectl( - f"-n ax-system get pods -l app.kubernetes.io/name={d} -o jsonpath='{{.items[0].spec.nodeName}}'" - ).strip() - assert node == "nas", f"{d} runs on {node}" - pv = kubectl( - "get pv -o jsonpath='{range .items[?(@.spec.claimRef.name==\"ax-redis-data\")]}{.spec.local.path}{.spec.hostPath.path}{end}'" - ) - assert pv.startswith("/mnt/nas/services/ax-fleet/local-path"), f"ax-redis volume at {pv!r}" - coordinator.wait_until_succeeds("curl -sf http://127.0.0.1:8080/healthz", timeout=300) - # No Claude credential, and no secret, in any ax object. - kubectl("-n ax-system get secrets -o name | (! grep -q .)") - -with subtest("T1 halogen: Completed, and still Completed after the hold"): - t1 = ax_smoke("halogen", "--hold", 90, "--keep") - assert t1["exit_code"] == 0 and t1["phase_after_hold"] == "Completed", t1 - # The golden-snapshot double execution judge 2 inferred: record, do not fail. - rc, count = worker.execute("curl -sf http://127.0.0.1:8731/stub/requests") - print(f"RECEIPT halogen stub requests after T1: rc={rc} {count.strip()!r}") - -with subtest("T2 pi: Completed with a schema-valid result read back through P1"): - t2 = ax_smoke("pi", "--hold", 30) - assert t2["result"]["valid"] is True, t2 - assert t2["result_bytes"] and t2["result_sha256"], t2 - -with subtest("T3 exit 3: Failed with ExitCode=3"): - t3 = ax_smoke("exit", 3, "--hold", 30) - assert t3["phase"] == "Failed" and t3["ready"]["message"].startswith("ExitCode=3"), t3 - -with subtest("T4 egress-deny: a non-allowlisted target is refused by the Gateway"): - # The coordinator's LAN address answers the worker in this run (checked - # first), so a failure inside the sandbox is the Gateway's refusal. - worker.succeed("curl -s -o /dev/null --max-time 10 http://10.42.0.2/") - t4 = ax_smoke("egress-deny", "http://10.42.0.2/") - assert t4["phase"] == "Failed", t4 - -with subtest("T5 floor 4: four Tasks in a row on the 2-worker pool, none ResourceExhausted"): - t5 = ax_smoke("floor", 4) - for t in t5["tasks"]: - assert t["phase"] == "Completed", t - occupancy = kubectl( - "-n ate-system get pods --field-selector spec.nodeName=coordinator -o name | grep -c ateom || true" - ).strip() - print(f"RECEIPT worker pods on coordinator after {t['task']}: {occupancy}") - -t1_name = t1["task"] - -with subtest("resilience: a restarted ax-controller leaves a finished Task Completed"): - kubectl("-n ax-system delete pod -l app.kubernetes.io/name=ax-controller --wait=true") - kubectl("-n ax-system rollout status deploy/ax-controller --timeout=300s") - coordinator.sleep(45) # three resync periods - assert task_phase(t1_name) == "Completed" - -with subtest("resilience: a restarted ax-redis keeps every Task (AOF on the volume)"): - before = coordinator.succeed(f"{AX} get tasks | tail -n +2 | wc -l").strip() - kubectl("-n ax-system delete pod -l app.kubernetes.io/name=ax-redis --wait=true") - kubectl("-n ax-system rollout status deploy/ax-redis --timeout=300s") - coordinator.wait_until_succeeds(f"{AX} get tasks >/dev/null", timeout=120) - after = coordinator.succeed(f"{AX} get tasks | tail -n +2 | wc -l").strip() - assert before == after and int(after) >= 1, f"tasks before={before} after={after}" - assert task_phase(t1_name) == "Completed" - -with subtest("resilience: the LAN leg flaps, worker pods keep their names, a new T1 completes"): - pods = lambda: kubectl( - "-n ate-system get pods --field-selector spec.nodeName=coordinator -o name | grep ateom | sort" - ) - before = pods() - coordinator.succeed("ip link set eth1 down") - coordinator.sleep(20) - coordinator.succeed("ip link set eth1 up") - nas.wait_until_succeeds( - "k3s kubectl get node coordinator -o jsonpath='{.status.conditions[?(@.type==\"Ready\")].status}' | grep -qx True", - timeout=300, - ) - assert pods() == before, f"worker pods changed: {before!r}" - ax_smoke("halogen") - -coordinator.succeed(f"{AX} delete task {t1_name}") diff --git a/tests/ax-fleet/phases/30-nop1.py b/tests/ax-fleet/phases/30-nop1.py new file mode 100644 index 000000000..21bb0472f --- /dev/null +++ b/tests/ax-fleet/phases/30-nop1.py @@ -0,0 +1,358 @@ +# probe/ax-fleet-nop1: is ax P1 (the completion write-back patch) necessary? +# +# This branch builds ax v0.3.0 with sandbox-class.patch only (P1 removed, the +# controller runs without --running-resync). Instead of the controller learning +# that a command exited, each Task's command reports its own completion to a +# stand-in floor (POST /floor/complete on the worker's stub, the one allowlisted +# egress target), and this driver, standing in for the link, reads the floor +# and then DELETES the Task through the stock ax client. Measured here: +# - the Task's phase just before the delete (stock ax: expected Running); +# - T5 shape without P1: 6 Tasks in a row, then 2 rounds of 2 at once, on the +# 2-worker pool; none ResourceExhausted, every result intact; +# - the delete frees the worker (the next Task gets one) and is idempotent; +# - leaks after delete: ax Tasks, Substrate actors and templates, PVs, the +# RustFS volume; +# - control: without the delete the 3rd Task is refused (the P1 failure); +# - side: SuspendTask instead of delete also frees a worker. +# Everything is recorded with record(); assertions are only the acceptance +# items above. +import base64 +import hashlib +import json +import re +from typing import Any + +AX = "AX_SERVER=http://127.0.0.1:8080 ax -a fleet" +ATE = "KUBECONFIG=/etc/rancher/k3s/k3s.yaml kubectl-ate" +STUB = "http://10.42.0.5:8731" +NS = "fleet" +TASK_TIMEOUT = 900 # test parameter: per-Task wait for a floor report + +TASK_SCRIPT = r""" +set -u +nonce="$(date +%s%N)-$$-$RANDOM" +started="$(date +%s.%N)" +ax-agent halogen-smoke >/tmp/agent.out 2>&1 +rc=$? +out="${AX_RESULT_PATH:-.ax/result.json}" +b64="" +sha="" +if [ -f "$out" ]; then + b64="$(base64 -w0 "$out")" + sha="$(sha256sum "$out" | cut -d' ' -f1)" +fi +body="$(jq -nc --arg task "$FLOOR_TASK" --arg nonce "$nonce" --arg started "$started" \ + --arg pwd "$PWD" --argjson rc "$rc" --arg b64 "$b64" --arg sha "$sha" \ + --arg kernel "$(cat /proc/version 2>/dev/null)" \ + '{task:$task, nonce:$nonce, started:$started, pwd:$pwd, rc:$rc, result_b64:$b64, result_sha256:$sha, kernel:$kernel}')" +for i in 1 2 3 4 5; do + curl -sf --max-time 20 -H 'content-type: application/json' -d "$body" "$FLOOR_URL/floor/complete" >/dev/null && break + sleep 2 +done +exit "$rc" +""" + + +def ax(args: str) -> tuple[int, str]: + rc, out = coordinator.execute(f"{AX} {args} 2>&1") + return rc, out.strip() + + +def apply_task(name: str) -> None: + task: dict[str, Any] = { + "apiVersion": "ax.io/v1alpha1", + "kind": "Task", + "metadata": {"name": name, "atespace": NS}, + "spec": { + "image": IMAGE, + "command": ["bash", "-c", TASK_SCRIPT], + "env": [ + {"name": "HALOGEN_URL", "value": STUB}, + {"name": "FLOOR_URL", "value": STUB}, + {"name": "FLOOR_TASK", "value": name}, + ], + "gateway": {"name": "halogen"}, + }, + } + b = base64.b64encode(json.dumps(task).encode()).decode() + coordinator.succeed(f"echo {b} | base64 -d | {AX} apply -f -") + + +def task_state(name: str) -> Any: + """(exists, phase, ready_reason, ready_message, raw) from `ax get task`.""" + rc, out = ax(f"get task {name}") + if rc != 0: + return {"exists": False, "rc": rc, "out": out[-300:]} + m = re.search(r"^\s+phase:\s*(\S+)", out, re.M) + phase = m.group(1).strip("\"'") if m else None + ready: Any = None + for block in re.split(r"\n\s*- ", out): + if re.search(r"type:\s*\"?Ready\"?", block): + r = re.search(r"reason:\s*(.+)", block) + msg = re.search(r"message:\s*(.+)", block) + ready = { + "reason": r.group(1).strip().strip("\"'") if r else None, + "message": msg.group(1).strip().strip("\"'")[:300] if msg else None, + } + return {"exists": True, "phase": phase, "ready": ready} + + +def floor_reports(name: Any = None) -> Any: + out = worker.succeed("curl -sf http://127.0.0.1:8731/floor/results") + reps = json.loads(out)["reports"] + if name is None: + return reps + return [r for r in reps if r["report"].get("task") == name] + + +def ate_json(args: str) -> Any: + rc, out = nas.execute(f"{ATE} {args} -o json 2>&1") + if rc != 0: + return {"rc": rc, "error": out.strip()[-400:]} + try: + return json.loads(out) + except json.JSONDecodeError: + return {"rc": rc, "unparsed": out.strip()[-400:]} + + +def actors() -> Any: + """Actors in the fleet atespace: [{name, state, worker}], or the error.""" + j = ate_json(f"get actors --atespace {NS}") + if isinstance(j, dict) and ("error" in j or "unparsed" in j): + return j + items = j if isinstance(j, list) else (j.get("actors") or j.get("items") or []) + res: list[Any] = [] + for a in items: + st = a.get("status") or {} + res.append( + { + "name": (a.get("metadata") or {}).get("name"), + "state": st.get("state") or st.get("phase"), + "worker": ((st.get("workerAssignment") or {}).get("worker") or {}).get("name"), + } + ) + return res + + +def templates() -> Any: + j = ate_json(f"get actor-template --atespace {NS}") + if isinstance(j, dict) and ("error" in j or "unparsed" in j): + return j + items = j if isinstance(j, list) else (j.get("actorTemplates") or j.get("items") or []) + return sorted((t.get("metadata") or {}).get("name") for t in items) + + +def rustfs_volume() -> Any: + pvs = json.loads(kubectl("get pv -o json"))["items"] + paths: list[str] = [] + for pv in pvs: + claim = (pv["spec"].get("claimRef") or {}).get("name", "") + path = (pv["spec"].get("local") or {}).get("path") or (pv["spec"].get("hostPath") or {}).get("path") + if "rustfs" in claim and path: + paths.append(path) + res: dict[str, Any] = {"pv_count": len(pvs), "rustfs_paths": paths} + for p in paths: + res[p] = nas.succeed(f"echo $(find {p} -type f | wc -l) $(du -sb {p} | cut -f1)").strip() + return res + + +def leak_snapshot(tag: str) -> Any: + snap: dict[str, Any] = { + "ax_tasks": ax("get tasks")[1], + "actors": actors(), + "templates": templates(), + "volumes": rustfs_volume(), + } + record(f"nop1_leaks_{tag}", snap) + return snap + + +def wait_report(name: str) -> Any: + """Poll the floor for NAME's report; stop early if the Task fails.""" + t0 = time.monotonic() + while time.monotonic() - t0 < TASK_TIMEOUT: + reps = floor_reports(name) + if reps: + return reps, task_state(name), round(time.monotonic() - t0, 1) + st = task_state(name) + if st.get("phase") == "Failed": + return [], st, round(time.monotonic() - t0, 1) + time.sleep(2) + return [], task_state(name), round(time.monotonic() - t0, 1) + + +def check_report(name: str, reps: Any) -> Any: + rep = reps[0]["report"] + raw = base64.b64decode(rep["result_b64"]) if rep.get("result_b64") else b"" + try: + result: Any = json.loads(raw) if raw else {} + except json.JSONDecodeError: + result = {} + intact = ( + rep.get("rc") == 0 + and raw != b"" + and hashlib.sha256(raw).hexdigest() == rep.get("result_sha256") + and result.get("ok") is True + and result.get("content") == "halogen-stub-ok" + ) + return { + "reports": len(reps), + "distinct_nonces": len({r["report"]["nonce"] for r in reps}), + "rc": rep["rc"], + "sha256": rep["result_sha256"], + "intact": intact, + "src": reps[0]["src"], + "gvisor": "gvisor" in rep.get("kernel", ""), + } + + +def delete_task(name: str) -> Any: + """Delete as the link would; measure idempotence and the Task's removal.""" + t0 = time.monotonic() + first = ax(f"delete task {name}") + again = ax(f"delete task {name}") # while Terminating + coordinator.wait_until_succeeds(f"! {AX} get task {name} >/dev/null 2>&1", timeout=300) + gone_s = round(time.monotonic() - t0, 1) + after = ax(f"delete task {name}") # once gone + return { + "first": {"rc": first[0], "out": first[1][-200:]}, + "while_terminating": {"rc": again[0], "out": again[1][-200:]}, + "after_gone": {"rc": after[0], "out": after[1][-200:]}, + "gone_seconds": gone_s, + } + + +def run_one(name: str) -> Any: + apply_task(name) + reps, before, secs = wait_report(name) + entry: dict[str, Any] = {"task": name, "report_seconds": secs, "before_delete": before} + if reps: + entry["floor"] = check_report(name, reps) + entry["actors_before_delete"] = actors() + entry["delete"] = delete_task(name) + entry["actors_after_delete"] = actors() + return entry + + +def resource_exhausted(entry: Any) -> bool: + msg = json.dumps(entry.get("before_delete", {})) + return "ResourceExhausted" in msg or "no free workers" in msg + + +with step("nop1: stock ax control plane up (no P1, no --running-resync)"): + for d in ["ax-redis", "ax-server", "ax-controller"]: + kubectl(f"-n ax-system rollout status deploy/{d} --timeout=600s") + coordinator.wait_until_succeeds("curl -sf http://127.0.0.1:8080/healthz", timeout=300) + args = kubectl("-n ax-system get deploy ax-controller -o jsonpath='{.spec.template.spec.containers[0].args}'") + record("nop1_controller_args", args) + assert "running-resync" not in args, args + rc, out = ax("result task nothing") + record("nop1_ax_result_verb", {"rc": rc, "out": out[-200:]}) + IMAGE = coordinator.succeed("ax-fleet-image-ref").strip() + record("nop1_image", IMAGE) + gw: dict[str, Any] = { + "apiVersion": "ax.io/v1alpha1", + "kind": "Gateway", + "metadata": {"name": "halogen", "atespace": NS}, + "spec": {"egress": {"allowlist": {"hosts": [{"host": "10.42.0.5/32", "port": 8731}]}}}, + } + coordinator.succeed(f"echo {base64.b64encode(json.dumps(gw).encode()).decode()} | base64 -d | {AX} apply -f -") + worker.succeed("curl -sf http://127.0.0.1:8731/floor/results") + record("nop1_workers_baseline", ate_json("get workers")) + leak_snapshot("baseline") + +with step("nop1 T5 sequential: 6 Tasks in a row on the 2-worker pool, reported to the floor, deleted by the driver"): + seq: list[Any] = [] + for i in range(1, 7): + e = run_one(f"nop1-seq-{i}") + record(f"nop1_seq_{i}", e) + seq.append(e) + for e in seq: + assert not resource_exhausted(e), e + assert e.get("floor", {}).get("intact") is True, e + assert e["before_delete"].get("exists") is True, e + assert e["delete"]["first"]["rc"] == 0, e + record("nop1_seq_phase_before_delete", [e["before_delete"].get("phase") for e in seq]) + +with step("nop1 T5 concurrent: 2 rounds of 2 Tasks at once"): + conc: list[Any] = [] + for rnd in (1, 2): + names = [f"nop1-conc-{rnd}-{j}" for j in (1, 2)] + for n in names: + apply_task(n) + entries: list[Any] = [] + for n in names: + reps, before, secs = wait_report(n) + e: dict[str, Any] = {"task": n, "report_seconds": secs, "before_delete": before} + if reps: + e["floor"] = check_report(n, reps) + entries.append(e) + acts = actors() + for e in entries: + e["actors_before_delete"] = acts + e["delete"] = delete_task(e["task"]) + record(f"nop1_conc_round_{rnd}", entries) + conc.extend(entries) + for e in conc: + assert not resource_exhausted(e), e + assert e.get("floor", {}).get("intact") is True, e + leak_snapshot("after_t5") + +with step("nop1 control: without the delete, the 3rd Task on 2 workers"): + ctl: dict[str, Any] = {} + for n in ("nop1-ctl-1", "nop1-ctl-2"): + apply_task(n) + for n in ("nop1-ctl-1", "nop1-ctl-2"): + reps, before, secs = wait_report(n) + ctl[n] = {"reported": bool(reps), "state": before, "seconds": secs} + # Both commands have exited and reported; stock ax still holds their workers. + time.sleep(30) + ctl["held_after_30s"] = {n: task_state(n) for n in ("nop1-ctl-1", "nop1-ctl-2")} + ctl["actors_held"] = actors() + apply_task("nop1-ctl-3") + reps, st, secs = wait_report("nop1-ctl-3") + ctl["nop1-ctl-3"] = {"reported": bool(reps), "state": st, "seconds": secs} + record("nop1_control", ctl) + for n in ("nop1-ctl-1", "nop1-ctl-2", "nop1-ctl-3"): + ctl[f"delete_{n}"] = delete_task(n) + # After deleting the holders, the pool serves again. + ctl["after"] = run_one("nop1-ctl-after") + record("nop1_control", ctl) + assert ctl["after"].get("floor", {}).get("intact") is True, ctl["after"] + +with step("nop1 side: SuspendTask instead of delete frees a worker"): + side: dict[str, Any] = {} + for n in ("nop1-sus-1", "nop1-sus-2"): + apply_task(n) + for n in ("nop1-sus-1", "nop1-sus-2"): + reps, before, secs = wait_report(n) + side[n] = {"reported": bool(reps), "state": before} + side["suspend"] = ax("suspend task nop1-sus-1") + t0 = time.monotonic() + while time.monotonic() - t0 < 300 and task_state("nop1-sus-1").get("phase") != "Suspended": + time.sleep(2) + side["suspended_state"] = task_state("nop1-sus-1") + side["suspended_seconds"] = round(time.monotonic() - t0, 1) + side["actors_after_suspend"] = actors() + apply_task("nop1-sus-3") + reps, st, secs = wait_report("nop1-sus-3") + side["nop1-sus-3"] = {"floor": check_report("nop1-sus-3", reps) if reps else None, "state": st, "seconds": secs} + side["sus1_after_60s"] = task_state("nop1-sus-1") + side["reports_sus1"] = len(floor_reports("nop1-sus-1")) + record("nop1_suspend", side) + leak_snapshot("suspend_before_delete") + for n in ("nop1-sus-1", "nop1-sus-2", "nop1-sus-3"): + side[f"delete_{n}"] = delete_task(n) + record("nop1_suspend", side) + +with step("nop1: leaks after every Task is deleted"): + time.sleep(30) + final = leak_snapshot("final") + all_reports = floor_reports() + per_task: dict[str, list[Any]] = {} + for r in all_reports: + per_task.setdefault(r["report"]["task"], []).append(r["report"]["nonce"]) + record("nop1_reports_per_task", {k: len(v) for k, v in sorted(per_task.items())}) + record("nop1_workers_final", ate_json("get workers")) + rc, out = ax("get tasks") + assert "nop1-" not in out, out diff --git a/tests/ax-fleet/phases/90-rollback.py b/tests/ax-fleet/phases/90-rollback.py deleted file mode 100644 index 987ad662e..000000000 --- a/tests/ax-fleet/phases/90-rollback.py +++ /dev/null @@ -1,52 +0,0 @@ -# Phase 6: rollback, the kill switch proven (DESIGN.md 12.1, 13). Track -# cluster. Reverse order: coordinator, then nas. Each host goes back to its -# base toplevel (ax off) and then ax-fleet-teardown runs, as Tom would. - -BASE = "/run/booted-system/bin/switch-to-configuration test" -LEFTOVER_RULES = "iptables-save 2>/dev/null | grep -E 'KUBE-|FLANNEL|CNI-'" - - -with step("rollback coordinator"): - coordinator.succeed(f"{BASE} >&2") - coordinator.fail("systemctl is-active k3s.service") - coordinator.succeed(f"{TEARDOWN} >&2") - coordinator.fail("ip link show cni0") - coordinator.fail("ip link show flannel.1") - coordinator.fail(LEFTOVER_RULES) - coordinator.fail("pgrep -f containerd-shim") - coordinator.fail("iptables -t mangle -S ax-fleet-guard") - after = sysctls(coordinator) - record("sysctl_coordinator_after_rollback", after) - assert after == base["sysctl_coordinator"], (after, base["sysctl_coordinator"]) - assert user_unit_pid("herdr-standin") == base["herdr_pid"], "herdr stand-in restarted" - assert nm_invocation() == base["nm_invocation"], "NetworkManager restarted" - coordinator.fail("test -e /etc/NetworkManager/conf.d/90-ax-fleet.conf") - worker.succeed("curl -sf --max-time 10 http://10.42.0.2/ | grep -x caddy-ok") - peer.succeed("curl -sf --max-time 10 http://100.105.121.73/ | grep -x caddy-ok") - - -with step("rollback nas"): - nas.succeed(f"{BASE} >&2") - nas.fail("systemctl is-active k3s.service") - nas.fail("systemctl is-active docker-registry.service") - nas.succeed(f"{TEARDOWN} >&2") - nas.fail("ip link show cni0") - nas.fail("ip link show flannel.1") - nas.fail("pgrep -f containerd-shim") - ruleset = nas.succeed("nft -s list ruleset") - for marker in ("KUBE-", "FLANNEL", "CNI-", "ax-fleet"): - assert marker not in ruleset, f"{marker} left in the NAS ruleset" - nixos_fw = nas.succeed("nft -s list table inet nixos-fw") - assert nixos_fw == base["nas_nixos_fw"], "the NAS firewall table differs from the baseline" - record("nas_ruleset_equal_baseline", ruleset == base["nas_nft"]) - if ruleset != base["nas_nft"]: - record("nas_ruleset_after_rollback", ruleset) - after = sysctls(nas) - record("sysctl_nas_after_rollback", after) - assert after == base["sysctl_nas"], (after, base["sysctl_nas"]) - assert unit_invocation(nas, "postgresql.service") == base["postgres_invocation"], "the shared PostgreSQL restarted" - dbs = sorted(nas.succeed("runuser -u postgres -- psql -Atc 'select datname from pg_database'").split()) - assert dbs == base["nas_databases"], dbs - # Left on disk on purpose; deleting it is Tom's call. - nas.succeed("test -d /mnt/fast/k3s/rancher/k3s") - nas.fail("findmnt /var/lib/rancher") From 80d93cf3fb21511b84b68ff6c3916fef2d906b06 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 11:00:43 +0200 Subject: [PATCH 22/37] probe(ax-fleet-nop1): run 2 diagnostics, requeue-aware waits, a shape check Co-Authored-By: Claude Opus 5.5 --- tests/ax-fleet/phases/30-nop1.py | 66 +++++++++++++++++++++++++++++++- 1 file changed, 64 insertions(+), 2 deletions(-) diff --git a/tests/ax-fleet/phases/30-nop1.py b/tests/ax-fleet/phases/30-nop1.py index 21bb0472f..21ef28e3d 100644 --- a/tests/ax-fleet/phases/30-nop1.py +++ b/tests/ax-fleet/phases/30-nop1.py @@ -27,6 +27,7 @@ STUB = "http://10.42.0.5:8731" NS = "fleet" TASK_TIMEOUT = 900 # test parameter: per-Task wait for a floor report +FAILED_FINAL = 240 # test parameter: how long Failed must persist to count as final TASK_SCRIPT = r""" set -u @@ -167,17 +168,47 @@ def leak_snapshot(tag: str) -> Any: return snap +DIAGNOSED: list[str] = [] + + +def diagnose(name: str) -> None: + """Once per Task: the logs that say why a resume failed.""" + if name in DIAGNOSED: + return + DIAGNOSED.append(name) + _, wp = nas.execute("k3s kubectl -n ate-system logs -l ax.mecattaf.dev/pool=ateom-gvisor --all-containers --tail=120 2>&1") + _, atelet = nas.execute("k3s kubectl -n ate-system logs -l app=atelet --all-containers --tail=80 2>&1") + _, ctl = nas.execute("k3s kubectl -n ax-system logs deploy/ax-controller --tail=60 2>&1") + _, stub = worker.execute("tail -n 20 /var/lib/halogen-stub/requests.jsonl 2>&1") + record(f"nop1_diag_{name}", {"workers": wp[-8000:], "atelet": atelet[-6000:], "controller": ctl[-5000:], "stub": stub[-3000:]}) + + def wait_report(name: str) -> Any: - """Poll the floor for NAME's report; stop early if the Task fails.""" + """Poll the floor for NAME's report. Stock ax marks a Task Failed on a + reconcile error and requeues it, so Failed is final only when it says + ResourceExhausted or has lasted FAILED_FINAL seconds.""" t0 = time.monotonic() + failed_since: Any = None + states: list[Any] = [] while time.monotonic() - t0 < TASK_TIMEOUT: reps = floor_reports(name) if reps: return reps, task_state(name), round(time.monotonic() - t0, 1) st = task_state(name) + key = (st.get("phase"), (st.get("ready") or {}).get("reason")) + if not states or states[-1]["key"] != list(key): + states.append({"t": round(time.monotonic() - t0, 1), "key": list(key)}) + record(f"nop1_states_{name}", states) if st.get("phase") == "Failed": - return [], st, round(time.monotonic() - t0, 1) + diagnose(name) + failed_since = failed_since or time.monotonic() + msg = json.dumps(st) + if "ResourceExhausted" in msg or "no free workers" in msg or time.monotonic() - failed_since > FAILED_FINAL: + return [], st, round(time.monotonic() - t0, 1) + else: + failed_since = None time.sleep(2) + diagnose(name) return [], task_state(name), round(time.monotonic() - t0, 1) @@ -261,6 +292,37 @@ def resource_exhausted(entry: Any) -> bool: record("nop1_workers_baseline", ate_json("get workers")) leak_snapshot("baseline") +with step("nop1 shape check: the P1 runs' command form, [ax-agent, halogen-smoke] (recorded, not asserted)"): + shape: dict[str, Any] = {"name": "nop1-shape"} + shape_task: dict[str, Any] = { + "apiVersion": "ax.io/v1alpha1", + "kind": "Task", + "metadata": {"name": "nop1-shape", "atespace": NS}, + "spec": { + "image": IMAGE, + "command": ["ax-agent", "halogen-smoke"], + "env": [{"name": "HALOGEN_URL", "value": STUB}], + "gateway": {"name": "halogen"}, + }, + } + coordinator.succeed(f"echo {base64.b64encode(json.dumps(shape_task).encode()).decode()} | base64 -d | {AX} apply -f -") + shape_states: list[Any] = [] + t0 = time.monotonic() + while time.monotonic() - t0 < 300: + st = task_state("nop1-shape") + key = [st.get("phase"), (st.get("ready") or {}).get("reason")] + if not shape_states or shape_states[-1]["key"] != key: + shape_states.append({"t": round(time.monotonic() - t0, 1), "key": key, "ready": st.get("ready")}) + if key == ["Running", "TaskRunning"]: + break + time.sleep(2) + shape["states"] = shape_states + shape["actors"] = actors() + if shape_states[-1]["key"] != ["Running", "TaskRunning"]: + diagnose("nop1-shape") + shape["delete"] = delete_task("nop1-shape") + record("nop1_shape", shape) + with step("nop1 T5 sequential: 6 Tasks in a row on the 2-worker pool, reported to the floor, deleted by the driver"): seq: list[Any] = [] for i in range(1, 7): From c880c1691df0cb6d85a77d0053ee50245a8f9080 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 11:07:01 +0200 Subject: [PATCH 23/37] probe(ax-fleet-nop1): pass the Task script as one line (stock runner refuses a multi-line command in AX_TASK_YAML) Co-Authored-By: Claude Opus 5.5 --- tests/ax-fleet/phases/30-nop1.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/tests/ax-fleet/phases/30-nop1.py b/tests/ax-fleet/phases/30-nop1.py index 21ef28e3d..5e6b81bd4 100644 --- a/tests/ax-fleet/phases/30-nop1.py +++ b/tests/ax-fleet/phases/30-nop1.py @@ -52,6 +52,7 @@ done exit "$rc" """ +TASK_SCRIPT_B64 = base64.b64encode(TASK_SCRIPT.encode()).decode() def ax(args: str) -> tuple[int, str]: @@ -66,7 +67,10 @@ def apply_task(name: str) -> None: "metadata": {"name": name, "atespace": NS}, "spec": { "image": IMAGE, - "command": ["bash", "-c", TASK_SCRIPT], + # One line: stock v0.3.0's runner refuses AX_TASK_YAML when the + # command carries a multi-line string (MEASURED run 2: "yaml: line + # 30: mapping values are not allowed in this context"). + "command": ["bash", "-c", f"echo {TASK_SCRIPT_B64} | base64 -d | bash"], "env": [ {"name": "HALOGEN_URL", "value": STUB}, {"name": "FLOOR_URL", "value": STUB}, From 7232ad8f399b109530ce1c1cdd3333a2cbd2e521 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 11:21:43 +0200 Subject: [PATCH 24/37] probe(ax-fleet-nop1): retry a failed resume as attempt n+1 (the link's requeue); ateapi diagnostics Co-Authored-By: Claude Opus 5.5 --- tests/ax-fleet/phases/30-nop1.py | 37 +++++++++++++++++++++++--------- 1 file changed, 27 insertions(+), 10 deletions(-) diff --git a/tests/ax-fleet/phases/30-nop1.py b/tests/ax-fleet/phases/30-nop1.py index 5e6b81bd4..688a1fff4 100644 --- a/tests/ax-fleet/phases/30-nop1.py +++ b/tests/ax-fleet/phases/30-nop1.py @@ -27,7 +27,8 @@ STUB = "http://10.42.0.5:8731" NS = "fleet" TASK_TIMEOUT = 900 # test parameter: per-Task wait for a floor report -FAILED_FINAL = 240 # test parameter: how long Failed must persist to count as final +FAILED_FINAL = 20 # test parameter: stock ax does not requeue a failed reconcile (MEASURED run 3: Failed held 5 min) +MAX_ATTEMPTS = 4 # the link's requeue: a Task whose resume failed is deleted and created again as attempt n+1 TASK_SCRIPT = r""" set -u @@ -183,8 +184,9 @@ def diagnose(name: str) -> None: _, wp = nas.execute("k3s kubectl -n ate-system logs -l ax.mecattaf.dev/pool=ateom-gvisor --all-containers --tail=120 2>&1") _, atelet = nas.execute("k3s kubectl -n ate-system logs -l app=atelet --all-containers --tail=80 2>&1") _, ctl = nas.execute("k3s kubectl -n ax-system logs deploy/ax-controller --tail=60 2>&1") + _, pods = nas.execute("k3s kubectl -n ate-system get pods -o wide 2>&1; k3s kubectl -n ate-system logs deploy/ateapi --all-containers --tail=40 2>&1") _, stub = worker.execute("tail -n 20 /var/lib/halogen-stub/requests.jsonl 2>&1") - record(f"nop1_diag_{name}", {"workers": wp[-8000:], "atelet": atelet[-6000:], "controller": ctl[-5000:], "stub": stub[-3000:]}) + record(f"nop1_diag_{name}", {"workers": wp[-8000:], "atelet": atelet[-6000:], "controller": ctl[-5000:], "stub": stub[-3000:], "ate_pods_api": pods[-6000:]}) def wait_report(name: str) -> Any: @@ -258,14 +260,23 @@ def delete_task(name: str) -> Any: def run_one(name: str) -> Any: - apply_task(name) - reps, before, secs = wait_report(name) - entry: dict[str, Any] = {"task": name, "report_seconds": secs, "before_delete": before} - if reps: - entry["floor"] = check_report(name, reps) - entry["actors_before_delete"] = actors() - entry["delete"] = delete_task(name) - entry["actors_after_delete"] = actors() + """One logical job: attempts name-a1.. until the floor has a report.""" + failed: list[Any] = [] + entry: dict[str, Any] = {} + for attempt in range(1, MAX_ATTEMPTS + 1): + n = f"{name}-a{attempt}" + apply_task(n) + reps, before, secs = wait_report(n) + entry = {"task": n, "attempt": attempt, "report_seconds": secs, "before_delete": before} + if reps: + entry["floor"] = check_report(n, reps) + entry["actors_before_delete"] = actors() + entry["delete"] = delete_task(n) + entry["actors_after_delete"] = actors() + if reps or resource_exhausted(entry): + break + failed.append({"task": n, "state": before, "delete_rc": entry["delete"]["first"]["rc"]}) + entry["failed_attempts"] = failed return entry @@ -357,6 +368,12 @@ def resource_exhausted(entry: Any) -> bool: for e in entries: e["actors_before_delete"] = acts e["delete"] = delete_task(e["task"]) + for i, e in enumerate(entries): + if "floor" not in e and not resource_exhausted(e): + # resume failed (not capacity): the link requeues it as a new attempt + retry = run_one(e["task"] + "-retry") + retry["concurrent_first_attempt"] = e + entries[i] = retry record(f"nop1_conc_round_{rnd}", entries) conc.extend(entries) for e in conc: From 1c9e01a7c451b252ca4d6d2fb2dc368fd6b3e3e6 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 10:48:34 +0200 Subject: [PATCH 25/37] probe(ax-fleet-zeropatch): stock ax v0.3.0 with zero patches (nop1 + nosc); port stock-ax sandboxClass control into 30-nop1; assert gVisor kernel per report Co-Authored-By: Claude Opus 5.5 --- modules/ax-fleet/ax-fleet-smoke.sh | 3 +- pkgs/ax/default.nix | 13 +- pkgs/ax/patches/sandbox-class.patch | 304 ---------------------------- tests/ax-fleet/phases/30-nop1.py | 18 ++ 4 files changed, 28 insertions(+), 310 deletions(-) delete mode 100644 pkgs/ax/patches/sandbox-class.patch diff --git a/modules/ax-fleet/ax-fleet-smoke.sh b/modules/ax-fleet/ax-fleet-smoke.sh index 082f260fa..b34a7a3f3 100644 --- a/modules/ax-fleet/ax-fleet-smoke.sh +++ b/modules/ax-fleet/ax-fleet-smoke.sh @@ -39,7 +39,8 @@ done [ "${#args[@]}" -ge 1 ] || { sed -n '2,21p' "$0" >&2; exit 64; } case_="${args[0]}" [ -n "$image" ] || image="$(ax-fleet-image-ref)" -[ "$case_" != probe ] || sandbox_class="${sandbox_class:-gvisor}" +# PROBE VARIANT probe/ax-fleet-nosc: stock ax has no spec.sandboxClass and +# rejects the unknown field, so the probe sends none (ax hardcodes gVisor). run_id="$(date +%s)-$$" ensure_gateway() { diff --git a/pkgs/ax/default.nix b/pkgs/ax/default.nix index c4d0864ee..cf274c382 100644 --- a/pkgs/ax/default.nix +++ b/pkgs/ax/default.nix @@ -96,11 +96,14 @@ buildGo127Module { # the resync, a crashed actor, the runner's endpoints, and the floor test: # four Tasks in a row on a 2-worker pool all Completed, next to the contrast # that without an exit report the third is refused ResourceExhausted. - patches = [ - ./patches/sandbox-class.patch - # probe/ax-fleet-nop1: p1-completion.patch removed; completion is reported - # by the Task to the floor and the link deletes the Task. - ]; + # PROBE VARIANT probe/ax-fleet-nosc: sandbox-class.patch removed to measure + # whether stock v0.3.0's hardcoded SANDBOX_CLASS_GVISOR / gvisor-default is + # enough (evals-2026-09-23/zero-patch/no-sandboxclass.md). P1 kept. + # PROBE VARIANT probe/ax-fleet-zeropatch: BOTH carried patches removed. + # p1-completion.patch: completion is reported by the Task to the floor and + # the link deletes the Task (no-p1.md). sandbox-class.patch: stock v0.3.0 + # already hardcodes SANDBOX_CLASS_GVISOR / gvisor-default (no-sandboxclass.md). + patches = [ ]; # subPackages left unset so all four commands build, matching upstream's # `make build-binaries` plus the cross-compiled runner. -s -w mirrors the diff --git a/pkgs/ax/patches/sandbox-class.patch b/pkgs/ax/patches/sandbox-class.patch deleted file mode 100644 index 68675ea94..000000000 --- a/pkgs/ax/patches/sandbox-class.patch +++ /dev/null @@ -1,304 +0,0 @@ -diff --git a/internal/controller/reconciler.go b/internal/controller/reconciler.go -index 31c399b..c4aebda 100644 ---- a/internal/controller/reconciler.go -+++ b/internal/controller/reconciler.go -@@ -164,11 +164,11 @@ func (r *TaskReconciler) Reconcile(ctx context.Context, task *v1alpha1.Task, gat - } - - // If a custom image, workspace, or extra environment is specified, provision or use a dedicated ActorTemplate -- if task.Spec != nil && (task.Spec.Image != "" || len(extraEnv) > 0) { -+ if task.Spec != nil && (task.Spec.Image != "" || task.Spec.GetSandboxClass() != "" || len(extraEnv) > 0) { - slog.Info("ensuring custom ActorTemplate for task", "image", task.Spec.Image) -- customTemplateName := taskTemplateName(task.Metadata.Name, task.Spec.Image, extraEnv) -+ customTemplateName := taskTemplateName(task.Metadata.Name, task.Spec.Image, task.Spec.GetSandboxClass(), extraEnv) - -- tmpl, err := r.client.EnsureActorTemplateWithImage(ctx, templateAtespace, templateName, atespace, customTemplateName, task.Spec.Image, extraEnv) -+ tmpl, err := r.client.EnsureActorTemplateWithImage(ctx, templateAtespace, templateName, atespace, customTemplateName, task.Spec.Image, task.Spec.GetSandboxClass(), extraEnv) - if err != nil { - slog.Warn("could not create custom ActorTemplate, falling back to default template", "error", err) - } else if tmpl != nil && tmpl.Metadata != nil { -@@ -386,9 +386,14 @@ func (r *TaskReconciler) lookupGeminiKey(ctx context.Context, atespace string) s - - // taskTemplateName derives the per-task ActorTemplate name from the task name and a - // digest of the image and container environment, so a spec change yields a new template. --func taskTemplateName(taskName, image string, env map[string]string) string { -+func taskTemplateName(taskName, image, sandboxClass string, env map[string]string) string { - h := sha256.New() - h.Write([]byte(image)) -+ // Only mixed in when set, so that a task that does not name a sandbox class -+ // keeps the template name it had before the field existed. -+ if sandboxClass != "" { -+ h.Write([]byte("sandboxClass=" + sandboxClass + ";")) -+ } - keys := make([]string, 0, len(env)) - for k := range env { - keys = append(keys, k) -diff --git a/internal/substrate/client.go b/internal/substrate/client.go -index bacb078..2840ce2 100644 ---- a/internal/substrate/client.go -+++ b/internal/substrate/client.go -@@ -210,7 +210,7 @@ const ( - ) - - // BuildActorTemplate constructs a Substrate ActorTemplate based on the standard ate-env specification. --func BuildActorTemplate(atespace, name, image string, envMap map[string]string, command []string, snapshotsBucket string) *ateapipb.ActorTemplate { -+func BuildActorTemplate(atespace, name, image string, envMap map[string]string, command []string, snapshotsBucket, sandboxClass string) *ateapipb.ActorTemplate { - if atespace == "" { - atespace = "default" - } -@@ -269,15 +269,38 @@ func BuildActorTemplate(atespace, name, image string, envMap map[string]string, - FromData: ateapipb.ResumeSource_RESUME_SOURCE_GOLDEN, - }, - }, -- SandboxConfig: &ateapipb.SandboxConfig{ -- SandboxClass: ateapipb.SandboxClass_SANDBOX_CLASS_GVISOR, -- ConfigName: "gvisor-default", -- }, -+ SandboxConfig: sandboxConfigFor(sandboxClass), -+ } -+} -+ -+// sandboxConfigFor maps an ax spec.sandboxClass onto the substrate's own -+// SandboxConfig. The empty string means gVisor, which is what this function's -+// caller hardcoded before the field existed, so the default is unchanged. -+// -+// config_name names a cluster-scoped substrate SandboxConfig object and must -+// match the class. "gvisor-default" is the name ax has always used; the microVM -+// name follows the same convention, and ax neither creates nor verifies it. -+// -+// An unrecognised value should never reach here: v1alpha1.ValidateTask refuses -+// it at admission. If one does, gVisor is the fallback because it is the -+// historical default and the only class whose config object ax has ever named -+// successfully. The substrate enum has no workerd member, so no value of -+// sandboxClass can select one. -+func sandboxConfigFor(sandboxClass string) *ateapipb.SandboxConfig { -+ if sandboxClass == v1alpha1.SandboxClassMicroVM { -+ return &ateapipb.SandboxConfig{ -+ SandboxClass: ateapipb.SandboxClass_SANDBOX_CLASS_MICROVM, -+ ConfigName: "microvm-default", -+ } -+ } -+ return &ateapipb.SandboxConfig{ -+ SandboxClass: ateapipb.SandboxClass_SANDBOX_CLASS_GVISOR, -+ ConfigName: "gvisor-default", - } - } - - // EnsureActorTemplateWithImage creates an ActorTemplate using the specified container image and optional environment variables. --func (c *Client) EnsureActorTemplateWithImage(ctx context.Context, baseAtespace, baseTemplate, targetAtespace, targetTemplate, image string, extraEnv ...map[string]string) (*ateapipb.ActorTemplate, error) { -+func (c *Client) EnsureActorTemplateWithImage(ctx context.Context, baseAtespace, baseTemplate, targetAtespace, targetTemplate, image, sandboxClass string, extraEnv ...map[string]string) (*ateapipb.ActorTemplate, error) { - existing, err := c.GetActorTemplate(ctx, targetAtespace, targetTemplate) - if err == nil && existing != nil { - return existing, nil -@@ -290,7 +313,7 @@ func (c *Client) EnsureActorTemplateWithImage(ctx context.Context, baseAtespace, - } - } - -- tmpl := BuildActorTemplate(targetAtespace, targetTemplate, image, envMap, nil, "") -+ tmpl := BuildActorTemplate(targetAtespace, targetTemplate, image, envMap, nil, "", sandboxClass) - req := &ateapipb.CreateActorTemplateRequest{ - ActorTemplate: tmpl, - } -diff --git a/pkg/apis/v1alpha1/ax.pb.go b/pkg/apis/v1alpha1/ax.pb.go -index b1adb26..9a71dc0 100644 ---- a/pkg/apis/v1alpha1/ax.pb.go -+++ b/pkg/apis/v1alpha1/ax.pb.go -@@ -15,7 +15,7 @@ - // Code generated by protoc-gen-go. DO NOT EDIT. - // versions: - // protoc-gen-go v1.36.11 --// protoc v7.34.1 -+// protoc v7.35.1 - // source: pkg/apis/v1alpha1/ax.proto - - package v1alpha1 -@@ -187,7 +187,23 @@ type TaskSpec struct { - Gateway *GatewayRef `protobuf:"bytes,8,opt,name=gateway,proto3" json:"gateway,omitempty"` - // debug enables the in-container guest services (process execution and file - // access) that back `ax ssh`. Off by default. -- Debug bool `protobuf:"varint,10,opt,name=debug,proto3" json:"debug,omitempty"` -+ Debug bool `protobuf:"varint,10,opt,name=debug,proto3" json:"debug,omitempty"` -+ // sandbox_class selects the sandbox runtime family the task's actor runs in. -+ // Empty means "gvisor", which is what every task got before this field -+ // existed, so the default is unchanged. -+ // -+ // The accepted values are exactly the members Agent Substrate's own -+ // SandboxClass enum offers, lowercased: "gvisor" and "microvm". Any other -+ // value is refused by ValidateTask rather than downgraded, because a task -+ // that silently runs in a weaker sandbox than it asked for is worse than a -+ // task that does not start. In particular there is no "workerd" value: the -+ // substrate enum has no such member, and ax cannot add one. -+ // -+ // "microvm" additionally requires a cluster-scoped substrate SandboxConfig -+ // object named "microvm-default", by analogy with the "gvisor-default" object -+ // the gVisor path has always named. ax neither creates nor verifies it; if it -+ // is absent the substrate rejects the ActorTemplate. -+ SandboxClass string `protobuf:"bytes,11,opt,name=sandbox_class,json=sandboxClass,proto3" json:"sandbox_class,omitempty"` - unknownFields protoimpl.UnknownFields - sizeCache protoimpl.SizeCache - } -@@ -278,6 +294,13 @@ func (x *TaskSpec) GetDebug() bool { - return false - } - -+func (x *TaskSpec) GetSandboxClass() string { -+ if x != nil { -+ return x.SandboxClass -+ } -+ return "" -+} -+ - type EnvVar struct { - state protoimpl.MessageState `protogen:"open.v1"` - Name string `protobuf:"bytes,1,opt,name=name,proto3" json:"name,omitempty"` -@@ -3165,7 +3188,7 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + - "\x04kind\x18\x02 \x01(\tR\x04kind\x123\n" + - "\bmetadata\x18\x03 \x01(\v2\x17.ax.v1alpha1.ObjectMetaR\bmetadata\x12)\n" + - "\x04spec\x18\x04 \x01(\v2\x15.ax.v1alpha1.TaskSpecR\x04spec\x12/\n" + -- "\x06status\x18\x05 \x01(\v2\x17.ax.v1alpha1.TaskStatusR\x06status\"\xd4\x02\n" + -+ "\x06status\x18\x05 \x01(\v2\x17.ax.v1alpha1.TaskStatusR\x06status\"\xf9\x02\n" + - "\bTaskSpec\x12\x18\n" + - "\asuspend\x18\x02 \x01(\bR\asuspend\x12\x14\n" + - "\x05image\x18\x03 \x01(\tR\x05image\x12\x18\n" + -@@ -3177,7 +3200,8 @@ const file_pkg_apis_v1alpha1_ax_proto_rawDesc = "" + - "workspaces\x121\n" + - "\agateway\x18\b \x01(\v2\x17.ax.v1alpha1.GatewayRefR\agateway\x12\x14\n" + - "\x05debug\x18\n" + -- " \x01(\bR\x05debugJ\x04\b\x01\x10\x02J\x04\b\t\x10\n" + -+ " \x01(\bR\x05debug\x12#\n" + -+ "\rsandbox_class\x18\v \x01(\tR\fsandboxClassJ\x04\b\x01\x10\x02J\x04\b\t\x10\n" + - "R\x04goalR\bpolicies\"2\n" + - "\x06EnvVar\x12\x12\n" + - "\x04name\x18\x01 \x01(\tR\x04name\x12\x14\n" + -diff --git a/pkg/apis/v1alpha1/ax.proto b/pkg/apis/v1alpha1/ax.proto -index 62bd57a..cf35316 100644 ---- a/pkg/apis/v1alpha1/ax.proto -+++ b/pkg/apis/v1alpha1/ax.proto -@@ -91,6 +91,22 @@ message TaskSpec { - // debug enables the in-container guest services (process execution and file - // access) that back `ax ssh`. Off by default. - bool debug = 10; -+ // sandbox_class selects the sandbox runtime family the task's actor runs in. -+ // Empty means "gvisor", which is what every task got before this field -+ // existed, so the default is unchanged. -+ // -+ // The accepted values are exactly the members Agent Substrate's own -+ // SandboxClass enum offers, lowercased: "gvisor" and "microvm". Any other -+ // value is refused by ValidateTask rather than downgraded, because a task -+ // that silently runs in a weaker sandbox than it asked for is worse than a -+ // task that does not start. In particular there is no "workerd" value: the -+ // substrate enum has no such member, and ax cannot add one. -+ // -+ // "microvm" additionally requires a cluster-scoped substrate SandboxConfig -+ // object named "microvm-default", by analogy with the "gvisor-default" object -+ // the gVisor path has always named. ax neither creates nor verifies it; if it -+ // is absent the substrate rejects the ActorTemplate. -+ string sandbox_class = 11; - } - - message EnvVar { -diff --git a/pkg/apis/v1alpha1/types.go b/pkg/apis/v1alpha1/types.go -index a65e626..820c19b 100644 ---- a/pkg/apis/v1alpha1/types.go -+++ b/pkg/apis/v1alpha1/types.go -@@ -36,6 +36,16 @@ const ( - - DefaultTaskImage = "gcr.io/ax-substrate/ate-images/ax-task-runner" - -+ // The sandbox classes TaskSpec.SandboxClass accepts. They are exactly the -+ // members of Agent Substrate's own SandboxClass enum, lowercased; ax does -+ // not own that enum and cannot add to it. -+ SandboxClassGVisor = "gvisor" -+ SandboxClassMicroVM = "microvm" -+ -+ // DefaultSandboxClass is what an empty spec.sandboxClass resolves to. It is -+ // the class every task ran in before the field existed. -+ DefaultSandboxClass = SandboxClassGVisor -+ - // PhaseTerminating marks a task whose deletion has been requested and whose - // actor is being torn down. The record disappears once cleanup completes. - PhaseTerminating = "Terminating" -@@ -252,6 +262,15 @@ func ValidateTask(t *Task) error { - if spec == nil { - return nil - } -+ switch spec.GetSandboxClass() { -+ case "", SandboxClassGVisor, SandboxClassMicroVM: -+ default: -+ // Refused, not downgraded: a task that quietly runs in a different -+ // sandbox class than it asked for is worse than a task that does not -+ // start. Agent Substrate offers no other class today. -+ return fmt.Errorf("spec.sandboxClass: %q is not a sandbox class Agent Substrate offers; supported values are %q and %q", -+ spec.GetSandboxClass(), SandboxClassGVisor, SandboxClassMicroVM) -+ } - refs := spec.WorkspaceRefs() - paths := spec.WorkspacePaths() - names := make(map[string]bool, len(refs)) -diff --git a/pkg/apis/v1alpha1/types_test.go b/pkg/apis/v1alpha1/types_test.go -index 83deaa4..2431b97 100644 ---- a/pkg/apis/v1alpha1/types_test.go -+++ b/pkg/apis/v1alpha1/types_test.go -@@ -137,6 +137,43 @@ func TestStrictDecoding_RejectsUnknownFields(t *testing.T) { - } - } - -+func TestTask_SandboxClass_YAML(t *testing.T) { -+ // The point of this test is that the field is real to protojson, not just -+ // present in the .proto: strict decoding rejects anything it does not know. -+ var task v1alpha1.Task -+ if err := yaml.Unmarshal([]byte("kind: Task\nspec:\n sandboxClass: microvm\n"), &task); err != nil { -+ t.Fatalf("decoding a manifest naming sandboxClass: %v", err) -+ } -+ if got := task.GetSpec().GetSandboxClass(); got != v1alpha1.SandboxClassMicroVM { -+ t.Errorf("spec.sandboxClass = %q, want %q", got, v1alpha1.SandboxClassMicroVM) -+ } -+ -+ out, err := yaml.Marshal(&task) -+ if err != nil { -+ t.Fatalf("marshalling: %v", err) -+ } -+ if !strings.Contains(string(out), "sandboxClass: microvm") { -+ t.Errorf("round trip lost the field, got:\n%s", out) -+ } -+ -+ // Empty stays empty rather than being rendered as the default, so existing -+ // manifests keep their existing YAML shape. -+ var plain v1alpha1.Task -+ if err := yaml.Unmarshal([]byte("kind: Task\nspec:\n image: img\n"), &plain); err != nil { -+ t.Fatalf("decoding a manifest without sandboxClass: %v", err) -+ } -+ if got := plain.GetSpec().GetSandboxClass(); got != "" { -+ t.Errorf("spec.sandboxClass = %q, want empty", got) -+ } -+ out, err = yaml.Marshal(&plain) -+ if err != nil { -+ t.Fatalf("marshalling: %v", err) -+ } -+ if strings.Contains(string(out), "sandboxClass") { -+ t.Errorf("unset field should not be rendered, got:\n%s", out) -+ } -+} -+ - func TestGateway_RoundTrip(t *testing.T) { - want := &v1alpha1.Gateway{ - ApiVersion: v1alpha1.APIVersion, -@@ -410,6 +447,19 @@ func TestValidateTask(t *testing.T) { - wantErr string - }{ - {name: "nil spec"}, -+ {name: "sandbox class unset", spec: &v1alpha1.TaskSpec{}}, -+ {name: "sandbox class gvisor", spec: &v1alpha1.TaskSpec{SandboxClass: v1alpha1.SandboxClassGVisor}}, -+ {name: "sandbox class microvm", spec: &v1alpha1.TaskSpec{SandboxClass: v1alpha1.SandboxClassMicroVM}}, -+ { -+ name: "sandbox class workerd", -+ spec: &v1alpha1.TaskSpec{SandboxClass: "workerd"}, -+ wantErr: `spec.sandboxClass: "workerd" is not a sandbox class Agent Substrate offers`, -+ }, -+ { -+ name: "sandbox class wrong case", -+ spec: &v1alpha1.TaskSpec{SandboxClass: "gVisor"}, -+ wantErr: `spec.sandboxClass: "gVisor" is not a sandbox class Agent Substrate offers`, -+ }, - {name: "no workspaces", spec: &v1alpha1.TaskSpec{}}, - {name: "list of one", spec: &v1alpha1.TaskSpec{Workspaces: []*v1alpha1.WorkspaceRef{{Name: "a"}}}}, - { diff --git a/tests/ax-fleet/phases/30-nop1.py b/tests/ax-fleet/phases/30-nop1.py index 688a1fff4..ad2b99264 100644 --- a/tests/ax-fleet/phases/30-nop1.py +++ b/tests/ax-fleet/phases/30-nop1.py @@ -270,6 +270,7 @@ def run_one(name: str) -> Any: entry = {"task": n, "attempt": attempt, "report_seconds": secs, "before_delete": before} if reps: entry["floor"] = check_report(n, reps) + assert entry["floor"]["gvisor"], reps entry["actors_before_delete"] = actors() entry["delete"] = delete_task(n) entry["actors_after_delete"] = actors() @@ -307,6 +308,23 @@ def resource_exhausted(entry: Any) -> bool: record("nop1_workers_baseline", ate_json("get workers")) leak_snapshot("baseline") +with step("zeropatch: the running ax is stock, it refuses spec.sandboxClass"): + # Negative control ported from probe/ax-fleet-nosc: with sandbox-class.patch + # removed, strict protojson decode must reject the unknown field. + ctl_task: dict[str, Any] = { + "apiVersion": "ax.io/v1alpha1", + "kind": "Task", + "metadata": {"name": "nosc-control", "atespace": NS}, + "spec": {"image": IMAGE, "sandboxClass": "gvisor", "command": ["true"]}, + } + nosc_rc, nosc_out = coordinator.execute( + f"echo {base64.b64encode(json.dumps(ctl_task).encode()).decode()} | base64 -d | {AX} apply -f - 2>&1" + ) + record("nosc_sandboxclass_refused", {"rc": nosc_rc, "out": nosc_out[-2000:]}) + coordinator.execute(f"{AX} delete task nosc-control 2>&1 || true") + assert nosc_rc != 0, nosc_out + + with step("nop1 shape check: the P1 runs' command form, [ax-agent, halogen-smoke] (recorded, not asserted)"): shape: dict[str, Any] = {"name": "nop1-shape"} shape_task: dict[str, Any] = { From fa10c3c990841756db0ecd9426154c5707b373c1 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 11:39:47 +0200 Subject: [PATCH 26/37] probe(ax-fleet-zeropatch): record each floor report's raw /proc/version Co-Authored-By: Claude Opus 5.5 --- tests/ax-fleet/phases/30-nop1.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/ax-fleet/phases/30-nop1.py b/tests/ax-fleet/phases/30-nop1.py index ad2b99264..712bf78d0 100644 --- a/tests/ax-fleet/phases/30-nop1.py +++ b/tests/ax-fleet/phases/30-nop1.py @@ -240,6 +240,7 @@ def check_report(name: str, reps: Any) -> Any: "intact": intact, "src": reps[0]["src"], "gvisor": "gvisor" in rep.get("kernel", ""), + "kernel": rep.get("kernel", ""), } From df190fa91adc20cb36b47077ecbcf3894f4bd46c Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 11:40:51 +0200 Subject: [PATCH 27/37] ax-fleet: fix round 1 (stale base, teardown PATH, boot race, LAN exposure, panic tunables, RBAC, RustFS credential, real flap) - Stacked on the live coordinator revision 3a658991 (merge commit before this one) plus f7b34239 (DF-6, check-only), so a coordinator switch adds only the ax-fleet delta: same home-manager generation as live, no tcp 8731 rule, no harness version rollback. - ax-fleet-teardown: inheritPath = false, refuse if tailscale is on PATH, so k3s-killall.sh never runs `tailscale set --advertise-routes=` (the NAS's 10.42.0.0/24 route). Also drops the NAS guard table. - NAS: docker-registry waits up to 30 s for 10.42.0.1 and restarts on failure; the seed wants (not requires) it and restarts on failure; k3s waits for the LAN address on both nodes. The boot test now adds the address 40 s late and proves the registry recovers. - NAS: inet ax-fleet-guard prerouting at priority raw drops routed LAN traffic to the pod and Service ranges before kube-proxy DNAT. - Coordinator guard: flannel.1 payloads must come from the pod CIDR; pod traffic out the LAN leg is dropped for every private range. - kubelet's kernel.panic, panic_on_oops, overcommit_memory are put back after each kubelet start (myAxFleet.kubelet.keepHostKernelTunables, default true, Tom's ruling pending). - ax-controller: no ClusterRole on Secrets, no automounted API token. - Substrate patch 0003: RustFS, ate-api-server, atelet and the bucket-init Job read a per-cluster credential from Secret ax-fleet-rustfs, generated once on the NAS by bootstrap step 25-rustfs-secret. - k3s bind mounts unmount lazily, so the rollback switch no longer fails on a busy /var/lib/kubelet. - VM test: the flap waits for Ready=Unknown and the unreachable taint; new subtests for the routed-LAN paths, the Freebox-style foreign /24, the tunables restore, the RBAC and credential surface, and the teardown never calling tailscale. Co-Authored-By: Claude Opus 5.5 --- modules/ax-fleet/ax.nix | 49 +++-------- modules/ax-fleet/control.nix | 66 +++++++++++++- modules/ax-fleet/harness.nix | 39 ++++++++- modules/ax-fleet/interface.nix | 14 +++ modules/ax-fleet/k3s.nix | 44 ++++++++++ modules/ax-fleet/substrate.nix | 33 +++++-- pkgs/ax-fleet-teardown/default.nix | 19 +++++ pkgs/substrate/default.nix | 6 +- .../0003-kind-rustfs-credential-secret.patch | 85 +++++++++++++++++++ tests/ax-fleet-boot/default.nix | 30 +++++++ tests/ax-fleet/default.nix | 32 +++++++ tests/ax-fleet/nodes.nix | 11 +++ tests/ax-fleet/phases/10-cluster.py | 45 ++++++++-- tests/ax-fleet/phases/30-ax.py | 29 +++++-- tests/ax-fleet/phases/35-lan-guard.py | 83 ++++++++++++++++++ tests/ax-fleet/phases/90-rollback.py | 6 ++ 16 files changed, 526 insertions(+), 65 deletions(-) create mode 100644 pkgs/substrate/patches/0003-kind-rustfs-credential-secret.patch create mode 100644 tests/ax-fleet/phases/35-lan-guard.py diff --git a/modules/ax-fleet/ax.nix b/modules/ax-fleet/ax.nix index efaa19dfc..bad5a52fe 100644 --- a/modules/ax-fleet/ax.nix +++ b/modules/ax-fleet/ax.nix @@ -279,45 +279,13 @@ let labels = labels "ax-controller"; }; } - { - apiVersion = "rbac.authorization.k8s.io/v1"; - kind = "ClusterRole"; - metadata = { - name = "ax-controller"; - labels = labels "ax-controller"; - }; - rules = [ - { - apiGroups = [ "" ]; - resources = [ "secrets" ]; - verbs = [ - "get" - "list" - "watch" - ]; - } - ]; - } - { - apiVersion = "rbac.authorization.k8s.io/v1"; - kind = "ClusterRoleBinding"; - metadata = { - name = "ax-controller"; - labels = labels "ax-controller"; - }; - subjects = [ - { - kind = "ServiceAccount"; - name = "ax-controller"; - namespace = "ax-system"; - } - ]; - roleRef = { - apiGroup = "rbac.authorization.k8s.io"; - kind = "ClusterRole"; - name = "ax-controller"; - }; - } + # No ClusterRole (fix round 1, 2026-09-23). Upstream grants get/list/watch + # on every Secret cluster-wide; the controller's only read is one GET of + # gemini-api-secret in the Task's atespace (reconciler.go lookupGeminiKey, + # REPORTED from the ax source), which falls back to env and then to "". + # Day one is pi on Halogen with no key, so the grant bought nothing. If a + # Gemini key is ever ruled in: a namespaced Role in that atespace with + # resourceNames [ "gemini-api-secret" ] and verbs [ "get" ] only. { apiVersion = "apps/v1"; kind = "Deployment"; @@ -336,6 +304,9 @@ let spec = { nodeSelector = controlSelector; serviceAccountName = "ax-controller"; + # No API token in the pod: it holds no RBAC and needs none. The + # projected ateapi token below is a separate volume and stays. + automountServiceAccountToken = false; containers = [ { name = "controller"; diff --git a/modules/ax-fleet/control.nix b/modules/ax-fleet/control.nix index ae97d842f..4cc5fc259 100644 --- a/modules/ax-fleet/control.nix +++ b/modules/ax-fleet/control.nix @@ -58,6 +58,37 @@ let kubectl = "${cfg.k3sPackage}/bin/kubectl"; + waitLanAddr = pkgs.writeShellScript "ax-fleet-wait-lan-addr" '' + for _ in $(${pkgs.coreutils}/bin/seq 30); do + ${pkgs.iproute2}/bin/ip -4 addr show dev ${cfg.lan.interface} 2>/dev/null \ + | ${pkgs.gnugrep}/bin/grep -qF 'inet ${cfg.lan.address}/' && exit 0 + ${pkgs.coreutils}/bin/sleep 1 + done + echo "ax-fleet: ${cfg.lan.address} not on ${cfg.lan.interface} after 30 s; starting anyway" >&2 + exit 0 + ''; + + # ── the cluster-range guard (fix round 1, 2026-09-23) ── + # The NAS is the house default gateway and forwards with policy accept, and + # kube-proxy's KUBE-SERVICES DNAT sits in PREROUTING for every interface. So + # without this, any LAN device that routes the pod or Service range via + # 10.42.0.1 reaches ClusterIPs and pods (MEASURED by the security review: + # ax-server 200, Redis INFO, RustFS 403; and, through VXLAN, pods on the + # coordinator). At priority raw, before any DNAT: destinations in the + # cluster ranges are accepted only from the pod-side interfaces. VXLAN + # outer packets target ${cfg.lan.address}, so flannel is unaffected, and + # traffic the NAS itself originates never passes prerouting. + rangeGuard = '' + chain prerouting { + type filter hook prerouting priority raw; policy accept; + iifname { "cni0", "flannel.1", "lo" } return + iifname "veth*" return + # The match carries the drop (never a bare drop): IPv6 and every other + # destination fall through to policy accept. + ip daddr { ${cfg.podCidr}, ${cfg.serviceCidr} } counter drop comment "ax-fleet: cluster ranges only from pods and flannel" + } + ''; + # ── the bootstrap steps this track owns ── bootstrapApi = '' # 10-api: wait for the apiserver and the admin kubeconfig. @@ -153,6 +184,12 @@ in inherit what where; type = "none"; options = "bind"; + # Lazy: at the kill-switch switch, pods and shims outlive k3s + # (KillMode=process) and can keep /var/lib/kubelet busy, which made the + # rollback switch exit 4 (MEASURED, fix round 1 VM run 4). Detaching + # lazily leaves the data on /mnt/fast untouched; ax-fleet-teardown then + # stops what still holds it. + mountConfig.LazyUnmount = true; requires = [ "ax-fleet-dirs.service" ]; after = [ "ax-fleet-dirs.service" ]; wantedBy = [ "k3s.service" ]; @@ -203,6 +240,28 @@ in wants = [ "ax-fleet-dirs.service" ]; after = [ "ax-fleet-dirs.service" ]; unitConfig.RequiresMountsFor = [ cfg.registryRoot ]; + # The explicit ${registryHost} bind races NetworkManager at boot: the + # static address reaches ${cfg.lan.interface} seconds after + # network(-online).target (MEASURED on the NAS, 2026-09-23; the same race + # hosts/nas/headscale.nix and modules/adguardhome.nix already guard). + # Wait up to 30 s for the address, then start anyway and let Restart + # cover a genuinely late interface. + serviceConfig = { + ExecStartPre = waitLanAddr; + Restart = "on-failure"; + RestartSec = 5; + }; + }; + + assertions = [ + { + assertion = config.networking.nftables.enable; + message = "modules/ax-fleet/control.nix: the control role's cluster-range guard is an nftables table; the NAS runs nftables."; + } + ]; + networking.nftables.tables.ax-fleet-guard = { + family = "inet"; + content = rangeGuard; }; # ── firewall: only what the coordinator and the pods need ── @@ -221,7 +280,10 @@ in systemd.services.ax-fleet-registry-seed = { description = "ax-fleet: seed the NAS registry from the store, digests preserved"; wantedBy = [ "multi-user.target" ]; - requires = [ "docker-registry.service" ]; + # wants, not requires: a registry that fails its first start (late LAN + # address) must not fail the seed on dependency, which Restart= would + # never retry. The script itself waits for /v2/, and Restart retries. + wants = [ "docker-registry.service" ]; after = [ "docker-registry.service" ]; path = [ pkgs.skopeo @@ -233,6 +295,8 @@ in Type = "oneshot"; RemainAfterExit = true; StateDirectory = "ax-fleet"; + Restart = "on-failure"; + RestartSec = 15; }; script = seedScript; }; diff --git a/modules/ax-fleet/harness.nix b/modules/ax-fleet/harness.nix index 96f412b65..ce3db0bf4 100644 --- a/modules/ax-fleet/harness.nix +++ b/modules/ax-fleet/harness.nix @@ -17,7 +17,15 @@ # wifi. A conf.d drop-in plus `nmcli general reload conf` instead. # - the tailnet: flannel and kube-proxy bind the LAN leg only, nothing is # published, and the guard chain below keeps pods, wifi and the tailnet -# apart even though k3s turns ip_forward on (judge 1's second risk). +# apart even though k3s turns ip_forward on (judge 1's second risk). The +# chain covers the direct path; the path routed through the NAS into +# VXLAN is closed twice, by the NAS's prerouting range guard +# (control.nix) and by the flannel.1 source rule here. Plain VXLAN on the +# wifi leg is accepted by source address only (spoofable over wifi); see +# DESIGN.md Unknowns. +# - the kernel's panic behaviour: kubelet's kernel.panic / panic_on_oops / +# overcommit values are put back after it starts +# (myAxFleet.kubelet.keepHostKernelTunables, k3s.nix). # - herdr and every user unit: only system units are added. # - no containerd template, no runsc on PATH, no RuntimeClass (Substrate # runs its own runsc in the worker pods). @@ -44,7 +52,18 @@ let # reply arrives on the LAN leg from a source that is not the NAS. Only NEW # flows are policed, which is the property the guard exists for. guardRules = - [ "-m conntrack --ctstate ESTABLISHED,RELATED -j RETURN" ] + [ + "-m conntrack --ctstate ESTABLISHED,RELATED -j RETURN" + # Into pods over VXLAN only from the pod network (fix round 1). Real + # peers, the NAS host included (its flannel.1 address), are sourced + # from the pod CIDR. Defence in depth, not the fix: LAN traffic the NAS + # routes into VXLAN (MEASURED bypass, worker -> nas -> flannel.1 -> + # harness pod, rc=0) is likely masqueraded by flannel's own rule to the + # NAS's flannel.1 address (INFERRED), so the NAS's prerouting range + # guard (control.nix) is what closes that path. This rule drops VXLAN + # payloads whose inner source is outside the pod CIDR. + "-i flannel.1 ! -s ${cfg.podCidr} -j DROP" + ] ++ lib.concatMap ( g: lib.concatMap (p: [ @@ -64,8 +83,20 @@ let # apiserver and the registry on the NAS; the internet (atelet's GCS # fetch) is unaffected. Everything else in-cluster rides flannel.1. "-i cni0 -o ${lan} -d ${cfg.serverAddress} -p tcp -m multiport --dports 6443,${registryPort} -j RETURN" - "-i cni0 -o ${lan} -d ${cfg.lan.cidr} -j DROP" - ]; + ] + # Every private range, not only the house /24 (fix round 1): when + # NetworkManager falls back to the Freebox profile on ${lan} + # (hosts/coordinator/uplink-nas.nix), the leg is a DHCP subnet this + # module does not know, and 100.64/10 is the tailnet's range. + ++ map (r: "-i cni0 -o ${lan} -d ${r} -j DROP") privateRanges; + + privateRanges = lib.unique [ + cfg.lan.cidr + "10.0.0.0/8" + "172.16.0.0/12" + "192.168.0.0/16" + "100.64.0.0/10" + ]; guardStart = '' # ax-fleet guard chain (idempotent) diff --git a/modules/ax-fleet/interface.nix b/modules/ax-fleet/interface.nix index 0b765635f..0c56097ed 100644 --- a/modules/ax-fleet/interface.nix +++ b/modules/ax-fleet/interface.nix @@ -153,6 +153,20 @@ in type = types.nullOr types.str; default = null; }; + keepHostKernelTunables = mkOption { + type = types.bool; + default = true; + description = '' + kubelet sets kernel.panic=10, kernel.panic_on_oops=1 and + vm.overcommit_memory=1 when it starts (REPORTED by the VM receipt). + On the desk that turns any kernel oops into a reboot 10 s later, + dropping herdr and every seat. true: ax-fleet-kernel-tunables puts + the three keys back to the values recorded before k3s first ran, + after every kubelet start. false: kubelet's values stay. Tom's + ruling is pending (Waiting on you, 2026-09-23); true is the default + because it changes nothing for Tom at switch. + ''; + }; }; workerPool = { diff --git a/modules/ax-fleet/k3s.nix b/modules/ax-fleet/k3s.nix index 20afb7905..53ad4641a 100644 --- a/modules/ax-fleet/k3s.nix +++ b/modules/ax-fleet/k3s.nix @@ -115,9 +115,53 @@ in # up, whatever order a live switch ran tmpfiles and mounts in. serviceConfig.ExecStartPre = [ "${config.systemd.package}/bin/systemd-tmpfiles --create --prefix=/var/lib/rancher/k3s" + # --node-ip and --flannel-iface name the LAN address, which + # NetworkManager adds after network.target at boot (MEASURED on the + # NAS). Bounded wait, then start anyway (k3s's Restart covers it). + "${pkgs.writeShellScript "ax-fleet-k3s-wait-lan-addr" '' + for _ in $(${pkgs.coreutils}/bin/seq 30); do + ${pkgs.iproute2}/bin/ip -4 addr show dev ${cfg.lan.interface} 2>/dev/null \ + | ${pkgs.gnugrep}/bin/grep -qF 'inet ${cfg.lan.address}/' && exit 0 + ${pkgs.coreutils}/bin/sleep 1 + done + echo "ax-fleet: ${cfg.lan.address} not on ${cfg.lan.interface} after 30 s; starting anyway" >&2 + exit 0 + ''}" ]; }; + # ── kubelet's kernel tunables, put back (myAxFleet.kubelet.keepHostKernelTunables) ── + # kubelet sets these once per start (container manager setup, + # protectKernelDefaults off; INFERRED from upstream kubelet, the VM test + # measures the result). PartOf k3s: every k3s restart re-runs this + # after kubelet has applied its values, which it waits for. + systemd.services.ax-fleet-kernel-tunables = lib.mkIf cfg.kubelet.keepHostKernelTunables { + description = "ax-fleet: restore the host's kernel.panic, kernel.panic_on_oops, vm.overcommit_memory after kubelet"; + wantedBy = [ "k3s.service" ]; + after = [ "k3s.service" ]; + partOf = [ "k3s.service" ]; + path = [ + pkgs.procps + pkgs.coreutils + pkgs.gnugrep + ]; + serviceConfig = { + Type = "oneshot"; + RemainAfterExit = true; + }; + script = '' + snap=/var/lib/ax-fleet/sysctl-before.conf + [ -s "$snap" ] || { echo "no $snap; leaving kernel tunables as they are"; exit 0; } + want() { [ "$(sysctl -n kernel.panic)" = 10 ] && [ "$(sysctl -n kernel.panic_on_oops)" = 1 ] && [ "$(sysctl -n vm.overcommit_memory)" = 1 ]; } + for _ in $(seq 900); do want && break; sleep 1; done + want || echo "kubelet has not set its tunables after 900 s; restoring anyway" + for k in kernel.panic kernel.panic_on_oops vm.overcommit_memory; do + v=$(grep -E "^$k = " "$snap" | cut -d' ' -f3) + if [ -n "$v" ]; then sysctl -w "$k=$v"; fi + done + ''; + }; + # ── sysctl snapshot, taken BEFORE this generation's sysctls apply ── # An activation snippet, not a unit: at a live switch the activation # script runs before systemd-sysctl is restarted with the new values, diff --git a/modules/ax-fleet/substrate.nix b/modules/ax-fleet/substrate.nix index a91e779a1..48f5df4bf 100644 --- a/modules/ax-fleet/substrate.nix +++ b/modules/ax-fleet/substrate.nix @@ -209,6 +209,28 @@ in ${kubectl} apply -f ${registrySvc} ''; + "25-rustfs-secret" = '' + # 25-rustfs-secret: the RustFS credential RustFS, ate-api and atelet + # read through secretKeyRef (pkgs/substrate patch 0003), in place of + # the kind overlay's literal default published upstream. Generated + # once on this host into /var/lib/ax-fleet/rustfs.env (0600 root), + # applied before ate-setup starts those pods. Never printed. + f=/var/lib/ax-fleet/rustfs.env + if [ ! -s "$f" ]; then + ( + umask 077 + tmp=$(mktemp /var/lib/ax-fleet/.rustfs.env.XXXXXX) + rnd() { head -c "$1" /dev/urandom | od -An -tx1 | tr -d ' \n'; } + printf 'access-key=ax%s\nsecret-key=%s\n' "$(rnd 10)" "$(rnd 32)" > "$tmp" + mv "$tmp" "$f" + ) + echo "generated $f" + fi + ${kubectl} -n ${ns} create secret generic ax-fleet-rustfs --from-env-file="$f" \ + --dry-run=client -o yaml | ${kubectl} apply -f - >/dev/null + echo "secret ${ns}/ax-fleet-rustfs applied" + ''; + "30-substrate" = '' # 30-substrate: upstream's own installer, once per (installer, images) # pair. The stamp lives in the cluster, so a wiped cluster re-installs. @@ -243,18 +265,17 @@ in "40-gvisor-asset" = '' # 40-gvisor-asset: put the pinned runsc tarball where atelet's S3 # fallback looks for gs://${images.gvisor.bucket}/${images.gvisor.key} - # (same bucket and key; the scheme is ignored). The upstream kind - # credential is read from the RustFS Deployment and kept off argv. + # (same bucket and key; the scheme is ignored). The credential is + # read from the Secret 25-rustfs-secret applied and kept off argv. ${kubectl} -n ${ns} rollout status deploy/rustfs --timeout=10m ip=$(${kubectl} -n ${ns} get svc rustfs -o jsonpath='{.spec.clusterIP}') base="http://$ip:9000" creds=$(mktemp) trap 'rm -f "$creds"' EXIT chmod 600 "$creds" - ${kubectl} -n ${ns} get deploy rustfs -o json | ${jq} -r ' - .spec.template.spec.containers[0].env - | map({(.name): .value}) | add - | "user = \"\(.RUSTFS_ACCESS_KEY):\(.RUSTFS_SECRET_KEY)\""' > "$creds" + ${kubectl} -n ${ns} get secret ax-fleet-rustfs -o json | ${jq} -r ' + .data + | "user = \"\(.["access-key"] | @base64d):\(.["secret-key"] | @base64d)\""' > "$creds" s3() { ${curl} -sS --aws-sigv4 "aws:amz:us-east-1:s3" -K "$creds" "$@"; } code=$(s3 -o /dev/null -w '%{http_code}' -I "$base/${images.gvisor.bucket}") if [ "$code" != 200 ]; then diff --git a/pkgs/ax-fleet-teardown/default.nix b/pkgs/ax-fleet-teardown/default.nix index 3365a2b1b..0a015442f 100644 --- a/pkgs/ax-fleet-teardown/default.nix +++ b/pkgs/ax-fleet-teardown/default.nix @@ -2,6 +2,7 @@ writeShellApplication, k3s, iptables, + nftables, iproute2, procps, coreutils, @@ -19,13 +20,23 @@ # (/var/lib/ax-fleet/sysctl-before.conf, written by the activation snippet in # modules/ax-fleet/k3s.nix). # +# PATH is pinned (inheritPath = false). The upstream killall's +# remove_interfaces runs `tailscale set --advertise-routes=` whenever a +# `tailscale` binary is on PATH; on the NAS that would withdraw the +# 10.42.0.0/24 subnet route (hosts/nas/headscale.nix). With the caller's PATH +# cut off, `command -v tailscale` fails and the call is skipped; the killall +# wrapper still prefixes its own dependencies. The explicit check below makes +# the teardown refuse to run if tailscale ever becomes reachable anyway. +# # Left on disk on purpose: /mnt/fast/k3s, the data-pool directories and # /var/lib/ax-fleet. Deleting them is Tom's call. writeShellApplication { name = "ax-fleet-teardown"; + inheritPath = false; runtimeInputs = [ k3s iptables + nftables iproute2 procps coreutils @@ -38,6 +49,11 @@ writeShellApplication { exit 1 fi + if command -v tailscale >/dev/null 2>&1; then + echo "ax-fleet-teardown: tailscale is on PATH; k3s-killall.sh would clear the advertised routes. Refusing." >&2 + exit 1 + fi + if systemctl is-enabled --quiet k3s.service 2>/dev/null; then echo "ax-fleet-teardown: note: k3s.service is still part of this generation;" >&2 echo " k3s-killall.sh stops it now, and it starts again at the next boot or switch." >&2 @@ -51,6 +67,9 @@ writeShellApplication { iptables -w -t mangle -F ax-fleet-guard 2>/dev/null || true iptables -w -t mangle -X ax-fleet-guard 2>/dev/null || true + echo "== NAS cluster-range guard table" + nft delete table inet ax-fleet-guard 2>/dev/null || true + snap=/var/lib/ax-fleet/sysctl-before.conf if [ -s "$snap" ]; then echo "== sysctl restore from $snap" diff --git a/pkgs/substrate/default.nix b/pkgs/substrate/default.nix index 153f62a47..63bd158af 100644 --- a/pkgs/substrate/default.nix +++ b/pkgs/substrate/default.nix @@ -21,11 +21,14 @@ # kagent-dev fork's ghcr images are ruled out), so every component image is # built here and seeded into the NAS registry by pkgs/substrate/images.nix. # -# The two patches touch only manifests/, never Go code: +# The three patches touch only manifests/, never Go code: # 0001 pauseImage -> localhost:5000/pause (atelet pulls it itself; the fleet # must not depend on registry.k8s.io at sandbox start). # 0002 third-party images -> their linux/amd64 child digests, so the NAS # seeds ~0.7 GB instead of every platform (see the patch header). +# 0003 the kind overlay's literal RustFS credential (public in the upstream +# repo) -> secretKeyRef to Secret ate-system/ax-fleet-rustfs, which the +# bootstrap step 25-rustfs-secret generates once on the NAS. # ate-setup reads the manifests from its working directory's repository root # (it walks up to go.mod). What it reads for `deploy ate-system` is go.mod, # manifests/ and hack/ (kustomize overlays, CSI manifests; MEASURED grep of @@ -52,6 +55,7 @@ let patches = [ ./patches/0001-sandboxconfig-pause-localhost.patch ./patches/0002-images-linux-amd64-digests.patch + ./patches/0003-kind-rustfs-credential-secret.patch ]; }; diff --git a/pkgs/substrate/patches/0003-kind-rustfs-credential-secret.patch b/pkgs/substrate/patches/0003-kind-rustfs-credential-secret.patch new file mode 100644 index 000000000..8ebef8952 --- /dev/null +++ b/pkgs/substrate/patches/0003-kind-rustfs-credential-secret.patch @@ -0,0 +1,85 @@ +--- a/manifests/ate-install/kind/atelet/kustomization.yaml ++++ b/manifests/ate-install/kind/atelet/kustomization.yaml +@@ -52,7 +52,15 @@ + - name: AWS_S3_USE_PATH_STYLE + value: "true" + # TODO: use a secret / identity management ++ # mecattaf fleet: the kind default credential is replaced by a ++ # per-cluster one the ax-fleet bootstrap generates on the NAS. + - name: AWS_ACCESS_KEY_ID +- value: rustfsadmin ++ valueFrom: ++ secretKeyRef: ++ name: ax-fleet-rustfs ++ key: access-key + - name: AWS_SECRET_ACCESS_KEY +- value: rustfsadmin ++ valueFrom: ++ secretKeyRef: ++ name: ax-fleet-rustfs ++ key: secret-key +--- a/manifests/ate-install/kind/kustomization.yaml ++++ b/manifests/ate-install/kind/kustomization.yaml +@@ -66,8 +66,16 @@ + - name: AWS_S3_USE_PATH_STYLE + value: "true" + # TODO: use a secret / identity management ++ # mecattaf fleet: the kind default credential is replaced by a ++ # per-cluster one the ax-fleet bootstrap generates on the NAS. + - name: AWS_ACCESS_KEY_ID +- value: rustfsadmin ++ valueFrom: ++ secretKeyRef: ++ name: ax-fleet-rustfs ++ key: access-key + - name: AWS_SECRET_ACCESS_KEY +- value: rustfsadmin ++ valueFrom: ++ secretKeyRef: ++ name: ax-fleet-rustfs ++ key: secret-key + +--- a/manifests/ate-install/kind/rustfs.yaml ++++ b/manifests/ate-install/kind/rustfs.yaml +@@ -78,10 +78,18 @@ + value: "true" + - name: RUSTFS_VOLUMES + value: "/data" ++ # mecattaf fleet: the kind default credential is replaced by a ++ # per-cluster one the ax-fleet bootstrap generates on the NAS. + - name: RUSTFS_ACCESS_KEY +- value: "rustfsadmin" ++ valueFrom: ++ secretKeyRef: ++ name: ax-fleet-rustfs ++ key: access-key + - name: RUSTFS_SECRET_KEY +- value: "rustfsadmin" ++ valueFrom: ++ secretKeyRef: ++ name: ax-fleet-rustfs ++ key: secret-key + volumeMounts: + - name: data + mountPath: /data +@@ -104,10 +112,18 @@ + - name: create-bucket + image: amazon/aws-cli:2.17.0@sha256:643507c10ada7964ca6157b3d799f030b90577643da9955d319a77399ed80d73 + env: ++ # mecattaf fleet: the kind default credential is replaced by a ++ # per-cluster one the ax-fleet bootstrap generates on the NAS. + - name: AWS_ACCESS_KEY_ID +- value: rustfsadmin ++ valueFrom: ++ secretKeyRef: ++ name: ax-fleet-rustfs ++ key: access-key + - name: AWS_SECRET_ACCESS_KEY +- value: rustfsadmin ++ valueFrom: ++ secretKeyRef: ++ name: ax-fleet-rustfs ++ key: secret-key + - name: AWS_REGION + value: us-east-1 + - name: AWS_ENDPOINT_URL diff --git a/tests/ax-fleet-boot/default.nix b/tests/ax-fleet-boot/default.nix index 77dfb9902..5dfbd809e 100644 --- a/tests/ax-fleet-boot/default.nix +++ b/tests/ax-fleet-boot/default.nix @@ -19,10 +19,31 @@ pkgs.testers.runNixOSTest { imports = [ nodes.nas ]; myAxFleet.enable = true; myAxFleet.kubelet.systemReserved = "cpu=1,memory=1Gi"; + # The real NAS gets 10.42.0.1 from NetworkManager seconds AFTER + # network(-online).target (MEASURED 2026-09-23: target at 11.78 s, address + # at 20.15 s). Reproduce that: no static address, then a unit nothing + # waits for adds it late. The delay is a test parameter longer than the + # units' 30 s address wait, so both the wait and Restart= are exercised. + networking.interfaces.eth1.ipv4.addresses = lib.mkOverride 10 [ ]; + systemd.services.late-lan-addr = { + wantedBy = [ "multi-user.target" ]; + after = [ "network.target" ]; + path = [ + pkgs.iproute2 + pkgs.coreutils + ]; + script = '' + sleep 40 + ip link set eth1 up + ip addr add 10.42.0.1/24 dev eth1 + date +%s > /run/late-lan-addr.done + ''; + }; }; globalTimeout = 3600; testScript = '' nas.start() + nas.wait_for_file("/run/late-lan-addr.done", timeout=600) nas.wait_for_unit("k3s.service", timeout=900) nas.wait_for_unit("ax-fleet-bootstrap.service", timeout=1800) for p in ("/var/lib/rancher/k3s", "/var/lib/kubelet", "/var/log/pods"): @@ -30,6 +51,15 @@ pkgs.testers.runNixOSTest { assert src.startswith("/dev/vdb"), f"{p} is on {src}, not /mnt/fast" nas.succeed("ls /mnt/fast/k3s/rancher/k3s/agent/images/ | grep -q airgap") nas.succeed("systemctl show ax-fleet-registry-seed.service -p Result --value | grep -x success") + # The registry came up on the late address and stays up. + nas.succeed("systemctl is-active docker-registry.service") + nas.succeed("curl -sf --max-time 10 http://10.42.0.1:5000/v2/") + addr_at = int(nas.succeed("cat /run/late-lan-addr.done").strip()) + reg_at = int(nas.succeed( + "date -d \"$(systemctl show docker-registry.service -p ActiveEnterTimestamp --value)\" +%s" + ).strip()) + assert reg_at >= addr_at, f"registry active at {reg_at}, before the address at {addr_at}" + print("AXFLEET-BOOT registry NRestarts=" + nas.succeed("systemctl show docker-registry.service -p NRestarts --value").strip()) nas.succeed("systemctl show ax-fleet-bootstrap.service -p Result --value | grep -x success") nas.wait_until_succeeds( "k3s kubectl get node nas -o jsonpath='{.status.conditions[?(@.type==\"Ready\")].status}' | grep -x True", diff --git a/tests/ax-fleet/default.nix b/tests/ax-fleet/default.nix index bff1722cf..c0b97bd20 100644 --- a/tests/ax-fleet/default.nix +++ b/tests/ax-fleet/default.nix @@ -91,6 +91,38 @@ let ) + KERNEL_KEYS = ("kernel.panic", "kernel.panic_on_oops", "vm.overcommit_memory") + + + def kernel_keys_back(machine, baseline): + """myAxFleet.kubelet.keepHostKernelTunables: kubelet's values are put back.""" + for k in KERNEL_KEYS: + machine.wait_until_succeeds(f"test \"$(sysctl -n {k})\" = '{baseline[k]}'", timeout=300) + + + def flap_until_unreachable(tag): + """Take the coordinator's LAN leg down until the control plane has + reacted (Ready=Unknown and the unreachable taint), then bring it back. + A test parameter, not an estimate: it waits for the transition.""" + t0 = time.monotonic() + coordinator.succeed("ip link set eth1 down") + try: + nas.wait_until_succeeds( + "k3s kubectl get node coordinator -o jsonpath='{.status.conditions[?(@.type==\"Ready\")].status}' | grep -qx Unknown", + timeout=900, + ) + nas.wait_until_succeeds( + "k3s kubectl get node coordinator -o jsonpath='{.spec.taints[*].key}' | grep -qw node.kubernetes.io/unreachable", + timeout=600, + ) + record(f"flap_{tag}_taints_while_down", nas.succeed("k3s kubectl get node coordinator -o jsonpath='{.spec.taints}'").strip()) + finally: + coordinator.succeed("ip link set eth1 up") + outage = round(time.monotonic() - t0, 1) + record(f"flap_{tag}_outage_seconds", outage) + return outage + + def user_unit_pid(unit): return coordinator.succeed( "runuser -u alice -- env XDG_RUNTIME_DIR=/run/user/$(id -u alice) " diff --git a/tests/ax-fleet/nodes.nix b/tests/ax-fleet/nodes.nix index a9d890763..c1a725a95 100644 --- a/tests/ax-fleet/nodes.nix +++ b/tests/ax-fleet/nodes.nix @@ -183,6 +183,17 @@ in }; }; + # A recording `tailscale` on the system PATH, as the real NAS has one: + # k3s-killall.sh's remove_interfaces runs `tailscale set + # --advertise-routes=` whenever the binary is reachable, which would + # withdraw the house subnet route. 90-rollback asserts the teardown + # never calls it. + environment.systemPackages = [ + (pkgs.writeShellScriptBin "tailscale" '' + echo "$*" >> /var/log/tailscale-stub.log + '') + ]; + # The bystander: the NAS's shared PostgreSQL (Paperless, Immich) must # not restart and must not change. services.postgresql = { diff --git a/tests/ax-fleet/phases/10-cluster.py b/tests/ax-fleet/phases/10-cluster.py index 6262b4fa1..7ff5d3d23 100644 --- a/tests/ax-fleet/phases/10-cluster.py +++ b/tests/ax-fleet/phases/10-cluster.py @@ -139,7 +139,9 @@ def apply_probe(name, role, host_port, pvc=None): nas.succeed("dig +short @10.42.0.1 only-nas.test | grep -x 10.42.0.77") ruleset = nas.succeed("nft list ruleset") record("nas_kube_services_in_nft", "KUBE-SERVICES" in ruleset) + nas.succeed("nft list chain inet ax-fleet-guard prerouting | grep -q 'ax-fleet: cluster ranges'") record("sysctl_nas_after_switch", sysctls(nas)) + kernel_keys_back(nas, base["sysctl_nas"]) with step("switch coordinator"): @@ -191,6 +193,9 @@ def apply_probe(name, role, host_port, pvc=None): ref = [l.split("Loaded image:")[1].strip() for l in loaded.splitlines() if "Loaded image" in l][0] coordinator.succeed(f"podman run --rm --network bridge {ref} curl -sf --max-time 10 http://10.88.0.1/ | grep -x caddy-ok") coordinator.succeed("lsmod | grep -q br_netfilter") + # kubelet's panic tunables are put back: an oops does not reboot the desk. + kernel_keys_back(coordinator, base["sysctl_coordinator"]) + record("sysctl_coordinator_after_restore", sysctls(coordinator)) with step("coordinator: the guards hold"): @@ -226,6 +231,32 @@ def apply_probe(name, role, host_port, pvc=None): record("guard_chain", coordinator.succeed("iptables -t mangle -S ax-fleet-guard").strip().splitlines()) +with step("coordinator: pods reach no private range on the LAN leg (the Freebox fallback case)"): + # A private subnet this module does not know, on the same leg, as when + # NetworkManager falls back to the Freebox profile. Discriminating: the + # coordinator host reaches it; a pod must not. + worker.succeed("ip addr add 192.168.77.5/24 dev eth1") + coordinator.succeed("ip route replace 192.168.77.0/24 dev eth1") + try: + coordinator.succeed("curl -sf --max-time 10 http://192.168.77.5:8731/health") + kubectl("exec probe-coord -- sh -c '! curl -s --max-time 5 -o /dev/null http://192.168.77.5:8731/health'") + finally: + coordinator.succeed("ip route del 192.168.77.0/24 dev eth1") + worker.succeed("ip addr del 192.168.77.5/24 dev eth1") + + +with step("coordinator: LAN traffic routed through the NAS never reaches a harness pod"): + # The house default gateway is the NAS. Before fix round 1 this path + # (worker -> nas -> flannel.1 -> harness pod) answered (MEASURED). + pod_ip = jsonpath("pod probe-coord", "{.status.podIP}") + nas.succeed(f"curl -sf --max-time 10 http://{pod_ip}:8000/ | grep -x pod-ok") # the path itself works + worker.succeed("ip route replace 10.200.0.0/16 via 10.42.0.1") + try: + worker.fail(f"curl -s --max-time 5 -o /dev/null http://{pod_ip}:8000/") + finally: + worker.succeed("ip route del 10.200.0.0/16 via 10.42.0.1") + + def diag(cmds): out = {} for name, (machine, cmd) in cmds.items(): @@ -265,12 +296,8 @@ def diag(cmds): kubectl("exec probe-nas -- nslookup -type=a only-nas.test | grep -q 10.42.0.77") -with step("resilience: coordinator link flap"): - uid = jsonpath("pod probe-coord", "{.metadata.uid}") - coordinator.succeed("ip link set eth1 down") - time.sleep(20) - coordinator.succeed("ip link set eth1 up") - node_ready("coordinator") - kubectl("wait --for=condition=Ready pod/probe-coord --timeout=300s") - assert jsonpath("pod probe-coord", "{.metadata.uid}") == uid, "the probe pod was replaced by the flap" - nas.wait_until_succeeds("curl -sf --max-time 5 http://10.42.0.2:18085/ | grep -x pod-ok", timeout=120) +# The coordinator link flap runs at the END of 30-ax (fix round 1), together +# with the Substrate/ax flap: a flap long enough to take the node NotReady +# can leave connections to the NAS stale (INFERRED), and the Task phases before it +# must run on a cluster that has not seen an outage. probe-coord stays up +# until then; that subtest asserts its uid survives. diff --git a/tests/ax-fleet/phases/30-ax.py b/tests/ax-fleet/phases/30-ax.py index 04ec7681f..9c7054cb0 100644 --- a/tests/ax-fleet/phases/30-ax.py +++ b/tests/ax-fleet/phases/30-ax.py @@ -100,19 +100,38 @@ def task_phase(name): assert before == after and int(after) >= 1, f"tasks before={before} after={after}" assert task_phase(t1_name) == "Completed" -with subtest("resilience: the LAN leg flaps, worker pods keep their names, a new T1 completes"): +with subtest("resilience: the LAN leg is down until the coordinator is NotReady; pods keep their names and uid, a new T1 completes"): pods = lambda: kubectl( "-n ate-system get pods --field-selector spec.nodeName=coordinator -o name | grep ateom | sort" ) before = pods() - coordinator.succeed("ip link set eth1 down") - coordinator.sleep(20) - coordinator.succeed("ip link set eth1 up") + probe_uid = kubectl("get pod probe-coord -o jsonpath='{.metadata.uid}'").strip() + flap_until_unreachable("ax") nas.wait_until_succeeds( "k3s kubectl get node coordinator -o jsonpath='{.status.conditions[?(@.type==\"Ready\")].status}' | grep -qx True", timeout=300, ) + nas.wait_until_succeeds( + "! k3s kubectl get node coordinator -o jsonpath='{.spec.taints[*].key}' | grep -qw node.kubernetes.io/unreachable", + timeout=300, + ) assert pods() == before, f"worker pods changed: {before!r}" - ax_smoke("halogen") + kubectl("wait --for=condition=Ready pod/probe-coord --timeout=300s") + assert kubectl("get pod probe-coord -o jsonpath='{.metadata.uid}'").strip() == probe_uid, "the probe pod was replaced by the flap" + nas.wait_until_succeeds("curl -sf --max-time 5 http://10.42.0.2:18085/ | grep -x pod-ok", timeout=120) + # Tasks after a real outage are recorded as they are. Fix round 1 + # MEASURED that each stale gRPC connection left by the outage costs one + # Task: ActorResumeFailed, Unavailable, "connection reset by peer", once + # NAS -> atelet :8085 and once atelet -> NAS :443 ("mint actor + # certificate"). ax does not retry Unavailable. The bound is a test + # parameter: the fleet must recover within it without a restart. + attempts = [] + for _ in range(6): + r = ax_smoke("halogen", expect_pass=False) + attempts.append({k: r.get(k) for k in ("pass", "phase", "ready")}) + if r.get("pass") is True: + break + record("post_outage_tasks", attempts) + assert attempts[-1]["pass"] is True, f"no Task completed after the outage: {attempts}" coordinator.succeed(f"{AX} delete task {t1_name}") diff --git a/tests/ax-fleet/phases/35-lan-guard.py b/tests/ax-fleet/phases/35-lan-guard.py new file mode 100644 index 000000000..c03dc0ecf --- /dev/null +++ b/tests/ax-fleet/phases/35-lan-guard.py @@ -0,0 +1,83 @@ +# Phase: the house LAN cannot reach the cluster ranges through the NAS, and +# the credential and RBAC surface is what fix round 1 left (2026-09-23). Runs +# after 30-ax, so ax-server, ax-redis and RustFS all exist. The worker is a +# plain LAN host; the NAS is its default gateway on the real LAN. + + +def tcp_open(ip, port): + return f"timeout 5 bash -c 'exec 3<>/dev/tcp/{ip}/{port}'" + + +with step("lan: routed LAN traffic never reaches a ClusterIP or pod IP"): + def svc_ip(ns, name): + return jsonpath(f"-n {ns} svc {name}", "{.spec.clusterIP}") + + ax_ip = svc_ip("ax-system", "ax-server") + redis_ip = svc_ip("ax-system", "ax-redis") + rustfs_ip = svc_ip("ate-system", "rustfs") + ax_pod = kubectl( + "-n ax-system get pods -l app.kubernetes.io/name=ax-server -o jsonpath='{.items[0].status.podIP}'" + ).strip() + record("lan_probe_targets", {"ax": ax_ip, "redis": redis_ip, "rustfs": rustfs_ip, "ax_pod": ax_pod}) + + # Positive controls from the NAS host (OUTPUT path, not prerouting): the + # endpoints are up, so a failure from the worker is the guard. + nas.succeed(f"curl -sf --max-time 10 http://{ax_ip}:8080/healthz") + nas.succeed(tcp_open(redis_ip, 6379)) + nas.succeed(f"curl -s --max-time 10 -o /dev/null http://{rustfs_ip}:9000/") + + worker.succeed("ip route replace 10.200.0.0/16 via 10.42.0.1") + worker.succeed("ip route replace 10.201.0.0/16 via 10.42.0.1") + try: + # curl without -f: rc 0 on ANY HTTP answer, so fail() means no answer. + worker.fail(f"curl -s --max-time 5 -o /dev/null http://{ax_ip}:8080/healthz") + worker.fail(f"curl -s --max-time 5 -o /dev/null http://{ax_pod}:8080/healthz") + worker.fail(f"curl -s --max-time 5 -o /dev/null http://{rustfs_ip}:9000/") + worker.fail(tcp_open(redis_ip, 6379)) + worker.fail("dig +time=2 +tries=1 @10.201.0.10 ax-server.ax-system.svc.cluster.local") + finally: + worker.succeed("ip route del 10.200.0.0/16 via 10.42.0.1") + worker.succeed("ip route del 10.201.0.0/16 via 10.42.0.1") + record("nas_range_guard", nas.succeed("nft list chain inet ax-fleet-guard prerouting").strip().splitlines()) + + +with step("security: ax-controller holds no Secret grant and no API token"): + nas.fail("k3s kubectl get clusterrole ax-controller") + nas.fail("k3s kubectl get clusterrolebinding ax-controller") + auto = jsonpath("-n ax-system deploy ax-controller", "{.spec.template.spec.automountServiceAccountToken}") + assert auto == "false", auto + vols = kubectl( + "-n ax-system get pods -l app.kubernetes.io/name=ax-controller -o jsonpath='{.items[0].spec.volumes[*].name}'" + ).split() + record("ax_controller_volumes", vols) + assert not any(v.startswith("kube-api-access") for v in vols), vols + can = nas.succeed( + "k3s kubectl auth can-i list secrets --all-namespaces --as=system:serviceaccount:ax-system:ax-controller || true" + ).strip() + assert can == "no", can + + +with step("security: RustFS and its clients read a generated credential, not the kind default"): + nas.succeed("stat -c '%a %U' /var/lib/ax-fleet/rustfs.env | grep -x '600 root'") + nas.succeed("k3s kubectl -n ate-system get secret ax-fleet-rustfs -o name") + wanted = {"RUSTFS_ACCESS_KEY", "RUSTFS_SECRET_KEY", "AWS_ACCESS_KEY_ID", "AWS_SECRET_ACCESS_KEY"} + seen = [] + # The bucket-init Job uses the credential too; fix round 1 first missed it + # and it crash-looped (MEASURED), so it must complete here. + kubectl("-n ate-system wait --for=condition=complete job/rustfs-bucket-init --timeout=300s") + for obj in ("deploy rustfs", "deploy ate-api-server", "ds atelet", "job rustfs-bucket-init"): + kind, name = obj.split() + names = kubectl(f"-n ate-system get {kind} -o name").split() + target = [n for n in names if n.split("/")[-1].startswith(name)] + for t in target: + doc = json.loads(kubectl(f"-n ate-system get {t} -o json")) + for c in doc["spec"]["template"]["spec"]["containers"]: + for e in c.get("env", []): + if e["name"] in wanted: + assert "value" not in e, f"{t} {e['name']} carries a literal value" + assert e["valueFrom"]["secretKeyRef"]["name"] == "ax-fleet-rustfs", (t, e["name"]) + seen.append(f"{t}:{e['name']}") + record("rustfs_credential_refs", sorted(seen)) + assert any(s.endswith("RUSTFS_SECRET_KEY") for s in seen), seen + assert any(s.endswith("AWS_SECRET_ACCESS_KEY") for s in seen), seen + assert any(s.startswith("job") and s.endswith("AWS_SECRET_ACCESS_KEY") for s in seen), seen diff --git a/tests/ax-fleet/phases/90-rollback.py b/tests/ax-fleet/phases/90-rollback.py index 987ad662e..79a1392e7 100644 --- a/tests/ax-fleet/phases/90-rollback.py +++ b/tests/ax-fleet/phases/90-rollback.py @@ -26,10 +26,16 @@ with step("rollback nas"): + # Diagnostic only: which processes hold the kubelet bind before the switch. + _, holders = nas.execute("ls -l /proc/[0-9]*/cwd /proc/[0-9]*/root 2>/dev/null | grep -c /var/lib/kubelet") + record("nas_kubelet_holders_before_rollback", holders.strip()) nas.succeed(f"{BASE} >&2") nas.fail("systemctl is-active k3s.service") nas.fail("systemctl is-active docker-registry.service") + nas.succeed("command -v tailscale") # the stub is reachable from a root shell nas.succeed(f"{TEARDOWN} >&2") + # The teardown's pinned PATH keeps k3s-killall.sh away from tailscale. + nas.fail("test -e /var/log/tailscale-stub.log") nas.fail("ip link show cni0") nas.fail("ip link show flannel.1") nas.fail("pgrep -f containerd-shim") From 6ada9d2eafa64a018613efcca0dda1de9e490f2d Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 12:27:58 +0200 Subject: [PATCH 28/37] probe(ax-fleet-zeropatch): capture controller, ateapi, atelet, redis stream state when a Task delete does not finish; fix ateapi deploy name in diagnose Run 2 failed with nop1-ctl-2 stuck in Terminating and none of these logs. Co-Authored-By: Claude Opus 5.5 --- tests/ax-fleet/phases/30-nop1.py | 45 +++++++++++++++++++++++++++++--- 1 file changed, 41 insertions(+), 4 deletions(-) diff --git a/tests/ax-fleet/phases/30-nop1.py b/tests/ax-fleet/phases/30-nop1.py index 712bf78d0..ea08d67fe 100644 --- a/tests/ax-fleet/phases/30-nop1.py +++ b/tests/ax-fleet/phases/30-nop1.py @@ -184,7 +184,7 @@ def diagnose(name: str) -> None: _, wp = nas.execute("k3s kubectl -n ate-system logs -l ax.mecattaf.dev/pool=ateom-gvisor --all-containers --tail=120 2>&1") _, atelet = nas.execute("k3s kubectl -n ate-system logs -l app=atelet --all-containers --tail=80 2>&1") _, ctl = nas.execute("k3s kubectl -n ax-system logs deploy/ax-controller --tail=60 2>&1") - _, pods = nas.execute("k3s kubectl -n ate-system get pods -o wide 2>&1; k3s kubectl -n ate-system logs deploy/ateapi --all-containers --tail=40 2>&1") + _, pods = nas.execute("k3s kubectl -n ate-system get pods -o wide 2>&1; k3s kubectl -n ate-system logs deploy/ate-api-server --all-containers --tail=40 2>&1") _, stub = worker.execute("tail -n 20 /var/lib/halogen-stub/requests.jsonl 2>&1") record(f"nop1_diag_{name}", {"workers": wp[-8000:], "atelet": atelet[-6000:], "controller": ctl[-5000:], "stub": stub[-3000:], "ate_pods_api": pods[-6000:]}) @@ -244,17 +244,54 @@ def check_report(name: str, reps: Any) -> Any: } +def delete_hang_diag(name: str, calls: Any) -> None: + """A delete that did not finish: what ax, the controller, ateapi and atelet + say, so a stuck Terminating Task is diagnosable (run 2 lost all of it).""" + diag: dict[str, Any] = {"calls": calls} + diag["task"] = ax(f"get task {name}")[1][-1500:] + diag["actors"] = actors() + diag["templates"] = templates() + _, diag["pods"] = nas.execute("k3s kubectl get pods -A -o wide 2>&1") + _, diag["controller"] = nas.execute("k3s kubectl -n ax-system logs deploy/ax-controller --since=30m 2>&1 | tail -n 150") + _, diag["controller_prev"] = nas.execute("k3s kubectl -n ax-system logs deploy/ax-controller --previous --tail=60 2>&1") + _, diag["ateapi"] = nas.execute( + "for p in $(k3s kubectl -n ate-system get pods -o name | grep ate-api-server); do echo \"== $p\"; " + f"k3s kubectl -n ate-system logs $p --all-containers --since=30m 2>&1 | grep -a -i -E '{name}|error|warn|terminate' | tail -n 80; done" + ) + _, diag["ate_controller"] = nas.execute("k3s kubectl -n ate-system logs deploy/ate-controller --since=30m 2>&1 | tail -n 60") + _, diag["atelet"] = nas.execute( + f"k3s kubectl -n ate-system logs -l app=atelet --all-containers --since=30m 2>&1 | grep -a -i -E '{name}|Terminate|error' | tail -n 80" + ) + _, diag["redis"] = nas.execute( + "k3s kubectl -n ax-system exec deploy/ax-redis -- sh -c " + "'redis-cli XINFO GROUPS ax:stream:tasks; redis-cli XINFO CONSUMERS ax:stream:tasks ax-controllers; " + "redis-cli XPENDING ax:stream:tasks ax-controllers; redis-cli XREVRANGE ax:stream:tasks + - COUNT 8' 2>&1" + ) + _, diag["host_load"] = coordinator.execute("cat /proc/loadavg; cat /proc/pressure/cpu 2>/dev/null") + diag = {k: (v[-8000:] if isinstance(v, str) else v) for k, v in diag.items()} + record(f"nop1_delete_hang_{name}", diag) + + def delete_task(name: str) -> Any: """Delete as the link would; measure idempotence and the Task's removal.""" t0 = time.monotonic() first = ax(f"delete task {name}") + first_s = round(time.monotonic() - t0, 1) + calls: dict[str, Any] = {"first": {"rc": first[0], "out": first[1][-300:], "done_at_s": first_s}} + if first[0] != 0 or first_s > 30: + # The client's own wait (5 min) did not see NotFound: capture now, while stuck. + delete_hang_diag(name, calls) again = ax(f"delete task {name}") # while Terminating - coordinator.wait_until_succeeds(f"! {AX} get task {name} >/dev/null 2>&1", timeout=300) + calls["while_terminating"] = {"rc": again[0], "out": again[1][-300:], "done_at_s": round(time.monotonic() - t0, 1)} + try: + coordinator.wait_until_succeeds(f"! {AX} get task {name} >/dev/null 2>&1", timeout=300) + except Exception: + delete_hang_diag(name, calls) + raise gone_s = round(time.monotonic() - t0, 1) after = ax(f"delete task {name}") # once gone return { - "first": {"rc": first[0], "out": first[1][-200:]}, - "while_terminating": {"rc": again[0], "out": again[1][-200:]}, + **calls, "after_gone": {"rc": after[0], "out": after[1][-200:]}, "gone_seconds": gone_s, } From 163b680238736d831133c3bb3a38347aee02b4da Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 12:41:28 +0200 Subject: [PATCH 29/37] ax-fleet: fix round 2 (pod-to-host INPUT, desk CPU weight, event-driven tunables, conntrack, teardown on PATH, read-only registry, split k3s token, NAS parity, boot snapshot, receipt) - harness: refuse NEW connections from cni0/flannel.1 at the head of nixos-fw (v4 and v6); user.slice and system.slice CPUWeight 10000 (deskCpuWeight). - k3s: kernel tunables watcher bound to k3s (no 900 s bound); boot snapshot takes non-ax declared values (NIXOS_ACTION unset at boot); kube-proxy conntrack args 0; conntrack keys snapshotted; teardown on PATH for the control and harness roles whatever the switch says; server and agent tokens split (agentTokenFile on the NAS). - control: registry read-only (maintenance.readonly, delete off); the seed pushes through a loopback writer that lives only for the seed run. - substrate: images built from inputs.nixpkgs, one digest set everywhere. - secrets: k3s-token.age re-minted for editors ++ nasOnly, new k3s-agent-token.age for editors ++ coordinatorOnly ++ nasOnly. - tests: ax-fleet-boot on nixpkgs-stable with the systemd initrd and the NAS kernel; VM coordinator on the desk kernel with sshd; new assertions; every phase subtest recorded in receipt.json. Co-Authored-By: Claude Opus 5.5 --- flake.nix | 9 +- hosts/coordinator/default.nix | 4 +- hosts/nas/default.nix | 13 +- modules/ax-fleet/control.nix | 52 +++- modules/ax-fleet/harness.nix | 112 +++++--- modules/ax-fleet/interface.nix | 20 ++ modules/ax-fleet/k3s.nix | 382 +++++++++++++++++--------- modules/ax-fleet/substrate.nix | 14 +- pkgs/ax-fleet-teardown/default.nix | 4 +- secrets.nix | 17 +- secrets/k3s-agent-token.age | 9 + secrets/k3s-token.age | 18 +- tests/ax-fleet-boot/default.nix | 36 ++- tests/ax-fleet-topology/default.nix | 83 +++++- tests/ax-fleet/default.nix | 7 + tests/ax-fleet/nodes.nix | 29 +- tests/ax-fleet/phases/10-cluster.py | 40 +++ tests/ax-fleet/phases/20-substrate.py | 18 +- tests/ax-fleet/phases/30-ax.py | 24 +- tests/ax-fleet/phases/35-lan-guard.py | 17 ++ tests/ax-fleet/phases/90-rollback.py | 8 +- 21 files changed, 675 insertions(+), 241 deletions(-) create mode 100644 secrets/k3s-agent-token.age diff --git a/flake.nix b/flake.nix index 17e9e127e..df96db44f 100644 --- a/flake.nix +++ b/flake.nix @@ -727,8 +727,13 @@ }; }; - # `sudo nix run ~/dotfiles#ax-fleet-teardown` after `myAxFleet.enable = - # false` and a switch: k3s-killall.sh, the guard chain, the sysctl restore. + # The same teardown the control and harness roles keep on their PATH + # (`sudo ax-fleet-teardown` after `myAxFleet.enable = false` and a + # switch). This app is for a generation rollback from a checkout, where + # the older generation predates the package: on the coordinator + # `sudo nix run ~/dotfiles#ax-fleet-teardown`; for the NAS, which holds + # no checkout, `nix copy --to ssh-ng://nas .#ax-fleet-teardown`, then + # `ssh -t nas sudo /bin/ax-fleet-teardown`. apps.${system}.ax-fleet-teardown = { type = "app"; program = "${self.packages.${system}.ax-fleet-teardown}/bin/ax-fleet-teardown"; diff --git a/hosts/coordinator/default.nix b/hosts/coordinator/default.nix index 070800e38..b42c8994c 100644 --- a/hosts/coordinator/default.nix +++ b/hosts/coordinator/default.nix @@ -93,8 +93,8 @@ # The HARNESS node (modules/ax-fleet/harness.nix): a k3s agent tainted # ate.dev/sandboxClass=gvisor:NoSchedule, so only atelet and the gVisor # WorkerPool land here. "agent harnesses on coordinator" (Tom, 2026-09-23). - # Switch the NAS first. `false`, switch, then - # `sudo nix run ~/dotfiles#ax-fleet-teardown` is the whole rollback. This + # Switch the NAS first. `false`, switch, then `sudo ax-fleet-teardown` + # (on PATH whatever the switch says) is the whole rollback. This # also turns myAxClient (kubectl, ax) on by mkDefault. myAxFleet = { enable = true; diff --git a/hosts/nas/default.nix b/hosts/nas/default.nix index 7e9d908d0..5ff4591d3 100644 --- a/hosts/nas/default.nix +++ b/hosts/nas/default.nix @@ -128,11 +128,14 @@ myNas.headscale.backup.enable = true; # ── ax on the fleet: THE kill switch for this host ───────────────────── - # One line. `false`, switch, then `sudo nix run ~/dotfiles#ax-fleet-teardown` - # (k3s-killall.sh plus the sysctl restore) removes every trace but the data - # left on purpose under /mnt/fast/k3s and /mnt/nas/services/ax-fleet. The - # k3s token is the agenix secret secrets/k3s-token.age (mySecrets is on - # here). Switch order: this host first, then the coordinator. + # One line. `false`, switch (from the coordinator, --target-host nas), then + # `ssh -t nas sudo ax-fleet-teardown` (k3s-killall.sh, the guard table and + # the sysctl restore). The teardown stays on this host's PATH with the + # switch off; there is no dotfiles checkout here. It removes every trace but + # the data left on purpose under /mnt/fast/k3s and /mnt/nas/services/ax-fleet. + # The k3s credentials are the agenix secrets secrets/k3s-token.age (server, + # this host only) and secrets/k3s-agent-token.age (mySecrets is on here). + # Switch order: this host first, then the coordinator. myAxFleet = { enable = true; role = "control"; diff --git a/modules/ax-fleet/control.nix b/modules/ax-fleet/control.nix index 4cc5fc259..883cff8fa 100644 --- a/modules/ax-fleet/control.nix +++ b/modules/ax-fleet/control.nix @@ -116,24 +116,58 @@ let ''; stepNames = lib.sort (a: b: a < b) (lib.attrNames cfg.bootstrap); - stepScript = name: pkgs.writeShellScript "ax-fleet-step-${name}" '' - set -euo pipefail - ${cfg.bootstrap.${name}} - ''; + stepScript = + name: + pkgs.writeShellScript "ax-fleet-step-${name}" '' + set -euo pipefail + ${cfg.bootstrap.${name}} + ''; + + # ── the registry: read-only to the network, writable only to the seed ── + # (fix round 2) The round-2 review MEASURED an unprivileged user on the + # coordinator pushing blobs (202), mounting across repos (201) and + # overwriting substrate/atelet:d277088b's tag (201): the source-address rule + # admits every uid and every pod on the coordinator (masqueraded to its LAN + # address). ate-setup resolves --image-tag to a digest at install time, so a + # rewritten tag is what would get pinned. Now the served instance on + # ${cfg.registry} runs with storage.maintenance.readonly, and delete is off; + # the seed pushes through a second, loopback-only instance on the same root + # directory that lives only for the seed run, as the registry user. + seedAddr = "127.0.0.1:5001"; + registryBase = { + version = "0.1"; + log.fields.service = "registry"; + storage = { + cache.blobdescriptor = "inmemory"; + delete.enabled = false; + filesystem.rootdirectory = cfg.registryRoot; + }; + http.headers.X-Content-Type-Options = [ "nosniff" ]; + }; + seedRegistryConfig = pkgs.writeText "ax-fleet-seed-registry.json" ( + builtins.toJSON (lib.recursiveUpdate registryBase { http.addr = seedAddr; }) + ); seedScript = '' set -euo pipefail + # The writable instance, loopback only, gone when this script exits. + setpriv --reuid=docker-registry --regid=docker-registry --init-groups \ + ${lib.getExe config.services.dockerRegistry.package} serve ${seedRegistryConfig} & + writer=$! + trap 'kill $writer 2>/dev/null || true; wait $writer 2>/dev/null || true' EXIT for _ in $(seq 1 120); do - curl -fsS -o /dev/null http://${cfg.registry}/v2/ && break + curl -fsS -o /dev/null http://${cfg.registry}/v2/ && curl -fsS -o /dev/null http://${seedAddr}/v2/ && break sleep 1 done curl -fsS -o /dev/null http://${cfg.registry}/v2/ + curl -fsS -o /dev/null http://${seedAddr}/v2/ ${lib.concatStrings ( lib.mapAttrsToList (name: s: '' echo "seed ${name}: ${s.repo}:${s.tag}" digest=$(tr -d '[:space:]' < ${s.oci}/digest) skopeo --insecure-policy copy --all --preserve-digests --dest-tls-verify=false \ - oci:${s.oci} docker://${cfg.registry}/${s.repo}:${s.tag} + oci:${s.oci} docker://${seedAddr}/${s.repo}:${s.tag} + # Read back through the served, read-only instance: what pods pull. skopeo --insecure-policy inspect --raw --tls-verify=false \ docker://${cfg.registry}/${s.repo}@"$digest" >/dev/null echo "seeded ${s.repo}@$digest" @@ -231,7 +265,10 @@ in listenAddress = registryHost; port = registryPort; storagePath = cfg.registryRoot; - enableDelete = true; + # Read-only to the network (see seedAddr); garbage collection is + # offline and needs no delete API. + enableDelete = false; + extraConfig.storage.maintenance.readonly.enabled = true; enableGarbageCollect = true; garbageCollectDates = "weekly"; # openFirewall NOT used: the source-scoped rule below is the access control. @@ -289,6 +326,7 @@ in pkgs.skopeo pkgs.curl pkgs.coreutils + pkgs.util-linux ]; environment.HOME = "/var/lib/ax-fleet"; serviceConfig = { diff --git a/modules/ax-fleet/harness.nix b/modules/ax-fleet/harness.nix index ce3db0bf4..54e030faf 100644 --- a/modules/ax-fleet/harness.nix +++ b/modules/ax-fleet/harness.nix @@ -18,6 +18,9 @@ # - the tailnet: flannel and kube-proxy bind the LAN leg only, nothing is # published, and the guard chain below keeps pods, wifi and the tailnet # apart even though k3s turns ip_forward on (judge 1's second risk). The +# guard chain polices FORWARD only; pods reaching the coordinator HOST +# (sshd, which accepts passwords) go through INPUT, so pod interfaces get +# their own refusal at the head of nixos-fw (fix round 2, see podInput). The # chain covers the direct path; the path routed through the NAS into # VXLAN is closed twice, by the NAS's prerouting range guard # (control.nix) and by the flannel.1 source rule here. Plain VXLAN on the @@ -40,6 +43,7 @@ let ]; ipt = "${pkgs.iptables}/bin/iptables -w"; + ip6t = "${pkgs.iptables}/bin/ip6tables -w"; registryPort = lib.last (lib.splitString ":" cfg.registry); # ── the guard chain (DESIGN 6.4), in mangle FORWARD, position 1 ── @@ -51,44 +55,43 @@ let # (atelet's anonymous GCS fetch, a worker pod reaching the LAN), because the # reply arrives on the LAN leg from a source that is not the NAS. Only NEW # flows are policed, which is the property the guard exists for. - guardRules = - [ - "-m conntrack --ctstate ESTABLISHED,RELATED -j RETURN" - # Into pods over VXLAN only from the pod network (fix round 1). Real - # peers, the NAS host included (its flannel.1 address), are sourced - # from the pod CIDR. Defence in depth, not the fix: LAN traffic the NAS - # routes into VXLAN (MEASURED bypass, worker -> nas -> flannel.1 -> - # harness pod, rc=0) is likely masqueraded by flannel's own rule to the - # NAS's flannel.1 address (INFERRED), so the NAS's prerouting range - # guard (control.nix) is what closes that path. This rule drops VXLAN - # payloads whose inner source is outside the pod CIDR. - "-i flannel.1 ! -s ${cfg.podCidr} -j DROP" - ] - ++ lib.concatMap ( - g: - lib.concatMap (p: [ - "-i ${g} -o ${p} -j DROP" - "-i ${p} -o ${g} -j DROP" - ]) podIfs - ++ [ - "-i ${lan} -o ${g} -j DROP" - "-i ${g} -o ${lan} -j DROP" - ] - ) cfg.guardInterfaces + guardRules = [ + "-m conntrack --ctstate ESTABLISHED,RELATED -j RETURN" + # Into pods over VXLAN only from the pod network (fix round 1). Real + # peers, the NAS host included (its flannel.1 address), are sourced + # from the pod CIDR. Defence in depth, not the fix: LAN traffic the NAS + # routes into VXLAN (MEASURED bypass, worker -> nas -> flannel.1 -> + # harness pod, rc=0) is likely masqueraded by flannel's own rule to the + # NAS's flannel.1 address (INFERRED), so the NAS's prerouting range + # guard (control.nix) is what closes that path. This rule drops VXLAN + # payloads whose inner source is outside the pod CIDR. + "-i flannel.1 ! -s ${cfg.podCidr} -j DROP" + ] + ++ lib.concatMap ( + g: + lib.concatMap (p: [ + "-i ${g} -o ${p} -j DROP" + "-i ${p} -o ${g} -j DROP" + ]) podIfs ++ [ - "-i ${lan} -o cni0 ! -s ${cfg.serverAddress} -j DROP" - # Pod traffic leaving on the LAN leg is masqueraded to this host's LAN - # address and would inherit every NAS rule that trusts the coordinator - # (ssh, NFS, media, paperless). From pods, the LAN gets only the - # apiserver and the registry on the NAS; the internet (atelet's GCS - # fetch) is unaffected. Everything else in-cluster rides flannel.1. - "-i cni0 -o ${lan} -d ${cfg.serverAddress} -p tcp -m multiport --dports 6443,${registryPort} -j RETURN" + "-i ${lan} -o ${g} -j DROP" + "-i ${g} -o ${lan} -j DROP" ] - # Every private range, not only the house /24 (fix round 1): when - # NetworkManager falls back to the Freebox profile on ${lan} - # (hosts/coordinator/uplink-nas.nix), the leg is a DHCP subnet this - # module does not know, and 100.64/10 is the tailnet's range. - ++ map (r: "-i cni0 -o ${lan} -d ${r} -j DROP") privateRanges; + ) cfg.guardInterfaces + ++ [ + "-i ${lan} -o cni0 ! -s ${cfg.serverAddress} -j DROP" + # Pod traffic leaving on the LAN leg is masqueraded to this host's LAN + # address and would inherit every NAS rule that trusts the coordinator + # (ssh, NFS, media, paperless). From pods, the LAN gets only the + # apiserver and the registry on the NAS; the internet (atelet's GCS + # fetch) is unaffected. Everything else in-cluster rides flannel.1. + "-i cni0 -o ${lan} -d ${cfg.serverAddress} -p tcp -m multiport --dports 6443,${registryPort} -j RETURN" + ] + # Every private range, not only the house /24 (fix round 1): when + # NetworkManager falls back to the Freebox profile on ${lan} + # (hosts/coordinator/uplink-nas.nix), the leg is a DHCP subnet this + # module does not know, and 100.64/10 is the tailnet's range. + ++ map (r: "-i cni0 -o ${lan} -d ${r} -j DROP") privateRanges; privateRanges = lib.unique [ cfg.lan.cidr @@ -98,6 +101,20 @@ let "100.64.0.0/10" ]; + # ── pods never open a connection to the coordinator host (fix round 2) ── + # MEASURED by the round-2 review: from a pod, `nc 10.200.0.1 22` and the + # node's LAN address answered SSH-2.0-OpenSSH, because nixos-fw accepts 22 on + # every interface and the guard chain above sees FORWARD only. Nothing on + # the Substrate path needs a NEW pod-to-host flow: hostPorts and ClusterIPs + # are DNATed through FORWARD, kubelet reaches pods (OUTPUT, replies are + # ESTABLISHED), atelet talks to kubelet and containerd over unix sockets, and + # kubectl exec/logs ride the agent tunnel. First rules of nixos-fw, so they + # run before its ESTABLISHED accept and every port rule; IPv6 too (link-local + # addresses on cni0 and the veths). + podInput = lib.concatMap (p: [ + "-I nixos-fw 1 -i ${p} -m conntrack --ctstate NEW -m comment --comment ax-fleet-pod-input -j nixos-fw-refuse" + ]) podIfs; + guardStart = '' # ax-fleet guard chain (idempotent) ${ipt} -t mangle -N ax-fleet-guard 2>/dev/null || true @@ -107,6 +124,10 @@ let ${lib.concatMapStringsSep "\n" (r: "${ipt} -t mangle -A ax-fleet-guard ${r}") guardRules} # flannel VXLAN from the NAS only; no TCP port is opened. ${ipt} -A nixos-fw -i ${lan} -s ${cfg.serverAddress} -p udp --dport 8472 -j nixos-fw-accept + # pods to the host: refused (podInput; nixos-fw is rebuilt on every reload) + ${lib.concatMapStringsSep "\n" ( + r: "${ipt} ${r}" + lib.optionalString config.networking.enableIPv6 "\n${ip6t} ${r}" + ) podInput} ''; guardStop = '' @@ -146,13 +167,28 @@ in { config = lib.mkIf on { myAxFleet.kubelet = { - # Tom's seats, Chrome and a coordinator Halogen feel pressure after the - # sandboxes are evicted, never before. + # Memory: Tom's seats, Chrome and a coordinator Halogen feel pressure + # after the sandboxes are evicted, never before. CPU is the slices below: + # system-reserved only shrinks kubepods.slice's weight, it protects + # nothing. systemReserved = lib.mkDefault "cpu=8,memory=32Gi"; kubeReserved = lib.mkDefault "cpu=1,memory=2Gi"; evictionHard = lib.mkDefault "memory.available<8Gi"; }; + # ── CPU: the desk outweighs the sandboxes (fix round 2) ── + # kubelet gives kubepods.slice cpu.weight = 1 + ((allocatable_mcpu * 1024 + # / 1000 - 2) * 9999) / 262142: MEASURED 274 for 7 allocatable CPUs in the + # review VM, INFERRED 899 for the desk's 32 - 8 - 1 = 23. user.slice and + # system.slice are 100 on the live box (MEASURED), so under contention the + # gVisor workers (no CPU limit, substrate.nix) would take about 90 % of the + # CPU from niri, herdr, the seats and Chrome. Weights are work-conserving: + # idle desk CPU still goes to the sandboxes. Both slices get the same + # weight, so their ratio to each other is unchanged. switch-to-configuration + # never restarts a slice; daemon-reload applies the property. + systemd.slices.user.sliceConfig.CPUWeight = lib.mkDefault cfg.kubelet.deskCpuWeight; + systemd.slices.system.sliceConfig.CPUWeight = lib.mkDefault cfg.kubelet.deskCpuWeight; + services.k3s = { role = "agent"; # The IP, not the name `nas`. diff --git a/modules/ax-fleet/interface.nix b/modules/ax-fleet/interface.nix index 0c56097ed..937982601 100644 --- a/modules/ax-fleet/interface.nix +++ b/modules/ax-fleet/interface.nix @@ -108,6 +108,17 @@ in ''; }; + k3sAgentTokenFile = mkOption { + type = types.nullOr types.str; + default = null; + description = '' + The agent join credential (server --agent-token-file, agent + --token-file). null (the default) means the agenix secret + secrets/k3s-agent-token.age, declared by ./k3s.nix. Tests set a path + to a plain file instead. + ''; + }; + stateRoot = mkOption { type = types.str; default = "/mnt/fast"; @@ -153,6 +164,15 @@ in type = types.nullOr types.str; default = null; }; + deskCpuWeight = mkOption { + type = types.ints.between 1 10000; + default = 10000; + description = '' + Harness role: cpu.weight for user.slice and system.slice, against + kubepods.slice's kubelet-computed weight (INFERRED 899 on the desk). + The desk wins CPU contention; idle CPU still goes to the sandboxes. + ''; + }; keepHostKernelTunables = mkOption { type = types.bool; default = true; diff --git a/modules/ax-fleet/k3s.nix b/modules/ax-fleet/k3s.nix index 53ad4641a..a2f37b2cd 100644 --- a/modules/ax-fleet/k3s.nix +++ b/modules/ax-fleet/k3s.nix @@ -23,17 +23,21 @@ # (./control.nix). let cfg = config.myAxFleet; - cluster = cfg.enable && (cfg.role == "control" || cfg.role == "harness"); + clusterRole = options.myAxFleet.role.isDefined && (cfg.role == "control" || cfg.role == "harness"); + cluster = cfg.enable && clusterRole; + + teardown = pkgs.callPackage ../../pkgs/ax-fleet-teardown { k3s = cfg.k3sPackage; }; # The feature gates Substrate's podcertcontroller needs, on all three # components (A5a and the probe MEASURED that kubelet accepts all three). gates = "ClusterTrustBundle=true,ClusterTrustBundleProjection=true,PodCertificateRequest=true"; - kubeletArgs = - [ "feature-gates=${gates}" ] - ++ lib.optional (cfg.kubelet.systemReserved != null) "system-reserved=${cfg.kubelet.systemReserved}" - ++ lib.optional (cfg.kubelet.kubeReserved != null) "kube-reserved=${cfg.kubelet.kubeReserved}" - ++ lib.optional (cfg.kubelet.evictionHard != null) "eviction-hard=${cfg.kubelet.evictionHard}"; + kubeletArgs = [ + "feature-gates=${gates}" + ] + ++ lib.optional (cfg.kubelet.systemReserved != null) "system-reserved=${cfg.kubelet.systemReserved}" + ++ lib.optional (cfg.kubelet.kubeReserved != null) "kube-reserved=${cfg.kubelet.kubeReserved}" + ++ lib.optional (cfg.kubelet.evictionHard != null) "eviction-hard=${cfg.kubelet.evictionHard}"; commonFlags = [ "--resolv-conf=/etc/ax-fleet/resolv.conf" @@ -42,6 +46,14 @@ let # kube-proxy never sets route_localnet. "--kube-proxy-arg=nodeport-addresses=${cfg.lan.address}/32" "--kube-proxy-arg=iptables-localhost-nodeports=false" + # kube-proxy leaves the host's conntrack table alone (fix round 2). Its + # defaults set nf_conntrack_tcp_timeout_established to 86400 and + # close_wait to 3600 (MEASURED in the review VM; the live NAS and desk + # read 432000 and 60), which on the house router would drop the SNAT + # entry of any NATed flow idle for a day. 0 = do not touch. + "--kube-proxy-arg=conntrack-tcp-timeout-established=0s" + "--kube-proxy-arg=conntrack-tcp-timeout-close-wait=0s" + "--kube-proxy-arg=conntrack-max-per-core=0" ] ++ map (a: "--kubelet-arg=${a}") kubeletArgs; @@ -73,133 +85,245 @@ let "kernel.panic" "kernel.panic_on_oops" "vm.overcommit_memory" + # kube-proxy's conntrack keys (fix round 2): the args above leave them, + # the snapshot and the teardown make sure. Absent when nf_conntrack is + # not loaded yet (boot), then simply not recorded. + "net.netfilter.nf_conntrack_max" + "net.netfilter.nf_conntrack_tcp_timeout_established" + "net.netfilter.nf_conntrack_tcp_timeout_close_wait" ]; + + # kubelet's values for the three kernel keys (upstream kubelet + # setupKernelTunables; MEASURED 10 1 1 in the VM tests). + kubeletKernel = { + "kernel.panic" = "10"; + "kernel.panic_on_oops" = "1"; + "vm.overcommit_memory" = "1"; + }; + + # ── the host's own declared value for a snapshot key (fix round 2) ── + # At boot the activation script runs before systemd-sysctl, so /proc holds + # kernel defaults, not the generation's values: the round-2 review MEASURED + # a snapshot of ip_forward = 0 on a NAS that declares 1 (the house router), + # which the teardown would then have applied. So at boot, a key some module + # OUTSIDE modules/ax-fleet declares is recorded from that declaration; the + # keys only this module declares (the harness's ip_forward and proxy_arp) + # and the undeclared ones are read from /proc, where they still hold the + # pre-ax value. null = no non-ax declaration, or conflicting ones. + unwrap = + v: + if builtins.isAttrs v && (v._type or null) == "override" then + unwrap v.content + else if builtins.isAttrs v && (v._type or null) == "if" then + (if v.condition then unwrap v.content else null) + else + v; + sysctlString = + v: + if builtins.isBool v then + (if v then "1" else "0") + else if v == null then + null + else + toString v; + hostDeclared = + k: + let + defs = lib.filter ( + d: !(lib.hasInfix "/modules/ax-fleet/" (toString d.file)) + ) options.boot.kernel.sysctl.definitionsWithLocations; + vals = lib.unique ( + lib.filter (v: v != null) ( + map (d: sysctlString (unwrap (d.value.${k} or null))) ( + lib.filter (d: builtins.isAttrs d.value) defs + ) + ) + ); + in + if builtins.length vals == 1 then builtins.head vals else null; + + serverToken = + if cfg.k3sTokenFile != null then cfg.k3sTokenFile else config.age.secrets.k3s-token.path; + agentToken = + if cfg.k3sAgentTokenFile != null then + cfg.k3sAgentTokenFile + else + config.age.secrets.k3s-agent-token.path; in { - config = lib.mkIf cluster ( - lib.mkMerge [ - { - services.k3s = { - enable = true; - package = cfg.k3sPackage; - tokenFile = - if cfg.k3sTokenFile != null then cfg.k3sTokenFile else config.age.secrets.k3s-token.path; - nodeIP = cfg.lan.address; - # k3s core images (pause, coredns, local-path and its helper) come - # from the pinned airgap tarball, never from the network. - images = [ cfg.k3sPackage.airgap-images ]; - extraFlags = commonFlags; - }; - - environment.etc."rancher/k3s/registries.yaml".text = registriesYaml; - # The kubelet's (and so CoreDNS's) upstream resolver: AdGuard on the - # NAS, never the coordinator's systemd-resolved stub. - environment.etc."ax-fleet/resolv.conf".text = "nameserver ${cfg.serverAddress}\n"; - - boot.kernelModules = [ - "overlay" - "br_netfilter" - ]; - # Both hosts define 512 / 524288 today (modules/common.nix and the - # nixpkgs default); a plain definition would be an evaluation - # conflict. mkOverride 99 raises them to the probe recipe's values. - boot.kernel.sysctl."fs.inotify.max_user_instances" = lib.mkOverride 99 8192; - boot.kernel.sysctl."fs.inotify.max_user_watches" = lib.mkOverride 99 1048576; - - # The k3s package carries k3s-killall.sh; `KillMode=process` means - # pods and shims outlive the unit, so teardown is a real step. - environment.systemPackages = [ (pkgs.callPackage ../../pkgs/ax-fleet-teardown { k3s = cfg.k3sPackage; }) ]; - - systemd.services.k3s = { - # Re-create the manifest and image links AFTER the bind mounts are - # up, whatever order a live switch ran tmpfiles and mounts in. - serviceConfig.ExecStartPre = [ - "${config.systemd.package}/bin/systemd-tmpfiles --create --prefix=/var/lib/rancher/k3s" - # --node-ip and --flannel-iface name the LAN address, which - # NetworkManager adds after network.target at boot (MEASURED on the - # NAS). Bounded wait, then start anyway (k3s's Restart covers it). - "${pkgs.writeShellScript "ax-fleet-k3s-wait-lan-addr" '' - for _ in $(${pkgs.coreutils}/bin/seq 30); do - ${pkgs.iproute2}/bin/ip -4 addr show dev ${cfg.lan.interface} 2>/dev/null \ - | ${pkgs.gnugrep}/bin/grep -qF 'inet ${cfg.lan.address}/' && exit 0 - ${pkgs.coreutils}/bin/sleep 1 - done - echo "ax-fleet: ${cfg.lan.address} not on ${cfg.lan.interface} after 30 s; starting anyway" >&2 - exit 0 - ''}" - ]; - }; - - # ── kubelet's kernel tunables, put back (myAxFleet.kubelet.keepHostKernelTunables) ── - # kubelet sets these once per start (container manager setup, - # protectKernelDefaults off; INFERRED from upstream kubelet, the VM test - # measures the result). PartOf k3s: every k3s restart re-runs this - # after kubelet has applied its values, which it waits for. - systemd.services.ax-fleet-kernel-tunables = lib.mkIf cfg.kubelet.keepHostKernelTunables { - description = "ax-fleet: restore the host's kernel.panic, kernel.panic_on_oops, vm.overcommit_memory after kubelet"; - wantedBy = [ "k3s.service" ]; - after = [ "k3s.service" ]; - partOf = [ "k3s.service" ]; - path = [ - pkgs.procps - pkgs.coreutils - pkgs.gnugrep + config = lib.mkMerge [ + # The kill switch's second half stays on the host whatever the switch says + # (fix round 2): the documented rollback is `enable = false`, switch, then + # `sudo ax-fleet-teardown` ON the host, and the NAS holds no dotfiles + # checkout (MEASURED). Inert until run by hand. `KillMode=process` means + # pods and shims outlive k3s, so the teardown is a real step. + (lib.mkIf clusterRole { environment.systemPackages = [ teardown ]; }) + + (lib.mkIf cluster ( + lib.mkMerge [ + { + services.k3s = { + enable = true; + package = cfg.k3sPackage; + # Two credentials (fix round 2). The server token is the NAS's + # alone; with no agent token k3s gives agents the server password + # (MEASURED deps.go getNodePass), i.e. the k3s:server role on + # /v1-k3s/token, /cacerts and /encrypt/config. The coordinator joins + # with the agent token only. + tokenFile = if cfg.role == "control" then serverToken else agentToken; + agentTokenFile = if cfg.role == "control" then agentToken else null; + nodeIP = cfg.lan.address; + # k3s core images (pause, coredns, local-path and its helper) come + # from the pinned airgap tarball, never from the network. + images = [ cfg.k3sPackage.airgap-images ]; + extraFlags = commonFlags; + }; + + environment.etc."rancher/k3s/registries.yaml".text = registriesYaml; + # The kubelet's (and so CoreDNS's) upstream resolver: AdGuard on the + # NAS, never the coordinator's systemd-resolved stub. + environment.etc."ax-fleet/resolv.conf".text = "nameserver ${cfg.serverAddress}\n"; + + boot.kernelModules = [ + "overlay" + "br_netfilter" ]; - serviceConfig = { - Type = "oneshot"; - RemainAfterExit = true; + # Both hosts define 512 / 524288 today (modules/common.nix and the + # nixpkgs default); a plain definition would be an evaluation + # conflict. mkOverride 99 raises them to the probe recipe's values. + boot.kernel.sysctl."fs.inotify.max_user_instances" = lib.mkOverride 99 8192; + boot.kernel.sysctl."fs.inotify.max_user_watches" = lib.mkOverride 99 1048576; + + systemd.services.k3s = { + # Re-create the manifest and image links AFTER the bind mounts are + # up, whatever order a live switch ran tmpfiles and mounts in. + serviceConfig.ExecStartPre = [ + "${config.systemd.package}/bin/systemd-tmpfiles --create --prefix=/var/lib/rancher/k3s" + # --node-ip and --flannel-iface name the LAN address, which + # NetworkManager adds after network.target at boot (MEASURED on the + # NAS). Bounded wait, then start anyway (k3s's Restart covers it). + "${pkgs.writeShellScript "ax-fleet-k3s-wait-lan-addr" '' + for _ in $(${pkgs.coreutils}/bin/seq 30); do + ${pkgs.iproute2}/bin/ip -4 addr show dev ${cfg.lan.interface} 2>/dev/null \ + | ${pkgs.gnugrep}/bin/grep -qF 'inet ${cfg.lan.address}/' && exit 0 + ${pkgs.coreutils}/bin/sleep 1 + done + echo "ax-fleet: ${cfg.lan.address} not on ${cfg.lan.interface} after 30 s; starting anyway" >&2 + exit 0 + ''}" + ]; + }; + + # ── kubelet's kernel tunables, put back (myAxFleet.kubelet.keepHostKernelTunables) ── + # kubelet sets these whenever its container manager starts, which is + # NOT tied to a k3s restart: an agent whose server is unreachable is + # `active` and runs no kubelet, and sets 10 1 1 whenever the server + # answers, with NRestarts 0 (MEASURED by the round-2 review). So a + # watcher, bound to k3s, not a one-shot after it (the round-1 unit gave + # up after 900 s). Per key: only a value equal to kubelet's that differs + # from the snapshot is put back, so a deliberate host change to any + # other value is left alone. + systemd.services.ax-fleet-kernel-tunables = lib.mkIf cfg.kubelet.keepHostKernelTunables { + description = "ax-fleet: keep the host's kernel.panic, kernel.panic_on_oops, vm.overcommit_memory against kubelet"; + wantedBy = [ "k3s.service" ]; + after = [ "k3s.service" ]; + partOf = [ "k3s.service" ]; + path = [ + pkgs.procps + pkgs.coreutils + pkgs.gnugrep + ]; + serviceConfig = { + Type = "simple"; + Restart = "always"; + RestartSec = 5; + }; + script = '' + snap=/var/lib/ax-fleet/sysctl-before.conf + [ -s "$snap" ] || { echo "no $snap; leaving kernel tunables as they are"; exec sleep infinity; } + while :; do + ${lib.concatStrings ( + lib.mapAttrsToList (k: kv: '' + want=$(grep -E '^${k} = ' "$snap" | cut -d' ' -f3) + cur=$(sysctl -n ${k}) + if [ -n "$want" ] && [ "$cur" = ${kv} ] && [ "$cur" != "$want" ]; then + sysctl -w "${k}=$want" + fi + '') kubeletKernel + )} + sleep 5 + done + ''; }; - script = '' - snap=/var/lib/ax-fleet/sysctl-before.conf - [ -s "$snap" ] || { echo "no $snap; leaving kernel tunables as they are"; exit 0; } - want() { [ "$(sysctl -n kernel.panic)" = 10 ] && [ "$(sysctl -n kernel.panic_on_oops)" = 1 ] && [ "$(sysctl -n vm.overcommit_memory)" = 1 ]; } - for _ in $(seq 900); do want && break; sleep 1; done - want || echo "kubelet has not set its tunables after 900 s; restoring anyway" - for k in kernel.panic kernel.panic_on_oops vm.overcommit_memory; do - v=$(grep -E "^$k = " "$snap" | cut -d' ' -f3) - if [ -n "$v" ]; then sysctl -w "$k=$v"; fi - done - ''; - }; - - # ── sysctl snapshot, taken BEFORE this generation's sysctls apply ── - # An activation snippet, not a unit: at a live switch the activation - # script runs before systemd-sysctl is restarted with the new values, - # and at boot it runs before systemd starts. A unit ordered before - # k3s would record ip_forward=1 that this very generation just set. - # Written once; later generations never overwrite it. - system.activationScripts.ax-fleet-sysctl-snapshot = { - text = '' - if [ ! -e /var/lib/ax-fleet/sysctl-before.conf ]; then - mkdir -p /var/lib/ax-fleet - { - ${lib.concatMapStringsSep "\n" ( - k: "v=$(cat /proc/sys/${lib.replaceStrings [ "." ] [ "/" ] k} 2>/dev/null) && echo \"${k} = $v\" || true" - ) snapshotKeys} - } > /var/lib/ax-fleet/sysctl-before.conf.tmp - mv /var/lib/ax-fleet/sysctl-before.conf.tmp /var/lib/ax-fleet/sysctl-before.conf - fi - ''; - }; - } - - # The token: the existing agenix secret (commit 643a4196), recipients - # editors ++ delivered ++ nasOnly. Never printed. Only declared where - # agenix is imported and no test token is given. - (lib.optionalAttrs (options ? age) { - age.secrets = lib.mkIf (cfg.k3sTokenFile == null) { - k3s-token = { - file = ../../secrets/k3s-token.age; - mode = "0400"; + + # ── sysctl snapshot, taken BEFORE this generation's sysctls apply ── + # An activation snippet, not a unit: at a live switch the activation + # script runs before systemd-sysctl is restarted with the new values; + # a unit ordered before k3s would record the ip_forward=1 this very + # generation just set on the harness. switch-to-configuration exports + # NIXOS_ACTION (switch, test) to the activation; the boot activation + # (initrd-nixos-activation on the systemd-initrd hosts, stage 2 + # otherwise) does not, and there /proc holds kernel defaults, so + # declared keys come from hostDeclared (fix round 2). /run/systemd/system + # is no signal: it exists in the systemd initrd. Written once; later + # generations never overwrite it. + system.activationScripts.ax-fleet-sysctl-snapshot = { + text = '' + if [ ! -e /var/lib/ax-fleet/sysctl-before.conf ]; then + mkdir -p /var/lib/ax-fleet + { + ${lib.concatMapStringsSep "\n" ( + k: + let + proc = "/proc/sys/${lib.replaceStrings [ "." ] [ "/" ] k}"; + d = hostDeclared k; + in + if d == null then + "v=$(cat ${proc} 2>/dev/null) && echo \"${k} = $v\" || true" + else + '' + if [ -n "''${NIXOS_ACTION:-}" ]; then + v=$(cat ${proc} 2>/dev/null) && echo "${k} = $v" || true + else + echo "${k} = ${d}" + fi'' + ) snapshotKeys} + } > /var/lib/ax-fleet/sysctl-before.conf.tmp + mv /var/lib/ax-fleet/sysctl-before.conf.tmp /var/lib/ax-fleet/sysctl-before.conf + fi + ''; }; - }; - assertions = [ - { - assertion = cfg.k3sTokenFile != null || config.mySecrets.enable or false; - message = "myAxFleet needs agenix delivery (mySecrets.enable) for secrets/k3s-token.age on ${config.networking.hostName}."; - } - ]; - }) - ] - ); + } + + # The tokens: agenix secrets, never printed. k3s-token.age is the + # server's (recipients editors ++ nasOnly); k3s-agent-token.age is the + # agent credential (editors ++ coordinatorOnly ++ nasOnly). Only + # declared where agenix is imported and no test token is given. + (lib.optionalAttrs (options ? age) { + age.secrets = lib.mkMerge [ + (lib.mkIf (cfg.role == "control" && cfg.k3sTokenFile == null) { + k3s-token = { + file = ../../secrets/k3s-token.age; + mode = "0400"; + }; + }) + (lib.mkIf (cfg.k3sAgentTokenFile == null) { + k3s-agent-token = { + file = ../../secrets/k3s-agent-token.age; + mode = "0400"; + }; + }) + ]; + assertions = [ + { + assertion = + (cfg.k3sAgentTokenFile != null && (cfg.role != "control" || cfg.k3sTokenFile != null)) + || config.mySecrets.enable or false; + message = "myAxFleet needs agenix delivery (mySecrets.enable) for secrets/k3s-token.age and secrets/k3s-agent-token.age on ${config.networking.hostName}."; + } + ]; + }) + ] + )) + ]; } diff --git a/modules/ax-fleet/substrate.nix b/modules/ax-fleet/substrate.nix index 48f5df4bf..cfd1cd522 100644 --- a/modules/ax-fleet/substrate.nix +++ b/modules/ax-fleet/substrate.nix @@ -24,12 +24,18 @@ let inherit (pkgs.stdenv.hostPlatform) system; - # The NAS evaluates on nixpkgs-stable; the Go toolchain still comes from the - # nixpkgs-go input, the same derivation pkgs/ax uses. - substrate = pkgs.callPackage ../../pkgs/substrate { + # ONE image set for every evaluator (fix round 2), as ax.nix already does: + # built from the flake's own nixpkgs pin, never the host's `pkgs`. The NAS + # evaluates on nixpkgs-stable, and the round-2 review MEASURED that the six + # component images the real NAS would seed (ateapi, atecontroller, atelet, + # atenet, ateom-gvisor, podcertcontroller) had other digests than the ones + # the VM test ran. The Go toolchain comes from the nixpkgs-go input, the + # same derivation pkgs/ax uses. ax-fleet-topology asserts the parity. + fleetPkgs = inputs.nixpkgs.legacyPackages.${system}; + substrate = fleetPkgs.callPackage ../../pkgs/substrate { go_1_27 = inputs.nixpkgs-go.legacyPackages.${system}.go_1_27; }; - images = pkgs.callPackage ../../pkgs/substrate/images.nix { inherit substrate; }; + images = fleetPkgs.callPackage ../../pkgs/substrate/images.nix { inherit substrate; }; kubectl = "${cfg.k3sPackage}/bin/kubectl"; jq = "${pkgs.jq}/bin/jq"; diff --git a/pkgs/ax-fleet-teardown/default.nix b/pkgs/ax-fleet-teardown/default.nix index 0a015442f..134f739ef 100644 --- a/pkgs/ax-fleet-teardown/default.nix +++ b/pkgs/ax-fleet-teardown/default.nix @@ -73,7 +73,9 @@ writeShellApplication { snap=/var/lib/ax-fleet/sysctl-before.conf if [ -s "$snap" ]; then echo "== sysctl restore from $snap" - sysctl -p "$snap" + # -e: a conntrack key recorded while nf_conntrack was loaded may be + # absent now; skip it rather than abort the restore. + sysctl -e -p "$snap" else echo "ax-fleet-teardown: no $snap; sysctls left as they are" >&2 fi diff --git a/secrets.nix b/secrets.nix index 0d62610c5..05b09be96 100644 --- a/secrets.nix +++ b/secrets.nix @@ -92,12 +92,17 @@ in # NAS private media HTTPS: zone-limited DNS-01 token, no broad Wrangler OAuth # authority. Ciphertext is provisioned before enabling personal-https.nix. "secrets/nas-cloudflare-dns.age".publicKeys = editors ++ nasOnly; - # The k3s cluster join token (ax on the fleet, modules/ax-fleet/k3s.nix). - # Read by the NAS (server) and the coordinator (agent). The ciphertext - # exists (commit 643a4196); rotate with: - # nix develop -c agenix -e secrets/k3s-token.age - # `delivered` also reaches the worker and the client, which never read it. - "secrets/k3s-token.age".publicKeys = editors ++ delivered ++ nasOnly; + # The k3s credentials (ax on the fleet, modules/ax-fleet/k3s.nix), split in + # fix round 2 (2026-09-23). k3s-token.age is the SERVER token: read by the + # NAS only; with it comes the k3s:server role (/v1-k3s/token, /cacerts, + # /encrypt/config). k3s-agent-token.age is the agent join credential: the + # NAS passes it as --agent-token-file, the coordinator joins with it. Both + # were minted fresh from /dev/urandom with `age -R` over these lists, never + # displayed; the 643a4196 ciphertext (also decryptable by the worker and the + # client) was replaced, not re-encrypted, before any cluster used it. + # Rotate with: nix develop -c agenix -e secrets/.age + "secrets/k3s-token.age".publicKeys = editors ++ nasOnly; + "secrets/k3s-agent-token.age".publicKeys = editors ++ coordinatorOnly ++ nasOnly; # --- wifi PSK tier: the coordinator, whose Freebox uplink # (wlp192s0) is now declarative too (migrated from an imperative profile on # flash night — refs #37). Rekey after this change: nix develop -c agenix -r diff --git a/secrets/k3s-agent-token.age b/secrets/k3s-agent-token.age new file mode 100644 index 000000000..091edc6bd --- /dev/null +++ b/secrets/k3s-agent-token.age @@ -0,0 +1,9 @@ +age-encryption.org/v1 +-> X25519 RYDnXPmLP7LRAQUOkbt8OfXKlRrAkxw3ifwxFPxO1U0 +lPdZbsFZ0Q+tIikHyjq4zuiSRO6D/BmWQw4ilwGcrG8 +-> ssh-ed25519 b4VnIg 1Ew3degzdeyANd3OUkRJ4zkCFgTMpeZUaPcvqaNiQXs +b4tGOKtANgMigmVK9kDTjxVRGFaooM0/X6BfFMAai5w +-> ssh-ed25519 sKUETw Yir5k5hDZ2wzoziKlhicGETCYMpKURtD6bv+0A4ykjI +Wb8h0aOXwq3VpycN8k9WjRitw8ooiz8hHirbcb1un20 +--- W0F7q+18w3EXzMQBF7OT5o5kLcHo30Pkarp3RMTJxPQ +ð)¿Ã¯ x€:�{ä¬T™÷À±Ÿ„zeh~5¿ólLʚÃvÞÒd‚ò�7ŠÃox:¨=ˆŽ¾ú̲/€£r9m? \ No newline at end of file diff --git a/secrets/k3s-token.age b/secrets/k3s-token.age index e186668d6..35d03103a 100644 --- a/secrets/k3s-token.age +++ b/secrets/k3s-token.age @@ -1,13 +1,7 @@ age-encryption.org/v1 --> X25519 y+gqc8HC/ZfWgnn9o8usBYyz46P008G4aLATi5Yhwxs -APBAungjujrBhT1RdhkYk19c7SSF0+yJ6Nw5vF9vA8A --> ssh-ed25519 60zAgA e7LN3K6I5q4UauscCH0kayD4sRgA/iDUYRnxJRZ4eDM -uz7fB+BFAz4jtzvlIiIrxD3pkFNj3TKAPwZ2twhcU6k --> ssh-ed25519 b4VnIg Jq/hXKs8BWavUOUA/K3J79+D8Fs5zsCSD4psuaQRyzk -bn0yo1YA9YfAZNTfz0y9bTbnCdO8YKGYkwk1uV7+iiw --> ssh-ed25519 TKMZIQ +mhfynCuc2kc80KiS+JWyMXtT9rw+akJTaE+nysgnkQ -zv2a3s/LeAL6ebGrex0KYlZXqLNm/eBk7lfU7Jm6pvk --> ssh-ed25519 sKUETw t1+9AuwS24H6IK215BXRLA5zxxS/rkzkfH9ScKxymwo -bJN7c35gAZkaudCtRN6djvs6dk8y69DoZwfNC4U1Ezw ---- id+7Z5DGCuTJwOwmVDtMp++Eq3/x/B6db/3hkL7JFKs -|À™a=7ý¬| fò(·:X„8dP„Úø}29⪚ )=§Á!ç‚M_¯Zœ=$¨ÿŸ5û¡7ò¥Ñ_¤h�ÉtÚÛ fnþÊQíärLã‹jBÂX�ø Ñ39Úw­¥-d \ No newline at end of file +-> X25519 dtiPpi1BLiotWQN+ZoSMCY16V39onbZjA2tXCedlGhw +O8vycEoqUfgd4RjUn2TC2CBweycUR9YHaus2/b3IuQg +-> ssh-ed25519 sKUETw zN+hojttTBmbuORJJbvGBJo+feofYkS9R8s7eHnllWw ++wLRUCZuZv7n1sDj4Djnuu2mTDi2sUfNaPTQ0MuVeD4 +--- hvbZ54nuJjxpSPeHNTegaKSMV92TALVThkKhTcK9pbw +4Ô|²œH×x®¥Óm±SWfHmeŽ;uŸҭ×p>aHM`À¦ìl~ï큿� À’  Â0ÒϝÛô{eݮ†ùH[6$%¯ί¾¶äF"µ­ëÔöW–ÑH2¬»¦$Ò÷: \ No newline at end of file diff --git a/tests/ax-fleet-boot/default.nix b/tests/ax-fleet-boot/default.nix index 5dfbd809e..954e4a760 100644 --- a/tests/ax-fleet-boot/default.nix +++ b/tests/ax-fleet-boot/default.nix @@ -9,15 +9,35 @@ # on the /mnt/fast side, the registry seed and the bootstrap succeed, and the # node is Ready and untainted. The substrate track adds its Available checks # here through the same bootstrap. +# +# Built from nixpkgs-STABLE, as the real NAS is (fix round 2): the round-2 +# review MEASURED that every other ax-fleet VM runs NixOS 26.11 and systemd +# 261 while the NAS runs 26.05 and systemd 260, so the NAS's own module set +# (k3s, docker-registry, nftables, systemd) had never run. The NAS boots with +# the systemd initrd, as the real one does (boot.initrd.systemd.enable is +# true on it, MEASURED nix eval), because that is where the boot-time +# activation runs. let - nodes = import ../ax-fleet/nodes.nix { inherit pkgs lib inputs; }; + stablePkgs = inputs.nixpkgs-stable.legacyPackages.x86_64-linux; + nodes = import ../ax-fleet/nodes.nix { + pkgs = stablePkgs; + inherit lib inputs; + }; in -pkgs.testers.runNixOSTest { +stablePkgs.testers.runNixOSTest { name = "ax-fleet-boot"; node.specialArgs = { inherit inputs; }; nodes.nas = { imports = [ nodes.nas ]; myAxFleet.enable = true; + boot.initrd.systemd.enable = true; + # The NAS's kernel (hosts/nas/kernel.nix: freshPkgs.linuxPackages_7_2); + # ax-fleet-topology asserts it is the same derivation. + boot.kernelPackages = + (import inputs.nixpkgs-fresh { + system = "x86_64-linux"; + config.allowUnfree = true; + }).linuxPackages_7_2; myAxFleet.kubelet.systemReserved = "cpu=1,memory=1Gi"; # The real NAS gets 10.42.0.1 from NetworkManager seconds AFTER # network(-online).target (MEASURED 2026-09-23: target at 11.78 s, address @@ -70,8 +90,18 @@ pkgs.testers.runNixOSTest { nas.wait_until_succeeds("test -z \"$(k3s kubectl get node nas -o jsonpath='{.spec.taints}')\"", timeout=300) nas.succeed("k3s kubectl -n kube-system wait --for=condition=Available deploy/coredns deploy/local-path-provisioner --timeout=600s") nas.succeed("test -s /etc/ax-fleet/admin.kubeconfig") - # The snapshot was taken at first activation, before k3s ever ran. + # The snapshot was taken at first activation, before k3s ever ran, and at + # boot it carries the host's declared values, not the kernel defaults the + # boot activation sees (fix round 2: it had recorded ip_forward = 0 on + # the router, which the teardown would have applied). + print("AXFLEET-BOOT snapshot:\n" + nas.succeed("cat /var/lib/ax-fleet/sysctl-before.conf")) + print("AXFLEET-BOOT release=" + nas.succeed("nixos-version").strip() + " systemd=" + nas.succeed("systemctl --version | head -1").strip()) + nas.succeed("test \"$(sysctl -n net.ipv4.ip_forward)\" = 1") + nas.succeed("grep -x 'net.ipv4.ip_forward = 1' /var/lib/ax-fleet/sysctl-before.conf") nas.succeed("grep -q '^kernel.panic = ' /var/lib/ax-fleet/sysctl-before.conf") + # kubelet ran; its panic values are not what the host is left with. + snap_panic = nas.succeed("sed -n 's/^kernel.panic = //p' /var/lib/ax-fleet/sysctl-before.conf").strip() + nas.wait_until_succeeds(f"test \"$(sysctl -n kernel.panic)\" = '{snap_panic}'", timeout=300) nas.fail("findmnt -n -T /var/lib/rancher/k3s -o SOURCE | grep -q vda") ''; } diff --git a/tests/ax-fleet-topology/default.nix b/tests/ax-fleet-topology/default.nix index 4fddcf49e..6e9dc6db7 100644 --- a/tests/ax-fleet-topology/default.nix +++ b/tests/ax-fleet-topology/default.nix @@ -64,8 +64,11 @@ let volatile = f: f == "--token-file" + || f == "--agent-token-file" || lib.hasSuffix "/k3s-token" f + || lib.hasSuffix "/k3s-agent-token" f || lib.hasSuffix "-ax-fleet-vm-token" f + || lib.hasSuffix "-ax-fleet-vm-agent-token" f || lib.hasInfix "reserved=" f || lib.hasInfix "eviction-hard=" f; normFlags = @@ -85,10 +88,17 @@ let guardText = subst: cfg: map (lib.replaceStrings (lib.attrNames subst) (lib.attrValues subst)) ( - lib.filter (l: lib.hasInfix "ax-fleet-guard" l || lib.hasInfix "8472" l) ( - lib.splitString "\n" cfg.networking.firewall.extraCommands - ) + lib.filter ( + l: lib.hasInfix "ax-fleet-guard" l || lib.hasInfix "8472" l || lib.hasInfix "ax-fleet-pod-input" l + ) (lib.splitString "\n" cfg.networking.firewall.extraCommands) ); + pkgNames = cfg: map (p: p.pname or p.name or "") cfg.environment.systemPackages; + hasTeardown = cfg: builtins.elem "ax-fleet-teardown" (pkgNames cfg); + + # every image the real NAS seeds is the store path the VM NAS seeds (fix round 2) + seedPaths = cfg: lib.mapAttrs (_: s: s.oci.outPath) cfg.myAxFleet.registrySeed; + nasSeeds = seedPaths nas; + vmSeeds = seedPaths (testOn "nas"); in # nas: the control node assert (ax nas).enable && (ax nas).role == "control"; @@ -112,14 +122,18 @@ assert builtins.length (axLines nas.networking.firewall.extraInputRules) == 4; assert builtins.all (l: lib.hasInfix "ip saddr" l && lib.hasInfix "iifname" l) ( axLines nas.networking.firewall.extraInputRules ); -assert !(builtins.any (lib.hasInfix "tailscale0") (axLines nas.networking.firewall.extraInputRules)); +assert + !(builtins.any (lib.hasInfix "tailscale0") (axLines nas.networking.firewall.extraInputRules)); assert nas.services.dockerRegistry.listenAddress == "10.42.0.1"; assert !nas.services.dockerRegistry.openFirewall; assert lib.hasPrefix "/mnt/nas/" nas.services.dockerRegistry.storagePath; assert builtins.all (m: lib.hasPrefix "/mnt/fast/" m.what) ( - lib.filter (m: m.where == "/var/lib/rancher" || m.where == "/var/lib/kubelet" || m.where == "/var/log/pods") nas.systemd.mounts + lib.filter ( + m: m.where == "/var/lib/rancher" || m.where == "/var/lib/kubelet" || m.where == "/var/log/pods" + ) nas.systemd.mounts ); -assert builtins.length (lib.filter (m: lib.hasPrefix "/mnt/fast/k3s" m.what) nas.systemd.mounts) == 3; +assert + builtins.length (lib.filter (m: lib.hasPrefix "/mnt/fast/k3s" m.what) nas.systemd.mounts) == 3; # the shared PostgreSQL is not touched: identical settings with the switch off assert nas.services.postgresql.settings == (offCfg "nas").services.postgresql.settings; assert nas.services.postgresql.authentication == (offCfg "nas").services.postgresql.authentication; @@ -142,11 +156,15 @@ assert !(coord.networking.firewall.interfaces ? cni0); assert coord.environment.etc ? "NetworkManager/conf.d/90-ax-fleet.conf"; # NO NetworkManager restart trigger: NetworkManager.conf renders byte-identical with the switch off assert - coord.environment.etc."NetworkManager/NetworkManager.conf".source - == (offCfg "coordinator").environment.etc."NetworkManager/NetworkManager.conf".source; -assert coord.networking.networkmanager.unmanaged == (offCfg "coordinator").networking.networkmanager.unmanaged; + coord.environment.etc."NetworkManager/NetworkManager.conf".source == (offCfg "coordinator") + .environment.etc."NetworkManager/NetworkManager.conf".source; +assert + coord.networking.networkmanager.unmanaged == (offCfg "coordinator") + .networking.networkmanager.unmanaged; assert coord.boot.kernel.sysctl."net.ipv4.conf.default.proxy_arp" == 1; -assert !(coord.boot.kernel.sysctl ? "net.ipv4.conf.all.proxy_arp") || coord.boot.kernel.sysctl."net.ipv4.conf.all.proxy_arp" == null; +assert + !(coord.boot.kernel.sysctl ? "net.ipv4.conf.all.proxy_arp") + || coord.boot.kernel.sysctl."net.ipv4.conf.all.proxy_arp" == null; assert (coord.boot.kernel.sysctl."net.ipv6.conf.all.forwarding" or 0) == 0; assert coord.services.tailscale.useRoutingFeatures == "none"; assert coord.systemd.sockets.ax-server-proxy.listenStreams == [ "127.0.0.1:8080" ]; @@ -160,7 +178,50 @@ assert builtins.elem 8731 worker.networking.firewall.interfaces.enp191s0.allowed assert !(client ? myAxFleet); assert !client.services.k3s.enable; -# the kill switch +# the registry is read-only to the network; the seed writes on loopback only +assert nas.services.dockerRegistry.extraConfig.storage.maintenance.readonly.enabled; +assert !nas.services.dockerRegistry.enableDelete; +# two k3s credentials: the server token stays on the NAS +assert lib.hasSuffix "/k3s-token" nas.services.k3s.tokenFile; +assert lib.hasSuffix "/k3s-agent-token" nas.services.k3s.agentTokenFile; +assert lib.hasSuffix "/k3s-agent-token" coord.services.k3s.tokenFile; +assert !(coord.age.secrets ? k3s-token); +# pods never open a connection to the desk itself; the desk outweighs kubepods +assert lib.hasInfix + "-i cni0 -m conntrack --ctstate NEW -m comment --comment ax-fleet-pod-input -j nixos-fw-refuse" + coord.networking.firewall.extraCommands; +assert coord.systemd.slices.user.sliceConfig.CPUWeight == 10000; +assert coord.systemd.slices.system.sliceConfig.CPUWeight == 10000; +# kube-proxy leaves the host's conntrack table as it is +assert builtins.all + ( + h: + has h "--kube-proxy-arg=conntrack-tcp-timeout-established=0s" + && has h "--kube-proxy-arg=conntrack-max-per-core=0" + ) + [ + nas + coord + ]; +# the VM coordinator runs the desk's kernel +assert + coord.boot.kernelPackages.kernel.outPath == (testOn "coordinator") + .boot.kernelPackages.kernel.outPath; +# the NAS-from-boot VM runs the NAS's release line and kernel +assert + self.checks.x86_64-linux.ax-fleet-boot.nodes.nas.system.nixos.release == nas.system.nixos.release; +assert + self.checks.x86_64-linux.ax-fleet-boot.nodes.nas.boot.kernelPackages.kernel.outPath + == nas.boot.kernelPackages.kernel.outPath; +# image parity with the VM +assert builtins.all (n: vmSeeds ? ${n} && vmSeeds.${n} == nasSeeds.${n}) (lib.attrNames nasSeeds); + +# the kill switch; the teardown stays on the host's PATH with the switch off +assert builtins.all (h: hasTeardown (offCfg h)) [ + "nas" + "coordinator" +]; +assert !(hasTeardown worker); assert builtins.all killed [ "nas" "coordinator" diff --git a/tests/ax-fleet/default.nix b/tests/ax-fleet/default.nix index c0b97bd20..d91976b36 100644 --- a/tests/ax-fleet/default.nix +++ b/tests/ax-fleet/default.nix @@ -26,6 +26,8 @@ let import time from contextlib import contextmanager + # 90-rollback runs the teardown from the host's PATH, as documented; this + # store path is only compared against it. TEARDOWN = "${teardown}/bin/ax-fleet-teardown" PROBE_IMAGE = "ax-fleet-probe:test" PROBE_TARBALL = "${nodes.probeImage}" @@ -39,6 +41,11 @@ let "kernel.panic", "kernel.panic_on_oops", "vm.overcommit_memory", + # kube-proxy's conntrack keys (fix round 2): the house router's NAT + # table must keep the host's timeouts. + "net.netfilter.nf_conntrack_max", + "net.netfilter.nf_conntrack_tcp_timeout_established", + "net.netfilter.nf_conntrack_tcp_timeout_close_wait", ] from typing import Any diff --git a/tests/ax-fleet/nodes.nix b/tests/ax-fleet/nodes.nix index c1a725a95..e5d5cb50c 100644 --- a/tests/ax-fleet/nodes.nix +++ b/tests/ax-fleet/nodes.nix @@ -15,6 +15,8 @@ let # A plain file, test-only, not a secret: the fleet reads agenix instead. token = pkgs.writeText "ax-fleet-vm-token" "ax-fleet-vm-test-token-0123456789abcdef"; + # The agent credential, distinct from the server token as on the fleet. + agentToken = pkgs.writeText "ax-fleet-vm-agent-token" "ax-fleet-vm-agent-token-fedcba9876543210"; # busybox (httpd, nslookup) plus curl, imported by k3s from the images # directory on both nodes, so probe pods need no registry and no network. @@ -80,6 +82,7 @@ let inherit role; lan = lanAddr address; k3sTokenFile = "${token}"; + k3sAgentTokenFile = "${agentToken}"; guardInterfaces = [ "eth2" ]; }; environment.systemPackages = [ @@ -109,7 +112,12 @@ let }; in { - inherit probeImage token claudeProbeImage; + inherit + probeImage + token + agentToken + claudeProbeImage + ; nas = { ... }: @@ -155,6 +163,9 @@ in }; }; boot.supportedFilesystems = [ "btrfs" ]; + # The house router, as hosts/nas/router.nix declares it (fix round 2): + # the snapshot and the teardown must keep it. + boot.kernel.sysctl."net.ipv4.ip_forward" = 1; # The NAS firewall as the real one is shaped: nftables, interface-scoped # extraInputRules, filterForward off, strict rpfilter, and a reload that @@ -241,6 +252,22 @@ in "eth2" ]; zramSwap.enable = true; + # The desk's kernel, where runsc runs (fix round 2: the VM had run the + # pin's default kernel). The same expression as modules/strix.nix, so + # the same derivation; ax-fleet-topology asserts it. + boot.kernelPackages = + (import inputs.nixpkgs-fresh { + system = "x86_64-linux"; + config.allowUnfree = true; + }).linuxPackages_7_2; + # The desk's sshd as modules/common.nix renders it: port 22 open on every + # interface, passwords accepted (fix round 2). Pods must not reach it. + services.openssh = { + enable = true; + openFirewall = true; + settings.PasswordAuthentication = true; + settings.KbdInteractiveAuthentication = true; + }; services.caddy = { enable = true; diff --git a/tests/ax-fleet/phases/10-cluster.py b/tests/ax-fleet/phases/10-cluster.py index 7ff5d3d23..8e34a7f67 100644 --- a/tests/ax-fleet/phases/10-cluster.py +++ b/tests/ax-fleet/phases/10-cluster.py @@ -91,6 +91,8 @@ def apply_probe(name, role, host_port, pvc=None): nas.succeed("stat -c '%a %U %G' /etc/ax-fleet/admin.kubeconfig | grep -x '640 root wheel'") nas.succeed("grep -q 'server: https://10.42.0.1:6443' /etc/ax-fleet/admin.kubeconfig") nas.succeed("test -s /var/lib/ax-fleet/sysctl-before.conf") + # The house router's forwarding is what the snapshot says (fix round 2). + nas.succeed("grep -x 'net.ipv4.ip_forward = 1' /var/lib/ax-fleet/sysctl-before.conf") with step("nas: node Ready, untainted, control labels"): @@ -198,6 +200,24 @@ def apply_probe(name, role, host_port, pvc=None): record("sysctl_coordinator_after_restore", sysctls(coordinator)) +with step("coordinator: kubelet's tunables are put back whenever they appear, not only after a k3s start"): + # Fix round 2. kubelet applies them when its container manager starts, + # which on an agent is when the server first answers, possibly long after + # k3s.service started (MEASURED by the review). Write kubelet's values with + # no k3s restart, well after the switch: the watcher must restore them. + nrestarts = coordinator.succeed("systemctl show k3s.service -p NRestarts --value").strip() + coordinator.succeed("sysctl -w kernel.panic=10 kernel.panic_on_oops=1 vm.overcommit_memory=1") + kernel_keys_back(coordinator, base["sysctl_coordinator"]) + assert coordinator.succeed("systemctl show k3s.service -p NRestarts --value").strip() == nrestarts + coordinator.succeed("systemctl is-active ax-fleet-kernel-tunables.service") + + +with step("coordinator: the desk outweighs kubepods for CPU"): + w = {s: int(coordinator.succeed(f"cat /sys/fs/cgroup/{s}/cpu.weight").strip()) for s in ("user.slice", "system.slice", "kubepods.slice")} + record("cpu_weights", w) + assert w["user.slice"] > w["kubepods.slice"] and w["system.slice"] > w["kubepods.slice"], w + + with step("coordinator: the guards hold"): pod_ip = jsonpath("pod probe-coord", "{.status.podIP}") record("probe_coord_ip", pod_ip) @@ -231,6 +251,26 @@ def apply_probe(name, role, host_port, pvc=None): record("guard_chain", coordinator.succeed("iptables -t mangle -S ax-fleet-guard").strip().splitlines()) +with step("coordinator: pods never reach the coordinator host (sshd accepts passwords)"): + # Fix round 2. Discriminating: sshd answers the worker on the LAN, and + # port 22 is open on every interface (openFirewall), so only the pod-input + # refusal stops a pod. + worker.succeed("timeout 10 bash -c 'exec 3<>/dev/tcp/10.42.0.2/22; head -c 7 <&3' | grep -x SSH-2.0") + gw = coordinator.succeed("ip -4 -o addr show dev cni0 | awk '{print $4}' | cut -d/ -f1").strip() + fl = coordinator.succeed("ip -4 -o addr show dev flannel.1 | awk '{print $4}' | cut -d/ -f1").strip() + record("pod_input_targets", {"cni0": gw, "flannel.1": fl}) + for ip in ("10.42.0.2", gw, fl, "100.105.121.73"): + kubectl(f"exec probe-coord -- sh -c '! (nc -w 5 {ip} 22 /dev/null | grep -q SSH)'") + # The NAS's pods, over VXLAN, neither. + kubectl(f"exec probe-nas -- sh -c '! (nc -w 5 {fl} 22 /dev/null | grep -q SSH)'") + # Pod reachability of the host's own hostPort and of Services is unchanged. + code = kubectl("exec probe-coord -- curl -sk -o /dev/null -w '%{http_code}' --max-time 10 https://10.201.0.1/readyz").strip() + assert code != "000", "a coordinator pod lost the apiserver Service" + refused = coordinator.succeed("iptables -S nixos-fw | grep -c ax-fleet-pod-input").strip() + assert refused == "2", refused + coordinator.succeed("ip6tables -S nixos-fw | grep -q 'cni0.*ax-fleet-pod-input'") + + with step("coordinator: pods reach no private range on the LAN leg (the Freebox fallback case)"): # A private subnet this module does not know, on the same leg, as when # NetworkManager falls back to the Freebox profile. Discriminating: the diff --git a/tests/ax-fleet/phases/20-substrate.py b/tests/ax-fleet/phases/20-substrate.py index f95a18724..a173dc933 100644 --- a/tests/ax-fleet/phases/20-substrate.py +++ b/tests/ax-fleet/phases/20-substrate.py @@ -27,7 +27,7 @@ def sub_json(args): return json.loads(sub_k(f"{args} -o json")) -with subtest("substrate: bootstrap steps 20-50 ran"): +with step("substrate: bootstrap steps 20-50 ran"): nas.wait_for_unit("ax-fleet-bootstrap.service", timeout=3600) boot_log = nas.succeed("journalctl -b -u ax-fleet-bootstrap.service --no-pager") # Not `step`: that name is the prelude's receipt context manager, which @@ -40,7 +40,7 @@ def sub_json(args): ).strip() assert stamp == "d277088b", f"install stamp version {stamp!r}" -with subtest("substrate: every control workload Available, on the NAS"): +with step("substrate: every control workload Available, on the NAS"): nas.wait_until_succeeds( f"KUBECONFIG=/etc/rancher/k3s/k3s.yaml kubectl -n {SUB_NS} wait " "--for=condition=Available deploy --all --timeout=10s", @@ -73,7 +73,7 @@ def sub_json(args): continue assert node == "nas", f"control pod {name} landed on {node!r}" -with subtest("substrate: nas keeps substrate-version=none, coordinator carries the version"): +with step("substrate: nas keeps substrate-version=none, coordinator carries the version"): nodes = {n["metadata"]["name"]: n for n in sub_json("get nodes")["items"]} assert nodes["nas"]["metadata"]["labels"].get("ate.dev/substrate-version") == "none" assert ( @@ -81,7 +81,7 @@ def sub_json(args): == "d277088b" ) -with subtest("substrate: atelet runs on the coordinator only"): +with step("substrate: atelet runs on the coordinator only"): # The atelet DaemonSet is version-keyed by ate-setup; find it by label. ds = sub_k(f"-n {SUB_NS} get ds -l app=atelet -o jsonpath='{{.items[0].metadata.name}}'").strip() sub_k(f"-n {SUB_NS} rollout status ds/{ds} --timeout=600s", timeout=660) @@ -96,12 +96,12 @@ def sub_json(args): f"atelet on {p['spec']['nodeName']}" ) -with subtest("substrate: ClusterTrustBundles and pod certificates served"): +with step("substrate: ClusterTrustBundles and pod certificates served"): res = sub_k("get --raw /apis/certificates.k8s.io/v1beta1") assert "clustertrustbundles" in res and "podcertificaterequests" in res, res assert sub_json("get clustertrustbundles")["items"], "no ClusterTrustBundle objects" -with subtest("substrate: WorkerPool ateom-gvisor Ready 2 on the coordinator"): +with step("substrate: WorkerPool ateom-gvisor Ready 2 on the coordinator"): nas.wait_until_succeeds( "test \"$(KUBECONFIG=/etc/rancher/k3s/k3s.yaml kubectl -n ate-system get " "workerpool ateom-gvisor -o jsonpath='{.status.readyReplicas}')\" = 2", @@ -119,7 +119,7 @@ def sub_json(args): lims = [c.get("resources", {}).get("limits", {}).get("memory") for c in p["spec"]["containers"]] assert any(lims), f"worker pod has no memory limit: {lims}" -with subtest("substrate: gVisor fetched through the RustFS fallback"): +with step("substrate: gVisor fetched through the RustFS fallback"): # No internet in the VM: atelet's anonymous GCS open of gs://gvisor/... # fails, then its S3 client reads the same bucket and key from the # in-cluster RustFS (40-gvisor-asset). The prewarmer logs "Sandbox assets @@ -135,7 +135,7 @@ def sub_json(args): print("\n".join(l for l in logs.splitlines() if "gvisor" in l.lower())[-4000:]) assert "gVisor release download complete" in logs or "gvisor" in logs.lower() -with subtest("substrate: every PersistentVolume on the data pool"): +with step("substrate: every PersistentVolume on the data pool"): pvs = sub_json("get pv")["items"] assert pvs, "no PersistentVolumes" for pv in pvs: @@ -153,7 +153,7 @@ def sub_json(args): ] assert "coordinator" not in values, f"PV {pv['metadata']['name']} on the coordinator" -with subtest("substrate: every image came from the NAS registry by digest"): +with step("substrate: every image came from the NAS registry by digest"): pods = sub_json(f"-n {SUB_NS} get pods")["items"] for p in pods: for c in p["spec"].get("containers", []) + p["spec"].get("initContainers", []): diff --git a/tests/ax-fleet/phases/30-ax.py b/tests/ax-fleet/phases/30-ax.py index 9c7054cb0..06204d6c0 100644 --- a/tests/ax-fleet/phases/30-ax.py +++ b/tests/ax-fleet/phases/30-ax.py @@ -8,6 +8,9 @@ import re +smoke_receipts = [] + + def ax_smoke(case, *args, expect_pass=True): cmd = "ax-fleet-smoke " + " ".join([case, *map(str, args)]) rc, out = coordinator.execute(cmd + " 2>/dev/null") @@ -15,6 +18,9 @@ def ax_smoke(case, *args, expect_pass=True): assert lines, f"{cmd}: no receipt (rc={rc}): {out!r}" receipt = json.loads(lines[-1]) print(f"RECEIPT {cmd}: {json.dumps(receipt)}") + # Into receipt.json too (fix round 2): every Task outcome, in order. + smoke_receipts.append({"cmd": cmd, "rc": rc, "receipt": receipt}) + record("ax_smoke", smoke_receipts) if expect_pass: assert rc == 0 and receipt.get("pass") is True, f"{cmd} failed: {receipt}" return receipt @@ -36,7 +42,7 @@ def task_phase(name): return m.group(1).strip("\"'") -with subtest("ax control plane is Available on the NAS, nowhere else"): +with step("ax control plane is Available on the NAS, nowhere else"): for d in ["ax-redis", "ax-server", "ax-controller"]: kubectl(f"-n ax-system rollout status deploy/{d} --timeout=600s") node = kubectl( @@ -51,30 +57,30 @@ def task_phase(name): # No Claude credential, and no secret, in any ax object. kubectl("-n ax-system get secrets -o name | (! grep -q .)") -with subtest("T1 halogen: Completed, and still Completed after the hold"): +with step("T1 halogen: Completed, and still Completed after the hold"): t1 = ax_smoke("halogen", "--hold", 90, "--keep") assert t1["exit_code"] == 0 and t1["phase_after_hold"] == "Completed", t1 # The golden-snapshot double execution judge 2 inferred: record, do not fail. rc, count = worker.execute("curl -sf http://127.0.0.1:8731/stub/requests") print(f"RECEIPT halogen stub requests after T1: rc={rc} {count.strip()!r}") -with subtest("T2 pi: Completed with a schema-valid result read back through P1"): +with step("T2 pi: Completed with a schema-valid result read back through P1"): t2 = ax_smoke("pi", "--hold", 30) assert t2["result"]["valid"] is True, t2 assert t2["result_bytes"] and t2["result_sha256"], t2 -with subtest("T3 exit 3: Failed with ExitCode=3"): +with step("T3 exit 3: Failed with ExitCode=3"): t3 = ax_smoke("exit", 3, "--hold", 30) assert t3["phase"] == "Failed" and t3["ready"]["message"].startswith("ExitCode=3"), t3 -with subtest("T4 egress-deny: a non-allowlisted target is refused by the Gateway"): +with step("T4 egress-deny: a non-allowlisted target is refused by the Gateway"): # The coordinator's LAN address answers the worker in this run (checked # first), so a failure inside the sandbox is the Gateway's refusal. worker.succeed("curl -s -o /dev/null --max-time 10 http://10.42.0.2/") t4 = ax_smoke("egress-deny", "http://10.42.0.2/") assert t4["phase"] == "Failed", t4 -with subtest("T5 floor 4: four Tasks in a row on the 2-worker pool, none ResourceExhausted"): +with step("T5 floor 4: four Tasks in a row on the 2-worker pool, none ResourceExhausted"): t5 = ax_smoke("floor", 4) for t in t5["tasks"]: assert t["phase"] == "Completed", t @@ -85,13 +91,13 @@ def task_phase(name): t1_name = t1["task"] -with subtest("resilience: a restarted ax-controller leaves a finished Task Completed"): +with step("resilience: a restarted ax-controller leaves a finished Task Completed"): kubectl("-n ax-system delete pod -l app.kubernetes.io/name=ax-controller --wait=true") kubectl("-n ax-system rollout status deploy/ax-controller --timeout=300s") coordinator.sleep(45) # three resync periods assert task_phase(t1_name) == "Completed" -with subtest("resilience: a restarted ax-redis keeps every Task (AOF on the volume)"): +with step("resilience: a restarted ax-redis keeps every Task (AOF on the volume)"): before = coordinator.succeed(f"{AX} get tasks | tail -n +2 | wc -l").strip() kubectl("-n ax-system delete pod -l app.kubernetes.io/name=ax-redis --wait=true") kubectl("-n ax-system rollout status deploy/ax-redis --timeout=300s") @@ -100,7 +106,7 @@ def task_phase(name): assert before == after and int(after) >= 1, f"tasks before={before} after={after}" assert task_phase(t1_name) == "Completed" -with subtest("resilience: the LAN leg is down until the coordinator is NotReady; pods keep their names and uid, a new T1 completes"): +with step("resilience: the LAN leg is down until the coordinator is NotReady; pods keep their names and uid, a new T1 completes"): pods = lambda: kubectl( "-n ate-system get pods --field-selector spec.nodeName=coordinator -o name | grep ateom | sort" ) diff --git a/tests/ax-fleet/phases/35-lan-guard.py b/tests/ax-fleet/phases/35-lan-guard.py index c03dc0ecf..ef54ba24d 100644 --- a/tests/ax-fleet/phases/35-lan-guard.py +++ b/tests/ax-fleet/phases/35-lan-guard.py @@ -41,6 +41,23 @@ def svc_ip(ns, name): record("nas_range_guard", nas.succeed("nft list chain inet ax-fleet-guard prerouting").strip().splitlines()) +with step("security: the registry is read-only to the coordinator; the seed wrote everything"): + # Fix round 2. The round-2 review MEASURED 202 for an upload and 201 for a + # tag overwrite from an unprivileged coordinator user. + reg = "http://10.42.0.1:5000" + alice = "runuser -u alice -- curl -s -o /dev/null -w '%{http_code}' --max-time 10" + assert coordinator.succeed(f"{alice} {reg}/v2/_catalog").strip() == "200" + codes = { + "upload": coordinator.succeed(f"{alice} -X POST {reg}/v2/secprobe/blobs/uploads/").strip(), + "mount": coordinator.succeed(f"{alice} -X POST '{reg}/v2/secprobe/blobs/uploads/?mount=sha256:0000000000000000000000000000000000000000000000000000000000000000&from=ax/ax-redis'").strip(), + "delete": coordinator.succeed(f"{alice} -X DELETE {reg}/v2/ax/ax-redis/manifests/sha256:0000000000000000000000000000000000000000000000000000000000000000").strip(), + } + record("registry_write_codes", codes) + assert all(c not in ("201", "202") for c in codes.values()), codes + nas.fail("ss -ltn | grep -q '127.0.0.1:5001'") # the seed's writer is gone + nas.succeed("ss -ltn | grep -q '10.42.0.1:5000'") + + with step("security: ax-controller holds no Secret grant and no API token"): nas.fail("k3s kubectl get clusterrole ax-controller") nas.fail("k3s kubectl get clusterrolebinding ax-controller") diff --git a/tests/ax-fleet/phases/90-rollback.py b/tests/ax-fleet/phases/90-rollback.py index 79a1392e7..23c614dc5 100644 --- a/tests/ax-fleet/phases/90-rollback.py +++ b/tests/ax-fleet/phases/90-rollback.py @@ -9,7 +9,10 @@ with step("rollback coordinator"): coordinator.succeed(f"{BASE} >&2") coordinator.fail("systemctl is-active k3s.service") - coordinator.succeed(f"{TEARDOWN} >&2") + # As documented (fix round 2): the teardown from the rolled-back host's + # own PATH, no checkout, no injected store path. + coordinator.succeed("test -x /run/current-system/sw/bin/ax-fleet-teardown") + coordinator.succeed("ax-fleet-teardown >&2") coordinator.fail("ip link show cni0") coordinator.fail("ip link show flannel.1") coordinator.fail(LEFTOVER_RULES) @@ -33,7 +36,8 @@ nas.fail("systemctl is-active k3s.service") nas.fail("systemctl is-active docker-registry.service") nas.succeed("command -v tailscale") # the stub is reachable from a root shell - nas.succeed(f"{TEARDOWN} >&2") + nas.succeed("test -x /run/current-system/sw/bin/ax-fleet-teardown") + nas.succeed("ax-fleet-teardown >&2") # The teardown's pinned PATH keeps k3s-killall.sh away from tailscale. nas.fail("test -e /var/log/tailscale-stub.log") nas.fail("ip link show cni0") From f1bba47e0f3b638884d9dfd685f226e704297577 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 13:41:29 +0200 Subject: [PATCH 30/37] ax-fleet: fix round 3 (deny-by-default guard over every LAN leg, NAS pod egress, role-scoped guards through the kill switch, proxy ARP by interface, API owner match, proxy off :8080, bounded tracing) - harness guard: pod egress policed by destination on any interface, nothing enters cni0/flannel.1 but the NAS host; enp191s0 is a LAN leg (lan.extraInterfaces) with NetworkManager route metric 700 - control: forward chain in inet ax-fleet-guard; NAS pods reach only Halogen's host:port among private addresses (the Gateway port is ignored upstream), never the tailnet or ve-* - guards, pod-input refusal and API owner match render for the role whatever enable says; the teardown leaves declared guards - proxy_arp on cni0, flannel.1, veth* by name; conf.default untouched - ax-server-proxy on 127.0.0.1:8099 as a static user; OUTPUT owner match admits only root, apiUsers and the proxy to it and the cluster ranges - substrate patch 0004: jaeger max-traces and limits, collector memory_limiter, limits, debug basic - VM: eth3 DHCP leg, worker :2222, T4b, proxy ARP, API owner, rollback pod-to-host probes; topology asserts Co-Authored-By: Claude Opus 5.5 --- docs/ax-conwip.md | 4 +- home/ax-conwip.nix | 3 + hosts/coordinator/default.nix | 4 + modules/ax-fleet/ax-fleet-smoke.sh | 22 ++- modules/ax-fleet/ax.nix | 7 +- modules/ax-fleet/control.nix | 61 ++++-- modules/ax-fleet/harness.nix | 182 ++++++++++++++---- modules/ax-fleet/interface.nix | 37 ++++ pkgs/ax-fleet-teardown/default.nix | 27 ++- pkgs/substrate/default.nix | 8 +- .../0004-kind-otel-memory-bounds.patch | 77 ++++++++ tests/ax-fleet-topology/default.nix | 64 +++++- tests/ax-fleet/nodes.nix | 39 +++- tests/ax-fleet/phases/10-cluster.py | 49 +++++ tests/ax-fleet/phases/25-harness-probe.py | 4 +- tests/ax-fleet/phases/30-ax.py | 35 +++- tests/ax-fleet/phases/35-lan-guard.py | 12 ++ tests/ax-fleet/phases/90-rollback.py | 27 ++- 18 files changed, 578 insertions(+), 84 deletions(-) create mode 100644 pkgs/substrate/patches/0004-kind-otel-memory-bounds.patch diff --git a/docs/ax-conwip.md b/docs/ax-conwip.md index 509f54220..e4be4e416 100644 --- a/docs/ax-conwip.md +++ b/docs/ax-conwip.md @@ -116,7 +116,9 @@ Three of those defaults are choices worth defending: - **`serverUrl` defaults to loopback**, specifically the address the mock stack listens on by default (`ax-mockstack -addr 127.0.0.1:8080`). No default here points at a live host and none ever should: an accidental enable on a box with - no mock stack running reaches nothing at all. Note that this is a gRPC target + no mock stack running reaches nothing at all. On the coordinator the live + ax-server proxy listens on `myAxFleet.apiListen` (127.0.0.1:8099), never on + this default; ax-fleet-topology asserts that. Note that this is a gRPC target and not a URL. The transport is plain h2c with insecure credentials, so there is no scheme to write. - **`wipCap` defaults to 1**, which is stricter than the program's own default diff --git a/home/ax-conwip.nix b/home/ax-conwip.nix index 1b1695c4d..706e165ef 100644 --- a/home/ax-conwip.nix +++ b/home/ax-conwip.nix @@ -119,6 +119,9 @@ in that an accidental enable on a box with no mock stack running reaches nothing at all, and on a box with one running reaches only the mock. No default here points at a live host, and none ever should. + On the coordinator the live ax-server proxy listens on + myAxFleet.apiListen (127.0.0.1:8099), never on this default + (ax-fleet-topology asserts they differ). ''; }; diff --git a/hosts/coordinator/default.nix b/hosts/coordinator/default.nix index b42c8994c..f0a7f5589 100644 --- a/hosts/coordinator/default.nix +++ b/hosts/coordinator/default.nix @@ -102,6 +102,10 @@ lan = { interface = "wlp192s0"; address = "10.42.0.2"; + # The wired port: "Wired connection 1" autoconnects with DHCP (MEASURED + # nmcli, 2026-09-23). The guard covers it and it never takes the LAN + # routes from the wifi (fix round 3). + extraInterfaces = [ "enp191s0" ]; }; }; diff --git a/modules/ax-fleet/ax-fleet-smoke.sh b/modules/ax-fleet/ax-fleet-smoke.sh index 082f260fa..30a3d32ff 100644 --- a/modules/ax-fleet/ax-fleet-smoke.sh +++ b/modules/ax-fleet/ax-fleet-smoke.sh @@ -1,12 +1,12 @@ # ax-fleet-smoke: run one proof case as a real ax Task and print a JSON receipt. # (Wrapped by writeShellApplication in ./ax.nix: strict mode, pinned PATH, and -# AX_FLEET_{ATESPACE,HALOGEN,HALOGEN_CIDR,RESYNC_SECONDS} from the module.) +# AX_FLEET_{ATESPACE,HALOGEN,HALOGEN_CIDR,RESYNC_SECONDS,API} from the module.) # # ax-fleet-smoke halogen [--hold N] one chat completion against Halogen # ax-fleet-smoke pi [--hold N] pi against Halogen, schema-valid result # ax-fleet-smoke exit N [--hold N] the command exits N (Failed ExitCode=N) # ax-fleet-smoke egress-deny [URL] GET a non-allowlisted URL (default the -# coordinator's LAN address); must fail +# coordinator's LAN address); must be refused # ax-fleet-smoke floor N N Tasks in a row; none ResourceExhausted # ax-fleet-smoke probe [--hold N] `claude --version` and a GET of Halogen's # /v1/models, sandboxClass gvisor; needs an @@ -18,7 +18,7 @@ # fleet image, ax-fleet-image-ref), --sandbox-class C (spec.sandboxClass; empty # means gVisor). Exit 0 only when the case passes. The receipt carries no secret: Tasks carry none (DESIGN.md 11). -export AX_SERVER="${AX_SERVER:-http://127.0.0.1:8080}" +export AX_SERVER="${AX_SERVER:-$AX_FLEET_API}" ns="$AX_FLEET_ATESPACE" hold=0 timeout=900 @@ -42,6 +42,12 @@ case_="${args[0]}" [ "$case_" != probe ] || sandbox_class="${sandbox_class:-gvisor}" run_id="$(date +%s)-$$" +# The Gateway's `port` is carried for the record only: ax v0.3.0 copies only +# the host into a CIDR rule and Substrate's evaluator never compares ports +# (REPORTED ax client.go:457-490, egresspolicy.go:133-158; MEASURED by the +# round-3 review: worker:2222 answered through this Gateway). The port is +# enforced on the NAS instead: its pods reach ${AX_FLEET_HALOGEN} and no other +# private address (modules/ax-fleet/control.nix, podEgress). ensure_gateway() { ax -a "$ns" apply -f - >/dev/null </dev/null || true @@ -122,18 +166,25 @@ let while ${ipt} -t mangle -D FORWARD -j ax-fleet-guard 2>/dev/null; do :; done ${ipt} -t mangle -I FORWARD 1 -j ax-fleet-guard ${lib.concatMapStringsSep "\n" (r: "${ipt} -t mangle -A ax-fleet-guard ${r}") guardRules} - # flannel VXLAN from the NAS only; no TCP port is opened. - ${ipt} -A nixos-fw -i ${lan} -s ${cfg.serverAddress} -p udp --dport 8472 -j nixos-fw-accept # pods to the host: refused (podInput; nixos-fw is rebuilt on every reload) ${lib.concatMapStringsSep "\n" ( r: "${ipt} ${r}" + lib.optionalString config.networking.enableIPv6 "\n${ip6t} ${r}" ) podInput} + # the ax API and the cluster ranges from this host: owner match (fix round 3) + ${ipt} -N ax-fleet-api 2>/dev/null || true + ${ipt} -F ax-fleet-api + while ${ipt} -D OUTPUT -j ax-fleet-api 2>/dev/null; do :; done + ${ipt} -I OUTPUT 1 -j ax-fleet-api + ${lib.concatMapStringsSep "\n" (r: "${ipt} -A ax-fleet-api ${r}") apiRules} ''; guardStop = '' while ${ipt} -t mangle -D FORWARD -j ax-fleet-guard 2>/dev/null; do :; done ${ipt} -t mangle -F ax-fleet-guard 2>/dev/null || true ${ipt} -t mangle -X ax-fleet-guard 2>/dev/null || true + while ${ipt} -D OUTPUT -j ax-fleet-api 2>/dev/null; do :; done + ${ipt} -F ax-fleet-api 2>/dev/null || true + ${ipt} -X ax-fleet-api 2>/dev/null || true ''; nmDropIn = '' @@ -141,6 +192,18 @@ let # NetworkManager's. Appended (+=) to whatever NetworkManager.conf lists. [keyfile] unmanaged-devices+=interface-name:cni0;interface-name:flannel*;interface-name:veth* + '' + # (fix round 3) The other LAN-capable NICs never take the house routes from + # ${lan} while it is up: a profile that leaves ipv4.route-metric at -1 (the + # desk's "Wired connection 1", MEASURED) takes this default instead of + # ethernet's 100, which beat the wifi's 600. The NAS admits the harness to + # 6443, the registry and VXLAN by ${lan}'s address only. + + lib.optionalString (cfg.lan.extraInterfaces != [ ]) '' + + [connection-ax-fleet-extra-lan] + match-device=${lib.concatMapStringsSep ";" (i: "interface-name:${i}") cfg.lan.extraInterfaces} + ipv4.route-metric=${toString cfg.lan.extraRouteMetric} + ipv6.route-metric=${toString cfg.lan.extraRouteMetric} ''; kubeconfigScript = pkgs.writeShellApplication { @@ -165,7 +228,31 @@ let }; in { - config = lib.mkIf on { + config = lib.mkMerge [ + (lib.mkIf roleOn { + networking.firewall.extraCommands = guardStart; + networking.firewall.extraStopCommands = guardStop; + # A fixed uid for the proxy, so the owner match can name it (DynamicUser + # allocates from a range any other DynamicUser unit shares). + users.users.ax-server-proxy = { + isSystemUser = true; + group = "ax-server-proxy"; + }; + users.groups.ax-server-proxy = { }; + # The teardown leaves a guard the generation declares (see pkgs/ax-fleet-teardown). + environment.etc."ax-fleet/guard-declared".text = "harness\n"; + assertions = [ + { + assertion = builtins.all (u: config.users.users ? ${u}) cfg.apiUsers; + message = "modules/ax-fleet/harness.nix: every myAxFleet.apiUsers entry must be a declared user (iptables resolves the name when the firewall starts)."; + } + { + assertion = !(builtins.elem lan cfg.lan.extraInterfaces); + message = "modules/ax-fleet/harness.nix: myAxFleet.lan.extraInterfaces must not repeat lan.interface."; + } + ]; + }) + (lib.mkIf on { myAxFleet.kubelet = { # Memory: Tom's seats, Chrome and a coordinator Halogen feel pressure # after the sandboxes are evicted, never before. CPU is the slices below: @@ -205,14 +292,24 @@ in # k3s turns ip_forward on at start anyway; the guard chain is what keeps # it safe. IPv6 forwarding stays 0 (the wifi leg keeps accepting RAs). - boot.kernel.sysctl."net.ipv4.ip_forward" = 1; - # `default`, not `all`: only interfaces created after this applies (cni0, - # veth*, flannel.1) get proxy ARP; wlp192s0 never answers ARP for - # addresses it routes elsewhere. `all` is Tom's call (DESIGN Unknowns 11). - boot.kernel.sysctl."net.ipv4.conf.default.proxy_arp" = 1; + # flannel VXLAN from the NAS only; no TCP port is opened. With the switch + # on only: an accept, unlike the guards above. + networking.firewall.extraCommands = '' + ${ipt} -A nixos-fw -i ${lan} -s ${cfg.serverAddress} -p udp --dport 8472 -j nixos-fw-accept + ''; - networking.firewall.extraCommands = guardStart; - networking.firewall.extraStopCommands = guardStop; + boot.kernel.sysctl."net.ipv4.ip_forward" = 1; + # Proxy ARP on the pod interfaces only (fix round 3). Round 2 set + # `conf.default`, but every NIC created after systemd-sysctl copies + # `default`, and the desk's NICs are renamed after it runs (MEASURED boot + # journal: sysctl 6.356 s, enp191s0 6.497 s, wlp192s0 7.493 s), so + # wlp192s0 would have answered ARP for the whole LAN after a reboot. + # systemd's udev rule (99-systemd.rules) runs systemd-sysctl for each new + # interface, which applies these by name and glob as each one appears. + # `flannel/1` is sysctl.d's spelling of flannel.1. + boot.kernel.sysctl."net.ipv4.conf.cni0.proxy_arp" = 1; + boot.kernel.sysctl."net.ipv4.conf.flannel/1.proxy_arp" = 1; + boot.kernel.sysctl."net.ipv4.conf.veth*.proxy_arp" = 1; # ── NetworkManager: a drop-in and a config reload, never a restart ── environment.etc."NetworkManager/conf.d/90-ax-fleet.conf".text = nmDropIn; @@ -231,25 +328,32 @@ in }; systemd.services.k3s.wants = [ "ax-fleet-nm-unmanaged.service" ]; - # ── ax-server on 127.0.0.1:8080, through kube-proxy's OUTPUT rules ── + # ── ax-server on ${cfg.apiListen}, through kube-proxy's OUTPUT rules ── # The ClusterIP is never a NodePort: ax's API has no authentication - # (upstream #376). Loopback only. + # (upstream #376). Loopback only, and only root, apiUsers and this proxy + # may connect (apiRules). Not 127.0.0.1:8080 (fix round 3): that is + # ax-conwip's default and ax-mockstack's, which must reach nothing live. systemd.sockets.ax-server-proxy = { - description = "ax-fleet: ax-server on 127.0.0.1:8080"; + description = "ax-fleet: ax-server on ${cfg.apiListen}"; wantedBy = [ "sockets.target" ]; - listenStreams = [ "127.0.0.1:8080" ]; + listenStreams = [ cfg.apiListen ]; }; systemd.services.ax-server-proxy = { - description = "ax-fleet: proxy 127.0.0.1:8080 to the ax-server ClusterIP"; + description = "ax-fleet: proxy ${cfg.apiListen} to the ax-server ClusterIP"; requires = [ "ax-server-proxy.socket" ]; after = [ "ax-server-proxy.socket" ]; serviceConfig = { ExecStart = "${config.systemd.package}/lib/systemd/systemd-socket-proxyd ${cfg.axServerClusterIP}:8080"; - DynamicUser = true; + User = "ax-server-proxy"; + Group = "ax-server-proxy"; PrivateTmp = true; + NoNewPrivileges = true; + ProtectSystem = "strict"; + ProtectHome = true; }; }; environment.systemPackages = [ kubeconfigScript ]; - }; + }) + ]; } diff --git a/modules/ax-fleet/interface.nix b/modules/ax-fleet/interface.nix index 937982601..fb2f1add9 100644 --- a/modules/ax-fleet/interface.nix +++ b/modules/ax-fleet/interface.nix @@ -52,6 +52,43 @@ in type = types.str; default = "10.42.0.0/24"; }; + extraInterfaces = mkOption { + type = types.listOf types.str; + default = [ ]; + example = [ "enp191s0" ]; + description = '' + Other NICs that can reach the house LAN (fix round 3): the + coordinator's wired port enp191s0 has an autoconnecting DHCP profile. + On the harness role the guard chain treats them as LAN legs, and a + NetworkManager drop-in gives them `extraRouteMetric`, so the LAN + routes stay on `interface` while it is up. Tests use [ "eth3" ]. + ''; + }; + extraRouteMetric = mkOption { + type = types.int; + default = 700; + description = "Route metric for `extraInterfaces` (NetworkManager's wifi default is 600, ethernet 100)."; + }; + }; + + apiListen = mkOption { + type = types.str; + default = "127.0.0.1:8099"; + description = '' + Loopback address of ax-server-proxy.socket on the harness (fix round 3: + not 127.0.0.1:8080, which ax-conwip's default and ax-mockstack own). + AX_SERVER points here. + ''; + }; + apiUsers = mkOption { + type = types.listOf types.str; + default = [ "tom" ]; + description = '' + Local users (besides root) that may open connections to `apiListen` + and to the cluster ranges from the harness host (fix round 3). The ax + API has no authentication (upstream #376); everyone else is refused by + an owner match in OUTPUT. + ''; }; guardInterfaces = mkOption { diff --git a/pkgs/ax-fleet-teardown/default.nix b/pkgs/ax-fleet-teardown/default.nix index 134f739ef..da6380f7e 100644 --- a/pkgs/ax-fleet-teardown/default.nix +++ b/pkgs/ax-fleet-teardown/default.nix @@ -14,8 +14,8 @@ # The k3s unit runs with KillMode=process, so pods and containerd shims outlive # it; switching `myAxFleet.enable = false` alone leaves them, plus cni0, # flannel.1, the KUBE-/FLANNEL-/CNI- rules and ip_forward=1. This wraps the -# pinned package's own k3s-killall.sh, removes the coordinator guard chain -# (the firewall reload of the disabled generation does not know it), then +# pinned package's own k3s-killall.sh, removes the guard chains only when the +# running generation does not declare them (a pre-ax generation), then # restores the sysctls recorded before k3s first ran on this host # (/var/lib/ax-fleet/sysctl-before.conf, written by the activation snippet in # modules/ax-fleet/k3s.nix). @@ -62,13 +62,24 @@ writeShellApplication { echo "== k3s-killall.sh (${k3s.version})" ${k3s}/bin/k3s-killall.sh || echo "k3s-killall.sh exited $?; continuing" >&2 - echo "== guard chain" - while iptables -w -t mangle -D FORWARD -j ax-fleet-guard 2>/dev/null; do :; done - iptables -w -t mangle -F ax-fleet-guard 2>/dev/null || true - iptables -w -t mangle -X ax-fleet-guard 2>/dev/null || true + # Fix round 3: a harness or control generation declares the guards + # whatever `enable` says (the kill switch leaves pods running until this + # script), so they stay; a generation from before ax never had them, and + # then they go. + if [ -e /etc/ax-fleet/guard-declared ]; then + echo "== guards: declared by this generation ($(cat /etc/ax-fleet/guard-declared)); left in place" + else + echo "== guard chains" + while iptables -w -t mangle -D FORWARD -j ax-fleet-guard 2>/dev/null; do :; done + iptables -w -t mangle -F ax-fleet-guard 2>/dev/null || true + iptables -w -t mangle -X ax-fleet-guard 2>/dev/null || true + while iptables -w -D OUTPUT -j ax-fleet-api 2>/dev/null; do :; done + iptables -w -F ax-fleet-api 2>/dev/null || true + iptables -w -X ax-fleet-api 2>/dev/null || true - echo "== NAS cluster-range guard table" - nft delete table inet ax-fleet-guard 2>/dev/null || true + echo "== NAS guard table" + nft delete table inet ax-fleet-guard 2>/dev/null || true + fi snap=/var/lib/ax-fleet/sysctl-before.conf if [ -s "$snap" ]; then diff --git a/pkgs/substrate/default.nix b/pkgs/substrate/default.nix index 63bd158af..2c186d350 100644 --- a/pkgs/substrate/default.nix +++ b/pkgs/substrate/default.nix @@ -21,7 +21,7 @@ # kagent-dev fork's ghcr images are ruled out), so every component image is # built here and seeded into the NAS registry by pkgs/substrate/images.nix. # -# The three patches touch only manifests/, never Go code: +# The four patches touch only manifests/, never Go code: # 0001 pauseImage -> localhost:5000/pause (atelet pulls it itself; the fleet # must not depend on registry.k8s.io at sandbox start). # 0002 third-party images -> their linux/amd64 child digests, so the NAS @@ -29,6 +29,11 @@ # 0003 the kind overlay's literal RustFS credential (public in the upstream # repo) -> secretKeyRef to Secret ate-system/ax-fleet-rustfs, which the # bootstrap step 25-rustfs-secret generates once on the NAS. +# 0004 the kind overlay's tracing stack, bounded (fix round 3): jaeger +# keeps at most 10000 traces in memory, the collector gets a +# memory_limiter and debug at basic, and both get limits.memory +# 512Mi. Upstream leaves them unbounded (MEASURED at d277088b), and +# they run on the NAS next to Immich, Paperless and NFS. # ate-setup reads the manifests from its working directory's repository root # (it walks up to go.mod). What it reads for `deploy ate-system` is go.mod, # manifests/ and hack/ (kustomize overlays, CSI manifests; MEASURED grep of @@ -56,6 +61,7 @@ let ./patches/0001-sandboxconfig-pause-localhost.patch ./patches/0002-images-linux-amd64-digests.patch ./patches/0003-kind-rustfs-credential-secret.patch + ./patches/0004-kind-otel-memory-bounds.patch ]; }; diff --git a/pkgs/substrate/patches/0004-kind-otel-memory-bounds.patch b/pkgs/substrate/patches/0004-kind-otel-memory-bounds.patch new file mode 100644 index 000000000..c097991d9 --- /dev/null +++ b/pkgs/substrate/patches/0004-kind-otel-memory-bounds.patch @@ -0,0 +1,77 @@ +--- a/manifests/ate-install/kind/otel-collector.yaml 2026-09-23 13:06:34.261419536 +0200 ++++ b/manifests/ate-install/kind/otel-collector.yaml 2026-09-23 13:06:45.024736968 +0200 +@@ -32,6 +32,13 @@ + http: + endpoint: 0.0.0.0:4318 + processors: ++ # mecattaf fleet: the collector runs on the NAS (22 GB RAM, shared with ++ # Immich, Paperless and NFS). memory_limiter refuses data before the ++ # container limit below is reached, instead of being OOM-killed. ++ memory_limiter: ++ check_interval: 1s ++ limit_mib: 400 ++ spike_limit_mib: 100 + batch: + + # The count connector below sends deltas, and the Prometheus exporter +@@ -95,17 +102,19 @@ + insecure: true + prometheus: + endpoint: 0.0.0.0:8889 ++ # mecattaf fleet: basic, not detailed; detailed writes every span and ++ # log record to the pod log. + debug: +- verbosity: detailed ++ verbosity: basic + service: + pipelines: + traces: + receivers: [otlp] +- processors: [batch] ++ processors: [memory_limiter, batch] + exporters: [otlp/jaeger, debug] + metrics: + receivers: [otlp] +- processors: [batch] ++ processors: [memory_limiter, batch] + exporters: [prometheus, debug] + + # Every ate event, plus any workload that sends OTLP logs. debug is the +@@ -115,7 +124,7 @@ + # record reports a stale state with no sign that anything is missing. + logs: + receivers: [otlp] +- processors: [batch] ++ processors: [memory_limiter, batch] + exporters: [debug] + + # The count pipelines read the same receiver as the pipelines above, +@@ -180,6 +189,12 @@ + image: otel/opentelemetry-collector-contrib:0.157.0@sha256:4eb842091c796156d4d3c994eb22ba793590f5723719dbf6b8436cb4dfc17f48 + args: + - --config=/conf/otel-collector-config.yaml ++ # mecattaf fleet: bounded (see memory_limiter above). ++ resources: ++ requests: ++ memory: 128Mi ++ limits: ++ memory: 512Mi + volumeMounts: + - name: config + mountPath: /conf +@@ -243,6 +258,15 @@ + containers: + - name: jaeger + image: jaegertracing/all-in-one:1.55@sha256:d5bbf80eb37e3a0d1b1644f17d1c3a7b88abd74177ec06198d4a289b58b41798 ++ # mecattaf fleet: in-memory storage keeps every trace with no bound ++ # by default. Cap the trace count and the container. ++ args: ++ - --memory.max-traces=10000 ++ resources: ++ requests: ++ memory: 128Mi ++ limits: ++ memory: 512Mi + ports: + - name: otlp-grpc + containerPort: 4317 diff --git a/tests/ax-fleet-topology/default.nix b/tests/ax-fleet-topology/default.nix index 6e9dc6db7..f64ad81d0 100644 --- a/tests/ax-fleet-topology/default.nix +++ b/tests/ax-fleet-topology/default.nix @@ -53,8 +53,25 @@ let && !(c.environment.etc ? "NetworkManager/conf.d/90-ax-fleet.conf") && !(c.environment.etc ? "rancher/k3s/registries.yaml") && !(c.system.activationScripts ? ax-fleet-sysctl-snapshot) - && !(lib.hasInfix "ax-fleet-guard" c.networking.firewall.extraCommands) - && !(lib.hasInfix "ax-fleet:" (c.networking.firewall.extraInputRules or "")); + && !(lib.hasInfix "ax-fleet:" (c.networking.firewall.extraInputRules or "")) + && !(c.boot.kernel.sysctl ? "net.ipv4.conf.veth*.proxy_arp"); + + # ...except the guards, which stay for the role whatever `enable` says (fix + # round 3): the kill switch leaves pods running until the teardown. + guardsKept = + host: + let + c = offCfg host; + in + if host == "coordinator" then + lib.hasInfix "ax-fleet-guard" c.networking.firewall.extraCommands + && lib.hasInfix "ax-fleet-pod-input" c.networking.firewall.extraCommands + && lib.hasInfix "ax-fleet-api" c.networking.firewall.extraCommands + else if host == "nas" then + lib.hasInfix "hook forward" c.networking.nftables.tables.ax-fleet-guard.content + else + !(lib.hasInfix "ax-fleet-guard" c.networking.firewall.extraCommands) + && !(c.networking.nftables.tables ? ax-fleet-guard); # ── parity with the VM test: flags and firewall text, interfaces substituted ── testNodes = self.checks.x86_64-linux.ax-fleet.nodes; @@ -84,12 +101,17 @@ let coordSubst = { "wlp192s0" = "eth1"; "tailscale0" = "eth2"; + "enp191s0" = "eth3"; }; guardText = subst: cfg: map (lib.replaceStrings (lib.attrNames subst) (lib.attrValues subst)) ( lib.filter ( - l: lib.hasInfix "ax-fleet-guard" l || lib.hasInfix "8472" l || lib.hasInfix "ax-fleet-pod-input" l + l: + lib.hasInfix "ax-fleet-guard" l + || lib.hasInfix "8472" l + || lib.hasInfix "ax-fleet-pod-input" l + || lib.hasInfix "ax-fleet-api" l ) (lib.splitString "\n" cfg.networking.firewall.extraCommands) ); pkgNames = cfg: map (p: p.pname or p.name or "") cfg.environment.systemPackages; @@ -161,13 +183,29 @@ assert assert coord.networking.networkmanager.unmanaged == (offCfg "coordinator") .networking.networkmanager.unmanaged; -assert coord.boot.kernel.sysctl."net.ipv4.conf.default.proxy_arp" == 1; -assert - !(coord.boot.kernel.sysctl ? "net.ipv4.conf.all.proxy_arp") - || coord.boot.kernel.sysctl."net.ipv4.conf.all.proxy_arp" == null; +# proxy ARP only on the pod interfaces, never through all/default (fix round 3) +assert coord.boot.kernel.sysctl."net.ipv4.conf.veth*.proxy_arp" == 1; +assert coord.boot.kernel.sysctl."net.ipv4.conf.cni0.proxy_arp" == 1; +assert coord.boot.kernel.sysctl."net.ipv4.conf.flannel/1.proxy_arp" == 1; +assert builtins.all ( + k: !(coord.boot.kernel.sysctl ? ${k}) || coord.boot.kernel.sysctl.${k} == null +) [ "net.ipv4.conf.all.proxy_arp" "net.ipv4.conf.default.proxy_arp" ]; +# the wired port is a guarded LAN leg with a route metric above the wifi's +assert (ax coord).lan.extraInterfaces == [ "enp191s0" ]; +assert lib.hasInfix "-i cni0 -o enp191s0 -j RETURN" coord.networking.firewall.extraCommands; +assert lib.hasInfix "-i cni0 -d 10.0.0.0/8 -j DROP" coord.networking.firewall.extraCommands; +assert lib.hasInfix "-o cni0 -j DROP" coord.networking.firewall.extraCommands; +assert lib.hasInfix "match-device=interface-name:enp191s0" coord.environment.etc."NetworkManager/conf.d/90-ax-fleet.conf".text; +assert lib.hasInfix "ipv4.route-metric=700" coord.environment.etc."NetworkManager/conf.d/90-ax-fleet.conf".text; +# the ax API: only root, tom and the proxy (fix round 3) +assert lib.hasInfix "--uid-owner tom -j RETURN" coord.networking.firewall.extraCommands; +assert coord.systemd.services.ax-server-proxy.serviceConfig.User == "ax-server-proxy"; assert (coord.boot.kernel.sysctl."net.ipv6.conf.all.forwarding" or 0) == 0; assert coord.services.tailscale.useRoutingFeatures == "none"; -assert coord.systemd.sockets.ax-server-proxy.listenStreams == [ "127.0.0.1:8080" ]; +assert coord.systemd.sockets.ax-server-proxy.listenStreams == [ "127.0.0.1:8099" ]; +assert coord.environment.sessionVariables.AX_SERVER == "http://127.0.0.1:8099"; +# never the address ax-conwip's default (and the mock stack) points at +assert !(builtins.elem coord.home-manager.users.tom.myAxConwip.serverUrl coord.systemd.sockets.ax-server-proxy.listenStreams); assert coord.myAxClient.enable; # worker: inference, nothing at runtime @@ -227,6 +265,16 @@ assert builtins.all killed [ "coordinator" "worker" ]; +assert builtins.all guardsKept [ + "nas" + "coordinator" + "worker" +]; +# the NAS's pods: Halogen on its port, no other private address (fix round 3) +assert lib.hasInfix "ip daddr 10.42.0.5 tcp dport 8731 return" nas.networking.nftables.tables.ax-fleet-guard.content; +assert + lib.replaceStrings [ "tailscale0" ] [ "eth2" ] nas.networking.nftables.tables.ax-fleet-guard.content + == (testOn "nas").networking.nftables.tables.ax-fleet-guard.content; # parity with the VM test assert normFlags nasSubst nas == normFlags { } (testOn "nas"); diff --git a/tests/ax-fleet/nodes.nix b/tests/ax-fleet/nodes.nix index e5d5cb50c..9feaa3892 100644 --- a/tests/ax-fleet/nodes.nix +++ b/tests/ax-fleet/nodes.nix @@ -226,10 +226,16 @@ in (setAddr "eth2" "100.105.121.73" 10) ]; networking.hostName = "coordinator"; + # The desk's wired port (fix round 3): eth3, NetworkManager-managed, + # DHCP from the worker's second leg, as enp191s0's "Wired connection 1". + myAxFleet.lan.extraInterfaces = [ "eth3" ]; + # myAxFleet.apiUsers defaults to [ "tom" ]; alice is the other local user. + users.users.tom.isNormalUser = true; virtualisation = { vlans = [ 1 2 + 3 ]; memorySize = 8192; cores = 4; @@ -298,15 +304,44 @@ in address = "10.42.0.5"; }) (setAddr "eth1" "10.42.0.5" 24) + (setAddr "eth2" "192.168.43.5" 24) ]; networking.hostName = "worker"; - virtualisation.vlans = [ 1 ]; + virtualisation.vlans = [ + 1 + 3 + ]; + # vlan 3: a second LAN segment for the coordinator's wired leg (fix + # round 3). DHCP with no router option, so no default route moves. + services.dnsmasq = { + enable = true; + resolveLocalQueries = false; + settings = { + port = 0; + interface = [ "eth2" ]; + bind-interfaces = true; + dhcp-range = [ "192.168.43.100,192.168.43.150,1h" ]; + dhcp-option = [ "3" ]; + }; + }; + # Another worker port (fix round 3): the real worker opens 22 with + # passwords on every interface; pods and the egress gateway must reach + # 8731 and nothing else. + systemd.services.worker-2222 = { + wantedBy = [ "multi-user.target" ]; + serviceConfig.ExecStart = "${pkgs.busybox}/bin/httpd -f -p 2222 -h ${pkgs.writeTextDir "index.html" "worker-port-2222-reached\n"}"; + }; # The worker is not switched in the motion; its fleet role is ON from # boot and renders only the 8731 assertion, as on the real host. myAxFleet.enable = true; networking.firewall = { enable = true; - interfaces.eth1.allowedTCPPorts = [ 8731 ]; + interfaces.eth1.allowedTCPPorts = [ + 8731 + 2222 + ]; + interfaces.eth2.allowedTCPPorts = [ 8731 ]; + interfaces.eth2.allowedUDPPorts = [ 67 ]; }; systemd.services.halogen-stub = { wantedBy = [ "multi-user.target" ]; diff --git a/tests/ax-fleet/phases/10-cluster.py b/tests/ax-fleet/phases/10-cluster.py index 8e34a7f67..ae9bed9cb 100644 --- a/tests/ax-fleet/phases/10-cluster.py +++ b/tests/ax-fleet/phases/10-cluster.py @@ -285,6 +285,55 @@ def apply_probe(name, role, host_port, pvc=None): worker.succeed("ip addr del 192.168.77.5/24 dev eth1") +with step("coordinator: a second LAN leg (the desk's wired port) is guarded and never takes the routes"): + # Fix round 3. eth3 stands in for enp191s0: NetworkManager-managed, DHCP + # from the worker's second leg (vlan 3). Round 2's guard named only eth1, + # so pod egress and inbound hostPorts over eth3 matched no DROP. + rc, _ = coordinator.execute("timeout 120 sh -c 'until ip -4 -o addr show dev eth3 | grep -q \"inet 192.168.43.\"; do sleep 2; done'") + if rc != 0: + record("eth3_profile_added", True) + coordinator.succeed("nmcli con add type ethernet ifname eth3 con-name wired-eth3 ipv4.method auto ipv6.method ignore") + record("eth3_routes_before_reactivation", coordinator.succeed("ip -4 route show dev eth3").strip().splitlines()) + # The drop-in's metric applies at the next activation, as on the desk, + # where the wired port is down today. + coordinator.succeed("nmcli device disconnect eth3 && nmcli device connect eth3") + coordinator.wait_until_succeeds("ip -4 -o addr show dev eth3 | grep -q 'inet 192.168.43.'", timeout=120) + routes = coordinator.succeed("ip -4 route show dev eth3").strip().splitlines() + record("eth3_routes", routes) + assert routes and all("metric 700" in r for r in routes), routes + addr3 = coordinator.succeed("ip -4 -o addr show dev eth3 | awk '{print $4}' | cut -d/ -f1").strip() + pod_ip = jsonpath("pod probe-coord", "{.status.podIP}") + # Out: the host reaches the worker's second leg; a pod does not. + coordinator.succeed("curl -sf --max-time 10 http://192.168.43.5:8731/health") + kubectl("exec probe-coord -- sh -c '! curl -s --max-time 5 -o /dev/null http://192.168.43.5:8731/health'") + # In: Caddy answers on the second leg; the pod hostPort and a routed + # path into the pod network do not. + worker.succeed(f"curl -sf --max-time 10 http://{addr3}/ | grep -x caddy-ok") + worker.fail(f"curl -s --max-time 5 -o /dev/null http://{addr3}:18085/") + worker.succeed(f"ip route replace 10.200.0.0/16 via {addr3}") + try: + worker.fail(f"curl -s --max-time 5 -o /dev/null http://{pod_ip}:8000/") + finally: + worker.succeed(f"ip route del 10.200.0.0/16 via {addr3}") + + +with step("coordinator: proxy ARP on the pod interfaces only, never on a LAN leg or a NIC that appears later"): + # Fix round 3. Round 2 set conf.default, which every later NIC copies + # (the desk's NICs appear after systemd-sysctl at boot). + veth = coordinator.succeed("ip -o link show type veth | awk -F': ' '{print $2}' | cut -d@ -f1 | head -1").strip() + coordinator.succeed("ip link add axprobe0 type dummy && udevadm settle") + try: + parp = { + i: coordinator.succeed(f"cat /proc/sys/net/ipv4/conf/{i}/proxy_arp").strip() + for i in ("default", "all", "eth1", "eth2", "eth3", "axprobe0", "cni0", "flannel.1", veth) + } + finally: + coordinator.succeed("ip link del axprobe0") + record("proxy_arp", parp) + assert parp["cni0"] == "1" and parp["flannel.1"] == "1" and parp[veth] == "1", parp + assert all(parp[i] == "0" for i in ("default", "all", "eth1", "eth2", "eth3", "axprobe0")), parp + + with step("coordinator: LAN traffic routed through the NAS never reaches a harness pod"): # The house default gateway is the NAS. Before fix round 1 this path # (worker -> nas -> flannel.1 -> harness pod) answered (MEASURED). diff --git a/tests/ax-fleet/phases/25-harness-probe.py b/tests/ax-fleet/phases/25-harness-probe.py index 1659e95d6..d94cbd47b 100644 --- a/tests/ax-fleet/phases/25-harness-probe.py +++ b/tests/ax-fleet/phases/25-harness-probe.py @@ -18,7 +18,7 @@ PROBE_HOLD = 60 PROBE_EVERY = 5 -PROBE_AX = "AX_SERVER=http://127.0.0.1:8080 ax -a fleet" +PROBE_AX = "AX_SERVER=http://127.0.0.1:8099 ax -a fleet" PROBE_STUB_LOG = "/var/lib/halogen-stub/requests.jsonl" @@ -45,7 +45,7 @@ def probe_stub_lines(): ) for d in ("ax-redis", "ax-server", "ax-controller"): kubectl(f"-n ax-system rollout status deploy/{d} --timeout=600s") - coordinator.wait_until_succeeds("curl -sf http://127.0.0.1:8080/healthz", timeout=300) + coordinator.wait_until_succeeds("curl -sf http://127.0.0.1:8099/healthz", timeout=300) record("probe_nodes", kubectl("get nodes -o wide").strip().splitlines()) record("probe_ax_pods", kubectl("-n ax-system get pods -o wide").strip().splitlines()) workers_wide = kubectl("-n ate-system get pods -l ax.mecattaf.dev/pool=ateom-gvisor -o wide") diff --git a/tests/ax-fleet/phases/30-ax.py b/tests/ax-fleet/phases/30-ax.py index 06204d6c0..8a5d73a1e 100644 --- a/tests/ax-fleet/phases/30-ax.py +++ b/tests/ax-fleet/phases/30-ax.py @@ -30,7 +30,7 @@ def kubectl(args): return nas.succeed(f"k3s kubectl {args}") -AX = "AX_SERVER=http://127.0.0.1:8080 ax -a fleet" +AX = "AX_SERVER=http://127.0.0.1:8099 ax -a fleet" def task_phase(name): @@ -53,7 +53,7 @@ def task_phase(name): "get pv -o jsonpath='{range .items[?(@.spec.claimRef.name==\"ax-redis-data\")]}{.spec.local.path}{.spec.hostPath.path}{end}'" ) assert pv.startswith("/mnt/nas/services/ax-fleet/local-path"), f"ax-redis volume at {pv!r}" - coordinator.wait_until_succeeds("curl -sf http://127.0.0.1:8080/healthz", timeout=300) + coordinator.wait_until_succeeds("curl -sf http://127.0.0.1:8099/healthz", timeout=300) # No Claude credential, and no secret, in any ax object. kubectl("-n ax-system get secrets -o name | (! grep -q .)") @@ -80,6 +80,37 @@ def task_phase(name): t4 = ax_smoke("egress-deny", "http://10.42.0.2/") assert t4["phase"] == "Failed", t4 + +with step("T4b egress-deny: the Gateway's host on another port is refused (the NAS enforces the port)"): + # Fix round 3. The Gateway names 10.42.0.5/32 port 8731, but ax drops the + # port and Substrate never compares one: the round-3 review MEASURED + # curl_rc 0 from worker:2222 through the egress gateway. The NAS's pod + # egress chain (control.nix) is what refuses it now. Discriminating: the + # NAS host reaches that port. + nas.succeed("curl -sf --max-time 10 http://10.42.0.5:2222/ | grep -q worker-port-2222-reached") + t4b = ax_smoke("egress-deny", "http://10.42.0.5:2222/") + # The egress gateway answers 503 itself (its upstream connect is dropped + # on the NAS); before the fix the target answered 200 (MEASURED by the + # round-3 review). + assert t4b["refused_by"] in ("gateway-connection", "egress-upstream"), t4b + assert t4b["result"]["http_code"] != "200", t4b + +with step("security: only root and apiUsers reach the ax API and the cluster ranges from the desk"): + # Fix round 3. The round-3 review MEASURED alice applying a Gateway with + # host 0.0.0.0/0 through the loopback proxy. The owner match also covers + # the ClusterIP and the pod, which kube-proxy's OUTPUT DNAT would carry. + ax_pod = kubectl( + "-n ax-system get pods -l app.kubernetes.io/name=ax-server -o jsonpath='{.items[0].status.podIP}'" + ).strip() + coordinator.succeed("runuser -u tom -- curl -sf --max-time 10 http://127.0.0.1:8099/healthz") + coordinator.succeed("curl -sf --max-time 10 http://10.201.0.80:8080/healthz") + for url in ("http://127.0.0.1:8099/healthz", "http://10.201.0.80:8080/healthz", f"http://{ax_pod}:8080/healthz"): + coordinator.fail(f"runuser -u alice -- curl -s --max-time 5 -o /dev/null {url}") + # 127.0.0.1:8080 is left to ax-conwip's default and the mock stack. + coordinator.fail("ss -ltn | grep -q '127.0.0.1:8080 '") + record("api_owner_chain", coordinator.succeed("iptables -S ax-fleet-api").strip().splitlines()) + + with step("T5 floor 4: four Tasks in a row on the 2-worker pool, none ResourceExhausted"): t5 = ax_smoke("floor", 4) for t in t5["tasks"]: diff --git a/tests/ax-fleet/phases/35-lan-guard.py b/tests/ax-fleet/phases/35-lan-guard.py index ef54ba24d..927bf7b36 100644 --- a/tests/ax-fleet/phases/35-lan-guard.py +++ b/tests/ax-fleet/phases/35-lan-guard.py @@ -41,6 +41,18 @@ def svc_ip(ns, name): record("nas_range_guard", nas.succeed("nft list chain inet ax-fleet-guard prerouting").strip().splitlines()) +with step("nas: pods reach Halogen on its port and no other private address"): + # Fix round 3. The round-3 review MEASURED postgres-0 reaching worker:2222; + # the NAS had no pod egress guard. Discriminating: the NAS host reaches + # both targets. + nas.succeed("curl -sf --max-time 10 http://10.42.0.5:2222/ | grep -q worker-port-2222-reached") + nas.succeed("curl -sf --max-time 10 http://10.42.0.2/ | grep -x caddy-ok") + kubectl("exec probe-nas -- curl -sf --max-time 10 http://10.42.0.5:8731/health") + kubectl("exec probe-nas -- sh -c '! curl -s --max-time 5 -o /dev/null http://10.42.0.5:2222/'") + kubectl("exec probe-nas -- sh -c '! curl -s --max-time 5 -o /dev/null http://10.42.0.2/'") + record("nas_pod_egress", nas.succeed("nft list chain inet ax-fleet-guard forward").strip().splitlines()) + + with step("security: the registry is read-only to the coordinator; the seed wrote everything"): # Fix round 2. The round-2 review MEASURED 202 for an upload and 201 for a # tag overwrite from an unprivileged coordinator user. diff --git a/tests/ax-fleet/phases/90-rollback.py b/tests/ax-fleet/phases/90-rollback.py index 23c614dc5..a11575735 100644 --- a/tests/ax-fleet/phases/90-rollback.py +++ b/tests/ax-fleet/phases/90-rollback.py @@ -6,9 +6,27 @@ LEFTOVER_RULES = "iptables-save 2>/dev/null | grep -E 'KUBE-|FLANNEL|CNI-'" +def pod_netns_pid(machine): + # A pod sandbox's pause process in a network namespace other than the host's. + return machine.succeed( + "host=$(readlink /proc/1/ns/net); for p in $(pgrep -x pause); do " + "[ \"$(readlink /proc/$p/ns/net)\" != \"$host\" ] && { echo $p; break; }; done" + ).strip() + + with step("rollback coordinator"): coordinator.succeed(f"{BASE} >&2") coordinator.fail("systemctl is-active k3s.service") + # Fix round 3: between the kill switch and the teardown the pods still + # run (KillMode=process). The role-scoped guards keep them off the host. + pid = pod_netns_pid(coordinator) + assert pid, "no pod network namespace survived the kill switch" + coordinator.succeed("timeout 10 bash -c 'exec 3<>/dev/tcp/10.42.0.2/22'") # sshd is up + coordinator.fail(f"nsenter -t {pid} -n timeout 5 bash -c 'exec 3<>/dev/tcp/10.42.0.2/22'") + coordinator.fail(f"nsenter -t {pid} -n timeout 5 bash -c 'exec 3<>/dev/tcp/100.105.121.73/22'") + coordinator.fail(f"nsenter -t {pid} -n timeout 5 bash -c 'exec 3<>/dev/tcp/10.42.0.5/8731'") + assert coordinator.succeed("iptables -S nixos-fw | grep -c ax-fleet-pod-input").strip() == "2" + coordinator.succeed("iptables -t mangle -S FORWARD 1 | grep -q ax-fleet-guard") # As documented (fix round 2): the teardown from the rolled-back host's # own PATH, no checkout, no injected store path. coordinator.succeed("test -x /run/current-system/sw/bin/ax-fleet-teardown") @@ -17,7 +35,10 @@ coordinator.fail("ip link show flannel.1") coordinator.fail(LEFTOVER_RULES) coordinator.fail("pgrep -f containerd-shim") - coordinator.fail("iptables -t mangle -S ax-fleet-guard") + # The guards belong to the harness role's every generation (fix round 3); + # the teardown leaves them, inert without cni0 and flannel.1. + coordinator.succeed("iptables -t mangle -S ax-fleet-guard | grep -q DROP") + coordinator.succeed("iptables -S OUTPUT 1 | grep -q ax-fleet-api") after = sysctls(coordinator) record("sysctl_coordinator_after_rollback", after) assert after == base["sysctl_coordinator"], (after, base["sysctl_coordinator"]) @@ -44,8 +65,10 @@ nas.fail("ip link show flannel.1") nas.fail("pgrep -f containerd-shim") ruleset = nas.succeed("nft -s list ruleset") - for marker in ("KUBE-", "FLANNEL", "CNI-", "ax-fleet"): + for marker in ("KUBE-", "FLANNEL", "CNI-"): assert marker not in ruleset, f"{marker} left in the NAS ruleset" + # The control role's guard table stays in every generation (fix round 3). + nas.succeed("nft list chain inet ax-fleet-guard forward | grep -q 'tcp dport 8731'") nixos_fw = nas.succeed("nft -s list table inet nixos-fw") assert nixos_fw == base["nas_nixos_fw"], "the NAS firewall table differs from the baseline" record("nas_ruleset_equal_baseline", ruleset == base["nas_nft"]) From a3009ead50b29ea0c04c78212fe48fad4c2d6376 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 22:20:03 +0200 Subject: [PATCH 31/37] ax-fleet: fix round 4 without the ax patch (from ax/fleet-r4-wip e82cb885) Takes round 4's non-patch items and leaves p2-egress-deny.patch out (Tom, 09-23: no ax patches): - NAS: chain output in inet ax-fleet-guard; only root and clusterClientUids open NEW connections from the NAS host to the pod and Service CIDRs and the seed registry (ax-server API, ax-redis). - kubelet evictionHard restores the nodefs and imagefs signals; WorkerPool pods get ephemeral-storage 1Gi request, 32Gi limit. - ax-fleet-guard-apply, re-run by ax-fleet-teardown after k3s-killall.sh strips flannel's rules; the pod-input rules are idempotent. - ax-fleet-smoke --gateway NAME|none. - VM tests for the above (10-cluster, 20-substrate, 35-lan-guard, 90-rollback, ax-fleet-boot) and the TEST-NET-2 public stand-in on the worker. 30-ax's T4c (which asserted p2's EgressDenied) is not taken; the gateway-default phase replaces it. Round 4 was unreviewed and its VM runs 1 and 2 failed (REPORTED evals-2026-09-23/ax-fleet/receipts/fix-r4); not re-run in this step. Co-Authored-By: Claude Opus 5.5 (1M context) --- modules/ax-fleet/ax-fleet-smoke.sh | 24 +++- modules/ax-fleet/control.nix | 16 +++ modules/ax-fleet/harness.nix | 38 ++++-- modules/ax-fleet/interface.nix | 23 ++++ modules/ax-fleet/substrate.nix | 9 +- pkgs/ax-fleet-teardown/default.nix | 23 ++++ tests/ax-fleet-boot/default.nix | 7 +- tests/ax-fleet/default.nix | 1 + tests/ax-fleet/nodes.nix | 182 ++++++++++++++++++-------- tests/ax-fleet/phases/10-cluster.py | 12 ++ tests/ax-fleet/phases/20-substrate.py | 4 + tests/ax-fleet/phases/35-lan-guard.py | 12 ++ tests/ax-fleet/phases/90-rollback.py | 127 ++++++++++++++---- 13 files changed, 386 insertions(+), 92 deletions(-) diff --git a/modules/ax-fleet/ax-fleet-smoke.sh b/modules/ax-fleet/ax-fleet-smoke.sh index 8d2bb250e..6d1bf7225 100644 --- a/modules/ax-fleet/ax-fleet-smoke.sh +++ b/modules/ax-fleet/ax-fleet-smoke.sh @@ -16,7 +16,10 @@ # Options: --hold N (re-read the phase after N seconds), --timeout N (per Task, # default 900), --keep (do not delete the Tasks), --image REF (default: the # fleet image, ax-fleet-image-ref), --sandbox-class C (spec.sandboxClass; empty -# means gVisor). Exit 0 only when the case passes. The receipt carries no secret: Tasks carry none (DESIGN.md 11). +# means gVisor), --gateway NAME (spec.gateway.name, default halogen; `none` +# omits the gateway; ax-fleet-gateway-default then points the Task at the +# atespace's default Gateway, since stock ax would give it allow-all). +# Exit 0 only when the case passes. The receipt carries no secret: Tasks carry none (DESIGN.md 11). export AX_SERVER="${AX_SERVER:-$AX_FLEET_API}" ns="$AX_FLEET_ATESPACE" @@ -25,6 +28,7 @@ timeout=900 keep=0 image="" sandbox_class="" +gateway=halogen args=() while [ $# -gt 0 ]; do case "$1" in @@ -33,6 +37,7 @@ while [ $# -gt 0 ]; do --keep) keep=1; shift ;; --image) image="$2"; shift 2 ;; --sandbox-class) sandbox_class="$2"; shift 2 ;; + --gateway) gateway="$2"; shift 2 ;; *) args+=("$1"); shift ;; esac done @@ -80,6 +85,9 @@ run_task() { cmd_json="$(jq -cn '$ARGS.positional' --args -- "$@")" local class_line="" [ -z "$sandbox_class" ] || class_line=" sandboxClass: \"$sandbox_class\"" + local gateway_lines="" + [ "$gateway" = none ] || gateway_lines=" gateway: + name: \"$gateway\"" ax -a "$ns" apply -f - >/dev/null </dev/null 2>&1 || true; fi @@ -150,11 +158,15 @@ egress-deny) # Refused either at the Gateway (the connection fails: curl rc != 0) or, # for an allowlisted host on a port the NAS drops (fix round 3), by the # egress gateway's own upstream error (502/503/504, the target never - # answered). Any other HTTP status is the target answering: fail. - jq -c --arg case "$case_" --arg url "$url" \ - '. + {case:$case, url:$url, + # answered), or by Substrate's egress router refusing the actor outright + # (403 "egress denied": no policy, or a policy with no rules). Any other + # HTTP status + # is the target answering: fail. The deny targets in the VM answer 200. + jq -c --arg case "$case_" --arg url "$url" --arg gw "$gateway" \ + '. + {case:$case, url:$url, gateway:$gw, refused_by:(if (.result.curl_rc // 0) != 0 then "gateway-connection" elif ((.result.http_code // "") | test("^50[234]$")) then "egress-upstream" + elif (.result.http_code // "") == "403" then "egress-policy" else null end)} | . + {pass:(.refused_by != null and .ready.reason=="CommandExited")}' <<<"$r" ;; diff --git a/modules/ax-fleet/control.nix b/modules/ax-fleet/control.nix index 1d8c68902..d6697da87 100644 --- a/modules/ax-fleet/control.nix +++ b/modules/ax-fleet/control.nix @@ -116,6 +116,22 @@ let oifname { ${lib.concatMapStringsSep ", " (i: ''"${i}"'') cfg.guardInterfaces} } counter drop comment "ax-fleet: pods never reach the tailnet" oifname "ve-*" counter drop comment "ax-fleet: pods never reach the NAS's containers" } + + # ── the cluster from the NAS's own processes: root only (fix round 4) ── + # The coordinator's owner match (harness.nix apiRules) had no counterpart + # here: prerouting never sees locally generated packets, so paperless, + # immich, atticd, headscale or nginx (MEASURED uid-map) could open the + # unauthenticated ax-server API and the password-less ax-redis, and, while + # a seed runs, push to the loopback writer. Only NEW connections are + # policed: replies from AdGuard or the registry to pods are established. + chain output { + type filter hook output priority filter; policy accept; + ct state != new return + meta skuid 0 return + ${lib.optionalString (cfg.clusterClientUids != [ ]) "meta skuid { ${lib.concatMapStringsSep ", " toString cfg.clusterClientUids} } return"} + ip daddr { ${cfg.podCidr}, ${cfg.serviceCidr} } counter reject comment "ax-fleet: the cluster ranges from this host, root only" + ip daddr 127.0.0.1 tcp dport ${lib.last (lib.splitString ":" seedAddr)} counter reject with tcp reset comment "ax-fleet: the seed's writable registry, root only" + } ''; # ── the bootstrap steps this track owns ── diff --git a/modules/ax-fleet/harness.nix b/modules/ax-fleet/harness.nix index 089e86db2..b9c3f2bd4 100644 --- a/modules/ax-fleet/harness.nix +++ b/modules/ax-fleet/harness.nix @@ -136,9 +136,16 @@ let # kubectl exec/logs ride the agent tunnel. First rules of nixos-fw, so they # run before its ESTABLISHED accept and every port rule; IPv6 too (link-local # addresses on cni0 and the veths). - podInput = lib.concatMap (p: [ - "-I nixos-fw 1 -i ${p} -m conntrack --ctstate NEW -m comment --comment ax-fleet-pod-input -j nixos-fw-refuse" - ]) podIfs; + # Idempotent (fix round 4): deleted, then inserted, so guardApply can re-run + # it over a live nixos-fw without duplicating a rule. + podInputSpecs = map ( + p: "-i ${p} -m conntrack --ctstate NEW -m comment --comment ax-fleet-pod-input -j nixos-fw-refuse" + ) podIfs; + podInputCmds = + t: + lib.concatMapStringsSep "\n" (spec: '' + while ${t} -D nixos-fw ${spec} 2>/dev/null; do :; done + ${t} -I nixos-fw 1 ${spec}'') podInputSpecs; # ── the ax API and the cluster ranges: root, apiUsers and the proxy only ── # (fix round 3) The round-3 review MEASURED an unprivileged user applying a @@ -166,10 +173,9 @@ let while ${ipt} -t mangle -D FORWARD -j ax-fleet-guard 2>/dev/null; do :; done ${ipt} -t mangle -I FORWARD 1 -j ax-fleet-guard ${lib.concatMapStringsSep "\n" (r: "${ipt} -t mangle -A ax-fleet-guard ${r}") guardRules} - # pods to the host: refused (podInput; nixos-fw is rebuilt on every reload) - ${lib.concatMapStringsSep "\n" ( - r: "${ipt} ${r}" + lib.optionalString config.networking.enableIPv6 "\n${ip6t} ${r}" - ) podInput} + # pods to the host: refused (podInputSpecs) + ${podInputCmds ipt} + ${lib.optionalString config.networking.enableIPv6 (podInputCmds ip6t)} # the ax API and the cluster ranges from this host: owner match (fix round 3) ${ipt} -N ax-fleet-api 2>/dev/null || true ${ipt} -F ax-fleet-api @@ -178,6 +184,15 @@ let ${lib.concatMapStringsSep "\n" (r: "${ipt} -A ax-fleet-api ${r}") apiRules} ''; + # (fix round 4) The same rules as a script the teardown runs after + # k3s-killall.sh: the killall's `iptables-save | grep -iv flannel | + # iptables-restore` (REPORTED k3s-killall.sh:90 in the pinned k3s) deletes + # every rule naming flannel.1, the guard's and the pod-input refusal's + # included, and nothing reloads the firewall before k3s starts again. + guardApply = pkgs.writeShellScript "ax-fleet-guard-apply" ('' + set -eu + '' + guardStart); + guardStop = '' while ${ipt} -t mangle -D FORWARD -j ax-fleet-guard 2>/dev/null; do :; done ${ipt} -t mangle -F ax-fleet-guard 2>/dev/null || true @@ -241,6 +256,7 @@ in users.groups.ax-server-proxy = { }; # The teardown leaves a guard the generation declares (see pkgs/ax-fleet-teardown). environment.etc."ax-fleet/guard-declared".text = "harness\n"; + environment.etc."ax-fleet/guard-apply".source = guardApply; assertions = [ { assertion = builtins.all (u: config.users.users ? ${u}) cfg.apiUsers; @@ -260,7 +276,13 @@ in # nothing. systemReserved = lib.mkDefault "cpu=8,memory=32Gi"; kubeReserved = lib.mkDefault "cpu=1,memory=2Gi"; - evictionHard = lib.mkDefault "memory.available<8Gi"; + # Every signal, not only memory (fix round 4): a set --eviction-hard + # REPLACES kubelet's whole default map (MEASURED in the round-3 VM log: + # HardEvictionThresholds=[memory.available] only), which dropped k3s's + # nodefs/imagefs defaults. / here is also /nix/store, journald and the + # coordinator's postgres (MEASURED findmnt: one nvme partition), so a + # sandbox filling its writable layer must be evicted before they ENOSPC. + evictionHard = lib.mkDefault "memory.available<8Gi,nodefs.available<10%,nodefs.inodesFree<5%,imagefs.available<15%,imagefs.inodesFree<5%"; }; # ── CPU: the desk outweighs the sandboxes (fix round 2) ── diff --git a/modules/ax-fleet/interface.nix b/modules/ax-fleet/interface.nix index fb2f1add9..668b80444 100644 --- a/modules/ax-fleet/interface.nix +++ b/modules/ax-fleet/interface.nix @@ -235,6 +235,18 @@ in type = types.str; default = "16Gi"; }; + # (fix round 4) Each worker pod's writable layer, emptyDirs and logs + # live on the harness node's / (the desk's /nix/store disk). The limit + # evicts one runaway sandbox's pod before the node-level nodefs + # threshold (kubelet.evictionHard) has to evict everything. + ephemeralStorageRequest = mkOption { + type = types.str; + default = "1Gi"; + }; + ephemeralStorageLimit = mkOption { + type = types.str; + default = "32Gi"; + }; unreachableTolerationSeconds = mkOption { type = types.ints.unsigned; default = 3600; @@ -242,6 +254,17 @@ in }; }; + clusterClientUids = mkOption { + type = types.listOf types.ints.unsigned; + default = [ ]; + description = '' + Control role (fix round 4): numeric uids, besides root, that may open + connections from the NAS host to the pod and Service ranges. Numeric + because nftables resolves user names at load time and the ruleset is + checked in the build sandbox. Default: none. + ''; + }; + halogenEndpoint = mkOption { type = types.str; default = "10.42.0.5:8731"; diff --git a/modules/ax-fleet/substrate.nix b/modules/ax-fleet/substrate.nix index cfd1cd522..d0178bb41 100644 --- a/modules/ax-fleet/substrate.nix +++ b/modules/ax-fleet/substrate.nix @@ -163,11 +163,18 @@ let tolerationSeconds = cfg.workerPool.unreachableTolerationSeconds; } ]; + # ephemeral-storage (fix round 4): passed through unchanged by + # atecontroller's applyWorkerPoolPodTemplate (REPORTED + # workerpool_apply.go:522-528 at d277088b). resources = { - limits.memory = cfg.workerPool.memoryLimit; + limits = { + memory = cfg.workerPool.memoryLimit; + ephemeral-storage = cfg.workerPool.ephemeralStorageLimit; + }; requests = { cpu = "250m"; memory = "1Gi"; + ephemeral-storage = cfg.workerPool.ephemeralStorageRequest; }; }; }; diff --git a/pkgs/ax-fleet-teardown/default.nix b/pkgs/ax-fleet-teardown/default.nix index da6380f7e..383700dc4 100644 --- a/pkgs/ax-fleet-teardown/default.nix +++ b/pkgs/ax-fleet-teardown/default.nix @@ -59,6 +59,7 @@ writeShellApplication { echo " k3s-killall.sh stops it now, and it starts again at the next boot or switch." >&2 fi + guard_failed=0 echo "== k3s-killall.sh (${k3s.version})" ${k3s}/bin/k3s-killall.sh || echo "k3s-killall.sh exited $?; continuing" >&2 @@ -67,6 +68,18 @@ writeShellApplication { # script), so they stay; a generation from before ax never had them, and # then they go. if [ -e /etc/ax-fleet/guard-declared ]; then + # Fix round 4: the killall's `grep -iv flannel` pipeline just deleted + # every guard rule naming flannel.1 (the VXLAN source rule, the + # cross-node returns, the pod-input refusal). Re-apply the generation's + # guards rather than trust them; the NAS's are an nftables table the + # iptables pipeline never touches. + if [ -x /etc/ax-fleet/guard-apply ]; then + echo "== guards: re-applying this generation's ($(cat /etc/ax-fleet/guard-declared))" + if ! /etc/ax-fleet/guard-apply; then + echo "ax-fleet-teardown: WARNING: re-applying the guards failed; pods of a restarted k3s would reach this host over flannel.1. Reload the firewall." >&2 + guard_failed=1 + fi + fi echo "== guards: declared by this generation ($(cat /etc/ax-fleet/guard-declared)); left in place" else echo "== guard chains" @@ -76,6 +89,12 @@ writeShellApplication { while iptables -w -D OUTPUT -j ax-fleet-api 2>/dev/null; do :; done iptables -w -F ax-fleet-api 2>/dev/null || true iptables -w -X ax-fleet-api 2>/dev/null || true + # The pod-input refusal, if this generation's firewall reload left it. + for ifc in cni0 flannel.1; do + for t in iptables ip6tables; do + while $t -w -D nixos-fw -i "$ifc" -m conntrack --ctstate NEW -m comment --comment ax-fleet-pod-input -j nixos-fw-refuse 2>/dev/null; do :; done + done + done echo "== NAS guard table" nft delete table inet ax-fleet-guard 2>/dev/null || true @@ -95,6 +114,10 @@ writeShellApplication { ip -br link show 2>/dev/null | grep -E '^(cni0|flannel\.1|veth)' || true iptables-save 2>/dev/null | grep -cE 'KUBE-|FLANNEL|CNI-' || true pgrep -a containerd-shim || true + if [ "$guard_failed" -ne 0 ]; then + echo "ax-fleet-teardown: done, with the guard re-apply FAILED (see above)" >&2 + exit 1 + fi echo "ax-fleet-teardown: done" ''; } diff --git a/tests/ax-fleet-boot/default.nix b/tests/ax-fleet-boot/default.nix index 954e4a760..27600d673 100644 --- a/tests/ax-fleet-boot/default.nix +++ b/tests/ax-fleet-boot/default.nix @@ -28,7 +28,12 @@ stablePkgs.testers.runNixOSTest { name = "ax-fleet-boot"; node.specialArgs = { inherit inputs; }; nodes.nas = { - imports = [ nodes.nas ]; + # The pre-ax NAS plus its fleet module, ON from boot (fix round 4: the + # ax-fleet test's base no longer imports the module). + imports = [ + nodes.nasBase + nodes.nasFleet + ]; myAxFleet.enable = true; boot.initrd.systemd.enable = true; # The NAS's kernel (hosts/nas/kernel.nix: freshPkgs.linuxPackages_7_2); diff --git a/tests/ax-fleet/default.nix b/tests/ax-fleet/default.nix index df7876465..ff7b622af 100644 --- a/tests/ax-fleet/default.nix +++ b/tests/ax-fleet/default.nix @@ -23,6 +23,7 @@ let prelude = '' import json import os + import re import time from contextlib import contextmanager diff --git a/tests/ax-fleet/nodes.nix b/tests/ax-fleet/nodes.nix index 5146e0415..9134a448e 100644 --- a/tests/ax-fleet/nodes.nix +++ b/tests/ax-fleet/nodes.nix @@ -9,9 +9,16 @@ # production one; only interface names (eth1, eth2), the token and the kubelet # reservations differ, and ax-fleet-topology pins that parity. # -# Each base config is "today", with ax OFF. `specialisation.ax-on` sets -# myAxFleet.enable = true, and the test script switches to it live, NAS first, -# exactly as Tom will. +# Each base config is "today": it does NOT import modules/ax-fleet at all, as +# origin/main's hosts/{nas,coordinator} do not (fix round 4: the round-3 base +# carried the role, so every role-scoped guard was already in place before the +# switch). Two specialisations import the fleet module: +# ax-on myAxFleet.enable = true; the test switches to it live, NAS first, +# exactly as Tom will. +# ax-off the role declared, enable = false: the kill switch (DESIGN 13). +# 90-rollback runs the kill switch on the coordinator, and the generation +# rollback (back to this base) on both hosts. The worker is not switched in the +# motion and keeps its role in the base. let # A plain file, test-only, not a secret: the fleet reads agenix instead. token = pkgs.writeText "ax-fleet-vm-token" "ax-fleet-vm-test-token-0123456789abcdef"; @@ -80,7 +87,8 @@ let networking.interfaces.${iface}.ipv4.addresses = lib.mkForce [ { inherit address prefixLength; } ]; }; - # Everything that makes a node a fleet node in the test. + # Everything that makes a node a fleet node in the test (the ax-on and + # ax-off specialisations import it; the base does not). fleetNode = { role, address }: { @@ -92,7 +100,6 @@ let ../../modules/ax-client.nix inputs.agenix.nixosModules.default ]; - system.switch.enable = true; myAxFleet = { inherit role; lan = lanAddr address; @@ -100,63 +107,72 @@ let k3sAgentTokenFile = "${agentToken}"; guardInterfaces = [ "eth2" ]; }; - environment.systemPackages = [ + }; + + # The test tools, in the base: identical in every state. + testBase = { + system.switch.enable = true; + environment.systemPackages = [ pkgs.jq pkgs.curl pkgs.iptables pkgs.nftables pkgs.iproute2 - pkgs.dnsutils + pkgs.dnsutils + ]; + }; + + # fleet: the host's fleet module (fleetNode plus host-specific settings). + # extra: ax-on only (test images, reservations). + axStates = fleet: extra: { + specialisation.ax-on.configuration = { + imports = [ + fleet + extra ]; + myAxFleet.enable = true; + services.k3s.images = [ probeImage ]; }; + specialisation.ax-off.configuration = { + imports = [ fleet ]; + }; + }; - axOn = extra: { - specialisation.ax-on.configuration = lib.mkMerge [ - { - myAxFleet.enable = true; - services.k3s.images = [ probeImage ]; - } - extra - ]; + nasFleet = fleetNode { + role = "control"; + address = "10.42.0.1"; }; - vmReservations = { - systemReserved = "cpu=500m,memory=512Mi"; - kubeReserved = "cpu=250m,memory=256Mi"; - evictionHard = "memory.available<256Mi"; + coordinatorFleet = { + imports = [ + (fleetNode { + role = "harness"; + address = "10.42.0.2"; + }) + ]; + # The desk's wired port (fix round 3): eth3, NetworkManager-managed, + # DHCP from the worker's second leg, as enp191s0's "Wired connection 1". + myAxFleet.lan.extraInterfaces = [ "eth3" ]; }; -in -{ - inherit - probeImage - token - agentToken - claudeProbeImage - kubectlAte - ; - nas = + # The NAS as it is today, before ax (shared with ax-fleet-boot, which adds + # nasFleet in its base and boots it). + nasBase = { ... }: { imports = [ - (fleetNode { - role = "control"; - address = "10.42.0.1"; - }) - (axOn { - myAxFleet.kubelet = { - systemReserved = "cpu=1,memory=1Gi"; - }; - myAxFleet.registrySeed.ax-agent-claude-probe = { - oci = claudeProbeImage; - repo = "ax/ax-agent-claude-probe"; - inherit (claudeProbeImage.passthru) tag; - }; - }) + testBase (setAddr "eth1" "10.42.0.1" 24) ]; networking.hostName = "nas"; - environment.systemPackages = [ kubectlAte ]; + # The internet stand-in (fix round 4): TEST-NET-2 lives on the worker. + networking.interfaces.eth1.ipv4.routes = [ + { + address = "198.51.100.0"; + prefixLength = 24; + via = "10.42.0.5"; + } + ]; virtualisation = { vlans = [ 1 ]; memorySize = 10240; @@ -217,6 +233,7 @@ in # withdraw the house subnet route. 90-rollback asserts the teardown # never calls it. environment.systemPackages = [ + kubectlAte (pkgs.writeShellScriptBin "tailscale" '' echo "$*" >> /var/log/tailscale-stub.log '') @@ -230,22 +247,54 @@ in }; }; - coordinator = + + vmReservations = { + systemReserved = "cpu=500m,memory=512Mi"; + kubeReserved = "cpu=250m,memory=256Mi"; + # The full signal set, as harness.nix renders it (fix round 4): a set flag + # replaces every kubelet default. + evictionHard = "memory.available<256Mi,nodefs.available<10%,nodefs.inodesFree<5%,imagefs.available<15%,imagefs.inodesFree<5%"; + }; +in +{ + inherit + probeImage + token + agentToken + claudeProbeImage + kubectlAte + nasBase + nasFleet + ; + + nas = { ... }: { imports = [ - (fleetNode { - role = "harness"; - address = "10.42.0.2"; + nasBase + (axStates nasFleet { + myAxFleet.kubelet = { + systemReserved = "cpu=1,memory=1Gi"; + }; + myAxFleet.registrySeed.ax-agent-claude-probe = { + oci = claudeProbeImage; + repo = "ax/ax-agent-claude-probe"; + inherit (claudeProbeImage.passthru) tag; + }; }) - (axOn { myAxFleet.kubelet = vmReservations; }) + ]; + }; + + coordinator = + { ... }: + { + imports = [ + testBase + (axStates coordinatorFleet { myAxFleet.kubelet = vmReservations; }) (setAddr "eth1" "10.42.0.2" 24) (setAddr "eth2" "100.105.121.73" 10) ]; networking.hostName = "coordinator"; - # The desk's wired port (fix round 3): eth3, NetworkManager-managed, - # DHCP from the worker's second leg, as enp191s0's "Wired connection 1". - myAxFleet.lan.extraInterfaces = [ "eth3" ]; # myAxFleet.apiUsers defaults to [ "tom" ]; alice is the other local user. users.users.tom.isNormalUser = true; virtualisation = { @@ -316,11 +365,25 @@ in { ... }: { imports = [ + testBase (fleetNode { role = "inference"; address = "10.42.0.5"; }) - (setAddr "eth1" "10.42.0.5" 24) + # 198.51.100.5 (TEST-NET-2): a public address for the egress tests + # (fix round 4), routed to the worker by the NAS. + { + networking.interfaces.eth1.ipv4.addresses = lib.mkForce [ + { + address = "10.42.0.5"; + prefixLength = 24; + } + { + address = "198.51.100.5"; + prefixLength = 32; + } + ]; + } (setAddr "eth2" "192.168.43.5" 24) ]; networking.hostName = "worker"; @@ -344,6 +407,18 @@ in # Another worker port (fix round 3): the real worker opens 22 with # passwords on every interface; pods and the egress gateway must reach # 8731 and nothing else. + # The public target (fix round 4): what a Task with no Gateway must not + # reach, while the NAS host does. + systemd.services.public-8000 = { + wantedBy = [ "multi-user.target" ]; + # All addresses: binding 198.51.100.5 raced its assignment (MEASURED, + # fix-round-4 run 1: the unit exited 1 before network-addresses-eth1 + # added the address). + serviceConfig = { + ExecStart = "${pkgs.busybox}/bin/httpd -f -p 8000 -h ${pkgs.writeTextDir "index.html" "public-reached\n"}"; + Restart = "always"; + }; + }; systemd.services.worker-2222 = { wantedBy = [ "multi-user.target" ]; serviceConfig.ExecStart = "${pkgs.busybox}/bin/httpd -f -p 2222 -h ${pkgs.writeTextDir "index.html" "worker-port-2222-reached\n"}"; @@ -356,6 +431,7 @@ in interfaces.eth1.allowedTCPPorts = [ 8731 2222 + 8000 ]; interfaces.eth2.allowedTCPPorts = [ 8731 ]; interfaces.eth2.allowedUDPPorts = [ 67 ]; diff --git a/tests/ax-fleet/phases/10-cluster.py b/tests/ax-fleet/phases/10-cluster.py index b78e74bad..312c2cefa 100644 --- a/tests/ax-fleet/phases/10-cluster.py +++ b/tests/ax-fleet/phases/10-cluster.py @@ -218,6 +218,18 @@ def apply_probe(name, role, host_port, pvc=None): assert w["user.slice"] > w["kubepods.slice"] and w["system.slice"] > w["kubepods.slice"], w +with step("coordinator: kubelet evicts on disk pressure, not only memory"): + # Fix round 4: a set --eviction-hard replaces kubelet's defaults, so the + # harness renders every signal. The kubelet logs the map it runs with. + line = coordinator.succeed( + "journalctl -u k3s --no-pager -o cat | grep -o 'HardEvictionThresholds.*' | tail -n1 | cut -c1-2000" + ) + signals = sorted(set(re.findall(r'"Signal":"([a-z.A-Z]+)"', line))) + record("coordinator_hard_eviction_signals", signals) + for s in ("memory.available", "nodefs.available", "nodefs.inodesFree", "imagefs.available"): + assert s in signals, (s, signals) + + with step("coordinator: the guards hold"): pod_ip = jsonpath("pod probe-coord", "{.status.podIP}") record("probe_coord_ip", pod_ip) diff --git a/tests/ax-fleet/phases/20-substrate.py b/tests/ax-fleet/phases/20-substrate.py index a173dc933..01a6e8633 100644 --- a/tests/ax-fleet/phases/20-substrate.py +++ b/tests/ax-fleet/phases/20-substrate.py @@ -118,6 +118,10 @@ def sub_json(args): assert p["spec"]["nodeName"] == "coordinator", p["spec"]["nodeName"] lims = [c.get("resources", {}).get("limits", {}).get("memory") for c in p["spec"]["containers"]] assert any(lims), f"worker pod has no memory limit: {lims}" + # Fix round 4: the writable layer is bounded per pod too. + eph = [c.get("resources", {}).get("limits", {}).get("ephemeral-storage") for c in p["spec"]["containers"]] + assert any(eph), f"worker pod has no ephemeral-storage limit: {eph}" + record("worker_ephemeral_storage_limit", eph) with step("substrate: gVisor fetched through the RustFS fallback"): # No internet in the VM: atelet's anonymous GCS open of gs://gvisor/... diff --git a/tests/ax-fleet/phases/35-lan-guard.py b/tests/ax-fleet/phases/35-lan-guard.py index 70cc7370b..e05010b83 100644 --- a/tests/ax-fleet/phases/35-lan-guard.py +++ b/tests/ax-fleet/phases/35-lan-guard.py @@ -40,6 +40,18 @@ def svc_ip(ns, name): worker.succeed("ip route del 10.201.0.0/16 via 10.42.0.1") record("nas_range_guard", nas.succeed("nft list chain inet ax-fleet-guard prerouting").strip().splitlines()) + # Fix round 4: the NAS's own non-root processes (paperless, immich, ...) + # get none of it; root, above, still does. `nobody` and the dnsmasq + # stand-in's user stand in for them. + for u in ("nobody", "dnsmasq"): + nas.fail(f"runuser -u {u} -- curl -s --max-time 5 -o /dev/null http://{ax_ip}:8080/healthz") + nas.fail(f"runuser -u {u} -- curl -s --max-time 5 -o /dev/null http://{ax_pod}:8080/healthz") + nas.fail(f"runuser -u {u} -- {tcp_open(redis_ip, 6379)}") + nas.succeed(f"curl -sf --max-time 10 http://{ax_ip}:8080/healthz") + # Pods still resolve through the NAS's AdGuard stand-in (replies are not NEW). + kubectl("exec probe-nas -- sh -c 'nslookup only-nas.test 10.42.0.1 | grep -q 10.42.0.77'") + record("nas_output_guard", nas.succeed("nft list chain inet ax-fleet-guard output").strip().splitlines()) + with step("nas: pods reach Halogen on its port and no other private address"): # Fix round 3. The round-3 review MEASURED postgres-0 reaching worker:2222; diff --git a/tests/ax-fleet/phases/90-rollback.py b/tests/ax-fleet/phases/90-rollback.py index a11575735..f6d52a1ea 100644 --- a/tests/ax-fleet/phases/90-rollback.py +++ b/tests/ax-fleet/phases/90-rollback.py @@ -1,21 +1,76 @@ -# Phase 6: rollback, the kill switch proven (DESIGN.md 12.1, 13). Track -# cluster. Reverse order: coordinator, then nas. Each host goes back to its -# base toplevel (ax off) and then ax-fleet-teardown runs, as Tom would. +# Phase 6: rollback, the kill switch and the generation rollback proven +# (DESIGN.md 12.1, 13). Track cluster. Fix round 4: the base is "today" +# (pre-ax, no modules/ax-fleet), so every path below lands where Tom would. +# coordinator (a) ax-fleet-teardown inside the ax-on generation: the guards +# are re-applied after k3s-killall.sh strips every flannel rule, +# and a k3s that starts again runs behind them; +# (b) the kill switch (ax-off: role declared, enable false), then +# the teardown from the host's PATH; +# (c) the generation rollback to the pre-ax base, then the +# flake's teardown (the pre-ax PATH has none). +# nas the generation rollback straight from ax-on, pods still +# running, then the flake's teardown. -BASE = "/run/booted-system/bin/switch-to-configuration test" +AX_OFF = "/run/booted-system/specialisation/ax-off/bin/switch-to-configuration test" +PRE_AX = "/run/booted-system/bin/switch-to-configuration test" LEFTOVER_RULES = "iptables-save 2>/dev/null | grep -E 'KUBE-|FLANNEL|CNI-'" +POD_NETNS_PID = ( + "host=$(readlink /proc/1/ns/net); for p in $(pgrep -x pause); do " + "[ \"$(readlink /proc/$p/ns/net)\" != \"$host\" ] && { echo $p; break; }; done" +) +EXPECTED_GUARDS = {"pod_input": 2, "pod_input_flannel_v4": 1, "pod_input_flannel_v6": 1, "vxlan_source": 1, "guard_flannel": 5} def pod_netns_pid(machine): # A pod sandbox's pause process in a network namespace other than the host's. - return machine.succeed( - "host=$(readlink /proc/1/ns/net); for p in $(pgrep -x pause); do " - "[ \"$(readlink /proc/$p/ns/net)\" != \"$host\" ] && { echo $p; break; }; done" - ).strip() + return machine.succeed(POD_NETNS_PID).strip() -with step("rollback coordinator"): - coordinator.succeed(f"{BASE} >&2") +def rule_count(machine, cmd): + return int(machine.succeed(f"{cmd} || true").strip() or "0") + + +def guards(machine): + return { + "pod_input": rule_count(machine, "iptables -S nixos-fw | grep -c ax-fleet-pod-input"), + "pod_input_flannel_v4": rule_count(machine, "iptables -S nixos-fw | grep -c 'flannel.1.*ax-fleet-pod-input'"), + "pod_input_flannel_v6": rule_count(machine, "ip6tables -S nixos-fw | grep -c 'flannel.1.*ax-fleet-pod-input'"), + # iptables -S prints the source before the interface (MEASURED run 2). + "vxlan_source": rule_count(machine, "iptables -t mangle -S ax-fleet-guard | grep -cE -- '! -s [0-9./]+ -i flannel.1 -j DROP'"), + "guard_flannel": rule_count(machine, "iptables -t mangle -S ax-fleet-guard | grep -c flannel.1"), + } + + +with step("rollback coordinator (a): the teardown inside ax-on re-applies the guards; k3s restarts behind them"): + # Fix round 4. k3s-killall.sh runs `iptables-save | grep -iv flannel | + # iptables-restore`, which deleted every guard rule naming flannel.1 while + # the teardown reported them "left in place" (MEASURED by the review). + coordinator.succeed("ax-fleet-teardown >&2") + coordinator.fail("ip link show flannel.1") + g = guards(coordinator) + record("guards_after_teardown_in_ax_on", g) + assert g == EXPECTED_GUARDS, g + # "It starts again at the next boot or switch": no firewall reload first. + coordinator.succeed("systemctl start k3s.service") + node_ready("coordinator") + coordinator.wait_until_succeeds("ip link show flannel.1", timeout=300) + kubectl("wait --for=condition=Ready pod/probe-coord --timeout=600s") + g = guards(coordinator) + record("guards_after_k3s_restart", g) + assert g == EXPECTED_GUARDS, g # re-applied once, never duplicated + # Discriminating: the NAS's pod reaches a coordinator pod over VXLAN, and + # the coordinator's sshd listens on its flannel.1 address, yet the pod + # gets no SSH banner from it. + coord_pod = jsonpath("pod probe-coord", "{.status.podIP}") + nas.wait_until_succeeds(f"k3s kubectl exec probe-nas -- curl -sf --max-time 5 http://{coord_pod}:8000/ | grep -x pod-ok", timeout=300) + fl = coordinator.succeed("ip -4 -o addr show dev flannel.1 | awk '{print $4}' | cut -d/ -f1").strip() + coordinator.succeed(f"timeout 10 bash -c 'exec 3<>/dev/tcp/{fl}/22'") + kubectl(f"exec probe-nas -- sh -c '! (nc -w 5 {fl} 22 /dev/null | grep -q SSH)'") + record("ip_forward_after_k3s_restart", coordinator.succeed("sysctl -n net.ipv4.ip_forward").strip()) + + +with step("rollback coordinator (b): the kill switch, then the teardown from PATH"): + coordinator.succeed(f"{AX_OFF} >&2") coordinator.fail("systemctl is-active k3s.service") # Fix round 3: between the kill switch and the teardown the pods still # run (KillMode=process). The role-scoped guards keep them off the host. @@ -25,20 +80,38 @@ def pod_netns_pid(machine): coordinator.fail(f"nsenter -t {pid} -n timeout 5 bash -c 'exec 3<>/dev/tcp/10.42.0.2/22'") coordinator.fail(f"nsenter -t {pid} -n timeout 5 bash -c 'exec 3<>/dev/tcp/100.105.121.73/22'") coordinator.fail(f"nsenter -t {pid} -n timeout 5 bash -c 'exec 3<>/dev/tcp/10.42.0.5/8731'") - assert coordinator.succeed("iptables -S nixos-fw | grep -c ax-fleet-pod-input").strip() == "2" + assert guards(coordinator) == EXPECTED_GUARDS, guards(coordinator) coordinator.succeed("iptables -t mangle -S FORWARD 1 | grep -q ax-fleet-guard") - # As documented (fix round 2): the teardown from the rolled-back host's - # own PATH, no checkout, no injected store path. + # As documented (fix round 2): the teardown from the host's own PATH. coordinator.succeed("test -x /run/current-system/sw/bin/ax-fleet-teardown") coordinator.succeed("ax-fleet-teardown >&2") coordinator.fail("ip link show cni0") coordinator.fail("ip link show flannel.1") coordinator.fail(LEFTOVER_RULES) coordinator.fail("pgrep -f containerd-shim") - # The guards belong to the harness role's every generation (fix round 3); - # the teardown leaves them, inert without cni0 and flannel.1. - coordinator.succeed("iptables -t mangle -S ax-fleet-guard | grep -q DROP") + # The guards belong to the harness role's every generation (fix round 3), + # re-applied whole after the killall (fix round 4). + g = guards(coordinator) + record("guards_after_kill_switch_teardown", g) + assert g == EXPECTED_GUARDS, g coordinator.succeed("iptables -S OUTPUT 1 | grep -q ax-fleet-api") + + +with step("rollback coordinator (c): the generation rollback to the pre-ax base, then the flake's teardown"): + coordinator.succeed(f"{PRE_AX} >&2") + coordinator.fail("test -e /etc/ax-fleet/guard-declared") + # The pre-ax PATH has no teardown; Tom runs the flake's (DESIGN 13). + coordinator.fail("command -v ax-fleet-teardown") + record("coordinator_pre_ax_stale", { + "mangle_guard": rule_count(coordinator, "iptables -t mangle -S | grep -c ax-fleet-guard"), + "api_chain": rule_count(coordinator, "iptables -S | grep -c ax-fleet-api"), + "pod_input": rule_count(coordinator, "iptables -S nixos-fw | grep -c ax-fleet-pod-input"), + }) + coordinator.succeed(f"{TEARDOWN} >&2") + coordinator.fail("iptables -t mangle -S ax-fleet-guard") + coordinator.fail("iptables -S ax-fleet-api") + coordinator.fail("iptables-save | grep -q ax-fleet") + coordinator.fail("ip6tables-save | grep -q ax-fleet") after = sysctls(coordinator) record("sysctl_coordinator_after_rollback", after) assert after == base["sysctl_coordinator"], (after, base["sysctl_coordinator"]) @@ -49,26 +122,34 @@ def pod_netns_pid(machine): peer.succeed("curl -sf --max-time 10 http://100.105.121.73/ | grep -x caddy-ok") -with step("rollback nas"): +with step("rollback nas: the generation rollback straight from ax-on, then the flake's teardown"): # Diagnostic only: which processes hold the kubelet bind before the switch. _, holders = nas.execute("ls -l /proc/[0-9]*/cwd /proc/[0-9]*/root 2>/dev/null | grep -c /var/lib/kubelet") record("nas_kubelet_holders_before_rollback", holders.strip()) - nas.succeed(f"{BASE} >&2") + nas.succeed(f"{PRE_AX} >&2") nas.fail("systemctl is-active k3s.service") nas.fail("systemctl is-active docker-registry.service") + nas.fail("test -e /etc/ax-fleet/guard-declared") + nas.fail("command -v ax-fleet-teardown") + # The window DESIGN 13 documents: pods outlive the switch until the + # teardown. Recorded, not asserted: which guard (if any) still holds them. + pid = pod_netns_pid(nas) + window = {"pod_survived": bool(pid)} + if pid: + rc, _ = nas.execute(f"nsenter -t {pid} -n timeout 5 bash -c 'exec 3<>/dev/tcp/10.42.0.5/2222'") + window["pod_reaches_worker_2222"] = rc == 0 + window["guard_table_present"] = nas.execute("nft list table inet ax-fleet-guard")[0] == 0 + record("nas_generation_rollback_window", window) nas.succeed("command -v tailscale") # the stub is reachable from a root shell - nas.succeed("test -x /run/current-system/sw/bin/ax-fleet-teardown") - nas.succeed("ax-fleet-teardown >&2") + nas.succeed(f"{TEARDOWN} >&2") # The teardown's pinned PATH keeps k3s-killall.sh away from tailscale. nas.fail("test -e /var/log/tailscale-stub.log") nas.fail("ip link show cni0") nas.fail("ip link show flannel.1") nas.fail("pgrep -f containerd-shim") ruleset = nas.succeed("nft -s list ruleset") - for marker in ("KUBE-", "FLANNEL", "CNI-"): + for marker in ("KUBE-", "FLANNEL", "CNI-", "ax-fleet"): assert marker not in ruleset, f"{marker} left in the NAS ruleset" - # The control role's guard table stays in every generation (fix round 3). - nas.succeed("nft list chain inet ax-fleet-guard forward | grep -q 'tcp dport 8731'") nixos_fw = nas.succeed("nft -s list table inet nixos-fw") assert nixos_fw == base["nas_nixos_fw"], "the NAS firewall table differs from the baseline" record("nas_ruleset_equal_baseline", ruleset == base["nas_nft"]) From e948734d9feaf0a4ee2875f4f6e339e79bc0eca6 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 22:22:07 +0200 Subject: [PATCH 32/37] ax-fleet: close the gateway-less egress fail-open without an ax patch Stock ax v0.3.0 gives a Task with no gateway, or naming a missing one, an allow-all EgressPolicy (reconciler.go "Default to allow all egress"; MEASURED from the pinned source). Instead of p2-egress-deny.patch: - modules/ax-fleet/gateways.nix: myAxFleet.ax.atespaces declares every atespace the fleet creates (fleet and ax's CLI default "default"), each with its Gateways and a default one (Halogen only). Assertions: the default Gateway is declared for every atespace, its allowlist is not empty and allows no "*", 0.0.0.0/0 or ::/0; the smoke atespace is declared. - bootstrap step 80-ax-gateways applies every declared Gateway. - ax-fleet-gateway-default (control role, root, poll 2s): points any Task in a declared atespace without a usable gateway at the default through ax's own UpdateTask. pkgs/ax-gateway-default is a separate client binary built from stock ax's module tree (same vendorHash); ax is unchanged. - tests/ax-fleet/phases/33-gateway-default.py: the default exists in both atespaces; a Task with no gateway and one naming a missing gateway are pointed at the default, report to the floor, and cannot reach the TEST-NET-2 public stand-in; a Gateway allowlisting it gets 200 (positive control). Records the repoint window. The link's capacity 0 on a missing gateway (B10) stays the second guard. Not run in this step. Co-Authored-By: Claude Opus 5.5 (1M context) --- modules/ax-fleet/default.nix | 1 + modules/ax-fleet/gateways.nix | 188 ++++++++++++++++++++ pkgs/ax-gateway-default/default.nix | 17 ++ pkgs/ax-gateway-default/main.go | 114 ++++++++++++ tests/ax-fleet/phases/33-gateway-default.py | 68 +++++++ 5 files changed, 388 insertions(+) create mode 100644 modules/ax-fleet/gateways.nix create mode 100644 pkgs/ax-gateway-default/default.nix create mode 100644 pkgs/ax-gateway-default/main.go create mode 100644 tests/ax-fleet/phases/33-gateway-default.py diff --git a/modules/ax-fleet/default.nix b/modules/ax-fleet/default.nix index 3ea282549..7c341d792 100644 --- a/modules/ax-fleet/default.nix +++ b/modules/ax-fleet/default.nix @@ -70,6 +70,7 @@ in ./inference.nix ./substrate.nix ./ax.nix + ./gateways.nix ]; config = lib.mkMerge [ diff --git a/modules/ax-fleet/gateways.nix b/modules/ax-fleet/gateways.nix new file mode 100644 index 000000000..fd6f8a38c --- /dev/null +++ b/modules/ax-fleet/gateways.nix @@ -0,0 +1,188 @@ +{ + config, + lib, + inputs, + ... +}: +# ax-fleet gateways: the egress default for every atespace the fleet declares. +# +# Stock ax v0.3.0 fails open: a Task with no `gateway:`, or naming a Gateway +# its atespace lacks, gets an allow-all EgressPolicy (reconciler.go "Default +# to allow all egress"; the client turns "*" into EgressRule{All}, and +# Substrate ignores the port). The fleet does not patch ax (Tom, 2026-09-23 +# 08:45Z), so the default is closed from outside it, in three layers: +# +# 1. Declared: every atespace in myAxFleet.ax.atespaces names a default +# Gateway with a non-empty allowlist and no allow-all host (assertion +# below). The bootstrap step 80-ax-gateways applies every declared +# Gateway on the NAS, idempotently. +# 2. Enforced: ax-fleet-gateway-default (control role) polls each declared +# atespace and points any Task without a usable gateway at the default, +# through ax's own UpdateTask; the controller then replaces the actor's +# egress policy with the default's allowlist. The window between a +# gateway-less Task's first reconcile and the repoint is bounded by the +# poll interval plus one reconcile; checks.ax-fleet measures it +# (33-gateway-default, record gateway_default_repoint). +# 3. The link: capacity 0 on a missing gateway (B10), so the floor never +# dispatches a Task that would need layer 2. +# +# Below all three, the NAS forward chain (control.nix) keeps pods off every +# private range but Halogen, whatever policy ax writes. +let + cfg = config.myAxFleet; + system = "x86_64-linux"; + fleetPkgs = inputs.nixpkgs.legacyPackages.${system}; + ax = inputs.self.packages.${system}.ax; + enforcer = fleetPkgs.callPackage ../../pkgs/ax-gateway-default { inherit ax; }; + on = cfg.enable && cfg.role == "control"; + + halogenHost = builtins.head (lib.splitString ":" cfg.halogenEndpoint); + halogenPort = lib.toInt (lib.last (lib.splitString ":" cfg.halogenEndpoint)); + halogenRule = { + host = "${halogenHost}/32"; + port = halogenPort; + }; + allowAll = [ + "*" + "0.0.0.0/0" + "::/0" + ]; + + hostRule = lib.types.submodule { + options = { + host = lib.mkOption { + type = lib.types.str; + description = "A hostname pattern or a CIDR, as ax's HostRule.host."; + }; + port = lib.mkOption { + type = lib.types.port; + description = "Carried for the record; ax v0.3.0 drops it and the NAS chain enforces ports."; + }; + }; + }; + + atespaceType = lib.types.submodule { + options = { + defaultGateway = lib.mkOption { + type = lib.types.str; + default = "default"; + description = "The Gateway a Task without a usable gateway is pointed at."; + }; + gateways = lib.mkOption { + type = lib.types.attrsOf (lib.types.listOf hostRule); + default = { }; + description = "Gateway name -> egress allowlist, applied by the bootstrap."; + }; + }; + }; + + gatewayDoc = + ns: name: hosts: + builtins.toJSON { + apiVersion = "ax.io/v1alpha1"; + kind = "Gateway"; + metadata = { + inherit name; + atespace = ns; + }; + spec.egress.allowlist.hosts = hosts; + }; + + axServer = "${cfg.axServerClusterIP}:8080"; + spacesArg = lib.concatStringsSep "," ( + lib.mapAttrsToList (ns: a: "${ns}=${a.defaultGateway}") cfg.ax.atespaces + ); +in +{ + options.myAxFleet.ax = { + atespaces = lib.mkOption { + type = lib.types.attrsOf atespaceType; + default = { + ${cfg.ax.atespace}.gateways = { + default = [ halogenRule ]; + halogen = [ halogenRule ]; + }; + # ax's CLI default when -a is not given. + default.gateways.default = [ halogenRule ]; + }; + defaultText = lib.literalExpression ''{ fleet.gateways = { default = [ halogen ]; halogen = [ halogen ]; }; default.gateways.default = [ halogen ]; }''; + description = "Every atespace the fleet creates, each with its Gateways and the default one."; + }; + gatewayDefaultInterval = lib.mkOption { + type = lib.types.str; + default = "2s"; + description = "ax-fleet-gateway-default's poll interval (a Go duration). A trade between the fail-open window and API load, not an estimate."; + }; + }; + + config = lib.mkMerge [ + (lib.mkIf cfg.enable { + assertions = lib.flatten ( + lib.mapAttrsToList (ns: a: [ + { + assertion = a.gateways ? ${a.defaultGateway}; + message = "myAxFleet.ax.atespaces.${ns}: the default Gateway \"${a.defaultGateway}\" is not declared in its gateways; stock ax would give a gateway-less Task allow-all egress."; + } + { + assertion = (a.gateways.${a.defaultGateway} or [ ]) != [ ]; + message = "myAxFleet.ax.atespaces.${ns}: the default Gateway \"${a.defaultGateway}\" has an empty allowlist, which stock ax treats as allow-all."; + } + { + assertion = lib.all (r: !(lib.elem r.host allowAll)) (a.gateways.${a.defaultGateway} or [ ]); + message = "myAxFleet.ax.atespaces.${ns}: the default Gateway \"${a.defaultGateway}\" must not allow every host."; + } + ]) cfg.ax.atespaces + ) + ++ [ + { + assertion = cfg.ax.atespaces ? ${cfg.ax.atespace}; + message = "myAxFleet.ax.atespaces must declare myAxFleet.ax.atespace (\"${cfg.ax.atespace}\")."; + } + ]; + }) + + (lib.mkIf on { + myAxFleet.bootstrap."80-ax-gateways" = '' + # 80-ax-gateways: every declared Gateway, default ones included + # (modules/ax-fleet/gateways.nix). Idempotent: ax apply updates. + export AX_SERVER=http://${axServer} + for _ in $(seq 1 300); do + ${lib.getExe fleetPkgs.curl} -fsS -o /dev/null --max-time 5 "$AX_SERVER/healthz" && break + sleep 2 + done + ${lib.concatStrings ( + lib.flatten ( + lib.mapAttrsToList ( + ns: a: + lib.mapAttrsToList (name: hosts: '' + echo "gateway ${ns}/${name}" + ${ax}/bin/ax -a ${ns} apply -f ${fleetPkgs.writeText "gateway-${ns}-${name}.json" (gatewayDoc ns name hosts)} + '') a.gateways + ) cfg.ax.atespaces + ) + )} + echo "ax gateways declared: ${spacesArg}" + ''; + + systemd.services.ax-fleet-gateway-default = { + description = "ax-fleet: point Tasks without a usable gateway at their atespace's default Gateway"; + wantedBy = [ "multi-user.target" ]; + after = [ + "k3s.service" + "ax-fleet-bootstrap.service" + ]; + serviceConfig = { + ExecStart = "${lib.getExe enforcer} -server ${axServer} -atespaces ${spacesArg} -interval ${cfg.ax.gatewayDefaultInterval}"; + Restart = "always"; + RestartSec = 5; + # Root: the NAS output chain lets only root reach the Service range. + NoNewPrivileges = true; + ProtectSystem = "strict"; + ProtectHome = true; + PrivateTmp = true; + CapabilityBoundingSet = ""; + }; + }; + }) + ]; +} diff --git a/pkgs/ax-gateway-default/default.nix b/pkgs/ax-gateway-default/default.nix new file mode 100644 index 000000000..a2a50f4ee --- /dev/null +++ b/pkgs/ax-gateway-default/default.nix @@ -0,0 +1,17 @@ +# ax-gateway-default: a separate client binary built from stock ax's module +# tree (same vendorHash, no ax patch). See main.go and +# modules/ax-fleet/gateways.nix. +{ ax }: +ax.overrideAttrs (old: { + pname = "ax-gateway-default"; + postPatch = (old.postPatch or "") + '' + mkdir -p cmd/ax-gateway-default + cp ${./main.go} cmd/ax-gateway-default/main.go + ''; + subPackages = [ "cmd/ax-gateway-default" ]; + doCheck = false; + meta = (old.meta or { }) // { + description = "Points ax Tasks without a usable gateway at their atespace's default Gateway"; + mainProgram = "ax-gateway-default"; + }; +}) diff --git a/pkgs/ax-gateway-default/main.go b/pkgs/ax-gateway-default/main.go new file mode 100644 index 000000000..d1587e6fc --- /dev/null +++ b/pkgs/ax-gateway-default/main.go @@ -0,0 +1,114 @@ +// ax-gateway-default: stock ax v0.3.0 gives a Task with no gateway, or one +// naming a Gateway its atespace lacks, an allow-all EgressPolicy +// (internal/controller/reconciler.go, "Default to allow all egress"). This +// fleet does not patch ax (Tom, 2026-09-23), so the default lives outside it: +// every declared atespace carries a default Gateway (the bootstrap applies +// it), and this loop points any Task in those atespaces that has no usable +// gateway at that default through ax's own UpdateTask. The controller then +// reconciles the Task again and replaces the actor's egress policy with the +// default Gateway's allowlist. +// +// It is a client of ax's public gRPC API, built from the same module tree; +// it changes nothing inside ax. +package main + +import ( + "context" + "flag" + "fmt" + "log/slog" + "os" + "strings" + "time" + + "github.com/google/ax/pkg/apis/v1alpha1" + "google.golang.org/grpc" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/credentials/insecure" + "google.golang.org/grpc/status" +) + +func main() { + server := flag.String("server", os.Getenv("AX_SERVER"), "ax-server address, host:port or http://host:port") + spaces := flag.String("atespaces", "", "comma-separated atespace=defaultGateway pairs") + every := flag.Duration("interval", 2*time.Second, "poll interval") + once := flag.Bool("once", false, "one pass, then exit (tests)") + flag.Parse() + + defaults := map[string]string{} + for _, kv := range strings.Split(*spaces, ",") { + if kv == "" { + continue + } + ns, gw, ok := strings.Cut(kv, "=") + if !ok || ns == "" || gw == "" { + fmt.Fprintf(os.Stderr, "bad -atespaces entry %q (want atespace=gateway)\n", kv) + os.Exit(64) + } + defaults[ns] = gw + } + if len(defaults) == 0 { + fmt.Fprintln(os.Stderr, "no atespaces declared") + os.Exit(64) + } + target := strings.TrimPrefix(strings.TrimPrefix(*server, "http://"), "https://") + conn, err := grpc.NewClient(target, grpc.WithTransportCredentials(insecure.NewCredentials())) + if err != nil { + slog.Error("connect", "target", target, "error", err) + os.Exit(1) + } + defer conn.Close() + client := v1alpha1.NewAXClient(conn) + + for { + for ns, gw := range defaults { + if err := pass(client, ns, gw); err != nil { + slog.Warn("pass failed", "atespace", ns, "error", err) + } + } + if *once { + return + } + time.Sleep(*every) + } +} + +func pass(client v1alpha1.AXClient, ns, def string) error { + ctx, cancel := context.WithTimeout(context.Background(), 20*time.Second) + defer cancel() + if _, err := client.GetGateway(ctx, &v1alpha1.GetGatewayRequest{Atespace: ns, Name: def}); err != nil { + // Without the default there is nothing safe to point at; the + // bootstrap re-applies it. Log loudly, change nothing. + return fmt.Errorf("default gateway %s/%s: %w", ns, def, err) + } + resp, err := client.ListTasks(ctx, &v1alpha1.ListTasksRequest{Atespace: ns}) + if err != nil { + return fmt.Errorf("list tasks: %w", err) + } + for _, t := range resp.Tasks { + if t.Spec == nil || t.Metadata == nil { + continue + } + name := "" + if t.Spec.Gateway != nil { + name = t.Spec.Gateway.Name + } + if name != "" { + _, gerr := client.GetGateway(ctx, &v1alpha1.GetGatewayRequest{Atespace: ns, Name: name}) + if gerr == nil { + continue + } + if status.Code(gerr) != codes.NotFound { + slog.Warn("gateway lookup", "task", t.Metadata.Name, "gateway", name, "error", gerr) + continue + } + } + t.Spec.Gateway = &v1alpha1.GatewayRef{Name: def} + if _, err := client.UpdateTask(ctx, &v1alpha1.UpdateTaskRequest{Task: t}); err != nil { + slog.Warn("repoint", "task", t.Metadata.Name, "error", err) + continue + } + slog.Info("pointed at the default gateway", "atespace", ns, "task", t.Metadata.Name, "was", name, "now", def) + } + return nil +} diff --git a/tests/ax-fleet/phases/33-gateway-default.py b/tests/ax-fleet/phases/33-gateway-default.py new file mode 100644 index 000000000..c659d64db --- /dev/null +++ b/tests/ax-fleet/phases/33-gateway-default.py @@ -0,0 +1,68 @@ +# Phase 33: egress without a gateway lands behind the atespace's default +# Gateway (modules/ax-fleet/gateways.nix). Stock ax v0.3.0 gives such a Task +# allow-all; the fleet closes it without an ax patch: the bootstrap declared a +# default Gateway (Halogen and the floor only) and ax-fleet-gateway-default +# points the Task at it. Discriminating: the NAS host reaches the public +# stand-in (TEST-NET-2 on the worker, fix round 4), and a Gateway that +# allowlists it gets its 200 through the same egress path. +import re + +PUBLIC = "http://198.51.100.5:8000/" + + +def task_gateway(name: str) -> Any: + rc, out = ax(f"get task {name}") + if rc != 0: + return None + m = re.search(r"gateway:\s*\n\s+name:\s*(\S+)", out) + return m.group(1).strip("\"'") if m else "" + + +def late_curl_body(url: str, wait: int) -> str: + # Waits WAIT seconds first: the repoint is asynchronous by design. + return f"sleep {wait}\n" + curl_body(url) + + +with step("gateways: the bootstrap declared a default Gateway in every declared atespace"): + for ns in ("fleet", "default"): + coordinator.succeed(f"AX_SERVER=http://127.0.0.1:8099 ax -a {ns} get gateway default") + nas.succeed("systemctl is-active ax-fleet-gateway-default.service") + record("gateway_default_unit", nas.succeed("systemctl show ax-fleet-gateway-default -p ExecStart --value").strip()) + +with step("gateways: a Task without a gateway, or naming a missing one, is pointed at the default and cannot reach the public stand-in"): + worker.wait_for_unit("public-8000.service") + nas.wait_until_succeeds(f"curl -sf --max-time 10 {PUBLIC} | grep -q public-reached", timeout=120) + results = {} + for label, gw in (("none", None), ("missing", "no-such-gateway")): + name = f"gwdef-{label}" + n = f"{name}-a1" + t0 = time.monotonic() + fleet_task(n, late_curl_body(PUBLIC, 30), gateway=gw) + coordinator.wait_until_succeeds( + f"{AX} get task {n} | grep -A1 -E '^\\s+gateway:' | grep -qE 'name:\\s*\"?default\"?'", timeout=120 + ) + repoint_s = round(time.monotonic() - t0, 1) + reps, before, secs = wait_report(n) + assert reps, f"{n}: no floor report (the default Gateway allows the floor): {before}" + rep = reps[0]["report"] + res = json.loads(base64.b64decode(rep["result_b64"]) or b"{}") + results[label] = {"repoint_seconds": repoint_s, "report_seconds": secs, "result": res, "state": before} + record("gateway_default_repoint", results) + assert task_gateway(n) == "default", n + assert res.get("http_code") != "200", (label, res) + delete_task(n) + +with step("gateways: positive control, a Gateway allowlisting the public stand-in reaches it"): + coordinator.succeed( + "AX_SERVER=http://127.0.0.1:8099 ax -a fleet apply -f - <<'EOF'\n" + "apiVersion: ax.io/v1alpha1\nkind: Gateway\nmetadata:\n name: public-test\n atespace: fleet\n" + "spec:\n egress:\n allowlist:\n hosts:\n" + " - host: \"198.51.100.5/32\"\n port: 8000\n" + " - host: \"10.42.0.5/32\"\n port: 8731\n" + "EOF" + ) + ok = fleet_run("gw-public", curl_body(PUBLIC), gateway="public-test") + record("gateway_public_positive_control", ok["result"]) + assert ok["result"]["http_code"] == "200", ok + assert task_gateway("gw-public-a1") in (None, "public-test") + coordinator.execute("AX_SERVER=http://127.0.0.1:8099 ax -a fleet delete gateway public-test") From 5b90c6a751c0e3c5c40ad2e57ac4ba79cbb34bca Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 22:24:06 +0200 Subject: [PATCH 33/37] substrate-link: vendor the link source (ax-conwip a021003 apps/link) and package it pkgs/substrate-link/src is `git archive a021003` of package.json, pnpm-lock.yaml, pnpm-workspace.yaml, tsconfig.json and apps/link from ax-conwip eval/2026-09-23-link (SYNC.md names the sha and the resync recipe). The package is the conwip-link derivation from evals-2026-09-23/link, renamed, with src = ./src and a real pnpmDeps hash (MEASURED: fakeHash build, then "got:"). Built: esbuild bundle of 2,103,690 bytes; run with no config it logs token-missing and exits 78. Co-Authored-By: Claude Opus 5.5 (1M context) --- pkgs/substrate-link/SYNC.md | 22 + pkgs/substrate-link/default.nix | 57 + .../src/apps/link/proto/ax-p1.proto | 406 +++++ pkgs/substrate-link/src/apps/link/src/ax.ts | 121 ++ .../src/apps/link/src/config.ts | 103 ++ .../src/apps/link/src/contract.ts | 181 +++ .../substrate-link/src/apps/link/src/floor.ts | 169 ++ pkgs/substrate-link/src/apps/link/src/jobs.ts | 120 ++ .../src/apps/link/src/journal.ts | 187 +++ pkgs/substrate-link/src/apps/link/src/link.ts | 921 +++++++++++ pkgs/substrate-link/src/apps/link/src/main.ts | 105 ++ .../src/apps/link/test/ax-grpc.test.ts | 91 ++ .../src/apps/link/test/critique-fixed.test.ts | 406 +++++ .../src/apps/link/test/fake-ax.ts | 84 + .../src/apps/link/test/fake-floor.ts | 304 ++++ .../src/apps/link/test/final-pass.test.ts | 70 + .../src/apps/link/test/fixtures/ax-p1.proto | 405 +++++ .../src/apps/link/test/link.test.ts | 357 +++++ .../src/apps/link/test/review-r1.test.ts | 391 +++++ .../src/apps/link/test/review-r2.test.ts | 419 +++++ .../src/apps/link/test/review-r3.test.ts | 484 ++++++ .../apps/link/test/review-r4-expired.test.ts | 86 + .../link/test/review-r4-fiber-defect.test.ts | 96 ++ .../src/apps/link/test/review-r4-harness.ts | 59 + .../test/review-r4-heartbeat-envelope.test.ts | 134 ++ .../link/test/review-r4-journal-key.test.ts | 48 + .../apps/link/test/review-r4-overtake.test.ts | 78 + .../apps/link/test/review-r4-p1stale.test.ts | 74 + .../test/review-r4-restart-replay.test.ts | 143 ++ .../link/test/review-r4-stale-lost.test.ts | 67 + .../link/test/workerd.integration.test.ts | 190 +++ pkgs/substrate-link/src/package.json | 22 + pkgs/substrate-link/src/pnpm-lock.yaml | 1401 +++++++++++++++++ pkgs/substrate-link/src/pnpm-workspace.yaml | 4 + pkgs/substrate-link/src/tsconfig.json | 16 + 35 files changed, 7821 insertions(+) create mode 100644 pkgs/substrate-link/SYNC.md create mode 100644 pkgs/substrate-link/default.nix create mode 100644 pkgs/substrate-link/src/apps/link/proto/ax-p1.proto create mode 100644 pkgs/substrate-link/src/apps/link/src/ax.ts create mode 100644 pkgs/substrate-link/src/apps/link/src/config.ts create mode 100644 pkgs/substrate-link/src/apps/link/src/contract.ts create mode 100644 pkgs/substrate-link/src/apps/link/src/floor.ts create mode 100644 pkgs/substrate-link/src/apps/link/src/jobs.ts create mode 100644 pkgs/substrate-link/src/apps/link/src/journal.ts create mode 100644 pkgs/substrate-link/src/apps/link/src/link.ts create mode 100644 pkgs/substrate-link/src/apps/link/src/main.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/ax-grpc.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/critique-fixed.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/fake-ax.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/fake-floor.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/final-pass.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/fixtures/ax-p1.proto create mode 100644 pkgs/substrate-link/src/apps/link/test/link.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r1.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r2.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r3.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r4-expired.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r4-fiber-defect.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r4-harness.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r4-heartbeat-envelope.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r4-journal-key.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r4-overtake.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r4-p1stale.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r4-restart-replay.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/review-r4-stale-lost.test.ts create mode 100644 pkgs/substrate-link/src/apps/link/test/workerd.integration.test.ts create mode 100644 pkgs/substrate-link/src/package.json create mode 100644 pkgs/substrate-link/src/pnpm-lock.yaml create mode 100644 pkgs/substrate-link/src/pnpm-workspace.yaml create mode 100644 pkgs/substrate-link/src/tsconfig.json diff --git a/pkgs/substrate-link/SYNC.md b/pkgs/substrate-link/SYNC.md new file mode 100644 index 000000000..f7a19de5d --- /dev/null +++ b/pkgs/substrate-link/SYNC.md @@ -0,0 +1,22 @@ +# pkgs/substrate-link/src: vendored source + +- Upstream: the ax-conwip repository (now named "substrate", Tom 2026-09-23 E10), local only at the time of + vendoring: `/home/tom/mecattaf/ax-conwip-wt-link`, branch `eval/2026-09-23-link`. +- Source sha: `a02100344205fcd20c93c0f9bdc87902adbbcd4e` (`a021003`), "link: final pass, fence every journaled + earlier attempt and re-time the heartbeat sleep". +- Taken with `git archive a021003 package.json pnpm-lock.yaml pnpm-workspace.yaml tsconfig.json apps/link`, unchanged. + The root `src/`, `test/` and `proto/` (the CONWIP scheduler) are not vendored: `apps/link` imports none of them. +- Verified upstream at that sha (REPORTED `evals-2026-09-23/link/LINK-FINAL-2026-09-23.md`): `tsc --noEmit` rc 0, + link suite 134 pass. +- The bundle uses `apps/link/proto/ax-p1.proto`, a superset of stock ax v0.3.0's API (it adds P1's + `GetTaskResult`). On this fleet ax carries no patch, so that RPC answers Unimplemented and the link's + `completion` defaults to `guest` (hosts/nas/substrate-link.nix). + +To resync: `git -C archive package.json pnpm-lock.yaml pnpm-workspace.yaml tsconfig.json +apps/link | tar -x -C pkgs/substrate-link/src`, update the sha above, then refresh `pnpmDeps.hash` in +`default.nix` (set it to `lib.fakeHash`, build, paste the `got:` value). + +## Unknowns and proposed defaults + +- When the substrate repository gets its remote (agency-agency/substrate), whether to switch this vendored copy to a + flake input. Default: keep the vendored copy (E1: the code that wraps ax lives in dotfiles) and resync by sha. diff --git a/pkgs/substrate-link/default.nix b/pkgs/substrate-link/default.nix new file mode 100644 index 000000000..e64c2a788 --- /dev/null +++ b/pkgs/substrate-link/default.nix @@ -0,0 +1,57 @@ +{ + lib, + stdenvNoCC, + nodejs_24, + pnpm_10, + esbuild, + makeWrapper, +}: +# substrate-link as one bundled file, built from the source vendored in ./src (SYNC.md names the +# upstream sha). Formerly conwip-link. +# N2: Node 24, the line the link suite ran on (24.18.0, 24.20.0 and 22.23.1 all 43/43 on 2026-09-23); the pinned +# nixpkgs b6c98e9e has nodejs_24 = 24.20.0 (read from its all-packages.nix and v24.nix). +# The bundling step itself was run by hand in the link worktree (esbuild, one 2,030,521 B .mjs that +# starts from / with only AX_CONWIP_PROTO_PATH set, drains on SIGTERM and exits 0). +stdenvNoCC.mkDerivation (finalAttrs: { + pname = "substrate-link"; + version = "0.1.0-unstable-2026-09-23"; + src = ./src; + + nativeBuildInputs = [ + nodejs_24 + pnpm_10.configHook + esbuild + makeWrapper + ]; + + pnpmDeps = pnpm_10.fetchDeps { + inherit (finalAttrs) pname version src; + # nix build with lib.fakeHash, then the "got:" value (MEASURED 2026-09-23). + hash = "sha256-7i4RxnkDbAV/tVZd8P+zXbhVAi7DfnZHDHq0BjnC0M4="; + fetcherVersion = 2; + }; + + buildPhase = '' + runHook preBuild + esbuild apps/link/src/main.ts --bundle --platform=node --format=esm --target=node22 \ + --outfile=dist/substrate-link.mjs \ + "--banner:js=import{createRequire as __cr}from'node:module';const require=__cr(import.meta.url);" + runHook postBuild + ''; + + installPhase = '' + runHook preInstall + install -Dm444 dist/substrate-link.mjs $out/lib/substrate-link/substrate-link.mjs + install -Dm444 apps/link/proto/ax-p1.proto $out/share/substrate-link/ax-p1.proto + makeWrapper ${lib.getExe nodejs_24} $out/bin/substrate-link \ + --add-flags $out/lib/substrate-link/substrate-link.mjs \ + --set LINK_AX_PROTO_PATH $out/share/substrate-link/ax-p1.proto + runHook postInstall + ''; + + meta = { + description = "Outbound-only link from the Cloudflare Substrate floor to ax on the NAS"; + mainProgram = "substrate-link"; + platforms = lib.platforms.linux; + }; +}) diff --git a/pkgs/substrate-link/src/apps/link/proto/ax-p1.proto b/pkgs/substrate-link/src/apps/link/proto/ax-p1.proto new file mode 100644 index 000000000..134a0d244 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/proto/ax-p1.proto @@ -0,0 +1,406 @@ +// RUNTIME PROTO of conwip-link (round 2): the vendored ax.proto (v0.3.0 d8ed0fe) plus the three hunks of carried patch +// P1 that the link reads: rpc GetTaskResult, messages GetTaskResultRequest and TaskResult, UsageStats.tool_calls. +// Copied from dotfiles branch ax/fleet-bringup pkgs/ax/patches/p1-completion.patch (lines 3240-3300) on 2026-09-23. +// A superset of the stock proto: against a stock server GetTaskResult reaches the server, which answers UNIMPLEMENTED, +// so the capability probe (B6) is decided by the server, never by the client's proto. Keep in sync with P1. +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +syntax = "proto3"; + +package ax.v1alpha1; + +import "google/protobuf/struct.proto"; +import "google/protobuf/timestamp.proto"; + +option go_package = "github.com/google/ax/pkg/apis/v1alpha1"; + +// AX defines the core control plane gRPC service for autonomous agent orchestration. +service AX { + // Manifests are parsed by the client (see `ax apply`) and submitted through the + // typed Update* RPCs below; the server never receives raw YAML. + + // Tasks + rpc GetTask(GetTaskRequest) returns (Task); + rpc ListTasks(ListTasksRequest) returns (ListTasksResponse); + rpc UpdateTask(UpdateTaskRequest) returns (Task); + rpc DeleteTask(DeleteTaskRequest) returns (DeleteTaskResponse); + rpc SuspendTask(SuspendTaskRequest) returns (Task); + rpc ResumeTask(ResumeTaskRequest) returns (Task); + rpc WatchTask(WatchTaskRequest) returns (stream WatchTaskResponse); + // GetTaskResult returns the result file the task's command wrote, as copied + // by the controller when the command exited. NotFound until then. + rpc GetTaskResult(GetTaskResultRequest) returns (TaskResult); + + // Gateways + rpc GetGateway(GetGatewayRequest) returns (Gateway); + rpc ListGateways(ListGatewaysRequest) returns (ListGatewaysResponse); + rpc UpdateGateway(UpdateGatewayRequest) returns (Gateway); + rpc DeleteGateway(DeleteGatewayRequest) returns (DeleteGatewayResponse); + + // Workspaces + rpc GetWorkspace(GetWorkspaceRequest) returns (Workspace); + rpc ListWorkspaces(ListWorkspacesRequest) returns (ListWorkspacesResponse); + rpc UpdateWorkspace(UpdateWorkspaceRequest) returns (Workspace); + rpc DeleteWorkspace(DeleteWorkspaceRequest) returns (DeleteWorkspaceResponse); + + // Models + rpc GetModel(GetModelRequest) returns (Model); + rpc ListModels(ListModelsRequest) returns (ListModelsResponse); + rpc UpdateModel(UpdateModelRequest) returns (Model); + rpc DeleteModel(DeleteModelRequest) returns (DeleteModelResponse); + +} + +// ObjectMeta is metadata for AX resources. +message ObjectMeta { + string name = 1; + string atespace = 2; + google.protobuf.Timestamp creation_timestamp = 3; +} + +// --- Task --- + +message Task { + string api_version = 1; + string kind = 2; + ObjectMeta metadata = 3; + TaskSpec spec = 4; + TaskStatus status = 5; +} + +message TaskSpec { + // Field 1 was `goal`, removed; the workspace goal lives on WorkspaceRef. + reserved 1; + reserved "goal"; + bool suspend = 2; + string image = 3; + repeated string command = 4; + repeated EnvVar env = 5; + ResourceReqs resources = 6; + // workspaces binds one or more Workspaces, each mounted at its own path + // under /workspace. The first entry is the task command's working directory. + repeated WorkspaceRef workspaces = 7; + GatewayRef gateway = 8; + // Field 9 was `policies` (budget and approval config), removed for now. + reserved 9; + reserved "policies"; + // debug enables the in-container guest services (process execution and file + // access) that back `ax ssh`. Off by default. + bool debug = 10; +} + +message EnvVar { + string name = 1; + string value = 2; +} + +message ResourceReqs { + ResourceList requests = 1; + ResourceList limits = 2; +} + +message ResourceList { + string cpu = 1; + string memory = 2; +} + +message WorkspaceRef { + string name = 1; + string path = 2; + string goal = 3; +} + +message GatewayRef { + string name = 1; +} + +message TaskStatus { + string phase = 1; + string id = 2; + string actor = 3; + // json_name keeps the established `workerIP` spelling in YAML and JSON. + string worker_ip = 4 [json_name = "workerIP"]; + PendingApproval pending_approval = 5; + UsageStats usage = 6; + repeated Condition conditions = 7; +} + +message PendingApproval { + string id = 1; + string action = 2; + google.protobuf.Timestamp requested_at = 3; +} + +message UsageStats { + int32 prompt_tokens = 1; + int32 completion_tokens = 2; + int32 tool_calls = 3; +} + +message Condition { + string type = 1; + string status = 2; + google.protobuf.Timestamp last_transition_time = 3; + string reason = 4; + string message = 5; +} + +// --- Gateway --- + +message Gateway { + string api_version = 1; + string kind = 2; + ObjectMeta metadata = 3; + GatewaySpec spec = 4; +} + +message GatewaySpec { + repeated Listener listeners = 1; + EgressConfig egress = 2; +} + +message Listener { + string name = 1; + int32 port = 2; + string protocol = 3; +} + +message EgressConfig { + EgressAllowlist allowlist = 1; +} + +message EgressAllowlist { + repeated HostRule hosts = 1; +} + +message HostRule { + string host = 1; + int32 port = 2; +} + +// --- Workspace --- + +message Workspace { + string api_version = 1; + string kind = 2; + ObjectMeta metadata = 3; + WorkspaceSpec spec = 4; +} + +message WorkspaceSpec { + repeated GitRepo git = 1; + MCPConfig mcp = 2; + SkillsConfig skills = 3; +} + +message GitRepo { + string name = 1; + string repo = 2; + string branch = 3; + string dir = 4; + int32 depth = 5; +} + +message MCPConfig { + repeated MCPRegistry registries = 1; + repeated MCPServer servers = 2; +} + +message MCPRegistry { + string provider = 1; + string project = 2; + string query = 3; + repeated MCPServer servers = 4; +} + +message MCPServer { + string name = 1; + string endpoint = 2; + string command = 3; + repeated string args = 4; +} + +message SkillsConfig { + repeated SkillRegistry registries = 1; + string path = 2; +} + +message SkillRegistry { + string provider = 1; + string project = 2; + string query = 3; +} + +// --- Model --- + +message Model { + string api_version = 1; + string kind = 2; + ObjectMeta metadata = 3; + ModelSpec spec = 4; +} + +message ModelSpec { + string provider = 1; + string model = 2; + // Fields 3-5 were typed temperature, max_tokens, and system_instruction; + // provider settings now live in the free-form `parameters` map. + reserved 3, 4, 5; + reserved "temperature", "max_tokens", "system_instruction"; + SecretKeyRef secret_key = 6; + // parameters are provider-specific generation settings passed through to the + // model API as-is, for example temperature or maxOutputTokens for Gemini. + google.protobuf.Struct parameters = 7; +} + +message SecretKeyRef { + string name = 1; + string key = 2; +} + +// --- RPC Request & Response Messages --- + +// Tasks +message GetTaskRequest { + string atespace = 1; + string name = 2; +} + +message GetTaskResultRequest { + string atespace = 1; + string name = 2; +} + +message TaskResult { + bytes content = 1; + string sha256 = 2; +} + +message ListTasksRequest { + string atespace = 1; + int64 limit = 2; + int64 offset = 3; +} + +message ListTasksResponse { + repeated Task tasks = 1; +} + +message UpdateTaskRequest { + Task task = 1; +} + +message DeleteTaskRequest { + string atespace = 1; + string name = 2; +} + +message DeleteTaskResponse {} + +message SuspendTaskRequest { + string atespace = 1; + string name = 2; +} + +message ResumeTaskRequest { + string atespace = 1; + string name = 2; +} + +message WatchTaskRequest { + string atespace = 1; + string name = 2; +} + +message WatchTaskResponse { + Task task = 1; + string action = 2; +} + +// Gateways +message GetGatewayRequest { + string atespace = 1; + string name = 2; +} + +message ListGatewaysRequest { + string atespace = 1; +} + +message ListGatewaysResponse { + repeated Gateway gateways = 1; +} + +message UpdateGatewayRequest { + Gateway gateway = 1; +} + +message DeleteGatewayRequest { + string atespace = 1; + string name = 2; +} + +message DeleteGatewayResponse {} + +// Workspaces +message GetWorkspaceRequest { + string atespace = 1; + string name = 2; +} + +message ListWorkspacesRequest { + string atespace = 1; +} + +message ListWorkspacesResponse { + repeated Workspace workspaces = 1; +} + +message UpdateWorkspaceRequest { + Workspace workspace = 1; +} + +message DeleteWorkspaceRequest { + string atespace = 1; + string name = 2; +} + +message DeleteWorkspaceResponse {} + +// Models +message GetModelRequest { + string atespace = 1; + string name = 2; +} + +message ListModelsRequest { + string atespace = 1; +} + +message ListModelsResponse { + repeated Model models = 1; +} + +message UpdateModelRequest { + Model model = 1; +} + +message DeleteModelRequest { + string atespace = 1; + string name = 2; +} + +message DeleteModelResponse {} + diff --git a/pkgs/substrate-link/src/apps/link/src/ax.ts b/pkgs/substrate-link/src/apps/link/src/ax.ts new file mode 100644 index 000000000..cdff0d5d9 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/src/ax.ts @@ -0,0 +1,121 @@ +// The link's view of ax: five calls over plain gRPC (h2c), no CreateTask (ax has none; UpdateTask is an upsert, so +// the link only ever calls it after GetTask answered NotFound: decision L4, create-only). +// ListTasks is paged: ax-server answers limit 0 with 50 rows, newest first (MEASURED upstream +// internal/server/server.go:91-104), so `listAll` walks pages until a short one (B1, critique C1). +import { createHash } from "node:crypto" +import { isIPv4, isIPv6 } from "node:net" +import { dirname, resolve } from "node:path" +import { fileURLToPath } from "node:url" +import * as grpc from "@grpc/grpc-js" +import * as protoLoader from "@grpc/proto-loader" +import { Effect } from "effect" +import type { AxTask } from "./contract.ts" + +export class AxError { + readonly _tag = "AxError" + constructor(readonly code: number, readonly message: string) {} + get unavailable() { return this.code === grpc.status.UNAVAILABLE || this.code === grpc.status.DEADLINE_EXCEEDED || this.code === grpc.status.INTERNAL } + get resourceExhausted() { return this.code === grpc.status.RESOURCE_EXHAUSTED } +} +export const AX_PAGE = 50 // ax-server's own default page size + +export interface Condition { readonly type?: string; readonly status?: string; readonly reason?: string; readonly message?: string } +export interface AxObserved { + readonly name: string + readonly phase: string // "" and Pending before the controller acts; Running; Completed/Failed (P1); Terminating + readonly spec: { image?: string; command?: ReadonlyArray; env?: ReadonlyArray<{ name?: string; value?: string }>; gateway?: { name?: string } | null } + readonly conditions: ReadonlyArray + readonly usage?: { readonly promptTokens?: number; readonly completionTokens?: number; readonly toolCalls?: number } | null +} +/** `digestOk` is false when the server's sha256 does not match the bytes received, undefined when it sent none (B4). */ +export interface AxResult { readonly content: string; readonly sha256: string; readonly digestOk?: boolean } +/** A Gateway's egress, as ax applies it (MEASURED upstream internal/controller/reconciler.go:188-198 and + * internal/substrate/client.go:450-462): no egress or no allowlist means `*:443`, an empty host list applies no + * policy, and a `*` or `0.0.0.0/0` host allows everything. */ +export interface AxGateway { readonly hosts: ReadonlyArray<{ readonly host: string; readonly port: number }>; readonly hasAllowlist: boolean } +/** Round 2: ax sends every host containing "/" to Substrate as a CIDR rule (MEASURED upstream client.go:456-491), so an + * allowlist can be open by arithmetic (`::/0`, `0.0.0.0/1` + `128.0.0.0/1`). Fail closed: a CIDR that does not parse, + * or one at or wider than /8 (IPv4) or /16 (IPv6), counts as open. */ +export const cidrIsOpen = (host: string): boolean => { + const [addr = "", len, ...rest] = host.split("/") + if (rest.length > 0 || len === undefined || !/^\d{1,3}$/.test(len)) return true + const n = Number(len) + if (isIPv4(addr)) return n > 32 || n <= 8 + if (isIPv6(addr)) return n > 128 || n <= 16 + return true +} +export const gatewayAllowsAll = (g: AxGateway) => + !g.hasAllowlist || g.hosts.length === 0 || g.hosts.some((x) => { + const h = x.host.trim().toLowerCase() + return h === "*" || h === "" || (h.includes("/") && cidrIsOpen(h)) + }) + +/** Round 2: the link's own runtime proto, P1 included (see its header). `LINK_AX_PROTO_PATH` overrides it. */ +export const LINK_PROTO_PATH = resolve(dirname(fileURLToPath(import.meta.url)), "..", "proto", "ax-p1.proto") + +export interface AxApi { + readonly getTask: (name: string) => Effect.Effect // NotFound -> undefined + readonly createTask: (task: AxTask) => Effect.Effect // UpdateTask, called only after NotFound + /** One page of the executor's list (Buildkite informer); `limit` 0 means the server's 50. */ + readonly listTasks: (limit: number, offset: number) => Effect.Effect, AxError> + readonly getGateway: (name: string) => Effect.Effect // NotFound -> undefined + readonly deleteTask: (name: string) => Effect.Effect // NotFound counts as done + /** P1's GetTaskResult. `unimplemented` means the server predates P1 (stock v0.3.0): completion is the guest's (L7). */ + readonly getTaskResult: (name: string) => Effect.Effect +} + +/** Every page, until a short one. Rows can shift between pages while ax inserts or deletes; the link therefore + * confirms any "absent" Task with GetTask before acting on it (B1). */ +export const listAll = (ax: AxApi, page = AX_PAGE, maxPages = 400): Effect.Effect, AxError> => Effect.gen(function*() { + const out = new Map() + for (let i = 0; i < maxPages; i++) { + const rows = yield* ax.listTasks(page, i * page) + for (const t of rows) out.set(t.name, t) + if (rows.length < page) return [...out.values()] + } + return yield* Effect.fail(new AxError(grpc.status.OUT_OF_RANGE, `more than ${maxPages * page} Tasks in the atespace`)) +}) + +const observe = (t: any): AxObserved => ({ + name: t?.metadata?.name ?? "", + phase: t?.status?.phase ?? "", + spec: t?.spec ?? {}, + conditions: t?.status?.conditions ?? [], + usage: t?.status?.usage ?? null +}) + +/** The production AxApi: `address` is ax-server's `host:port` (the ClusterIP 10.201.0.80:8080 on the NAS). */ +export function grpcAx(address: string, atespace: string, deadlineMs = 10_000, protoPath: string = LINK_PROTO_PATH): AxApi & { close: () => void } { + const def = protoLoader.loadSync(protoPath, { keepCase: false, longs: String, enums: String, defaults: true, oneofs: true }) + const Ctor = (grpc.loadPackageDefinition(def) as any).ax?.v1alpha1?.AX + if (typeof Ctor !== "function") throw new Error(`ax.v1alpha1.AX not found in ${protoPath}`) + // Round 2: a proto without GetTaskResult would answer the P1 probe "unimplemented" locally, never asking the server + if (typeof Ctor.prototype?.GetTaskResult !== "function") throw new Error(`${protoPath} declares no GetTaskResult; the link needs its P1 proto`) + const client = new Ctor(address, grpc.credentials.createInsecure()) + const unary = (method: string, req: unknown) => Effect.callback((resume) => { + if (typeof client[method] !== "function") return resume(Effect.fail(new AxError(grpc.status.UNIMPLEMENTED, `${method} not in proto`))) + client[method](req, { deadline: new Date(Date.now() + deadlineMs) }, (err: grpc.ServiceError | null, res: T) => + resume(err ? Effect.fail(new AxError(err.code ?? grpc.status.UNKNOWN, err.details ?? err.message)) : Effect.succeed(res))) + }) + const notFound = (e: Effect.Effect, dflt: A) => + e.pipe(Effect.catchIf((x: AxError) => x.code === grpc.status.NOT_FOUND, () => Effect.succeed(dflt))) + return { + getTask: (name) => notFound(unary("GetTask", { atespace, name }).pipe(Effect.map(observe)), undefined), + createTask: (task) => unary("UpdateTask", { task }).pipe(Effect.asVoid), + listTasks: (limit, offset) => unary("ListTasks", { atespace, limit, offset }).pipe(Effect.map((r) => (r?.tasks ?? []).map(observe))), + getGateway: (name) => notFound(unary("GetGateway", { atespace, name }).pipe(Effect.map((g): AxGateway => { + const al = g?.spec?.egress?.allowlist + return { hasAllowlist: al != null, hosts: (al?.hosts ?? []).map((h: any) => ({ host: String(h?.host ?? ""), port: Number(h?.port ?? 0) })) } + })), undefined), + deleteTask: (name) => notFound(unary("DeleteTask", { atespace, name }).pipe(Effect.asVoid), undefined), + getTaskResult: (name) => notFound(unary("GetTaskResult", { atespace, name }).pipe( + Effect.map((r): AxResult => { + const bytes = Buffer.from(r?.content ?? "") + const sha256 = String(r?.sha256 ?? "") + return { content: bytes.toString("utf8"), sha256, ...(sha256 ? { digestOk: createHash("sha256").update(bytes).digest("hex") === sha256.toLowerCase() } : {}) } + }), + Effect.catchIf((x: AxError) => x.code === grpc.status.UNIMPLEMENTED, () => Effect.succeed("unimplemented" as const)) + ), undefined), + close: () => client.close() + } +} diff --git a/pkgs/substrate-link/src/apps/link/src/config.ts b/pkgs/substrate-link/src/apps/link/src/config.ts new file mode 100644 index 000000000..24198b656 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/src/config.ts @@ -0,0 +1,103 @@ +// conwip-link configuration from the environment (round 1, finding "config"). Every numeric key must be a plain +// non-negative integer inside its range and every enum key one of its literals; anything else is ConfigInvalid, which +// main.ts turns into exit 78 so systemd's RestartPreventExitStatus holds the unit down instead of letting it spin +// (Effect.sleep(NaN) is a zero sleep) or crash-loop. Errors name the key and the rule, never the value. + +export class ConfigInvalid extends Error { + readonly _tag = "ConfigInvalid" + constructor(readonly key: string, readonly why: string) { super(`${key}: ${why}`) } +} + +type Env = Readonly> + +const str = (env: Env, key: string, d?: string): string => { + const v = env[key] + if (v !== undefined && v !== "") return v + if (d !== undefined) return d + throw new ConfigInvalid(key, "missing") +} + +export const int = (env: Env, key: string, d: number, min: number, max = 1_000_000): number => { + const v = str(env, key, String(d)).trim() + if (!/^\d+$/.test(v)) throw new ConfigInvalid(key, "not a non-negative integer") + const n = Number(v) + if (n < min || n > max) throw new ConfigInvalid(key, `outside ${min}..${max}`) + return n +} + +export const literal = (env: Env, key: string, d: L, allowed: ReadonlyArray): L => { + const v = str(env, key, d) + if (!(allowed as ReadonlyArray).includes(v)) throw new ConfigInvalid(key, `not one of ${allowed.join(", ")}`) + return v as L +} + +export const seatCommands = (env: Env, key: string, d: string): Record> => { + let parsed: unknown + try { parsed = JSON.parse(str(env, key, d)) } catch { throw new ConfigInvalid(key, "not JSON") } + if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) throw new ConfigInvalid(key, "not a JSON object") + const out: Record> = {} + for (const [seat, cmd] of Object.entries(parsed)) { + if (!Array.isArray(cmd) || cmd.length === 0 || !cmd.every((c) => typeof c === "string" && c !== "")) + throw new ConfigInvalid(key, `seat ${JSON.stringify(seat)} is not a non-empty array of non-empty strings`) + out[seat] = cmd as Array + } + return out +} + +export interface LinkEnv { + readonly holder: string + readonly maxInFlight: number + readonly servedLabels: Array + readonly atespace: string + readonly image: string + readonly gateway: string + readonly axServer: string + readonly guestCompleteUrl: string | undefined + readonly seatCommands: Record> + readonly completion: "auto" | "p1" | "guest" + readonly resyncSeconds: number + readonly pendingTimeoutSeconds: number + readonly deleteAfterSeconds: number + readonly deadlineBackstopSeconds: number + readonly createAttempts: number + readonly outboxBackoffSeconds: readonly [number, number] + readonly fenceTimeoutSeconds: number + readonly resultReadTries: number + readonly verdictAttempts: number // round 2 + readonly maxOutputBytes: number // round 2: below the floor's 2 MB DO row limit + readonly internalHosts: Array // round 2: single-label names a guest Complete URL may use + readonly axProtoPath: string | undefined // round 2: the link's ax proto (must declare GetTaskResult) +} + +/** Throws ConfigInvalid on the first bad key. Defaults are the NixOS module's. */ +export const readLinkEnv = (env: Env): LinkEnv => { + const outMin = int(env, "LINK_OUTBOX_BACKOFF_MIN_SECONDS", 60, 1, 86_400) + const outMax = int(env, "LINK_OUTBOX_BACKOFF_MAX_SECONDS", 900, 1, 86_400) + if (outMax < outMin) throw new ConfigInvalid("LINK_OUTBOX_BACKOFF_MAX_SECONDS", "below LINK_OUTBOX_BACKOFF_MIN_SECONDS") + const servedLabels = str(env, "LINK_SERVED_LABELS", "seat:halogen,runtime:gvisor").split(",").map((l) => l.trim()).filter((l) => l !== "") + if (servedLabels.length === 0) throw new ConfigInvalid("LINK_SERVED_LABELS", "empty") + return { + holder: str(env, "LINK_HOLDER", "nas-link-1"), + maxInFlight: int(env, "LINK_MAX_IN_FLIGHT", 2, 1, 1000), + servedLabels, + atespace: str(env, "AX_ATESPACE", "fleet"), + image: str(env, "LINK_IMAGE", "ax-agent"), + gateway: str(env, "LINK_GATEWAY", "halogen"), + axServer: str(env, "AX_SERVER", "10.201.0.80:8080"), + guestCompleteUrl: env.LINK_GUEST_COMPLETE_URL ? env.LINK_GUEST_COMPLETE_URL : undefined, + seatCommands: seatCommands(env, "LINK_SEAT_COMMANDS", '{"halogen":["ax-agent","pi"]}'), + completion: literal(env, "LINK_COMPLETION", "auto", ["auto", "p1", "guest"]), + resyncSeconds: int(env, "LINK_RESYNC_SECONDS", 15, 1, 3600), + pendingTimeoutSeconds: int(env, "LINK_PENDING_TIMEOUT_SECONDS", 900, 1, 86_400), + deleteAfterSeconds: int(env, "LINK_DELETE_AFTER_SECONDS", 600, 0, 604_800), + deadlineBackstopSeconds: int(env, "LINK_DEADLINE_BACKSTOP_SECONDS", 300, 0, 86_400), + createAttempts: int(env, "LINK_CREATE_ATTEMPTS", 7, 1, 100), + outboxBackoffSeconds: [outMin, outMax], + fenceTimeoutSeconds: int(env, "LINK_FENCE_TIMEOUT_SECONDS", 120, 1, 86_400), + resultReadTries: int(env, "LINK_RESULT_READ_TRIES", 5, 1, 1000), + verdictAttempts: int(env, "LINK_VERDICT_ATTEMPTS", 8, 1, 1000), + maxOutputBytes: int(env, "LINK_MAX_OUTPUT_BYTES", 1_000_000, 1024, 2_000_000), + internalHosts: (env.LINK_INTERNAL_HOSTS ?? "").split(",").map((h) => h.trim()).filter((h) => h !== ""), + axProtoPath: env.LINK_AX_PROTO_PATH ? env.LINK_AX_PROTO_PATH : undefined + } +} diff --git a/pkgs/substrate-link/src/apps/link/src/contract.ts b/pkgs/substrate-link/src/apps/link/src/contract.ts new file mode 100644 index 000000000..40e93d8b1 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/src/contract.ts @@ -0,0 +1,181 @@ +// The floor <-> link wire contract (LINK-DESIGN.md section 3). Starting point: the option-A prototype +// (/home/tom/today/evals-2026-09-23/arc/proto/contract.ts, decision L1). Every L1 name is kept as it was. +// Five keys are ADDED, each optional so an L1-only peer ignores them, each named after its source: +// Grant.supersedes L3 (two-stage expiry): the old attempt the link must DeleteTask first +// Grant.leaseToken L7 (completion before ax P1): a per-lease token for the guest (Buildkite job token) +// Grant.deadline Temporal start-to-close, as an absolute epoch ms the link can hold across a restart +// Lease.endpoint Buildkite ping `endpoint`: move the link to a new floor URL after a test call +// Complete.withdrew round 2: the leaseId of attempt n+1 that rule 4b withdrew when it accepted attempt n's verdict. +// The link releases n+1 only when it is named here; it never infers a withdrawal (a duplicate +// answer or a delivered retryable failure withdrew nothing) +// Complete.withdrewTransitions round 3: the leaseTransitions of the generation rule 4b withdrew. A floor that +// requeues the withdrawn attempt under the same leaseId grants it again with a higher value; the +// link treats a grant as withdrawn only when its generation matches +// Nothing here is GitHub-hosted at runtime. +import { Schema } from "effect" +import { Rpc, RpcGroup } from "effect/unstable/rpc" + +// Actions job result values (GitHub/Gitea job `result`). +export const Result = Schema.Literals(["success", "failure", "cancelled", "skipped"]) +export type Result = typeof Result.Type + +// ---- AgentJob: an ultracode agent() node as an Actions-style job in a Kubernetes object envelope (unchanged) ---- +const Labels = Schema.Record(Schema.String, Schema.String) +export const PromptRef = Schema.Struct({ + sha256: Schema.String, // content address; the guest refuses on mismatch (FIELD-MAP 5a AX_CONWIP_PROMPT_SHA256) + bytes: Schema.Int, + uri: Schema.String // journal:////prompt.md +}) +export const AgentJob = Schema.Struct({ + apiVersion: Schema.Literal("ultracode.mecattaf.dev/v1alpha1"), + kind: Schema.Literal("AgentJob"), + metadata: Schema.Struct({ name: Schema.String, labels: Schema.optionalKey(Labels), annotations: Schema.optionalKey(Labels) }), + spec: Schema.Struct({ + // CONWIP admission key, e.g. ["seat:halogen","runtime:gvisor"]. B7 asked for Schema.NonEmptyArray here; it stays + // Array on the wire because one malformed job would then fail the decode of a whole Lease reply and strand every + // grant in it (INFERRED). The link refuses it per job instead (`validRunsOn`, jobs.ts), and the floor at enqueue. + "runs-on": Schema.Array(Schema.String), + "timeout-minutes": Schema.optionalKey(Schema.Int), // Temporal start-to-close, enforced by the floor + "continue-on-error": Schema.optionalKey(Schema.Boolean), + environment: Schema.optionalKey(Schema.NullOr(Schema.String)), // Tom's ack gate + with: Schema.Struct({ + prompt: Schema.optionalKey(Schema.String), + prompt_ref: PromptRef, + model: Schema.String, + effort: Schema.optionalKey(Schema.String), + schema: Schema.optionalKey(Schema.Unknown), + isolation: Schema.optionalKey(Schema.String), + "agent-type": Schema.optionalKey(Schema.String) + }), + needs: Schema.optionalKey(Schema.Record(Schema.String, Schema.Struct({ result: Result, outputs_ref: Schema.String }))), + outputs: Schema.optionalKey(Schema.Array(Schema.String)) + }) +}) +export type AgentJob = typeof AgentJob.Type + +// ---- ax Task JSON (protojson camelCase of ax.proto v0.3.0 d8ed0fe). The link emits it; it never rides the wire ---- +// apiVersion is ax's own constant `ax.io/v1alpha1` (ax pkg/apis/v1alpha1/types.go:31); the prototype's +// `ax/v1alpha1` passed only because ax stores what it is given. +export const AxTask = Schema.Struct({ + apiVersion: Schema.Literal("ax.io/v1alpha1"), + kind: Schema.Literal("Task"), + metadata: Schema.Struct({ name: Schema.String, atespace: Schema.String }), + spec: Schema.Struct({ + image: Schema.String, + command: Schema.Array(Schema.String), + env: Schema.Array(Schema.Struct({ name: Schema.String, value: Schema.String })), + gateway: Schema.optionalKey(Schema.Struct({ name: Schema.String })) + }) +}) +export type AxTask = typeof AxTask.Type + +// ---- the link contract ---- +// Kubernetes coordination.k8s.io/v1 LeaseSpec field names. +export const LeaseSpec = Schema.Struct({ + holderIdentity: Schema.String, leaseDurationSeconds: Schema.Int, + acquireTime: Schema.Number, renewTime: Schema.Number, leaseTransitions: Schema.Int +}) +// Absolute counts on every response plus an increasing seq (ARC scaleset statistics). +export const Stats = Schema.Struct({ seq: Schema.Int, cap: Schema.Int, wip: Schema.Int, queued: Schema.Int, done: Schema.Int }) +export const Grant = Schema.Struct({ + leaseId: Schema.String, // `-a`; it is also the ax Task name (the idempotency key on ax) + attempt: Schema.Int, + job: AgentJob, + lease: LeaseSpec, + supersedes: Schema.optionalKey(Schema.Array(Schema.String)), // L3 + leaseToken: Schema.optionalKey(Schema.String), // L7, only while the guest completes (pre-P1) + deadline: Schema.optionalKey(Schema.Number) // epoch ms; acquireTime + timeout-minutes +}) +export type Grant = typeof Grant.Type +export const Usage = Schema.Struct({ prompt_tokens: Schema.Int, completion_tokens: Schema.Int, tool_calls: Schema.Int }) +export type Usage = typeof Usage.Type + +// Temporal PollActivityTaskQueue + ARC free capacity on every poll + Forgejo request key + Buildkite intervals. +export const Lease = Rpc.make("Lease", { + payload: { holderIdentity: Schema.String, capacity: Schema.Int, requestKey: Schema.String }, + success: Schema.Struct({ + grants: Schema.Array(Grant), stats: Stats, nextPollSeconds: Schema.Int, heartbeatSeconds: Schema.Int, + endpoint: Schema.optionalKey(Schema.String) // Buildkite ping endpoint switch + }) +}) +// Temporal RecordActivityTaskHeartbeat with cancel_requested, batched per session. `leaseIds` is the COMPLETE set +// the holder believes it holds (level-triggered, ARC): an orphaned lease it omits is requeued (L3 a). +// Round 4: `pendingRequestKey` (Forgejo request key, B8) is the Lease requestKey the holder journaled and has not +// seen answered. The holder cannot list leaseIds it never received, so it vouches for the key instead: the floor +// renews every lease granted under that key and never releases one as an omitted orphan (L3 a) while it is named. +export const Heartbeat = Rpc.make("Heartbeat", { + payload: { holderIdentity: Schema.String, leaseIds: Schema.Array(Schema.String), pendingRequestKey: Schema.optionalKey(Schema.String) }, + success: Schema.Struct({ + renewed: Schema.Array(Schema.String), lost: Schema.Array(Schema.String), + cancelRequested: Schema.Array(Schema.String), stats: Stats + }) +}) +// Temporal RespondActivityTask{Completed,Failed,Canceled}. The verdict travels here, never as an exit code (ARC). +// The two errors are terminal: stop retrying (Buildkite 422). +export const CompleteError = Schema.Struct({ code: Schema.Literals(["unknown-lease", "stale-attempt"]) }) +export const Complete = Rpc.make("Complete", { + payload: { + leaseId: Schema.String, attempt: Schema.Int, result: Result, + output: Schema.optionalKey(Schema.Unknown), usage: Schema.optionalKey(Usage) + }, + success: Schema.Struct({ duplicate: Schema.Boolean, stats: Stats, withdrew: Schema.optionalKey(Schema.String), + withdrewTransitions: Schema.optionalKey(Schema.Int) }), + error: CompleteError +}) +export class FloorLink extends RpcGroup.make(Lease, Heartbeat, Complete) {} + +// The link's own view of the same wire (round 1, finding "poison grant"): identical tags, payloads and JSON, but each +// grant is decoded by the link one at a time, so one malformed job fails only its own lease (pre-start/invalid-spec) +// instead of turning the whole Lease reply into a decode defect that the journaled requestKey would replay for ever. +// Round 3: the whole envelope is lenient too. A drifted interval (the prototype floor sends leaseMs / 3000, e.g. +// 3.333), a null endpoint or a changed stats shape would otherwise fail the decode of every replay of the journaled +// requestKey and strand the grants in it. The link bounds the intervals itself (link.ts `boundedSeconds`). +export const LeaseLenient = Rpc.make("Lease", { + payload: { holderIdentity: Schema.String, capacity: Schema.Int, requestKey: Schema.String }, + success: Schema.Struct({ + grants: Schema.Array(Schema.Unknown), stats: Schema.optionalKey(Schema.Unknown), + nextPollSeconds: Schema.optionalKey(Schema.Unknown), heartbeatSeconds: Schema.optionalKey(Schema.Unknown), + endpoint: Schema.optionalKey(Schema.Unknown) + }) +}) +// Round 2: the link decodes Complete's error code as any string, so a code the floor adds later is a permanent refusal +// (dropped, logged with the code), never a decode defect retried for ever at the head of the outbox. +// Round 3: its success is lenient as well. Any 2xx Exit Success means the floor applied the verdict; a drifted key +// (`withdrew: null`, a missing stats) must not keep an applied verdict at the head of the outbox. +export const CompleteLenient = Rpc.make("Complete", { + payload: Complete.payloadSchema.fields, + success: Schema.Struct({ duplicate: Schema.optionalKey(Schema.Unknown), stats: Schema.optionalKey(Schema.Unknown), + withdrew: Schema.optionalKey(Schema.Unknown), withdrewTransitions: Schema.optionalKey(Schema.Unknown) }), + error: Schema.Struct({ code: Schema.String }) +}) +// Round 4: Heartbeat is lenient as well. The strict reply schema turned a drifted `stats` into a heartbeat-error on +// every call, so `cancelRequested`, `lost` and `renewed` were never read while the floor kept renewing (a cancel was +// never carried out, a journaled grant never resumed). Every key is optional Unknown; floor.ts keeps the strings. +export const HeartbeatLenient = Rpc.make("Heartbeat", { + payload: Heartbeat.payloadSchema.fields, + success: Schema.Struct({ renewed: Schema.optionalKey(Schema.Unknown), lost: Schema.optionalKey(Schema.Unknown), + cancelRequested: Schema.optionalKey(Schema.Unknown), stats: Schema.optionalKey(Schema.Unknown) }) +}) +export class FloorLinkClient extends RpcGroup.make(LeaseLenient, HeartbeatLenient, CompleteLenient) {} + +// On result "failure" the link's `output` is this shape (Buildkite finish signal_reason, ARC failure reason). +// The reasons in RETRYABLE are the floor's to retry as attempt+1 (Temporal retry policy, Buildkite finish -1 on a +// pre-start failure); every other reason is final: a deterministic refusal, or the agent's own failure, which goes +// to the interpreter (null-on-failure). +export const FailureOutput = Schema.Struct({ + reason: Schema.String, // pre-start/env-over-budget, pre-start/name-conflict, infra/task-lost, deadline-exceeded, ... + message: Schema.optionalKey(Schema.String), + phase: Schema.optionalKey(Schema.String), + exitCode: Schema.optionalKey(Schema.Int) +}) +export type FailureOutput = typeof FailureOutput.Type +export const RETRYABLE = new Set([ + "pre-start/ax-unavailable", "pre-start/pending-timeout", "pre-start/resource-exhausted", + "infra/task-lost", "infra/actor-crashed", "infra/result-unreadable", + "pre-start/superseded-verdict-pending", // round 1: n+1 refused while attempt n's verdict could not be delivered + "infra/link-defect" // round 4: a dispatch fiber died on a defect before the Task was created; the floor requeues +]) +// Round 2, final (not in RETRYABLE): `infra/verdict-undeliverable` (the floor answered every delivery of the real +// verdict with a server error; a rerun would most likely produce the same bytes) and `agent/output-too-large` (the +// result is over the link's cap, below the floor's 2 MB row limit). +export const isRetryableReason = (reason: string) => RETRYABLE.has(reason) diff --git a/pkgs/substrate-link/src/apps/link/src/floor.ts b/pkgs/substrate-link/src/apps/link/src/floor.ts new file mode 100644 index 000000000..0884c3192 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/src/floor.ts @@ -0,0 +1,169 @@ +// The link's view of the floor: the three L1 RPCs over Effect's stock HTTP RPC client (decision L2), outbound only. +// HTTP statuses the floor's Worker entry answers before any RPC runs are classified here (Buildkite's retry +// classifier, api/retryable.go): 401/403 fatal (token), 409 fatal (a second session for this identity, L5), +// 429 back off honouring Retry-After. Round 2: any 3xx is fatal (`redirect`: redirects are never followed, so the only +// way to another floor URL is the Lease `endpoint` checked against the Nix list, B18); 404 and 405 are fatal +// (`misrouted`: a wrong route answers every call so, and dropping verdicts for it would lose them); any other 4xx except 408 is +// `rejected` (permanent for these bytes: Complete drops the verdict); a 500, 501 or 505+ and an RPC Defect in the reply +// are `server-error` (the floor answered and refused these bytes; the outbox counts them per verdict); 502, 503, 504, +// 408, network errors and timeouts stay `transient`. Every call has a ceiling (B9). +import { Effect, Layer, Schema, Scope } from "effect" +import { FetchHttpClient, HttpClient, HttpClientRequest } from "effect/unstable/http" +import { RpcClient, RpcSerialization } from "effect/unstable/rpc" +import { Complete, FloorLinkClient, Grant } from "./contract.ts" +import type { Usage, Result } from "./contract.ts" + +/** A grant the link could not decode. `leaseId` and `attempt` are set when they alone decode, so the lease can be + * failed back to the floor as pre-start/invalid-spec; otherwise the floor's own expiry is the only way out. */ +export interface InvalidGrant { readonly leaseId?: string; readonly attempt?: number; readonly message: string } +/** Round 3: the reply as the link uses it, after the lenient envelope decode. Intervals and endpoint are kept only + * when they have the right JSON type; the link bounds the intervals (link.ts `boundedSeconds`). */ +export type LeaseReply = { + readonly grants: ReadonlyArray; readonly invalid?: ReadonlyArray + readonly stats?: unknown; readonly nextPollSeconds?: number; readonly heartbeatSeconds?: number; readonly endpoint?: string +} +/** Round 4: the Heartbeat reply after the lenient decode: each list keeps only its strings, a missing one is []. */ +type HeartbeatReply = { readonly renewed: ReadonlyArray; readonly lost: ReadonlyArray; readonly cancelRequested: ReadonlyArray; readonly stats?: unknown } +const strings = (v: unknown): Array => Array.isArray(v) ? v.filter((x): x is string => typeof x === "string") : [] +const heartbeatReply = (r: { readonly renewed?: unknown; readonly lost?: unknown; readonly cancelRequested?: unknown; readonly stats?: unknown }): HeartbeatReply => + ({ renewed: strings(r.renewed), lost: strings(r.lost), cancelRequested: strings(r.cancelRequested), stats: r.stats }) +type CompleteReply = { readonly duplicate: boolean; readonly withdrew?: string; readonly withdrewTransitions?: number; readonly stats?: unknown } +const num = (v: unknown) => typeof v === "number" ? v : undefined +const leaseReply = (r: { readonly grants: ReadonlyArray; readonly nextPollSeconds?: unknown; readonly heartbeatSeconds?: unknown; readonly endpoint?: unknown; readonly stats?: unknown }): LeaseReply => { + const s = splitGrants(r) + const poll = num(r.nextPollSeconds), hb = num(r.heartbeatSeconds) + return { grants: s.grants, invalid: s.invalid, stats: r.stats, ...(poll !== undefined ? { nextPollSeconds: poll } : {}), + ...(hb !== undefined ? { heartbeatSeconds: hb } : {}), ...(typeof r.endpoint === "string" ? { endpoint: r.endpoint } : {}) } +} +const completeReply = (r: { readonly duplicate?: unknown; readonly withdrew?: unknown; readonly withdrewTransitions?: unknown; readonly stats?: unknown }): CompleteReply => { + const wt = num(r.withdrewTransitions) + return { duplicate: r.duplicate === true, stats: r.stats, ...(typeof r.withdrew === "string" ? { withdrew: r.withdrew } : {}), + ...(wt !== undefined && Number.isInteger(wt) ? { withdrewTransitions: wt } : {}) } +} +const encodeComplete = Schema.encodeUnknownExit(Complete.payloadSchema) + +export type FloorErrorKind = "transient" | "auth" | "session-conflict" | "rate-limited" | "redirect" | "misrouted" | "rejected" | "server-error" +export class FloorError { + readonly _tag = "FloorError" + constructor(readonly kind: FloorErrorKind, readonly message: string, readonly retryAfterMs?: number, readonly status?: number) {} + get fatal() { return this.kind === "auth" || this.kind === "session-conflict" || this.kind === "redirect" || this.kind === "misrouted" } +} +/** The floor refused these bytes for good: one of its CompleteError codes (any string, round 2), or an HTTP 4xx. */ +export class CompleteRefused { + readonly _tag = "CompleteRefused" + constructor(readonly code: string) {} +} + +export interface CompletePayload { leaseId: string; attempt: number; result: Result; output?: unknown; usage?: Usage } +export interface FloorApi { + readonly lease: (p: { holderIdentity: string; capacity: number; requestKey: string }) => Effect.Effect + readonly heartbeat: (p: { holderIdentity: string; leaseIds: ReadonlyArray; pendingRequestKey?: string }) => Effect.Effect + readonly complete: (p: CompletePayload) => Effect.Effect +} + +export interface FloorOptions { + readonly url: string // e.g. https://conwip-floor..workers.dev (no trailing /rpc) + readonly token: string // the per-link bearer, read from $CREDENTIALS_DIRECTORY; never logged + readonly sessionId: string // stable per state directory: a restart is the same session, a second replica is not + readonly fetch?: typeof globalThis.fetch // tests hand in an in-process handler; production uses globalThis.fetch + /** B9: a ceiling on every call (Buildkite's 330 s on acquire, shortened): Lease and Heartbeat 20 s, Complete 60 s. */ + readonly timeoutsMs?: { readonly lease: number; readonly heartbeat: number; readonly complete: number } +} +export const DEFAULT_FLOOR_TIMEOUTS_MS = { lease: 20_000, heartbeat: 20_000, complete: 60_000 } as const + +/** B9 (critique R9): a non-2xx answer is thrown by the fetch wrapper of the call that received it, so its status + * travels inside that call's own error chain; nothing is shared between concurrent calls. */ +class HttpStatus extends Error { + readonly linkHttpStatus = true + constructor(readonly status: number, readonly retryAfterMs: number | undefined) { super(`floor answered ${status}`) } +} +const statusIn = (e: unknown): HttpStatus | undefined => { + let x: any = e + for (let i = 0; i < 10 && x != null && typeof x === "object"; i++) { + if (x instanceof HttpStatus || x.linkHttpStatus === true) return x as HttpStatus + x = x.reason ?? x.cause ?? x.error + } + return undefined +} +const decodeGrant = Schema.decodeUnknownExit(Grant) +const Envelope = Schema.decodeUnknownExit(Schema.Struct({ leaseId: Schema.String, attempt: Schema.Int })) +/** Decode each grant on its own: the good ones are dispatched, the bad ones are refused one by one. */ +export const splitGrants = }>(r: R): Omit & { grants: Array; invalid: Array } => { + const grants: Array = [], invalid: Array = [] + for (const raw of r.grants) { + const g = decodeGrant(raw) + if (g._tag === "Success") { grants.push(g.value); continue } + const env = Envelope(raw) + const message = String(g.cause).slice(0, 500) + invalid.push(env._tag === "Success" ? { leaseId: env.value.leaseId, attempt: env.value.attempt, message } : { message }) + } + return { ...r, grants, invalid } +} + +export const classifyFloorError = (e: unknown): FloorError => { + const st = statusIn(e) + const s = st?.status ?? 0 + const msg = e instanceof Error ? e.message : String((e as any)?.message ?? e) + if (s >= 300 && s < 400) return new FloorError("redirect", `floor answered ${s}: redirects are never followed; fix LINK_FLOOR_URL`, undefined, s) + if (s === 401 || s === 403) return new FloorError("auth", `floor answered ${s}`, undefined, s) + if (s === 409) return new FloorError("session-conflict", "floor answered 409: another session holds this link identity", undefined, s) + if (s === 429) return new FloorError("rate-limited", "floor answered 429", st?.retryAfterMs, s) + // 404/405: the route is wrong for every call, not these bytes; fatal (exit 78), so no verdict is dropped for it + if (s === 404 || s === 405) return new FloorError("misrouted", `floor answered ${s}: fix LINK_FLOOR_URL or the Worker route`, undefined, s) + if (s >= 400 && s < 500 && s !== 408) return new FloorError("rejected", `floor answered ${s}`, undefined, s) + if (s >= 500 && s !== 502 && s !== 503 && s !== 504) return new FloorError("server-error", `floor answered ${s}`, undefined, s) + return new FloorError("transient", s ? `floor answered ${s}` : msg, undefined, s || undefined) +} +/** Round 3: a defect raised by a call is always the reply's: the link encodes the Complete payload itself before it + * sends (a failure there is `transient`, the link's own bug), and the envelopes it decodes are lenient. A defect is + * then the floor's RPC `Defect` reply (e.g. SQLITE_TOOBIG) or a reply that is not the protocol at all: the server + * refusing these bytes (`server-error`), counted per verdict on Complete and per requestKey on Lease. */ +export const rpcFloor = (o: FloorOptions): Effect.Effect => Effect.gen(function*() { + const base = o.fetch ?? globalThis.fetch + const t = o.timeoutsMs ?? DEFAULT_FLOOR_TIMEOUTS_MS + const observing: typeof globalThis.fetch = async (input, init) => { + // Round 2: never follow a redirect. A followed 307 lands on an origin the Nix list never declared, and the link + // would act on its replies (B18 bypass); an Access 302 to a login page would read as a transient decode error. + const res = await base(input, { ...init, redirect: "manual" }) + if (res.type === "opaqueredirect") throw new HttpStatus(307, undefined) + if (!res.ok) { + const ra = Number(res.headers.get("retry-after")) + throw new HttpStatus(res.status, Number.isFinite(ra) && ra > 0 ? ra * 1000 : undefined) + } + return res + } + const protocol = RpcClient.layerProtocolHttp({ + url: `${o.url.replace(/\/$/, "")}/rpc`, + transformClient: HttpClient.mapRequest((r) => r.pipe( + HttpClientRequest.bearerToken(o.token), + HttpClientRequest.setHeader("x-link-session", o.sessionId))) + }).pipe(Layer.provide([FetchHttpClient.layer, RpcSerialization.layerJson]), + Layer.provide(Layer.succeed(FetchHttpClient.Fetch, observing))) + const client = yield* RpcClient.make(FloorLinkClient).pipe(Effect.provide(protocol)) + // The timeout interrupts the call, which aborts its fetch (FetchHttpClient passes an AbortSignal). + const within = (ms: number, what: string) => (e: Effect.Effect) => + e.pipe(Effect.timeoutOrElse({ duration: ms, orElse: () => Effect.fail(new FloorError("transient", `${what} timed out after ${ms} ms`)) })) + return { + // Round 1: a reply that does not decode is a typed transient failure, never a defect that kills the link. + lease: (p) => client.Lease(p).pipe(Effect.catchDefect((d) => Effect.fail(new FloorError("server-error", `Lease reply undecodable: ${String(d).slice(0, 300)}`))), + Effect.mapError((e) => e instanceof FloorError ? e : classifyFloorError(e)), Effect.map(leaseReply), within(t.lease, "Lease")), + // Round 4: lenient like Lease and Complete; a reply that still does not decode is the server's (server-error). + heartbeat: (p) => client.Heartbeat({ holderIdentity: p.holderIdentity, leaseIds: p.leaseIds, ...(p.pendingRequestKey !== undefined ? { pendingRequestKey: p.pendingRequestKey } : {}) }).pipe( + Effect.catchDefect((d) => Effect.fail(new FloorError("server-error", `Heartbeat reply undecodable: ${String(d).slice(0, 300)}`))), + Effect.mapError((e) => e instanceof FloorError ? e : classifyFloorError(e)), Effect.map(heartbeatReply), within(t.heartbeat, "Heartbeat")), + // L9: omit absent optional keys; `usage: undefined` in an optionalKey field is an encode defect, not a failure. + complete: (p) => { + const payload = { leaseId: p.leaseId, attempt: p.attempt, result: p.result, + ...(p.output !== undefined ? { output: p.output } : {}), ...(p.usage !== undefined ? { usage: p.usage } : {}) } + const enc = encodeComplete(payload) + if (enc._tag === "Failure") return Effect.fail(new FloorError("transient", `encode defect: ${String(enc.cause).slice(0, 300)}`)) + return client.Complete(payload).pipe(Effect.map(completeReply), + Effect.catchDefect((d) => Effect.fail(new FloorError("server-error", `server defect: ${String((d as any)?.message ?? d).slice(0, 300)}`))), + Effect.mapError((e: any) => e instanceof FloorError ? e + : e && typeof e === "object" && "code" in e && !("_tag" in e) ? new CompleteRefused(String(e.code)) : classifyFloorError(e)), + // Round 2: a 4xx other than 401/403/408/409/429 refuses these bytes for good (Buildkite: not retryable) + Effect.mapError((e) => e instanceof FloorError && e.kind === "rejected" ? new CompleteRefused(`http-${e.status}`) : e), + within(t.complete, "Complete")) + } + } +}) diff --git a/pkgs/substrate-link/src/apps/link/src/jobs.ts b/pkgs/substrate-link/src/apps/link/src/jobs.ts new file mode 100644 index 000000000..c8b4753c1 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/src/jobs.ts @@ -0,0 +1,120 @@ +// Pure mapping AgentJob -> ax Task JSON (FIELD-MAP-2026-09-23.md 5a keys). Ported from the option-A prototype +// (arc/proto/jobs.ts) with ARC-DECISION X1 applied: atespace `fleet`, image `ax-agent`, Gateway `halogen`, the names +// ax-fleet DESIGN.md 10.2 uses. No I/O: the link and any converter call the same function. +import { createHash } from "node:crypto" +import type { AgentJob, AxTask, Grant } from "./contract.ts" + +export const ENV_BUDGET = 16384 // FIELD-MAP 5a hard pre-dispatch check (stock v0.3.0, before P5) +const A = "ultracode.mecattaf.dev/" +const utf8Len = (s: string) => new TextEncoder().encode(s).length +/** FIELD-MAP 5a journal key: `:` (ax-conwip taskspec.ts journalKey). */ +export const JOURNAL_KEY = /^[0-9a-f]{64}:\d+$/ + +export interface TaskShape { + readonly atespace: string // `fleet` + readonly image: string // `localhost:5000/ax-agent@sha256:...` (ax-fleet-image-ref) + readonly gateway: string // `halogen` + readonly command: (job: AgentJob) => ReadonlyArray | undefined // e.g. ["ax-agent", "pi"] for seat:halogen + readonly completeUrl?: string // set only in guest-completion mode (L7, pre-P1) + readonly holder?: string // round 1: AX_CONWIP_HOLDER, the ownership mark a journal-less link checks before a fence +} + +/** B7: no default seat. `undefined` unless `runs-on` names exactly one `seat:` label. */ +export const seatOf = (job: AgentJob): string | undefined => { + const seats = job.spec["runs-on"].filter((l) => l.startsWith("seat:")) + return seats.length === 1 ? seats[0]!.slice(5) : undefined +} + +/** B7: `runs-on` must be non-empty, name exactly one `seat:` label, and be wholly served by this link. */ +export const validRunsOn = (job: AgentJob, served: ReadonlyArray): { reason: string; message: string } | undefined => { + const on = job.spec["runs-on"] + if (on.length === 0) return { reason: "pre-start/runs-on-invalid", message: "runs-on is empty" } + const seats = on.filter((l) => l.startsWith("seat:")) + if (seats.length !== 1) return { reason: "pre-start/runs-on-invalid", message: `runs-on names ${seats.length} seat: labels, not 1` } + const unserved = on.filter((l) => !served.includes(l)) + if (unserved.length > 0) return { reason: "pre-start/label-not-served", message: unserved.join(",") } + return undefined +} + +export type Built = { readonly _tag: "ok"; readonly task: AxTask } | { readonly _tag: "refused"; readonly reason: string; readonly message: string } + +export function axTaskFromGrant(g: Grant, shape: TaskShape): Built { + const job = g.job + // Round 1 (fail closed): a guest Task without its per-lease token has no completion route; never create it. + if (shape.completeUrl !== undefined && (g.leaseToken === undefined || g.leaseToken === "")) + return { _tag: "refused", reason: "pre-start/lease-token-missing", message: "guest completion needs a leaseToken; the floor's token binding for this holder is not guest" } + const ann = job.metadata.annotations ?? {} + const lab = job.metadata.labels ?? {} + if (ann[A + "prompt-unresolved"] !== undefined) // L8: never dispatch an item whose prompt was not resolved + return { _tag: "refused", reason: "pre-start/prompt-unresolved", message: ann[A + "prompt-unresolved"]! } + // Round 4 (FIELD-MAP 5a): the journal key is sha256 of canonical [prompt, opts minus label] plus the occurrence, and + // only the floor-side converter can compute it (it holds opts and the occurrence). It rides the AgentJob as an + // annotation; without it the grant is refused, never replaced by the prompt digest (two calls sharing one prompt). + const journalKey = ann[A + "journal-key"] + if (journalKey === undefined || !JOURNAL_KEY.test(journalKey)) + return { _tag: "refused", reason: "pre-start/invalid-spec", message: journalKey === undefined ? `annotation ${A}journal-key is missing` : `annotation ${A}journal-key is not :` } + const w = job.spec.with + const seat = seatOf(job) + if (seat === undefined) return { _tag: "refused", reason: "pre-start/runs-on-invalid", message: "runs-on does not name exactly one seat: label" } + const command = shape.command(job) + if (command === undefined || command.length === 0) return { _tag: "refused", reason: "pre-start/seat-command-missing", message: `no command for seat ${seat}` } + const env: Array<{ name: string; value: string }> = [ + { name: "AX_CONWIP_RUN_ID", value: ann[A + "run-id-raw"] ?? "" }, + { name: "AX_CONWIP_LABEL", value: ann[A + "label"] ?? "" }, + { name: "AX_CONWIP_MODEL", value: w.model }, + { name: "AX_CONWIP_ITEM_KEY", value: ann[A + "item-key"] ?? "" }, + { name: "AX_CONWIP_JOURNAL_KEY", value: journalKey }, + { name: "AX_CONWIP_WORKFLOW", value: lab[A + "workflow"] ?? "" }, + { name: "AX_CONWIP_PHASE_INDEX", value: lab[A + "phase-index"] ?? "" }, + { name: "AX_CONWIP_PHASE_TITLE", value: ann[A + "phase-title"] ?? "" }, + { name: "AX_CONWIP_SEAT", value: seat }, + { name: "AX_CONWIP_ATTEMPT", value: String(g.attempt) }, + { name: "AX_CONWIP_LEASE_ID", value: g.leaseId }, + ...(shape.holder !== undefined ? [{ name: "AX_CONWIP_HOLDER", value: shape.holder }] : []), + { name: "AX_CONWIP_PROMPT_SHA256", value: w.prompt_ref.sha256 }, + { name: "AX_CONWIP_RESULT_PATH", value: "/workspace/.ax/result.json" }, + { name: "AX_CONWIP_USAGE_PATH", value: "/workspace/.ax/usage.json" } + ] + if (w.effort !== undefined) env.push({ name: "AX_CONWIP_EFFORT", value: w.effort }) + if (w.schema !== undefined) env.push({ name: "AX_CONWIP_SCHEMA_JSON", value: JSON.stringify(w.schema) }) + if (w.isolation !== undefined) env.push({ name: "AX_CONWIP_ISOLATION", value: w.isolation }) + if (w["agent-type"] !== undefined) env.push({ name: "AX_CONWIP_AGENT_TYPE", value: w["agent-type"] }) + if (job.spec["timeout-minutes"] !== undefined) + env.push({ name: "AX_CONWIP_TIMEOUT_MINUTES", value: String(job.spec["timeout-minutes"]) }) + if (shape.completeUrl !== undefined && g.leaseToken !== undefined) { // L7 (a missing token was refused above): a per-lease token, never the fleet bearer + env.push({ name: "AX_CONWIP_COMPLETE_URL", value: shape.completeUrl }) + env.push({ name: "AX_CONWIP_LEASE_TOKEN", value: g.leaseToken }) + } + const used = env.reduce((n, e) => n + utf8Len(e.value), 0) + // B12 (critique R2): the file route (AX_CONWIP_PROMPT_FILE) needs a Workspace that writes the file, and the Task + // carries none yet. Until that route exists a prompt that does not fit inline is refused, never sent to a dead path. + if (used > ENV_BUDGET) // FIELD-MAP D4 and probe test 4: over budget ax falls back silently, so refuse here + return { _tag: "refused", reason: "pre-start/env-over-budget", message: `env values ${used} B > ${ENV_BUDGET} B before the prompt` } + if (w.prompt === undefined || used + utf8Len(w.prompt) > ENV_BUDGET) + return { _tag: "refused", reason: "pre-start/prompt-file-route-missing", + message: w.prompt === undefined ? "no inline prompt and no Workspace route" : `prompt ${utf8Len(w.prompt)} B does not fit the ${ENV_BUDGET - used} B left inline` } + env.push({ name: "AX_CONWIP_PROMPT", value: w.prompt }) + const total = env.reduce((n, e) => n + utf8Len(e.value), 0) + if (total > ENV_BUDGET) // FIELD-MAP D4 and probe test 4: over budget ax falls back silently, so refuse here + return { _tag: "refused", reason: "pre-start/env-over-budget", message: `env values ${total} B > ${ENV_BUDGET} B` } + return { + _tag: "ok", + task: { + apiVersion: "ax.io/v1alpha1", + kind: "Task", + metadata: { name: g.leaseId, atespace: shape.atespace }, + spec: { image: shape.image, command: [...command], env, gateway: { name: shape.gateway } } + } + } +} + +// The spec digest used by create-only adoption (L4): found with the same digest = adopt, another = conflict. +// Only the fields the link writes are hashed, in a fixed order, so ax defaults added on the server do not count. +export const specDigest = (spec: { image?: string; command?: ReadonlyArray; env?: ReadonlyArray<{ name?: string; value?: string }>; gateway?: { name?: string } | null }) => + createHash("sha256").update(JSON.stringify([ + spec.image ?? "", [...(spec.command ?? [])], + [...(spec.env ?? [])].map((e) => [e.name ?? "", e.value ?? ""]), + spec.gateway?.name ?? "" + ])).digest("hex") + +export const envBytes = (t: AxTask) => t.spec.env.reduce((n, e) => n + utf8Len(e.value), 0) diff --git a/pkgs/substrate-link/src/apps/link/src/journal.ts b/pkgs/substrate-link/src/apps/link/src/journal.ts new file mode 100644 index 000000000..8ee3e03bc --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/src/journal.ts @@ -0,0 +1,187 @@ +// The link's crash-safe local state: one append-only JSONL file, fsynced per line, written BEFORE the action it +// records (the tally uplink's "offline is a queue" outbox, kept verbatim in spirit: done / keep / drop). The ax +// store is the ledger of Tasks (Buildkite: the cluster is the ledger); this file only covers what ax cannot know: +// grants not yet created, and Complete reports not yet acknowledged by the floor. +import { appendFileSync, closeSync, existsSync, fsyncSync, mkdirSync, openSync, readFileSync, renameSync, writeSync } from "node:fs" +import { join } from "node:path" +import type { Grant, Result, Usage } from "./contract.ts" + +export type Entry = + | { ev: "grant"; leaseId: string; grant: Grant; regrant?: true } // round 2: `regrant` replaces a released record + | { ev: "creating"; leaseId: string; digest?: string } // round 1: written BEFORE the first UpdateTask; the Task may exist from here on. Round 3: the spec digest the link will write, so a cleanup can tell its own Task from a foreign one + | { ev: "not-mine"; leaseId: string } // round 3: a Task of this name exists and is not the link's; `creating` no longer holds + | { ev: "created"; leaseId: string; digest: string } + | { ev: "grant-invalid"; leaseId: string; attempt: number; message: string } // round 1: a grant that did not decode + | { ev: "report"; leaseId: string; attempt: number; result: Result; output?: unknown; usage?: Usage; replace?: true } // outbox; round 2: `replace` swaps an undeliverable verdict + | { ev: "reported"; leaseId: string; duplicate: boolean; withdrew?: string; withdrewGen?: Gen } // done: 2xx from the floor; round 2: what rule 4b withdrew; round 3: which generation of it + | { ev: "dropped"; leaseId: string; code: string } // drop: the floor refused these bytes for good + | { ev: "released"; leaseId: string; why: string } // no longer held (lost, cancelled, superseded) + | { ev: "deleting"; leaseId: string; why: string } // DeleteTask accepted; waiting for NotFound (ax deletes in two phases) + | { ev: "deleted"; leaseId: string } + | { ev: "lease-key"; requestKey: string } // B8: a Lease is in flight with this key; reuse it until its reply lands + | { ev: "lease-replied"; requestKey: string; abandoned?: true } // the reply's grants are journaled; the next Lease draws a new key + | { ev: "tomb"; leaseId: string; gen: Gen } // round 3: written by compaction for a finished lease; refuses its redelivery + +/** Round 3: a grant generation. `acquireTime` is absent when only the floor's `withdrewTransitions` is known. */ +export interface Gen { readonly leaseTransitions: number; readonly acquireTime?: number } +export const genOf = (g: Grant): Gen => ({ leaseTransitions: g.lease.leaseTransitions, acquireTime: g.lease.acquireTime }) +/** `g` is a strictly later generation than `than` (leaseTransitions first, then acquireTime). */ +export const laterGen = (g: Gen, than: Gen) => g.leaseTransitions > than.leaseTransitions || + (g.leaseTransitions === than.leaseTransitions && g.acquireTime !== undefined && than.acquireTime !== undefined && g.acquireTime > than.acquireTime) +/** `g` is the generation `w` names: equal leaseTransitions, and equal acquireTime when both are known. */ +export const sameGen = (g: Gen, w: Gen) => g.leaseTransitions === w.leaseTransitions && + (g.acquireTime === undefined || w.acquireTime === undefined || g.acquireTime === w.acquireTime) + +export interface LeaseRec { + grant: Grant + created?: string // spec digest + creating?: boolean // an UpdateTask may have landed (write-ahead); cleanup must DeleteTask by name and see NotFound + creatingDigest?: string // round 3: the spec digest the link was about to write + notMine?: boolean // round 3: the Task under this name is someone else's; never read, delete or report it + createdAt?: number + report?: Extract + reported?: boolean + duplicate?: boolean // round 2: the floor answered this verdict as a duplicate (rule 4c) + withdrew?: string // round 2: the leaseId the floor withdrew when it accepted this verdict (rule 4b), never inferred + withdrewGen?: Gen // round 3: the generation withdrawn; undefined when this link never held it (it then matches nothing) + dropped?: string + released?: string + deleting?: string // why: cancel, lost, dropped, janitor, pre-start, deadline, superseded + deleted?: boolean + at: number // when this record was last touched (ms) +} + +export class Journal { + readonly path: string + readonly recs = new Map() + /** Grants that did not decode: refused to the floor as pre-start/invalid-spec through the same outbox. */ + readonly invalid = new Map() + pendingLeaseKey: string | undefined + /** Round 3: finished leases, kept through compaction for TOMB_MS so a redelivered grant of one is never run again. */ + readonly tombs = new Map() + static TOMB_MS = 24 * 3600_000 // well past the floor's replay window (rule 2b: lease + grace, 390 s in production) + private readonly dir: string + private constructor(dir: string) { this.dir = dir; this.path = join(dir, "journal.jsonl") } + + static open(dir: string, now = Date.now()): Journal { + mkdirSync(dir, { recursive: true, mode: 0o700 }) + const j = new Journal(dir) + if (existsSync(j.path)) { + for (const line of readFileSync(j.path, "utf8").split("\n")) { + if (line.trim() === "") continue + let e: Entry & { t?: number } + try { e = JSON.parse(line) } catch { continue } // a torn last line from a crash mid-write is skipped + j.apply(e, e.t ?? now) + } + j.compact() + } + return j + } + + private apply(e: Entry, t: number) { + if (e.ev === "lease-key") { this.pendingLeaseKey = e.requestKey; return } + if (e.ev === "lease-replied") { if (this.pendingLeaseKey === e.requestKey) this.pendingLeaseKey = undefined; return } + if (e.ev === "tomb") { this.tombs.set(e.leaseId, { gen: e.gen, at: t }); return } + if (e.ev === "grant") { + if (this.recs.has(e.leaseId) && !e.regrant) return + this.recs.set(e.leaseId, { grant: e.grant, at: t }) + // round 2: the withdrawal recorded against the previous generation of this leaseId does not apply to the new one + if (e.regrant) for (const x of this.recs.values()) if (x.withdrew === e.leaseId) { delete x.withdrew; delete x.withdrewGen } + return + } + if (e.ev === "grant-invalid") { if (!this.recs.has(e.leaseId) && !this.invalid.has(e.leaseId)) this.invalid.set(e.leaseId, { attempt: e.attempt, message: e.message, at: t }); return } + const r = this.recs.get(e.leaseId) + if (!r) { + const x = this.invalid.get(e.leaseId) + if (x && e.ev === "reported") x.reported = true + if (x && e.ev === "dropped") x.dropped = e.code + if (x && e.ev === "released") x.released ??= e.why + return + } + r.at = t + switch (e.ev) { + case "creating": r.creating = true; delete r.notMine; if (e.digest !== undefined) r.creatingDigest = e.digest; break + case "not-mine": r.notMine = true; delete r.creating; break + case "created": r.created = e.digest; r.createdAt ??= t; break + case "report": if (!r.report || e.replace) r.report = e; break // first verdict wins (the floor dedupes as well), unless replaced + case "reported": r.reported = true; r.duplicate = e.duplicate; if (e.withdrew !== undefined) { r.withdrew = e.withdrew; if (e.withdrewGen !== undefined) r.withdrewGen = e.withdrewGen } break + case "dropped": r.dropped = e.code; break + case "released": r.released ??= e.why; break + case "deleting": r.deleting ??= e.why; break + case "deleted": r.deleted = true; break + } + } + + append(e: Entry, t = Date.now()) { + appendFileSync(this.path, JSON.stringify({ t, ...e }) + "\n", { mode: 0o600 }) + const fd = openSync(this.path, "r") + try { fsyncSync(fd) } finally { closeSync(fd) } + this.apply(e, t) + if (++this.appends >= Journal.COMPACT_EVERY) this.compact() // idle Lease keys must not grow the file for ever + } + static COMPACT_EVERY = 2000 + private appends = 0 + + /** Finished = nothing left to tell the floor and nothing left in ax. */ + static finished(r: LeaseRec) { + // Round 2: a verdict read after the lease was released (lost, superseded) is still owed to the floor + if (Journal.pending(r)) return false + const toldFloor = r.reported === true || r.dropped !== undefined || r.released !== undefined + const axClean = !Journal.maybeCreated(r) || r.deleted === true + return toldFloor && axClean + } + /** Round 1: a Task may exist in ax once `creating` is journaled, even if `created` never was (a crash, or a + * DEADLINE_EXCEEDED after the write landed). Every cleanup path treats such a record as created. */ + static maybeCreated(r: LeaseRec) { return r.created !== undefined || r.creating === true } + /** A verdict is journaled and neither acknowledged nor dropped. */ + static pending(r: LeaseRec) { return r.report !== undefined && !r.reported && r.dropped === undefined } + + /** Rewrite the file with only unfinished leases (atomic rename), so it never grows without bound. */ + compact() { + const tmp = this.path + ".tmp" + const fd = openSync(tmp, "w", 0o600) + try { + if (this.pendingLeaseKey !== undefined) writeSync(fd, JSON.stringify({ t: Date.now(), ev: "lease-key", requestKey: this.pendingLeaseKey }) + "\n") + for (const [id, r] of this.recs) { + if (Journal.finished(r)) { this.recs.delete(id); this.tombs.set(id, { gen: genOf(r.grant), at: r.at }); continue } + const lines: Array = [{ ev: "grant", leaseId: id, grant: r.grant }] + if (r.creating) lines.push({ ev: "creating", leaseId: id, ...(r.creatingDigest !== undefined ? { digest: r.creatingDigest } : {}) }) + if (r.notMine) lines.push({ ev: "not-mine", leaseId: id }) + if (r.created !== undefined) lines.push({ ev: "created", leaseId: id, digest: r.created }) + if (r.report) lines.push(r.report) + if (r.reported) lines.push({ ev: "reported", leaseId: id, duplicate: r.duplicate ?? false, ...(r.withdrew !== undefined ? { withdrew: r.withdrew } : {}), ...(r.withdrewGen !== undefined ? { withdrewGen: r.withdrewGen } : {}) }) + if (r.dropped !== undefined) lines.push({ ev: "dropped", leaseId: id, code: r.dropped }) + if (r.released !== undefined) lines.push({ ev: "released", leaseId: id, why: r.released }) + if (r.deleting !== undefined) lines.push({ ev: "deleting", leaseId: id, why: r.deleting }) + for (const l of lines) writeSync(fd, JSON.stringify({ t: l.ev === "created" || l.ev === "grant" ? (r.createdAt ?? r.at) : r.at, ...l }) + "\n") + } + const now = Date.now() + for (const [id, x] of this.tombs) { + if (now - x.at > Journal.TOMB_MS) { this.tombs.delete(id); continue } + writeSync(fd, JSON.stringify({ t: x.at, ev: "tomb", leaseId: id, gen: x.gen }) + "\n") + } + for (const [id, x] of this.invalid) { // never created, so finished once the floor is told + if (x.reported || x.dropped !== undefined || x.released !== undefined) { this.invalid.delete(id); continue } + writeSync(fd, JSON.stringify({ t: x.at, ev: "grant-invalid", leaseId: id, attempt: x.attempt, message: x.message }) + "\n") + } + fsyncSync(fd) + } finally { closeSync(fd) } + renameSync(tmp, this.path) + this.appends = 0 + const d = openSync(this.dir, "r") // B20: make the rename itself durable + try { fsyncSync(d) } finally { closeSync(d) } + } + + /** The complete set this holder believes it holds: sent in every Heartbeat (level-triggered). A lease whose + * verdict is still in the outbox stays held, so a heartbeat that races the outbox drain cannot orphan it. */ + held(): Array { + return [...[...this.recs].filter(([, r]) => !r.reported && r.dropped === undefined && r.released === undefined).map(([id]) => id), + ...[...this.invalid].filter(([, x]) => !x.reported && x.dropped === undefined && x.released === undefined).map(([id]) => id)] + } + /** Leases whose verdict is written but not yet acknowledged; they are still held until the floor says done. */ + outbox(): Array> { + return [...[...this.recs.values()].filter(Journal.pending).map((r) => r.report!), + ...[...this.invalid].filter(([, x]) => !x.reported && x.dropped === undefined && x.released === undefined) + .map(([leaseId, x]) => ({ ev: "report" as const, leaseId, attempt: x.attempt, result: "failure" as const, output: { reason: "pre-start/invalid-spec", message: x.message } }))] + } +} diff --git a/pkgs/substrate-link/src/apps/link/src/link.ts b/pkgs/substrate-link/src/apps/link/src/link.ts new file mode 100644 index 000000000..d8de5b991 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/src/link.ts @@ -0,0 +1,921 @@ +// The NAS link (LINK-DESIGN.md section 5): one outbound-only process between the Cloudflare floor and ax. +// Lease (free capacity on every poll, ARC; the outbox drains first, B15; the requestKey is journaled, B8) +// -> journal the grant (write-ahead) -> fence every superseded attempt: DeleteTask and wait for NotFound (L3, B2) +// -> wait for a sandbox slot (B11) -> GetTask, UpdateTask only on NotFound (L4, create-only), read it back (B12) +// -> resync by paging ListTasks (Buildkite: the executor's list is the ledger; B1) and confirming every absence with +// GetTask -> read the outcome (P1: terminal phase + GetTaskResult, retried and digest-checked, B4; guest mode only +// when configured, L7, B6) -> Complete through the journal's outbox (L6) -> DeleteTask after the floor acknowledged. +// One batched Heartbeat per session carries the COMPLETE held set; its reply carries cancel and lost (Temporal). +// Nothing is leased while ax is unreachable, the completion mode is unproven (B6) or the egress Gateway is missing or +// open (B10). +import { createHash, randomUUID } from "node:crypto" +import { isIPv4, isIPv6 } from "node:net" +import { Cause, Deferred, Effect, Schedule, Semaphore } from "effect" +import { gatewayAllowsAll, listAll } from "./ax.ts" +import type { AxApi, AxError, AxObserved } from "./ax.ts" +import type { FailureOutput, Grant, Result, Usage } from "./contract.ts" +import type { FloorApi, FloorError, LeaseReply } from "./floor.ts" +import { axTaskFromGrant, specDigest, validRunsOn } from "./jobs.ts" +import type { TaskShape } from "./jobs.ts" +import { genOf, Journal, laterGen, sameGen } from "./journal.ts" +import type { Gen, LeaseRec } from "./journal.ts" + +export interface LinkConfig { + readonly holder: string // holderIdentity; bound to this link's token on the floor + readonly maxInFlight: number // sandbox cap: the WorkerPool replicas this link may fill + readonly servedLabels: ReadonlyArray // defence in depth: the floor already routes by holder + readonly shape: TaskShape + /** `p1` and `auto` both require P1 on the server, proven by a probe (B6); `auto` never selects guest. `guest` is + * the L7 reserve and needs a fleet-internal `shape.completeUrl` (critique A5). */ + readonly completion: "auto" | "p1" | "guest" + readonly secondMs: number // 1000 in production; tests shrink every server-set interval with it + readonly resyncMs: number // 15 s, the P1 --running-resync period + readonly pendingTimeoutMs: number // 15 min, agent-stack-k8s: a Pod stuck Pending fails the job + readonly deleteAfterMs: number // 10 min, agent-stack-k8s TTLSecondsAfterFinished + readonly deadlineBackstopMs: number // local stop past the floor's deadline, if the floor cannot say cancel + readonly createAttempts: number // UpdateTask tries while ax is unavailable before pre-start failure + readonly outboxBackoffMs: readonly [number, number] // uplink shut-door backoff: min, max + readonly fenceTimeoutMs?: number // B2: how long a superseded attempt may take to reach NotFound (default 120 s) + readonly resultReadTries?: number // B4: resyncs a Completed Task's result may fail to read (default 5) + readonly floorUrls?: ReadonlyArray // B18: the only URLs a Lease `endpoint` may move this link to + readonly initialPollSeconds?: number // until the first Lease reply sets them (defaults 15 and 30, section 3.7) + readonly initialHeartbeatSeconds?: number + /** Round 2: a result larger than this is refused `agent/output-too-large` before it is reported (the floor's DO row + * limit is 2 MB, MEASURED as SQLITE_TOOBIG at 2.2 MB; default 1 000 000 B). */ + readonly maxOutputBytes?: number + /** Round 2: server errors one verdict may collect before it is replaced by `infra/verdict-undeliverable`, and that + * replacement before it is dead-lettered (journaled `dropped`). Default 8. */ + readonly verdictAttempts?: number + /** Round 2: single-label hosts a guest Complete URL may name (fleet DNS names without a dot). Default none. */ + readonly internalHosts?: ReadonlyArray + /** Round 3: consecutive undecodable or server-error Lease replies for one journaled requestKey before the key is + * abandoned (its grants are then released by the floor's own expiry and requeued). Default 5. */ + readonly leaseKeyAttempts?: number +} + +export type Log = (ev: string, fields?: Record) => void +export interface LinkDeps { + readonly ax: AxApi; readonly floor: FloorApi; readonly journal: Journal; readonly log: Log; readonly stop: Deferred.Deferred + readonly floorUrl?: string // the URL in use, to tell a real endpoint switch from an echo + readonly onEndpoint?: (url: string) => void // B18: called once for an allowed switch; main persists it and restarts +} + +const TERMINAL = new Set(["Completed", "Failed"]) +const jitter = (ms: number) => Math.round(ms * (0.8 + 0.4 * Math.random())) +const NOT_FOUND = 5 // gRPC status: DeleteTask of a name that is already gone +const MAX_POLL_SECONDS = 300, MAX_HEARTBEAT_SECONDS = 300 +/** Round 3: a server-driven interval (Buildkite: the agent falls back to its defaults). Missing, non-finite or not + * positive gives `def`; anything else is clamped to [1, hi]. `Effect.sleep(0)` would otherwise be a tight loop. */ +export const boundedSeconds = (v: number | undefined, def: number, hi: number) => + v === undefined || !Number.isFinite(v) || v <= 0 ? Math.min(def, hi) : Math.max(1, Math.min(v, hi)) + +/** Critique A5 and N3: a guest's Complete URL must stay inside the fleet, never a public Workers host. + * Round 2: private ranges are matched only on IP literals (after WHATWG normalisation, so `10.evil.example` and + * `127.0.0.1.nip.io` are names, not addresses); a name must end in a non-public suffix; a single-label name only when + * it is listed. */ +export const isFleetInternalUrl = (raw: string | undefined, internalHosts: ReadonlyArray = []): boolean => { + if (raw === undefined) return false + let u: URL + try { u = new URL(raw) } catch { return false } + let h = u.hostname.toLowerCase() + if (h.endsWith(".workers.dev") || h.endsWith(".pages.dev")) return false + if (h.startsWith("[") && h.endsWith("]")) h = h.slice(1, -1) + if (isIPv4(h)) { + const [a, b] = h.split(".").map(Number) as [number, number] + return a === 127 || a === 10 || (a === 192 && b === 168) || (a === 172 && b >= 16 && b <= 31) || (a === 100 && b >= 64 && b <= 127) + } + if (isIPv6(h)) return h === "::1" || /^f[cd][0-9a-f]{2}:/.test(h) // loopback, ULA fc00::/7 (the tailnet's fd7a:115c:a1e0::/48) + if (h === "localhost" || internalHosts.map((x) => x.toLowerCase()).includes(h)) return true + return /^([a-z0-9-]+\.)+(internal|lan|local|home\.arpa)$/.test(h) +} + +export const runLink = (cfg: LinkConfig, deps: LinkDeps) => Effect.gen(function*() { + const { ax, floor, journal: j, log } = deps + const fatal = yield* Deferred.make() + const fenceTimeoutMs = cfg.fenceTimeoutMs ?? 120_000 + const resultReadTries = cfg.resultReadTries ?? 5 + const maxOutputBytes = cfg.maxOutputBytes ?? 1_000_000 + const verdictAttempts = cfg.verdictAttempts ?? 8 + const defaultPoll = cfg.initialPollSeconds ?? 15, defaultHb = cfg.initialHeartbeatSeconds ?? 30 + let pollSeconds = defaultPoll, hbAsked: number | undefined // round 3: the floor's raw heartbeatSeconds, bounded at use + let clampLogged = "" + const leaseKeyAttempts = cfg.leaseKeyAttempts ?? 5 + let leaseKeyFailures = 0 + let axUp = false, resynced = false + let serverP1: boolean | undefined // the probe's answer; cleared on every ax-up transition (B6) + let gatewayOk: boolean | undefined // B10 + const guestUrlOk = cfg.completion !== "guest" || isFleetInternalUrl(cfg.shape.completeUrl, cfg.internalHosts) + if (!guestUrlOk) log("guest-url-invalid", { completion: "guest" }) + let last = new Map() + let outboxWaitUntil = 0, outboxBackoff = cfg.outboxBackoffMs[0] + let outboxNotBefore = 0 // round 3: the floor's own Retry-After on Complete; Lease does not cut it short (B15 does not) + const dispatching = new Map() // round 2: keyed by leaseId, valued by the grant generation it runs + const admitted = new Set() // B11: grants that passed the slot gate in this process + const abandoned = new Map() // B3: set by the Heartbeat reply, read by dispatch + const absent = new Map() // B1: consecutive resyncs a held Task answered NotFound + const resultFailures = new Map() // B4 + const verdictFailures = new Map() // round 2: server errors per verdict (a restart grants a fresh budget) + const verdictNotBefore = new Map() // round 2: per-verdict backoff after a server error + const pendingResume = new Set() // B13: journaled, never created; resumed only once the floor renews them + /** Round 4 (Buildkite reserve-with-expiry): the local time of the last call the floor answered by renewing this + * grant generation (the Lease that granted it, then every Heartbeat listing it in `renewed`). The send time is taken, + * so it is never later than the floor's own renewTime. Keyed by the grant object: a regrant starts afresh. */ + const renewedAt = new WeakMap() + /** Round 4: a create may start only while the floor still holds the lease: the last renewal is less than 3/4 of a + * lease old. A grant never renewed in this process (journaled by a previous one) is not live until a Heartbeat says so. */ + const leaseLive = (g: Grant) => { + const at = renewedAt.get(g) + if (at === undefined) return false + const ms = g.lease.leaseDurationSeconds * cfg.secondMs + return Date.now() < at + ms - ms / 4 + } + const deadlinePassed = (g: Grant) => g.deadline !== undefined && Date.now() >= g.deadline + let wake = yield* Deferred.make() // B17: a verdict that frees a slot leases at once + let hbWake = yield* Deferred.make() // final pass (R3-1): a Lease reply re-times a heartbeat sleep already begun + const failFatal = (e: FloorError) => Deferred.fail(fatal, e).pipe(Effect.asVoid) + + /** The mode the link may create in, or undefined while it is unproven (capacity 0). */ + const mode = (): "p1" | "guest" | undefined => { + if (cfg.completion === "guest") return guestUrlOk ? "guest" : undefined + return serverP1 === true ? "p1" : undefined + } + const canCreate = () => axUp && resynced && gatewayOk === true && mode() !== undefined + + // ---------------------------------------------------------------- verdicts (write-ahead, then the outbox drains) + /** `salvage` (round 2): the verdict was read from a Task whose lease is already released (lost, superseded). It is + * still owed to the floor, which accepts it by rule 4b or answers stale-attempt (then it is dropped). */ + /** Round 4: `g`, when given, is the grant generation the verdict is about; a record that holds another generation + * by now (a regrant of the same leaseId) is not given it. */ + const report = (id: string, result: Result, output?: unknown, usage?: Usage, salvage = false, g?: Grant) => Effect.gen(function*() { + const r = j.recs.get(id) + if (g !== undefined && r?.grant !== g) return log("stale-generation-verdict", { leaseId: id, result }) + if (!r || r.report || (r.released !== undefined && !salvage)) return + j.append({ ev: "report", leaseId: id, attempt: r.grant.attempt, result, ...(output !== undefined ? { output } : {}), ...(usage !== undefined ? { usage } : {}) }) + log("verdict", { leaseId: id, result, ...(result === "failure" ? { reason: (output as FailureOutput | undefined)?.reason } : {}), ...(salvage ? { salvaged: r.released } : {}) }) + yield* Deferred.succeed(wake, undefined) + }) + const fail = (id: string, o: FailureOutput) => report(id, "failure", o) + const failG = (g: Grant, o: FailureOutput) => report(g.leaseId, "failure", o, undefined, false, g) // round 4 + + // Round 1: single-flight. The main loop, the resync fiber and a dispatch waiting on a superseded verdict all drain; + // two concurrent drains read the same outbox before either journals `reported` and send every verdict twice. + const outboxLock = Semaphore.makeUnsafe(1) + const drainOutbox = outboxLock.withPermit(Effect.gen(function*() { + if (Date.now() < outboxWaitUntil || Date.now() < outboxNotBefore) return + for (const p of j.outbox()) { + if ((verdictNotBefore.get(p.leaseId) ?? 0) > Date.now()) continue + // Round 4: the generation of attempt n+1 this link held when the verdict was SENT. A regrant of n+1 taken while + // the reply is in flight is later than anything this Complete can have withdrawn. + const succId = `${p.leaseId.replace(/-a\d+$/, "")}-a${p.attempt + 1}` + const succAtSend = j.recs.get(succId)?.grant + const r = yield* floor.complete({ leaseId: p.leaseId, attempt: p.attempt, result: p.result, ...(p.output !== undefined ? { output: p.output } : {}), ...(p.usage !== undefined ? { usage: p.usage } : {}) }).pipe(Effect.result) + if (r._tag === "Success") { + const withdrew = r.success.withdrew + // Round 3: a withdrawal names a GENERATION, not a leaseId. The floor may requeue the withdrawn attempt under the + // same leaseId (a retryable verdict of n), and that later grant is live. The generation is the floor's + // `withdrewTransitions` when it sends one, else the one this link holds; a generation this link never received + // (n+1 was still queued) matches no later grant. + const w = withdrew !== undefined ? j.recs.get(withdrew) : undefined + const withdrewGen: Gen | undefined = withdrew === undefined ? undefined + : r.success.withdrewTransitions !== undefined ? { leaseTransitions: r.success.withdrewTransitions } + : succAtSend !== undefined && withdrew === succId ? genOf(succAtSend) + : w !== undefined ? genOf(w.grant) : undefined + j.append({ ev: "reported", leaseId: p.leaseId, duplicate: r.success.duplicate, ...(withdrew !== undefined ? { withdrew } : {}), ...(withdrewGen !== undefined ? { withdrewGen } : {}) }) + log("complete", { leaseId: p.leaseId, result: p.result, duplicate: r.success.duplicate, ...(withdrew !== undefined ? { withdrew } : {}) }) + if (withdrew !== undefined && withdrewGen === undefined) log("withdrawn-unheld", { leaseId: withdrew, by: p.leaseId }) + if (w !== undefined && withdrewGen !== undefined && sameGen(genOf(w.grant), withdrewGen) && w.released === undefined && !Journal.maybeCreated(w)) { + j.append({ ev: "released", leaseId: withdrew!, why: "withdrawn" }) // round 2: released at once, so a regrant + log("withdrawn", { leaseId: withdrew, by: p.leaseId }) // of it is not taken for a duplicate + } + outboxBackoff = cfg.outboxBackoffMs[0] + verdictFailures.delete(p.leaseId); verdictNotBefore.delete(p.leaseId) + } else if (r.failure._tag === "CompleteRefused") { // done/keep/drop: the floor refused these bytes for good + j.append({ ev: "dropped", leaseId: p.leaseId, code: r.failure.code }) + log("complete-dropped", { leaseId: p.leaseId, code: r.failure.code }) + } else if (r.failure.kind === "server-error") { + // Round 2: the floor answered and refused THESE bytes (a 500, an RPC Defect such as SQLITE_TOOBIG). Counted per + // verdict; the rest of the outbox still goes out. At the ceiling the verdict is replaced by a small final + // failure the floor can store; if that fails as well it is dead-lettered, so one verdict never gates Lease. + const n = (verdictFailures.get(p.leaseId) ?? 0) + 1 + verdictFailures.set(p.leaseId, n) + if (n < verdictAttempts) { + const wait = jitter(Math.min(cfg.outboxBackoffMs[0] * 2 ** (n - 1), cfg.outboxBackoffMs[1])) + verdictNotBefore.set(p.leaseId, Date.now() + wait) + log("outbox-keep", { leaseId: p.leaseId, kind: r.failure.kind, try: n, waitMs: wait }) + continue + } + verdictFailures.delete(p.leaseId); verdictNotBefore.delete(p.leaseId) + const rec = j.recs.get(p.leaseId) + if (rec !== undefined && !p.replace) { + const digest = createHash("sha256").update(JSON.stringify(p.output ?? null)).digest("hex") + j.append({ ev: "report", leaseId: p.leaseId, attempt: p.attempt, result: "failure", replace: true, + output: { reason: "infra/verdict-undeliverable", message: `${p.result} verdict (output sha256 ${digest}) refused ${n} times: ${r.failure.message.slice(0, 200)}` } }) + log("verdict-replaced", { leaseId: p.leaseId, result: p.result, tries: n, outputSha256: digest }) + } else { + j.append({ ev: "dropped", leaseId: p.leaseId, code: "undeliverable" }) + log("verdict-dead-letter", { leaseId: p.leaseId, tries: n, message: r.failure.message.slice(0, 200) }) + } + } else { + if (r.failure.fatal) return yield* failFatal(r.failure) + outboxWaitUntil = Date.now() + Math.min(r.failure.retryAfterMs ?? jitter(outboxBackoff), MAX_POLL_SECONDS * cfg.secondMs) + if (r.failure.retryAfterMs !== undefined) outboxNotBefore = outboxWaitUntil + outboxBackoff = Math.min(outboxBackoff * 2, cfg.outboxBackoffMs[1]) + log("outbox-keep", { leaseId: p.leaseId, kind: r.failure.kind, waitMs: outboxWaitUntil - Date.now() }) + return // keep: oldest first, stop at the first transient failure + } + } + })) + + // ---------------------------------------------------------------- ax calls + // Round 3 (token scope): a record that is only `creating` names a Task the link MAY have written. The floor chose + // that name, so before anything reads or deletes it the Task must carry this link's write: the spec digest the + // link built for this grant, or this lease id and this holder in its env. Anything else is someone else's Task. + const envMarks = (id: string, t: AxObserved) => { + const env = new Map((t.spec.env ?? []).map((e) => [e.name ?? "", e.value ?? ""])) + return env.get("AX_CONWIP_LEASE_ID") === id && env.get("AX_CONWIP_HOLDER") === cfg.holder + } + const owns = (id: string, r: LeaseRec, t: AxObserved) => + r.created !== undefined || (r.creatingDigest !== undefined && specDigest(t.spec) === r.creatingDigest) || envMarks(id, t) + /** Journal that the Task under this name is not the link's: nothing reads, deletes or reports it any more. A cancel + * already asked for is answered `cancelled`, since nothing of this lease runs in ax. */ + const markNotMine = (id: string) => Effect.gen(function*() { + const r = j.recs.get(id) + if (!r || r.notMine) return + j.append({ ev: "not-mine", leaseId: id }) + log("not-mine", { leaseId: id }) + if (!r.report && r.released === undefined && (abandoned.get(id) === "cancel" || r.deleting === "cancel")) yield* report(id, "cancelled") + }) + /** GetTask a `creating`-only record's name: owned, absent, foreign (journaled not-mine) or unknown (ax did not answer). */ + const ownership = (id: string, r: LeaseRec) => Effect.gen(function*() { + const got = yield* ax.getTask(id).pipe(Effect.result) + if (got._tag === "Failure") return "unknown" as const + if (got.success === undefined) return "absent" as const + if (owns(id, r, got.success)) return "owned" as const + yield* markNotMine(id) + return "foreign" as const + }) + + // Round 3: the intent is journaled BEFORE DeleteTask (`deleting`), so a DeleteTask that lands and whose reply is lost + // (DEADLINE_EXCEEDED) still reads as a delete on the next resync (a cancel is reported `cancelled`, not task-lost), + // and one that did not land is sent again by reconcile. NotFound means already gone. + const del = (id: string, why: string): Effect.Effect => Effect.gen(function*() { + const rec = j.recs.get(id) + if (rec?.notMine) return true + if (rec !== undefined && rec.created === undefined && rec.creating) { + const o = yield* ownership(id, rec) + if (o === "unknown") { log("delete-deferred", { task: id, why, until: "ownership" }); return false } + if (o === "foreign") return true + } + if (rec !== undefined && rec.deleting === undefined) j.append({ ev: "deleting", leaseId: id, why }) + return yield* ax.deleteTask(id).pipe( + Effect.tap(() => Effect.sync(() => log("delete", { task: id, why }))), + Effect.as(true), + Effect.catch((e: AxError) => Effect.sync(() => { + if (e.code === NOT_FOUND) { log("delete", { task: id, why, absent: true }); return true } + log("delete-deferred", { task: id, why, code: e.code }); return false + }))) + }) + + /** B2: an old attempt is gone only when GetTask answers NotFound; until then attempt n+1 is not created. */ + const fence = (old: string) => Effect.gen(function*() { + const until = Date.now() + fenceTimeoutMs + let asked = false + for (;;) { + if (j.recs.get(old)?.notMine) return true // round 3: not this link's Task; nothing of attempt n runs under it + const r = yield* ax.getTask(old).pipe(Effect.result) + if (r._tag === "Success" && r.success === undefined) { + if (j.recs.has(old) && !j.recs.get(old)!.deleted) j.append({ ev: "deleted", leaseId: old }) + return true + } + if (r._tag === "Success" && (!asked || r.success!.phase !== "Terminating")) asked = (yield* del(old, "superseded")) || asked + if (Date.now() >= until) return false + yield* Effect.sleep(Math.min(cfg.resyncMs, 2 * cfg.secondMs)) + } + }) + + /** B11: never more live Tasks than maxInFlight. Busy = live Tasks in the atespace (anyone's) U admitted leases + * not known finished. Synchronous, so two dispatch fibers cannot both take the last slot. */ + const busy = () => { + const b = new Set() + for (const [name, t] of last) if (!TERMINAL.has(t.phase)) b.add(name) + for (const id of admitted) { + const r = j.recs.get(id), t = last.get(id) + if (!r || r.report || r.released !== undefined || r.dropped !== undefined || r.deleted || (t && TERMINAL.has(t.phase))) { admitted.delete(id); b.delete(id); continue } + b.add(id) + } + return b + } + const admit = (id: string) => Effect.sync(() => { + if (!canCreate()) return false + const b = busy() + if (b.has(id) || b.size < cfg.maxInFlight) { admitted.add(id); return true } + return false + }) + + /** A grant this link will not create any more (B3): a cancel is reported once any Task that may exist is gone. */ + const abandon = (id: string, why: "cancel" | "lost", where: string) => Effect.gen(function*() { + log(where, { leaseId: id, why }) + const rec = j.recs.get(id) + if (rec && Journal.maybeCreated(rec)) return yield* del(id, why) // round 1: reconcile reports cancelled on NotFound + if (why === "cancel") yield* report(id, "cancelled") + }) + + let guestTokenWarned = false + const createOnly = (g: Grant) => Effect.gen(function*() { + const id = g.leaseId + let waited = false + let unconfirmed = false + for (;;) { // round 1: the capacity gate is re-checked before every UpdateTask; a closed gate goes back to waiting + // Round 4: and the lease must still be live (renewed within the lease) before the slot is taken + while (!leaseLive(g) || !(yield* admit(id))) { // B11: journaled and heartbeated while it waits + if (abandoned.has(id) || deadlinePassed(g)) break + if (!leaseLive(g)) { if (!unconfirmed) { log("lease-unconfirmed", { leaseId: id }); unconfirmed = true } } + else if (!waited) { log("waiting-for-slot", { leaseId: id }); waited = true } + yield* Effect.sleep(cfg.resyncMs) + } + const early = abandoned.get(id) + if (early !== undefined) return yield* abandon(id, early, "abandoned-before-create") + const now = j.recs.get(id) // round 2: withdrawn, or replaced by a newer grant of the same leaseId: never create + if (now?.grant !== g || now.released !== undefined) return log("grant-superseded-before-create", { leaseId: id }) + if (deadlinePassed(g)) return yield* failG(g, { reason: "deadline-exceeded", message: "the grant's deadline passed before its Task was created" }) // round 4 + const m = mode() + const built = axTaskFromGrant(g, { ...cfg.shape, holder: cfg.holder, ...(m === "guest" ? {} : { completeUrl: undefined }) }) + if (built._tag === "refused") { + if (built.reason === "pre-start/lease-token-missing" && !guestTokenWarned) { + guestTokenWarned = true + log("guest-token-missing", { leaseId: id, hint: "floor and link disagree: the link completes as guest, the floor mints no lease token for this holder" }) + } + return yield* failG(g, { reason: built.reason, message: built.message }) + } + const digest = specDigest(built.task.spec) + for (let i = 1; ; i++) { + const why = abandoned.get(id) // B3: checked before every UpdateTask + if (why !== undefined) return yield* abandon(id, why, "abandoned-before-create") + const cur = j.recs.get(id) + if (cur?.grant !== g || cur.released !== undefined) return log("grant-superseded-before-create", { leaseId: id }) + if (!canCreate() || mode() !== m) { log("gate-closed-before-create", { leaseId: id, try: i }); break } // round 1 + if (deadlinePassed(g)) return yield* failG(g, { reason: "deadline-exceeded", message: "the grant's deadline passed before its Task was created" }) + if (!leaseLive(g)) { log("lease-unconfirmed-before-create", { leaseId: id, try: i }); break } // round 4: back to waiting + if (!j.recs.get(id)?.creating) j.append({ ev: "creating", leaseId: id, digest }) // round 1: write-ahead; round 3: with the digest + const r = yield* Effect.gen(function*() { + const found = yield* ax.getTask(id) + if (found === undefined) { yield* ax.createTask(built.task); return "created" as const } + if (specDigest(found.spec) === digest) return "adopted" as const + return envMarks(id, found) ? "conflict" as const : "foreign" as const + }).pipe(Effect.result) + if (r._tag === "Success") { + // L4: never overwrite a Task this link did not write. Round 3: a Task without this link's marks is journaled + // not-mine first, so no cleanup path (janitor, lost, cancel, salvage) ever reads or deletes it. + if (r.success === "foreign") yield* markNotMine(id) + if (r.success === "conflict" || r.success === "foreign") + return yield* failG(g, { reason: "pre-start/name-conflict", message: "a Task with this name and another spec exists" }) + j.append({ ev: "created", leaseId: id, digest }) + log(r.success, { leaseId: id, attempt: g.attempt }) + const after = abandoned.get(id) // B3: and again once UpdateTask returned + if (after !== undefined) { yield* del(id, after); return } + const back = yield* ax.getTask(id).pipe(Effect.result) // B12: FIELD-MAP 5a read-back + if (back._tag === "Success" && back.success !== undefined && specDigest(back.success.spec) !== digest) { + yield* del(id, "pre-start") + return yield* failG(g, { reason: "pre-start/readback-mismatch", message: "ax stored another spec than the link wrote" }) + } + return + } + const e = r.failure + if (e.resourceExhausted) return yield* failG(g, { reason: "pre-start/resource-exhausted", message: e.message }) // B5 + if (!e.unavailable) return yield* failG(g, { reason: "pre-start/invalid-spec", message: e.message }) + if (i >= cfg.createAttempts) return yield* failG(g, { reason: "pre-start/ax-unavailable", message: e.message }) + yield* Effect.sleep(jitter(2 ** i * 100)) + } + } + }) + + /** Round 1: an unjournaled name may be fenced only if it is `-a` with k < this attempt and the Task's own + * env says this lease id and this holder (AX_CONWIP_HOLDER, written by this link at create). */ + const ownEarlierAttempt = (old: string, g: Grant, t: AxObserved) => { + const base = g.leaseId.replace(/-a\d+$/, ""), m = /^(.*)-a(\d+)$/.exec(old) + if (m === null || m[1] !== base || Number(m[2]) >= g.attempt) return false + return envMarks(old, t) + } + + const dispatch = (g: Grant) => Effect.gen(function*() { + if (j.recs.get(g.leaseId)?.grant === g) dispatching.set(g.leaseId, g) // round 2: only the current generation + const bad = validRunsOn(g.job, cfg.servedLabels) // B7 + if (bad) return yield* failG(g, bad) + // Final pass (VERIFY-LINK V1): the fence set is `supersedes` plus every earlier attempt of the same job this link + // journaled, that may exist in ax and is not confirmed deleted. A floor that omits `supersedes` (or a Heartbeat + // whose `lost` arrives after this grant) no longer lets attempt n+1 start beside a running attempt n. + const olds = [...new Set([...(g.supersedes ?? []), ...[...j.recs.values()] + .filter((r) => r.grant.leaseId !== g.leaseId && r.grant.job.metadata.name === g.job.metadata.name && r.grant.attempt < g.attempt + && Journal.maybeCreated(r) && r.deleted !== true && !r.notMine) + .map((r) => r.grant.leaseId)])] + if (olds.length > (g.supersedes ?? []).length) log("fence-set-extended", { leaseId: g.leaseId, supersedes: g.supersedes ?? [], fence: olds }) + // Round 2: an earlier attempt that may exist and has no verdict yet is READ before anything fences it. A Task that + // finished while its lease expired carries a result the floor can still accept (rule 4b); deleting it first loses + // the success and runs the job again. + for (const o of olds) { + const until = Date.now() + fenceTimeoutMs + for (;;) { + const s = yield* salvage(o) + if (s !== "unknown") break + const why = abandoned.get(g.leaseId) + if (why !== undefined) return yield* abandon(g.leaseId, why, "abandoned-before-create") + if (Date.now() >= until) + return yield* failG(g, { reason: "pre-start/ax-unavailable", message: `superseded attempt ${o} could not be read before its fence` }) + yield* Effect.sleep(Math.min(cfg.resyncMs, 2 * cfg.secondMs)) + } + } + // Round 1 (B15 at dispatch): an earlier attempt whose verdict is still in the outbox is delivered BEFORE it is + // fenced. The floor accepts it (rule 4b: n+1 is queued, or leased to this holder and not created) and withdraws + // this grant; deleting the finished Task first would throw its result away and rerun the job. + const undelivered = () => olds.filter((o) => { const r = j.recs.get(o); return r?.report !== undefined && !r.reported && r.dropped === undefined }) + if (undelivered().length > 0) { + log("supersede-waits-for-verdict", { leaseId: g.leaseId, old: undelivered() }) + const until = Date.now() + fenceTimeoutMs + for (;;) { + if (undelivered().length === 0) break // delivered (maybe by another fiber): decided below + const mine = j.recs.get(g.leaseId) + if (mine?.grant !== g || mine.released !== undefined) return log("grant-superseded-before-create", { leaseId: g.leaseId }) + const why = abandoned.get(g.leaseId) + if (why !== undefined) return yield* abandon(g.leaseId, why, "abandoned-before-create") + yield* drainOutbox + if (undelivered().length === 0) break + if (Date.now() >= until) + return yield* failG(g, { reason: "pre-start/superseded-verdict-pending", message: `the verdict of ${undelivered().join(",")} is not delivered yet` }) + yield* Effect.sleep(Math.min(cfg.resyncMs, 2 * cfg.secondMs)) + } + } + // Round 2: withdrawn only when the floor SAID so (Complete.withdrew names this grant). A duplicate answer (rule 4c) + // or a delivered retryable failure withdrew nothing: this grant is the retry, so fence the old attempt and create. + // Round 3: and only for the generation it withdrew (see drainOutbox). + const by = olds.find((o) => { const x = j.recs.get(o); return x?.withdrew === g.leaseId && x.withdrewGen !== undefined && sameGen(genOf(g), x.withdrewGen) }) + if (by !== undefined) { + // Round 4: only this generation's record; a regrant taken meanwhile is the floor's newer word and stays live + const cur = j.recs.get(g.leaseId) + if (cur?.grant !== g) return log("grant-superseded-before-create", { leaseId: g.leaseId }) + if (cur.released === undefined) j.append({ ev: "released", leaseId: g.leaseId, why: "withdrawn" }) + return log("withdrawn", { leaseId: g.leaseId, by }) + } + const self = j.recs.get(g.leaseId) + if (self?.grant !== g || self.released !== undefined) return log("grant-superseded-before-create", { leaseId: g.leaseId }) + for (const old of olds) { // L3 and B2: the old attempt is gone before attempt n+1 is created + const rec = j.recs.get(old) + if (rec && rec.released === undefined) j.append({ ev: "released", leaseId: old, why: "superseded" }) + if (rec && Journal.maybeCreated(rec)) { + if (!(yield* fence(old))) { + log("fence-failed", { leaseId: g.leaseId, old }) + return yield* failG(g, { reason: "pre-start/ax-unavailable", message: `superseded attempt ${old} not confirmed deleted` }) + } + continue + } + // Round 1 (token scope): a name this link has no journal record for is never deleted on the floor's word alone. + // Absent is fine. Present is fenced only when it is an earlier attempt of THIS job and the Task itself carries + // this link's marks (the journal-lost case, F8); anything else is someone else's Task and this grant is refused. + const r = yield* ax.getTask(old).pipe(Effect.result) + if (r._tag === "Failure") + return yield* failG(g, { reason: "pre-start/ax-unavailable", message: `superseded attempt ${old} not confirmed absent` }) + if (r.success !== undefined && ownEarlierAttempt(old, g, r.success)) { + log("supersedes-unjournaled-own", { leaseId: g.leaseId, old }) + if (!(yield* fence(old))) { + log("fence-failed", { leaseId: g.leaseId, old }) + return yield* failG(g, { reason: "pre-start/ax-unavailable", message: `superseded attempt ${old} not confirmed deleted` }) + } + continue + } + if (r.success !== undefined) { + log("supersedes-foreign", { leaseId: g.leaseId, old }) + return yield* failG(g, { reason: "pre-start/supersedes-foreign", message: `supersedes names ${old}, a Task this link did not create` }) + } + } + yield* createOnly(g) + }).pipe( + Effect.ensuring(Effect.sync(() => { if (dispatching.get(g.leaseId) === g) dispatching.delete(g.leaseId) })), + // L9. Round 4: a defect is not swallowed while the lease is held. The grant is failed back (infra/link-defect, + // retryable: the floor requeues it and the next attempt fences anything this one may have created); if even that + // cannot be journaled, the process dies (exit 1) and a restart resumes from the journal (B13). + Effect.catchCause((c) => Cause.hasInterruptsOnly(c) ? Effect.interrupt : Effect.gen(function*() { + log("job-fiber-died", { leaseId: g.leaseId, cause: Cause.pretty(c) }) + const rec = j.recs.get(g.leaseId) + if (rec?.grant !== g || rec.report !== undefined || rec.released !== undefined || rec.notMine) return + yield* failG(g, { reason: "infra/link-defect", message: Cause.pretty(c).slice(0, 300) }) + }).pipe(Effect.catchCause((c2) => Cause.hasInterruptsOnly(c2) ? Effect.interrupt + : Effect.sync(() => log("link-defect-fatal", { leaseId: g.leaseId, cause: Cause.pretty(c2).slice(0, 300) })).pipe( + Effect.andThen(Deferred.die(fatal, Cause.squash(c2))), Effect.asVoid))))) + + // ---------------------------------------------------------------- resync: the executor's list is the ledger + const outcomeOf = (id: string, t: AxObserved, salvage = false) => Effect.gen(function*() { + const fail = (id: string, o: FailureOutput) => report(id, "failure", o, undefined, salvage) + if (t.phase === "Completed") { + const res = yield* ax.getTaskResult(id).pipe(Effect.result) + // Round 4: UNIMPLEMENTED is the server losing P1 (a rollout), not this Task's result: close the gate at once and + // keep the Task, uncharged, to read it once P1 is back. Guest mode does not read results through ax. + if (res._tag === "Success" && res.success === "unimplemented" && cfg.completion !== "guest") { + setServerP1(false) + return log("result-read-deferred", { leaseId: id, why: "server without P1" }) + } + const unreadable = res._tag === "Failure" ? `GetTaskResult: ${res.failure.message}` + : res.success === undefined ? "no result stored" + : res.success === "unimplemented" ? "server without P1" + : res.success.digestOk === false ? "sha256 mismatch" : undefined + if (unreadable !== undefined) { // B4: a transient read is retried at the next resync, bounded + const n = (resultFailures.get(id) ?? 0) + 1 + resultFailures.set(id, n) + const permanent = res._tag === "Failure" && !res.failure.unavailable + if (!permanent && n < resultReadTries) return log("result-read-retry", { leaseId: id, try: n, why: unreadable }) + return yield* fail(id, { reason: "infra/result-unreadable", phase: t.phase, message: unreadable }) + } + const ok = (res as { success: { content: string; sha256: string } }).success + const bytes = Buffer.byteLength(ok.content, "utf8") + if (bytes > maxOutputBytes) // round 2: refused here, typed, before the floor answers SQLITE_TOOBIG for ever + return yield* fail(id, { reason: "agent/output-too-large", phase: t.phase, + message: `result ${bytes} B > ${maxOutputBytes} B (sha256 ${ok.sha256 || createHash("sha256").update(ok.content).digest("hex")})` }) + let output: unknown + try { output = JSON.parse(ok.content) } catch { return yield* fail(id, { reason: "agent/result-not-json", phase: t.phase }) } + const u = t.usage + return yield* report(id, "success", output, u ? { prompt_tokens: u.promptTokens ?? 0, completion_tokens: u.completionTokens ?? 0, tool_calls: u.toolCalls ?? 0 } : undefined, salvage) + } + const ready = t.conditions.find((c) => c.type === "Ready") + const why = `${ready?.reason ?? ""} ${ready?.message ?? ""}` + const code = /ExitCode=(\d+)/.exec(why) + const reason = /ResourceExhausted/i.test(why) ? "pre-start/resource-exhausted" // B5: a capacity race, retryable (#367) + : /ActorTemplateRejected/.test(why) ? "pre-start/actor-template-rejected" + : /CRASH/i.test(why) ? "infra/actor-crashed" : code ? "agent/exit-code" : "agent/failed" + yield* fail(id, { reason, phase: t.phase, ...(ready?.message ? { message: ready.message } : {}), ...(code ? { exitCode: Number(code[1]) } : {}) }) + }) + + /** B1: a Task missing from the list is confirmed with GetTask. `gone` is true only on NotFound. */ + const confirm = (id: string) => ax.getTask(id).pipe(Effect.map((t) => ({ t, gone: t === undefined })), Effect.orElseSucceed(() => undefined)) + + /** Round 2: before any DeleteTask of a Task that may exist and whose verdict is not journaled (heartbeat `lost`, the + * janitor, a supersede fence), GetTask it; a terminal Task has its outcome read and journaled first. `unknown` means + * ax did not answer or the result read is being retried: delete nothing yet. */ + const salvage = (id: string) => Effect.gen(function*() { + const r = j.recs.get(id) + if (!r || r.report !== undefined || !Journal.maybeCreated(r) || r.deleting !== undefined || r.deleted) return "none" as const + const t = yield* ax.getTask(id).pipe(Effect.result) + if (t._tag === "Failure") return "unknown" as const + if (t.success === undefined) return "gone" as const + if (!owns(id, r, t.success)) { yield* markNotMine(id); return "foreign" as const } // round 3: never read another's result + if (!TERMINAL.has(t.success.phase)) return "live" as const + yield* outcomeOf(id, t.success, true) + return j.recs.get(id)?.report !== undefined ? "read" as const : "unknown" as const + }) + + const reconcile = (id: string, r: LeaseRec, listed: AxObserved | undefined, now: number) => Effect.gen(function*() { + if (!Journal.maybeCreated(r)) return // not created yet: dispatch owns it (or a restart resumes it) + let t = listed + let gone = false + if (t === undefined && !r.deleted) { + const c = yield* confirm(id) + if (c === undefined) return // ax did not answer; decide nothing this round + t = c.t; gone = c.gone + } + if (!gone) absent.delete(id) + if (t !== undefined && !gone && !owns(id, r, t)) return yield* markNotMine(id) // round 3: token scope + if (r.deleting !== undefined) { + if (gone) { + j.append({ ev: "deleted", leaseId: id }) + if (r.deleting === "cancel" && !r.report && r.released === undefined) yield* report(id, "cancelled") + } else if (t !== undefined && t.phase !== "Terminating") { // round 3: a journaled DeleteTask that did not land + if (r.released !== undefined && r.report === undefined && !r.reported && r.dropped === undefined && TERMINAL.has(t.phase)) { + yield* outcomeOf(id, t, true) // a released lease's finished Task is read before it is deleted + if (j.recs.get(id)?.report === undefined) return + } + yield* del(id, r.deleting) + } + return + } + if (r.reported || r.dropped !== undefined || r.released !== undefined) { // the floor is told: janitor only + if (gone) return j.append({ ev: "deleted", leaseId: id }) + if (t === undefined) return + if (r.released !== undefined && r.report === undefined && !r.reported && r.dropped === undefined && TERMINAL.has(t.phase)) { + yield* outcomeOf(id, t, true) // round 2: a released lease's finished Task is read before the janitor deletes it + if (j.recs.get(id)?.report === undefined) return // read retried at the next resync; nothing deleted before + } + const live = !TERMINAL.has(t.phase) // B3: a told lease's sandbox that still runs goes at once + if (live || r.released !== undefined || r.dropped !== undefined || now - r.at >= cfg.deleteAfterMs) + yield* del(id, r.released ? "released" : r.dropped ? "dropped" : live ? "told-still-running" : "janitor") + return + } + if (r.created === undefined) return // round 1: `creating` only: cleanup above, otherwise dispatch or a resume owns it + if (r.report) return // verdict written, waiting for the outbox + if (gone) { + const n = (absent.get(id) ?? 0) + 1 + absent.set(id, n) + if (n < 2) return log("task-absent", { leaseId: id, resyncs: n }) // B1: two NotFound resyncs in a row + return yield* fail(id, { reason: "infra/task-lost", message: "created by this link, NotFound on two resyncs" }) + } + if (t === undefined) return + if (TERMINAL.has(t.phase)) return yield* outcomeOf(id, t) + if ((t.phase === "" || t.phase === "Pending") && now - (r.createdAt ?? r.at) > cfg.pendingTimeoutMs) { + yield* del(id, "pre-start") + return yield* fail(id, { reason: "pre-start/pending-timeout", phase: t.phase || "Pending" }) + } + const deadline = r.grant.deadline + if (t.phase === "Running" && deadline !== undefined && now > deadline + cfg.deadlineBackstopMs) { + yield* del(id, "deadline") + return yield* fail(id, { reason: "deadline-exceeded", phase: t.phase }) + } + }) + + /** Round 4: "P1 proven" is a live condition. Logged on every change; p1-missing closes the gate (capacity 0). */ + const setServerP1 = (v: boolean) => { + if (v === serverP1) return + serverP1 = v + log("completion-probe", { serverP1, completion: cfg.completion, mode: mode() ?? "none" }) + if (!v && cfg.completion !== "guest") log("p1-missing", { completion: cfg.completion }) + } + const probe = Effect.gen(function*() { // P1 adds GetTaskResult; stock v0.3.0 answers UNIMPLEMENTED + const p = yield* ax.getTaskResult("conwip-link-capability-probe").pipe(Effect.result) + if (p._tag === "Failure") return // not answered: keep what is known (undefined after an outage: capacity 0) + setServerP1(p.success !== "unimplemented") + }) + + const checkGateway = Effect.gen(function*() { // B10 + const g = yield* ax.getGateway(cfg.shape.gateway).pipe(Effect.result) + const ok = g._tag === "Success" && g.success !== undefined && !gatewayAllowsAll(g.success) + if (ok !== gatewayOk) log(ok ? "gateway-ok" : "gateway-missing", { gateway: cfg.shape.gateway, ...(g._tag === "Failure" ? { code: g.failure.code } : g.success === undefined ? { found: false } : { allowsAll: true }) }) + gatewayOk = ok + }) + + const resync = Effect.gen(function*() { + const listed = yield* listAll(ax).pipe(Effect.result) + if (listed._tag === "Failure") { + if (axUp) log("ax-down", { code: listed.failure.code }) + axUp = false + // Round 2: nothing read before the outage opens the gate after it. Guest mode does not depend on serverP1, so + // the Gateway verdict and the resync flag are cleared too; canCreate() opens only after a full resync. + resynced = false + gatewayOk = undefined + return + } + const wasDown = !axUp + if (wasDown) serverP1 = undefined // B6: nothing proven before the outage holds after it + yield* probe // round 4: on every resync, since a rollout replaces ax-server with no failed ListTasks + yield* checkGateway + last = new Map(listed.success.map((t) => [t.name, t])) + if (wasDown) log("ax-up", { tasks: listed.success.length }) + axUp = true // round 2: only once probe, Gateway and the occupancy snapshot are all from this side of the outage + const now = Date.now() + for (const [id, r] of [...j.recs]) { + if (dispatching.has(id)) continue + yield* reconcile(id, r, last.get(id), now) + } + resynced = true + }) + + // Occupancy = |live Tasks in the atespace (ours or not) U leases held and not known terminal| (limiter rebuilt from + // the executor's list, agent-stack-k8s limiter.go). A P1-terminal Task is suspended and holds no worker. + const occupancy = () => { + const b = new Set() + for (const [name, t] of last) if (!TERMINAL.has(t.phase)) b.add(name) + for (const id of j.held()) { const t = last.get(id); if (!(t && TERMINAL.has(t.phase))) b.add(id) } + return b.size + } + + // ---------------------------------------------------------------- heartbeat: level-triggered, complete held set + const heartbeat = Effect.gen(function*() { + const ids = j.held() + // Round 4: the reply is about the generations this call vouched for. A reply that lands after a regrant of the + // same leaseId (gen n+1) says nothing about it: `lost`, `cancelRequested` and `renewed` apply only while the record + // still holds the grant object that was sent. + const sent = new Map(ids.map((id) => [id, j.recs.get(id)?.grant])) + const current = (id: string) => sent.has(id) && j.recs.get(id)?.grant === sent.get(id) + const sentAt = Date.now() + const pendingRequestKey = j.pendingLeaseKey // round 4: vouch for grants under a key whose reply never landed (B8) + const r = yield* floor.heartbeat({ holderIdentity: cfg.holder, leaseIds: ids, ...(pendingRequestKey !== undefined ? { pendingRequestKey } : {}) }).pipe(Effect.result) + if (r._tag === "Failure") { + if (r.failure.fatal) { yield* failFatal(r.failure); return undefined } + log("heartbeat-error", { kind: r.failure.kind }) + return undefined + } + for (const id of r.success.renewed) { const g = sent.get(id); if (g !== undefined && current(id)) renewedAt.set(g, Math.max(renewedAt.get(g) ?? 0, sentAt)) } + for (const id of r.success.lost) { // fencing: the lease is gone, so the Task goes; no report + if (!current(id)) { log("stale-heartbeat-reply", { leaseId: id, about: "lost" }); continue } + const rec = j.recs.get(id) + pendingResume.delete(id) + if (!rec) { const x = j.invalid.get(id); if (x && x.released === undefined) j.append({ ev: "released", leaseId: id, why: "lost" }); continue } + if (rec.released !== undefined) continue + j.append({ ev: "released", leaseId: id, why: "lost" }) + abandoned.set(id, "lost") + log("lost", { leaseId: id }) + if (Journal.maybeCreated(rec)) { // round 1: `creating` counts; the janitor confirms NotFound + const s = yield* salvage(id) // round 2: a finished Task's verdict is read and journaled before the delete + if (s === "unknown") log("delete-deferred", { task: id, why: "lost", until: "verdict-read" }) // the janitor reads, then deletes + else if (s !== "gone" && s !== "foreign") yield* del(id, "lost") + } + } + for (const id of r.success.cancelRequested) { // Temporal cancel_requested; ax cancel is DeleteTask, two-phase + if (!current(id)) { log("stale-heartbeat-reply", { leaseId: id, about: "cancel" }); continue } + const rec = j.recs.get(id) + if (!rec || rec.report || rec.deleting !== undefined) continue + log("cancel", { leaseId: id }) + pendingResume.delete(id) + abandoned.set(id, "cancel") // B3: a running dispatch reports or deletes; no later resume creates it + if (dispatching.has(id)) continue + // Round 1: a Task that may exist (`creating`) is deleted and seen NotFound before `cancelled` is reported + if (Journal.maybeCreated(rec)) yield* del(id, "cancel") + else yield* report(id, "cancelled") + } + const renewed = new Set(r.success.renewed) + for (const id of [...pendingResume]) { // B13: resume only what the floor still says is ours + if (!renewed.has(id) || !current(id)) continue + pendingResume.delete(id) + const rec = j.recs.get(id) + if (rec && rec.created === undefined && !rec.report && rec.released === undefined && rec.dropped === undefined) { + log("resume", { leaseId: id }) + yield* Effect.forkScoped(dispatch(rec.grant)) + } + } + return renewed + }) + + // ---------------------------------------------------------------- the loops + const every = (name: string, ms: () => number, body: Effect.Effect) => Effect.forever( + Effect.suspend(() => Effect.sleep(ms())).pipe(Effect.andThen(body), // re-read: the floor may change the interval + Effect.catchCause((c) => Cause.hasInterruptsOnly(c) ? Effect.interrupt : Effect.sync(() => log(`${name}-died`, { cause: Cause.pretty(c) }))))) + + /** Round 3: the heartbeat interval, at most a third of the shortest lease held, so a large floor value can never let + * a held lease expire between two heartbeats. */ + const hbSeconds = () => { + let hi = MAX_HEARTBEAT_SECONDS + for (const id of j.held()) { + const d = j.recs.get(id)?.grant.lease.leaseDurationSeconds + if (d !== undefined && Number.isFinite(d) && d > 0) hi = Math.min(hi, d / 3) + } + return boundedSeconds(hbAsked, defaultHb, Math.max(1, hi)) + } + + /** Final pass (VERIFY-LINK R3-1). The heartbeat sleep is measured from the last heartbeat and is cut short when a + * Lease reply shrinks the interval or journals a grant with a shorter lease: a sleep begun at the 30 s default while + * nothing was held no longer outlives a 12 s lease granted during it. */ + let lastBeatAt = Date.now() + const heartbeatLoop = Effect.forever(Effect.gen(function*() { + let target = jitter(hbSeconds() * cfg.secondMs) + for (;;) { + const left = lastBeatAt + target - Date.now() + if (left <= 0) break + const w = hbWake + yield* Effect.raceFirst(Effect.sleep(left), Deferred.await(w)) + if (yield* Deferred.isDone(w)) { + hbWake = yield* Deferred.make() + target = Math.min(target, jitter(hbSeconds() * cfg.secondMs)) + } + } + lastBeatAt = Date.now() + yield* heartbeat + }).pipe(Effect.catchCause((c) => Cause.hasInterruptsOnly(c) ? Effect.interrupt : Effect.sync(() => log("heartbeat-died", { cause: Cause.pretty(c) }))))) + + let endpointSeen: string | undefined + const leaseOnce = Effect.gen(function*() { + outboxWaitUntil = Math.min(outboxWaitUntil, Date.now()) // B15: a verdict goes out before new work comes in + yield* drainOutbox + // Round 1 (B15): while any verdict is still undelivered, ask for nothing. A grant that arrived now could supersede + // that verdict's attempt, and once n+1 is leased the floor answers n's verdict stale-attempt. + // Round 2: a verdict the floor has answered with a server error is not an outage; it no longer gates Lease (its + // own budget replaces or dead-letters it, and a grant that supersedes it waits at dispatch, bounded). + const undelivered = j.outbox().filter((p) => !verdictFailures.has(p.leaseId)).length + if (undelivered > 0) log("lease-gated-by-outbox", { undelivered }) + const free = undelivered === 0 && canCreate() ? Math.max(0, cfg.maxInFlight - occupancy()) : 0 + // B8: a key is journaled before the call and reused until a reply is journaled, so a lost reply is replayed, + // not stranded. A capacity-0 call can grant nothing, so it is not journaled. + let requestKey = j.pendingLeaseKey + if (requestKey === undefined) { + requestKey = randomUUID() + if (free > 0) j.append({ ev: "lease-key", requestKey }) + } + const sentAt = Date.now() + const r = yield* callLease(free, requestKey) + if (r._tag === "Failure") { + if (r.failure.fatal) return yield* failFatal(r.failure) + leaseFailed(r.failure, requestKey) + // Round 3: bounded, and only this loop waits; the heartbeat and resync fibers run on their own intervals + return yield* Effect.sleep(Math.min(r.failure.retryAfterMs ?? jitter(pollSeconds * cfg.secondMs), MAX_POLL_SECONDS * cfg.secondMs)) + } + yield* applyLease(r.success, requestKey, sentAt) + }) + + const callLease = (capacity: number, requestKey: string) => floor.lease({ holderIdentity: cfg.holder, capacity, requestKey }).pipe( + Effect.retry({ times: 2, while: (e: FloorError) => e.kind === "transient", schedule: Schedule.exponential(Math.max(1, cfg.secondMs / 5)).pipe(Schedule.jittered) }), Effect.result) + const leaseFailed = (e: FloorError, requestKey: string) => { + log("lease-error", { kind: e.kind, pendingKey: j.pendingLeaseKey !== undefined, message: e.message.slice(0, 200) }) + // Round 3: a journaled key is replayed through outages (B8), but not through replies the floor keeps answering + // with a server error or that do not decode: those would strand the key's grants for ever. After + // `leaseKeyAttempts` in a row the key is abandoned; the floor's expiry requeues what it granted under it. + if (e.kind === "server-error" && j.pendingLeaseKey === requestKey) { + if (++leaseKeyFailures >= leaseKeyAttempts) { + j.append({ ev: "lease-replied", requestKey, abandoned: true }) + log("lease-key-abandoned", { tries: leaseKeyFailures, message: e.message.slice(0, 200) }) + leaseKeyFailures = 0 + } + } + } + /** Round 4 (B8 at startup): a key journaled by a previous process whose reply never landed is replayed BEFORE the + * first Heartbeat, so its grants are journaled and in the held set when the Heartbeat re-adopts orphans; replayed + * after it, an orphan granted under the key is omitted and released (L3 a). Capacity 0: a key the floor never saw + * grants nothing new. A failed replay leaves the key pending; the Heartbeat then vouches for it. */ + const replayPendingLease = Effect.gen(function*() { + const requestKey = j.pendingLeaseKey + if (requestKey === undefined) return + const sentAt = Date.now() + const r = yield* callLease(0, requestKey) + if (r._tag === "Failure") { + if (r.failure.fatal) return yield* failFatal(r.failure) + return leaseFailed(r.failure, requestKey) + } + log("lease-key-replayed", { grants: r.success.grants.length }) + yield* applyLease(r.success, requestKey, sentAt) + }) + + const applyLease = (reply: LeaseReply, requestKey: string, sentAt: number) => Effect.gen(function*() { + leaseKeyFailures = 0 + // Round 3: the server-driven intervals are bounded (Buildkite falls back to its defaults): 0, a negative or a + // missing value would make both loops tight loops of billed requests. + pollSeconds = boundedSeconds(reply.nextPollSeconds, defaultPoll, MAX_POLL_SECONDS) + hbAsked = reply.heartbeatSeconds + const hbNow = hbSeconds() + const clamp = `${reply.nextPollSeconds}->${pollSeconds} ${reply.heartbeatSeconds}->${hbNow}` + if ((pollSeconds !== reply.nextPollSeconds || hbNow !== reply.heartbeatSeconds) && clamp !== clampLogged) { + clampLogged = clamp + log("interval-clamped", { nextPollSeconds: reply.nextPollSeconds ?? null, usedPoll: pollSeconds, heartbeatSeconds: reply.heartbeatSeconds ?? null, usedHeartbeat: hbNow }) + } + for (const g of reply.grants) { + const prior = j.recs.get(g.leaseId) + // Round 3: a grant of a lease this link finished (compacted away since) is at-least-once redelivery, not work. + const tomb = j.tombs.get(g.leaseId) + if (prior === undefined && tomb !== undefined && !laterGen(genOf(g), tomb.gen)) { log("duplicate-grant", { leaseId: g.leaseId, finished: true }); continue } + if (prior !== undefined) { + // The floor already said it withdrew this generation (Complete.withdrew journaled on an earlier attempt) but its + // dispatch has not run yet: release it now, so its regrant is not taken for a duplicate. + const pg = genOf(prior.grant) + if (prior.released === undefined && !Journal.maybeCreated(prior) && [...j.recs.values()].some((x) => x.withdrew === g.leaseId && x.withdrewGen !== undefined && sameGen(pg, x.withdrewGen))) { + j.append({ ev: "released", leaseId: g.leaseId, why: "withdrawn" }) + log("withdrawn", { leaseId: g.leaseId, by: "journal" }) + } + // Round 2: dedupe on the grant's identity, not its leaseId. A floor that withdrew n+1 (rule 4b) and then requeued + // it for n's retryable verdict grants the same leaseId again with a new acquireTime and leaseTransitions; that + // is a new grant generation, dispatched, when the old record is released and holds nothing in ax or the outbox. + const newer = laterGen(genOf(g), pg) // round 3: strictly later, so a stale redelivery is never a regrant + // Round 4: the floor's generation is authoritative. A strictly later grant means the floor already withdrew + // the one held (its Complete reply may still be in flight, or lost). If the held one never reached ax and owes + // no verdict, it is released now and the regrant taken; the floor does not redeliver it. Its dispatch sees the + // record change and stops (grant-superseded-before-create). + if (newer && prior.released === undefined && !Journal.maybeCreated(prior) && !Journal.pending(prior)) { + j.append({ ev: "released", leaseId: g.leaseId, why: "superseded-by-regrant" }) + log("superseded-by-regrant", { leaseId: g.leaseId, was: pg.leaseTransitions, now: g.lease.leaseTransitions }) + } + const reusable = prior.released !== undefined && !Journal.pending(prior) && (!Journal.maybeCreated(prior) || prior.deleted === true) + if (!(newer && reusable)) { log("duplicate-grant", { leaseId: g.leaseId }); continue } // at-least-once delivery + for (const m of [abandoned, absent, resultFailures, verdictFailures, verdictNotBefore]) m.delete(g.leaseId) + admitted.delete(g.leaseId); pendingResume.delete(g.leaseId) + j.append({ ev: "grant", leaseId: g.leaseId, grant: g, regrant: true }) + log("regrant", { leaseId: g.leaseId, was: prior.released }) + } else j.append({ ev: "grant", leaseId: g.leaseId, grant: g }) + renewedAt.set(g, sentAt) // round 4: the floor acquired it no earlier than this call was sent + log("leased", { leaseId: g.leaseId, attempt: g.attempt, supersedes: g.supersedes ?? [] }) + yield* Effect.forkScoped(dispatch(g)) + } + for (const b of reply.invalid ?? []) { // round 1: refuse one bad grant, keep the rest of the reply + if (b.leaseId === undefined || b.attempt === undefined) { log("grant-undecodable", { message: b.message }); continue } // the floor's expiry frees it + if (j.recs.has(b.leaseId) || j.invalid.has(b.leaseId)) continue + j.append({ ev: "grant-invalid", leaseId: b.leaseId, attempt: b.attempt, message: b.message }) + log("grant-invalid", { leaseId: b.leaseId, attempt: b.attempt }) + yield* Deferred.succeed(wake, undefined) + } + if (j.pendingLeaseKey === requestKey) j.append({ ev: "lease-replied", requestKey }) + yield* Deferred.succeed(hbWake, undefined) // final pass (R3-1): the interval or the held set may have shrunk + const ep = reply.endpoint // B18: only an exact match in the Nix-declared list, and only once + if (ep !== undefined && ep !== deps.floorUrl && ep !== endpointSeen) { + endpointSeen = ep + if ((cfg.floorUrls ?? []).includes(ep) && ep.startsWith("https://")) { log("endpoint-accepted", { endpoint: ep }); deps.onEndpoint?.(ep) } + else log("endpoint-ignored", { endpoint: ep }) + } + }) + + const body = Effect.gen(function*() { + yield* resync // limiter and outcomes rebuilt before the first heartbeat or lease + for (const [id, r] of j.recs) // grants journaled by a previous process but never created (B13: resumed on renewal) + if (r.created === undefined && !r.report && r.released === undefined && r.dropped === undefined && !r.notMine) pendingResume.add(id) + yield* replayPendingLease // round 4: a key left pending by a kill is answered before the first Heartbeat + yield* heartbeat // vouch first: re-adopt or release orphans (L3), then resume only renewed grants + lastBeatAt = Date.now() + // Round 3: renewal and resync start now, before any Complete or Lease can stall (a 429 Retry-After, a timeout + // chain): held leases keep being renewed whatever the first Lease does. Both re-read their interval every time. + yield* Effect.forkScoped(heartbeatLoop) + yield* Effect.forkScoped(every("resync", () => cfg.resyncMs, resync.pipe(Effect.andThen(drainOutbox)))) + yield* drainOutbox // replay before new work (uplink) + yield* Effect.raceFirst(leaseOnce, Deferred.await(deps.stop)) // its reply sets the server-driven intervals + while (!(yield* Deferred.isDone(deps.stop))) { + const w = wake + yield* Effect.raceFirst(Effect.raceFirst(Effect.suspend(() => Effect.sleep(jitter(pollSeconds * cfg.secondMs))), Deferred.await(w)), Deferred.await(deps.stop)) + if (yield* Deferred.isDone(deps.stop)) break + if (yield* Deferred.isDone(w)) wake = yield* Deferred.make() + yield* Effect.raceFirst(leaseOnce, Deferred.await(deps.stop)) + } + // drain (Buildkite stop before disconnect): no more leases; ax Tasks keep running and are re-adopted on restart + outboxWaitUntil = 0 + yield* drainOutbox + log("drained", { held: j.held().length }) + }) + + return yield* Effect.raceFirst(Effect.scoped(body), Deferred.await(fatal)) +}) diff --git a/pkgs/substrate-link/src/apps/link/src/main.ts b/pkgs/substrate-link/src/apps/link/src/main.ts new file mode 100644 index 000000000..fa4306e5e --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/src/main.ts @@ -0,0 +1,105 @@ +// conwip-link: the NAS-side process. Configuration is environment only (the NixOS module sets it); the bearer is a +// file under $CREDENTIALS_DIRECTORY (systemd LoadCredential from the agenix secret), read once and never logged. +// Exit codes: 0 drained on SIGTERM; 75 another session holds this identity (L5), or an allowed endpoint switch (B18) +// was persisted and the unit should restart onto it; 78 the floor refused the token, answered a redirect (round 2), or +// the configuration is invalid. +import { randomUUID } from "node:crypto" +import { existsSync, readFileSync, writeFileSync } from "node:fs" +import { join } from "node:path" +import { Deferred, Effect } from "effect" +import { grpcAx } from "./ax.ts" +import { ConfigInvalid, readLinkEnv } from "./config.ts" +import type { LinkEnv } from "./config.ts" +import { rpcFloor } from "./floor.ts" +import { seatOf } from "./jobs.ts" +import { Journal } from "./journal.ts" +import { runLink } from "./link.ts" + +const env = (k: string, d?: string) => { + const v = process.env[k] + if (v !== undefined && v !== "") return v + if (d !== undefined) return d + console.error(JSON.stringify({ ev: "config-missing", key: k })) + process.exit(78) +} +const log = (ev: string, f: Record = {}) => console.log(JSON.stringify({ t: new Date().toISOString(), ev, ...f })) +// Round 1: every numeric and enum key is validated up front; a bad one is exit 78 naming the key, never the value. +let c: LinkEnv +try { c = readLinkEnv(process.env) } catch (e) { + if (e instanceof ConfigInvalid) { console.error(JSON.stringify({ ev: "config-invalid", key: e.key, why: e.why })); process.exit(78) } + throw e +} + +const stateDir = env("LINK_STATE_DIR", process.env.STATE_DIRECTORY ?? "/var/lib/conwip-link") +const tokenPath = env("LINK_TOKEN_FILE", join(process.env.CREDENTIALS_DIRECTORY ?? "/nonexistent", "floor-link-token")) +if (!existsSync(tokenPath)) { log("token-missing", { path: tokenPath }); process.exit(78) } +const token = readFileSync(tokenPath, "utf8").trim() +const sessionPath = join(stateDir, "session-id") // stable across restarts of this unit, distinct for a second replica +const journal = Journal.open(stateDir) +if (!existsSync(sessionPath)) writeFileSync(sessionPath, randomUUID() + "\n", { mode: 0o600 }) +const sessionId = readFileSync(sessionPath, "utf8").trim() + +// B18: the floor URLs this link may use, declared in Nix. A switch the floor asked for is persisted here and used on +// the next start only while it is still in the declared list. +const declaredUrl = env("LINK_FLOOR_URL") +const floorUrls = (process.env.LINK_FLOOR_URLS ?? declaredUrl).split(",").map((u) => u.trim()).filter((u) => u !== "") +const endpointPath = join(stateDir, "floor-endpoint") +const persisted = existsSync(endpointPath) ? readFileSync(endpointPath, "utf8").trim() : "" +const floorUrl = persisted !== "" && floorUrls.includes(persisted) ? persisted : declaredUrl +let restartForEndpoint = false + +// Round 2: the link loads its own P1 proto (apps/link/proto/ax-p1.proto), never the stock one: with the stock proto +// GetTaskResult is answered "unimplemented" locally and completion auto/p1 would lease nothing for ever. +let ax: ReturnType +try { ax = grpcAx(c.axServer, c.atespace, undefined, c.axProtoPath) } catch (e) { + console.error(JSON.stringify({ ev: "config-invalid", key: "LINK_AX_PROTO_PATH", why: String((e as Error)?.message ?? e).slice(0, 200) })) + process.exit(78) +} +const program = Effect.gen(function*() { + const stop = yield* Deferred.make() + for (const sig of ["SIGTERM", "SIGINT"] as const) + process.once(sig, () => { log("stop-requested", { signal: sig }); Effect.runFork(Deferred.succeed(stop, undefined)) }) + const floor = yield* rpcFloor({ url: floorUrl, token, sessionId }) + yield* runLink({ + holder: c.holder, + maxInFlight: c.maxInFlight, + servedLabels: c.servedLabels, + shape: { + atespace: c.atespace, + image: c.image, + gateway: c.gateway, + command: (job) => { const seat = seatOf(job); return seat === undefined ? undefined : c.seatCommands[seat] }, // B7: no default seat + ...(c.guestCompleteUrl !== undefined ? { completeUrl: c.guestCompleteUrl } : {}) + }, + completion: c.completion, + secondMs: 1000, + resyncMs: c.resyncSeconds * 1000, + pendingTimeoutMs: c.pendingTimeoutSeconds * 1000, + deleteAfterMs: c.deleteAfterSeconds * 1000, + deadlineBackstopMs: c.deadlineBackstopSeconds * 1000, + createAttempts: c.createAttempts, + outboxBackoffMs: [c.outboxBackoffSeconds[0] * 1000, c.outboxBackoffSeconds[1] * 1000], + fenceTimeoutMs: c.fenceTimeoutSeconds * 1000, + resultReadTries: c.resultReadTries, + verdictAttempts: c.verdictAttempts, + maxOutputBytes: c.maxOutputBytes, + internalHosts: c.internalHosts, + floorUrls + }, { + ax, floor, journal, log, stop, floorUrl, + onEndpoint: (url) => { // B18: persist, drain, exit 75 so systemd restarts onto the new URL + writeFileSync(endpointPath, url + "\n", { mode: 0o600 }) + restartForEndpoint = true + Effect.runFork(Deferred.succeed(stop, undefined)) + } + }) +}) + +Effect.runPromise(Effect.scoped(program)).then( + () => { ax.close(); process.exit(restartForEndpoint ? 75 : 0) }, + (e) => { + ax.close() + const kind = (e as { kind?: string })?.kind + log("fatal", { kind: kind ?? "defect", message: String((e as { message?: string })?.message ?? e) }) + process.exit(kind === "session-conflict" ? 75 : kind === "auth" || kind === "redirect" || kind === "misrouted" ? 78 : 1) + }) diff --git a/pkgs/substrate-link/src/apps/link/test/ax-grpc.test.ts b/pkgs/substrate-link/src/apps/link/test/ax-grpc.test.ts new file mode 100644 index 000000000..32a137f37 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/ax-grpc.test.ts @@ -0,0 +1,91 @@ +// The production AxApi against a real gRPC server loaded from the vendored ax.proto (v0.3.0 d8ed0fe), on loopback. +// Proves the wire mapping the doubles skip: protojson field names survive UpdateTask, NotFound maps to undefined, +// DeleteTask NotFound is success. Round 2: the link loads its own P1 proto (apps/link/proto/ax-p1.proto), so the +// capability probe is decided by the SERVER: a stock server answers UNIMPLEMENTED over the wire (pre-P1), a P1 server +// answers; a proto without GetTaskResult is refused at construction instead of answering "unimplemented" locally. +import { createHash } from "node:crypto" +import * as grpc from "@grpc/grpc-js" +import * as protoLoader from "@grpc/proto-loader" +import { Effect } from "effect" +import { afterAll, beforeAll, expect, it } from "vitest" +import { resolveProtoPath, VENDORED_PROTO_PATH } from "../../../src/axclient.ts" +import { gatewayAllowsAll, grpcAx, LINK_PROTO_PATH, listAll } from "../src/ax.ts" +import { axTaskFromGrant } from "../src/jobs.ts" + +const stored = new Map() +const gateways = new Map([ + ["halogen", { apiVersion: "ax.io/v1alpha1", kind: "Gateway", metadata: { name: "halogen", atespace: "fleet" }, spec: { egress: { allowlist: { hosts: [{ host: "worker", port: 8731 }] } } } }], + ["open", { apiVersion: "ax.io/v1alpha1", kind: "Gateway", metadata: { name: "open", atespace: "fleet" }, spec: {} }] +]) +let server: grpc.Server, port = 0 +const nf = (name: string) => ({ code: grpc.status.NOT_FOUND, details: `task "${name}" not found` }) + +beforeAll(async () => { + const def = protoLoader.loadSync(resolveProtoPath(), { keepCase: false, longs: String, enums: String, defaults: true, oneofs: true }) + const svc = (grpc.loadPackageDefinition(def) as any).ax.v1alpha1.AX.service + server = new grpc.Server() + const impl: Record> = { + GetTask: (c, cb) => { const t = stored.get(c.request.name); t ? cb(null, t) : cb(nf(c.request.name)) }, + UpdateTask: (c, cb) => { stored.set(c.request.task.metadata.name, { ...c.request.task, status: { phase: "Pending" } }); cb(null, c.request.task) }, + // upstream server.go:91-104: limit <= 0 means 50, newest first (store.go ZRevRange) + ListTasks: (c, cb) => { const n = Number(c.request.limit) > 0 ? Number(c.request.limit) : 50, o = Number(c.request.offset) || 0; cb(null, { tasks: [...stored.values()].reverse().slice(o, o + n) }) }, + GetGateway: (c, cb) => { const g = gateways.get(c.request.name); g ? cb(null, g) : cb({ code: grpc.status.NOT_FOUND, details: "gateway not found" }) }, + DeleteTask: (c, cb) => stored.delete(c.request.name) ? cb(null, {}) : cb(nf(c.request.name)) + } + server.addService(svc, impl) + port = await new Promise((res, rej) => server.bindAsync("127.0.0.1:0", grpc.ServerCredentials.createInsecure(), (e, p) => e ? rej(e) : res(p))) +}) +afterAll(() => { server.forceShutdown() }) + +it("G1 grpcAx speaks ax.proto: create-only round trip, NotFound as undefined, a stock server answers GetTaskResult UNIMPLEMENTED", async () => { + const ax = grpcAx(`127.0.0.1:${port}`, "fleet", 3000) + const job = { + apiVersion: "ultracode.mecattaf.dev/v1alpha1" as const, kind: "AgentJob" as const, + metadata: { name: "wf-g-1-abababab", annotations: { "ultracode.mecattaf.dev/item-key": "wf_g#1", "ultracode.mecattaf.dev/journal-key": `${"cd".repeat(32)}:1` } }, + spec: { "runs-on": ["seat:halogen"], with: { prompt: "hi", prompt_ref: { sha256: "ab".repeat(32), bytes: 2, uri: "journal://wf_g/1/prompt.md" }, model: "halogen-qwen3.8-flash-next" } } + } + const built = axTaskFromGrant({ leaseId: "wf-g-1-abababab-a1", attempt: 1, job, lease: { holderIdentity: "nas-link-1", leaseDurationSeconds: 90, acquireTime: 0, renewTime: 0, leaseTransitions: 0 } }, + { atespace: "fleet", image: "localhost:5000/ax-agent@sha256:00", gateway: "halogen", command: () => ["ax-agent", "pi"] }) + if (built._tag !== "ok") throw new Error(built.reason) + const run = (e: Effect.Effect) => Effect.runPromise(e) + expect(await run(ax.getTask("wf-g-1-abababab-a1"))).toBeUndefined() + await run(ax.createTask(built.task)) + const back = stored.get("wf-g-1-abababab-a1") + expect([back.apiVersion, back.metadata.atespace, back.spec.gateway.name, back.spec.command]).toEqual(["ax.io/v1alpha1", "fleet", "halogen", ["ax-agent", "pi"]]) + expect(back.spec.env.find((e: any) => e.name === "AX_CONWIP_ITEM_KEY").value).toBe("wf_g#1") + const seen = await run(ax.getTask("wf-g-1-abababab-a1")) + expect(seen?.phase).toBe("Pending") + expect((await run(ax.listTasks(0, 0))).map((t) => t.name)).toEqual(["wf-g-1-abababab-a1"]) + for (let i = 0; i < 120; i++) stored.set(`bulk-${i}`, { apiVersion: "ax.io/v1alpha1", kind: "Task", metadata: { name: `bulk-${i}`, atespace: "fleet" }, spec: {}, status: { phase: "Completed" } }) + expect((await run(ax.listTasks(0, 0))).length).toBe(50) // B1: one call is one 50-row page + const all = await run(listAll(ax)) + expect([all.length, all.some((t) => t.name === "wf-g-1-abababab-a1")]).toEqual([121, true]) + for (let i = 0; i < 120; i++) stored.delete(`bulk-${i}`) + const gw = await run(ax.getGateway("halogen")) // B10 wire mapping + expect([gw?.hasAllowlist, gw?.hosts, gatewayAllowsAll(gw!)]).toEqual([true, [{ host: "worker", port: 8731 }], false]) + const open = await run(ax.getGateway("open")) + expect([open?.hasAllowlist, gatewayAllowsAll(open!)]).toEqual([false, true]) // no egress: ax applies *:443 + expect(await run(ax.getGateway("missing"))).toBeUndefined() + await run(ax.deleteTask("wf-g-1-abababab-a1")) + await run(ax.deleteTask("wf-g-1-abababab-a1")) // NotFound is success: delete is idempotent for the janitor + expect(await run(ax.getTaskResult("anything"))).toBe("unimplemented") // v0.3.0 server has no GetTaskResult: pre-P1 mode + ax.close() +}) + +it("R2-10 the packaged link (default proto) proves P1 against a P1 server and reads a result; a stock proto is refused", async () => { + let calls = 0 + const def = protoLoader.loadSync(LINK_PROTO_PATH, { keepCase: false, longs: String, enums: String, defaults: true, oneofs: true }) + const svc = (grpc.loadPackageDefinition(def) as any).ax.v1alpha1.AX.service + const p1 = new grpc.Server() + const body = Buffer.from('{"ok":true}') + p1.addService(svc, { + GetTaskResult: (c: any, cb: any) => { calls++; c.request.name === "t1" ? cb(null, { content: body, sha256: createHash("sha256").update(body).digest("hex") }) : cb({ code: grpc.status.NOT_FOUND, details: "nf" }) } + }) + const p1Port = await new Promise((res, rej) => p1.bindAsync("127.0.0.1:0", grpc.ServerCredentials.createInsecure(), (e, p) => e ? rej(e) : res(p))) + const ax = grpcAx(`127.0.0.1:${p1Port}`, "fleet", 3000) // exactly what main.ts builds when LINK_AX_PROTO_PATH is unset + expect(await Effect.runPromise(ax.getTaskResult("conwip-link-capability-probe"))).toBeUndefined() // NotFound: P1 present + expect(await Effect.runPromise(ax.getTaskResult("t1"))).toEqual({ content: '{"ok":true}', sha256: createHash("sha256").update(body).digest("hex"), digestOk: true }) + expect(calls).toBe(2) + ax.close(); p1.forceShutdown() + expect(() => grpcAx("127.0.0.1:1", "fleet", 1000, resolveProtoPath({ AX_CONWIP_PROTO_PATH: VENDORED_PROTO_PATH }))).toThrow(/GetTaskResult/) +}) diff --git a/pkgs/substrate-link/src/apps/link/test/critique-fixed.test.ts b/pkgs/substrate-link/src/apps/link/test/critique-fixed.test.ts new file mode 100644 index 000000000..12a7799a7 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/critique-fixed.test.ts @@ -0,0 +1,406 @@ +// The nine defects the 2026-09-23 critique reproduced (LINK-DESIGN.md "Critique applied", section B), each INVERTED +// into the fixed behaviour now that its build item is in (commit history: the characterization versions are in +// 6342476 and b900624). Plus the new cases the build list names for B1, B6, B9 to B13, B15, B17 and B18. +// Same doubles as link.test.ts; the floor double implements section H's amended rules (B16). +import { mkdtempSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import { Deferred, Effect, Fiber } from "effect" +import { afterEach, beforeEach, describe, expect, it } from "vitest" +import { AxError } from "../src/ax.ts" +import type { AgentJob, AxTask } from "../src/contract.ts" +import { rpcFloor } from "../src/floor.ts" +import type { FloorOptions } from "../src/floor.ts" +import { Journal } from "../src/journal.ts" +import { isFleetInternalUrl, runLink } from "../src/link.ts" +import type { LinkConfig } from "../src/link.ts" +import { FakeAx } from "./fake-ax.ts" +import { FakeFloor } from "./fake-floor.ts" +import type { FloorConfig } from "./fake-floor.ts" + +const SEC = 20 +const UNAVAILABLE = 14 +const A = "ultracode.mecattaf.dev/" +const JK = `${"cd".repeat(32)}:1` // round 4: FIELD-MAP 5a journal key (jobs.ts refuses a grant without it) +const job = (n: string, spec: Partial = {}): AgentJob => ({ + apiVersion: "ultracode.mecattaf.dev/v1alpha1", kind: "AgentJob", + metadata: { + name: `wf-test-${n}`, + labels: { [A + "run-id"]: "wf-test", [A + "workflow"]: "link-test", [A + "phase-index"]: "1" }, + annotations: { [A + "run-id-raw"]: "wf_test", [A + "label"]: `probe:${n}`, [A + "item-key"]: `wf_test#${n}`, [A + "journal-key"]: JK, [A + "phase-title"]: "Probe" } + }, + spec: { + "runs-on": ["seat:halogen", "runtime:gvisor"], + with: { prompt: `say ${n}`, prompt_ref: { sha256: "ab".repeat(32), bytes: 5, uri: `journal://wf_test/${n}/prompt.md` }, model: "halogen-qwen3.8-flash-next" }, + ...spec + } +}) +const world = (o: Partial = {}) => ({ + floor: new FakeFloor({ cap: 2, leaseSeconds: 6, graceSeconds: 15, pollSeconds: 1, heartbeatSeconds: 2, maxAttempts: 3, secondMs: SEC, tokens: { "tok-nas": "nas-link-1" }, ...o }), + ax: new FakeAx() +}) +type World = ReturnType +const foreign = (name: string): AxTask => ({ apiVersion: "ax.io/v1alpha1", kind: "Task", metadata: { name, atespace: "fleet" }, spec: { image: "someone-else", command: ["true"], env: [] } }) + +interface Extra { fetch?: typeof globalThis.fetch; timeoutsMs?: FloorOptions["timeoutsMs"]; onEndpoint?: (u: string) => void } +function start(w: World, dir: string, over: Partial = {}, x: Extra = {}) { + const logs: Array> = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { + holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 1500, deleteAfterMs: 0, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, fenceTimeoutMs: 300, resultReadTries: 3, ...over + } + const fiber = Effect.runFork(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "tok-nas", sessionId: `s:${dir}`, fetch: x.fetch ?? w.floor.fetch, ...(x.timeoutsMs ? { timeoutsMs: x.timeoutsMs } : {}) }) + return yield* runLink(cfg, { ax: w.ax, floor, journal: Journal.open(dir), log: (ev, f) => logs.push({ ev, ...f }), stop, floorUrl: "http://floor.test", ...(x.onEndpoint ? { onEndpoint: x.onEndpoint } : {}) }) + }))) + return { + logs, + has: (ev: string, leaseId?: string) => logs.some((l) => l.ev === ev && (leaseId === undefined || l.leaseId === leaseId)), + verdict: (reason: string) => logs.some((l) => l.ev === "verdict" && l.reason === reason), + crash: () => Effect.runPromise(Fiber.interrupt(fiber)), + drain: () => { Effect.runSync(Deferred.succeed(stop, undefined)); return Promise.race([Effect.runPromise(Fiber.await(fiber)), sleep(1500)]) } + } +} +const until = async (what: string, pred: () => boolean, ms = 5000) => { + const t0 = Date.now() + while (!pred()) { if (Date.now() - t0 > ms) throw new Error(`timeout waiting for: ${what}`); await new Promise((r) => setTimeout(r, 5)) } +} +const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const J = (w: World, n = "1") => w.floor.jobs.get(`wf-test-${n}`)! +const liveOwn = (w: World) => [...w.ax.tasks.keys()].filter((k) => k.startsWith("wf-test-") && w.ax.live(k) && !["Completed", "Failed"].includes(w.ax.tasks.get(k)!.phase)).length + +describe("critique 2026-09-23: the defects, fixed", () => { + let dir: string + let worlds: Array = [] + const W = (o: Partial = {}) => { const w = world(o); worlds.push(w); return w } + beforeEach(() => { dir = mkdtempSync(join(tmpdir(), "conwip-link-critique-")) }) + afterEach(() => { for (const w of worlds) w.floor.close(); worlds = []; rmSync(dir, { recursive: true, force: true }) }) + + it("C1 (B1) a live Task off ListTasks' first 50-row page is neither lost nor rerun", async () => { + const w = W(); w.floor.enqueue(job("1")) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + for (let i = 0; i < 50; i++) w.ax.put(foreign(`foreign-${i}`), "Completed") // newer than ours: ours is on page 2 + const calls = w.ax.listCalls + await until("ten resyncs", () => w.ax.listCalls >= calls + 10) + expect([l.verdict("infra/task-lost"), l.has("task-absent"), w.ax.tasks.has("wf-test-1-a2"), w.ax.deletes]).toEqual([false, false, false, []]) + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).output, J(w).attempt]).toEqual(["success", { answer: 42 }, 1]) + await l.drain() + }) + + it("C1b (B1) 120 Tasks: occupancy counts a live foreign Task on page 3, and a truly gone Task needs two NotFound resyncs", async () => { + const w = W() + for (let i = 0; i < 2; i++) w.ax.put(foreign(`busy-${i}`), "Running") // oldest: on the last page + for (let i = 0; i < 118; i++) w.ax.put(foreign(`done-${i}`), "Completed") + w.floor.enqueue(job("1")) + const l = start(w, dir) + await until("a Lease was sent", () => w.floor.calls.some((c) => c.rpc === "Lease")) + await sleep(10 * SEC) + expect(w.floor.calls.filter((c) => c.rpc === "Lease").every((c) => c.capacity === 0)).toBe(true) // 2 busy of 2 + w.ax.tasks.delete("busy-0"); w.ax.tasks.delete("busy-1") + await until("created once the sandboxes free", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.tasks.delete("wf-test-1-a1") // the store forgets it (F19) + await until("task-lost after two NotFound resyncs", () => l.verdict("infra/task-lost")) + const absent = l.logs.filter((x) => x.ev === "task-absent" && x.leaseId === "wf-test-1-a1") + expect(absent.length).toBe(1) + await until("attempt 2 created", () => w.ax.tasks.has("wf-test-1-a2")) + await l.drain() + }) + + it("C2 (B2) a superseded attempt that will not delete blocks attempt n+1: retryable pre-start/ax-unavailable, never two at once", async () => { + const w = W(); w.floor.enqueue(job("1")); w.ax.holdPending.add("wf-test-1-a1") + const del = w.ax.deleteTask + w.ax.deleteTask = (name) => name === "wf-test-1-a1" ? Effect.fail(new AxError(UNAVAILABLE, "unavailable")) : del(name) + const l = start(w, dir, { pendingTimeoutMs: 200 }) + await until("fence failed", () => l.has("fence-failed", "wf-test-1-a2"), 8000) + expect(l.verdict("pre-start/ax-unavailable")).toBe(true) + expect(w.ax.updates.has("wf-test-1-a2")).toBe(false) + await until("the infra budget is spent", () => J(w).state === "done", 10000) + expect([J(w).result, (J(w).output as { reason: string }).reason]).toEqual(["failure", "infra/retry-budget-spent"]) + expect([...w.ax.updates.keys()]).toEqual(["wf-test-1-a1"]) // only ever one attempt in ax + expect(J(w).supersedes.length).toBeGreaterThan(1) // rule 7 amended: every unconfirmed old attempt is listed + await l.drain() + }) + + it("C3 (B3) a cancel that lands while UpdateTask is retrying stops the create; nothing runs on", async () => { + const w = W(); w.floor.enqueue(job("1")) + let refusals = 2 + const create = w.ax.createTask + w.ax.createTask = (t) => refusals-- > 0 ? Effect.fail(new AxError(UNAVAILABLE, "unavailable")) : create(t) + const l = start(w, dir, { deleteAfterMs: 60_000, createAttempts: 5 }) + await until("leased", () => l.has("leased", "wf-test-1-a1")) + w.floor.cancel("wf-test-1", "tom") + await until("floor done", () => J(w).state === "done") + expect(J(w).result).toBe("cancelled") + await sleep(600) + expect([w.ax.live("wf-test-1-a1"), l.has("abandoned-before-create", "wf-test-1-a1")]).toEqual([false, true]) + await l.drain() + }) + + it("C3b (B3) a cancel that lands while UpdateTask is in flight deletes the Task as soon as it returns, then reports cancelled", async () => { + const w = W(); w.floor.enqueue(job("1")) + let gate!: () => void + const create = w.ax.createTask + w.ax.createTask = (t) => create(t).pipe(Effect.andThen(Effect.callback((resume) => { gate = () => resume(Effect.void) }))) + const l = start(w, dir, { deleteAfterMs: 60_000 }) + await until("create in flight", () => w.ax.tasks.has("wf-test-1-a1") && gate !== undefined) + w.floor.cancel("wf-test-1", "tom") + await until("cancel seen", () => l.has("cancel", "wf-test-1-a1")) + gate() + await until("floor done", () => J(w).state === "done") + expect([J(w).result, w.ax.live("wf-test-1-a1")]).toEqual(["cancelled", false]) + await l.drain() + }) + + it("C4 (B4) one transient GetTaskResult error is retried at the next resync: the success lands, nothing reruns", async () => { + const w = W(); w.floor.enqueue(job("1")) + let blips = 1 + const res = w.ax.getTaskResult + w.ax.getTaskResult = (n) => n === "wf-test-1-a1" && blips-- > 0 ? Effect.fail(new AxError(UNAVAILABLE, "unavailable")) : res(n) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).output, J(w).attempt, l.has("result-read-retry", "wf-test-1-a1")]).toEqual(["success", { answer: 42 }, 1, true]) + expect(J(w).history.some((h) => h.startsWith("requeue"))).toBe(false) + await l.drain() + }) + + it("C4b (B4) a result whose sha256 never matches is infra/result-unreadable after the bounded tries", async () => { + const w = W(); w.floor.enqueue(job("1")) + const res = w.ax.getTaskResult + w.ax.getTaskResult = (n) => res(n).pipe(Effect.map((r) => r && r !== "unimplemented" && n === "wf-test-1-a1" ? { ...r, digestOk: false } : r)) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + await until("attempt 2 created", () => w.ax.tasks.has("wf-test-1-a2")) + expect(J(w).history).toContain("requeue:infra/result-unreadable") + expect(l.logs.filter((x) => x.ev === "result-read-retry" && x.leaseId === "wf-test-1-a1").length).toBe(2) + await l.drain() + }) + + it("C5 (B5) Failed with ResourceExhausted is the retryable pre-start/resource-exhausted", async () => { + const w = W(); w.floor.enqueue(job("1")) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + const t = w.ax.tasks.get("wf-test-1-a1")! + t.phase = "Failed" + t.conditions = [{ type: "Ready", status: "False", reason: "ActorCreateFailed", message: "ResourceExhausted: no free workers available" }] + await until("attempt 2 created", () => w.ax.tasks.has("wf-test-1-a2")) + expect(J(w).history).toContain("requeue:pre-start/resource-exhausted") + await l.drain() + }) + + it("C6 (B6) configured p1 on a server without P1 leases nothing and says p1-missing", async () => { + const w = W(); w.floor.enqueue(job("1")); w.ax.p1 = false + const l = start(w, dir, { completion: "p1" }) + await until("probe", () => l.has("p1-missing")) + await sleep(10 * SEC) + expect(w.floor.calls.filter((c) => c.rpc === "Lease").every((c) => c.capacity === 0)).toBe(true) + expect([w.ax.updates.size, J(w).state]).toEqual([0, "queued"]) + await l.drain() + }) + + it("C6b (B6) auto never selects guest, and re-probes when ax comes back: P1 appears, work starts", async () => { + const w = W(); w.floor.enqueue(job("1")); w.ax.p1 = false + const l = start(w, dir, { completion: "auto" }) + await until("probe", () => l.has("p1-missing")) + await sleep(5 * SEC) + expect(w.ax.updates.size).toBe(0) + w.ax.up = false + await until("ax down", () => l.has("ax-down")) + w.ax.p1 = true; w.ax.up = true // the operator rolled out P1 + await until("created in p1 mode", () => w.ax.tasks.has("wf-test-1-a1")) + const env = Object.fromEntries(w.ax.tasks.get("wf-test-1-a1")!.task.spec.env.map((e) => [e.name, e.value])) + expect(env.AX_CONWIP_COMPLETE_URL).toBeUndefined() + await l.drain() + }) + + it("C6c (B6, A5) guest mode with a workers.dev Complete URL leases nothing", async () => { + expect([isFleetInternalUrl("https://conwip-floor.x.workers.dev/guest"), isFleetInternalUrl("http://link.fleet.internal/guest"), isFleetInternalUrl("http://10.201.0.9:8732/g"), isFleetInternalUrl("https://example.com/")]).toEqual([false, true, true, false]) + const w = W(); w.floor.enqueue(job("1")); w.ax.p1 = false + const l = start(w, dir, { completion: "guest", shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"], completeUrl: "https://conwip-floor.x.workers.dev/guest" } }) + await until("refused", () => l.has("guest-url-invalid")) + await sleep(5 * SEC) + expect(w.floor.calls.filter((c) => c.rpc === "Lease").every((c) => c.capacity === 0)).toBe(true) + await l.drain() + }) + + it("C7 (B7) an empty runs-on, or two seats, is refused by the link (and by the floor's enqueue)", async () => { + const w = W({ cap: 3 }) + expect(w.floor.enqueue(job("1", { "runs-on": [] }))).toBe(false) + w.floor.enqueueUnchecked(job("1", { "runs-on": [] })) + w.floor.o.tokens["tok-nas"] = { holder: "nas-link-1", labels: ["seat:halogen", "seat:cc", "runtime:gvisor"] } + w.floor.enqueueUnchecked(job("2", { "runs-on": ["seat:halogen", "seat:cc"] })) + const l = start(w, dir, { maxInFlight: 3, servedLabels: ["seat:halogen", "seat:cc", "runtime:gvisor"] }) + await until("both done", () => J(w).state === "done" && J(w, "2").state === "done") + expect([(J(w).output as { reason: string }).reason, (J(w, "2").output as { reason: string }).reason]).toEqual(["pre-start/runs-on-invalid", "pre-start/runs-on-invalid"]) + expect(w.ax.updates.size).toBe(0) + await l.drain() + }) + + it("C8 (B8) a Lease reply lost on all three tries is replayed by the journaled requestKey: attempt 1 runs", async () => { + const w = W(); w.floor.enqueue(job("1")) + let lose = 3 + const lossy: typeof globalThis.fetch = async (input, init) => { + const res = await w.floor.fetch(input, init) + if (lose > 0 && (await res.clone().text()).includes("\"leaseId\":\"wf-test-1-a1\"")) { lose--; throw new TypeError("fetch failed: reply lost") } + return res + } + const l = start(w, dir, {}, { fetch: lossy }) + await until("attempt 1 created", () => w.ax.tasks.has("wf-test-1-a1"), 8000) + expect(J(w).history.some((h) => h.startsWith("requeue"))).toBe(false) + expect(w.ax.tasks.has("wf-test-1-a2")).toBe(false) + await l.drain() + }) + + it("C9 (B9) a hung Heartbeat times out; the next one renews; the running attempt survives", async () => { + const w = W(); w.floor.enqueue(job("1")) + let hang = false + const hanging: typeof globalThis.fetch = async (input, init) => { + const req = new Request(input as string | URL | Request, init) + if (hang && (await req.clone().text()).includes("\"Heartbeat\"")) { hang = false; return new Promise(() => {}) } + return w.floor.fetch(req) + } + const l = start(w, dir, {}, { fetch: hanging, timeoutsMs: { lease: 150, heartbeat: 150, complete: 300 } }) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + hang = true + await until("the hang timed out", () => l.logs.some((x) => x.ev === "heartbeat-error")) + await sleep(25 * SEC) // past lease plus grace + expect([J(w).history.includes("requeue:grace"), w.ax.deletes.includes("wf-test-1-a1"), J(w).attempt]).toEqual([false, false, 1]) + w.ax.finish("wf-test-1-a1", 0, { ok: 1 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).attempt]).toEqual(["success", 1]) + await l.drain() + }) + + it("C9b (B9) concurrent 401 and 503 are each charged to their own call", async () => { + const w = W() + const f: typeof globalThis.fetch = async (input, init) => { + const body = await new Request(input as string | URL | Request, init).text() + if (body.includes("\"Lease\"")) { await sleep(30); return new Response("busy", { status: 503 }) } + return new Response("no", { status: 401 }) + } + const out = await Effect.runPromise(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "tok-nas", sessionId: "s", fetch: f }) + return yield* Effect.all([ + floor.lease({ holderIdentity: "nas-link-1", capacity: 1, requestKey: "k" }).pipe(Effect.flip), + floor.heartbeat({ holderIdentity: "nas-link-1", leaseIds: [] }).pipe(Effect.flip) + ], { concurrency: "unbounded" }) + }))) + expect(out.map((e) => e.kind)).toEqual(["transient", "auth"]) + void w + }) + + it("B10 a missing Gateway, or one that allows everything, leases nothing; a restricted one does", async () => { + const w = W(); w.floor.enqueue(job("1")) + const halogen = w.ax.gateways.get("halogen")! + w.ax.gateways.delete("halogen") + const l = start(w, dir) + await until("gateway-missing", () => l.has("gateway-missing")) + w.ax.gateways.set("halogen", { hasAllowlist: true, hosts: [{ host: "*", port: 443 }] }) + await sleep(5 * SEC) + expect([w.ax.updates.size, w.floor.calls.filter((c) => c.rpc === "Lease").every((c) => c.capacity === 0)]).toEqual([0, true]) + w.ax.gateways.set("halogen", halogen) + await until("created", () => w.ax.tasks.has("wf-test-1-a1")) + expect(l.has("gateway-ok")).toBe(true) + await l.drain() + }) + + it("B11 a floor that over-grants never makes the link over-create: extra grants wait, journaled and heartbeated", async () => { + const w = W({ cap: 10, overGrant: 2 }) + for (const n of ["1", "2", "3"]) w.floor.enqueue(job(n)) + const l = start(w, dir, { maxInFlight: 1 }) + let peak = 0 + const sample = setInterval(() => { peak = Math.max(peak, liveOwn(w)) }, 2) + await until("three grants", () => ["1", "2", "3"].every((n) => J(w, n).state === "leased")) + await until("waiting", () => l.has("waiting-for-slot")) + for (const n of ["1", "2", "3"]) { + await until(`job ${n} running`, () => w.ax.tasks.get(`wf-test-${n}-a1`)?.phase === "Running") + w.ax.finish(`wf-test-${n}-a1`, 0, { n }) + await until(`job ${n} done`, () => J(w, n).state === "done") + } + clearInterval(sample) + expect(peak).toBe(1) + expect(["1", "2", "3"].map((n) => [J(w, n).result, J(w, n).attempt])).toEqual([["success", 1], ["success", 1], ["success", 1]]) + await l.drain() + }) + + it("B12 a Task that reads back with another spec is deleted and refused", async () => { + const w = W(); w.floor.enqueue(job("1")) + const create = w.ax.createTask + w.ax.createTask = (t) => create({ ...t, spec: { ...t.spec, env: t.spec.env.slice(0, 3) } }) // the store dropped env rows + const l = start(w, dir) + await until("floor done", () => J(w).state === "done") + expect([(J(w).output as { reason: string }).reason, w.ax.live("wf-test-1-a1")]).toEqual(["pre-start/readback-mismatch", false]) + await l.drain() + }) + + it("B13 a journaled grant the floor released while the link was dead is never created on restart", async () => { + const w = W(); w.floor.enqueue(job("1")); w.ax.blockCreate = true + const l1 = start(w, dir) + await until("grant journaled", () => l1.has("leased", "wf-test-1-a1")) + await l1.crash() + w.ax.blockCreate = false + await until("requeued after the grace", () => J(w).history.includes("requeue:grace"), 3000) + const l2 = start(w, dir) + await until("attempt 2 created", () => w.ax.tasks.has("wf-test-1-a2")) + expect([w.ax.updates.has("wf-test-1-a1"), l2.has("resume", "wf-test-1-a1")]).toEqual([false, false]) + w.ax.tick(); w.ax.finish("wf-test-1-a2", 0, { ok: 2 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).attempt]).toEqual(["success", 2]) + await l2.drain() + }) + + it("B15 floor down past lease plus grace: the queued success goes out before the next Lease and is accepted (rule 4b)", async () => { + const w = W(); w.floor.enqueue(job("1")) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + w.ax.finish("wf-test-1-a1", 0, { v: 1 }) + await until("outbox keeps", () => l.has("outbox-keep", "wf-test-1-a1")) + await until("the alarm requeued it while the floor was unreachable", () => J(w).history.includes("requeue:grace"), 3000) + w.floor.down = false + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).output, J(w).attempt]).toEqual(["success", { v: 1 }, 1]) + expect(J(w).history).toContain("withdrawn:a2") + expect(w.ax.tasks.has("wf-test-1-a2")).toBe(false) + await l.drain() + }) + + it("B17 a verdict that frees a slot leases at once, not at the next poll", async () => { + const w = W({ pollSeconds: 100 }) // 2 s between polls here + w.floor.enqueue(job("1")); w.floor.enqueue(job("2")) + const l = start(w, dir, { maxInFlight: 1 }) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running", 4000) + await sleep(200) + const t0 = Date.now() + w.ax.finish("wf-test-1-a1", 0, { ok: 1 }) + await until("job 2 created", () => w.ax.tasks.has("wf-test-2-a1"), 1500) + expect(Date.now() - t0).toBeLessThan(1000) + await l.drain() + }) + + it("B18 a Lease endpoint is followed only when it is in the declared list", async () => { + const w = W() + let ep = "https://evil.example/floor" + const f: typeof globalThis.fetch = async (input, init) => { + const res = await w.floor.fetch(input, init) + const text = await res.text() + const h = new Headers(res.headers); h.delete("content-length") + return new Response(text.replace("\"nextPollSeconds\":", `"endpoint":"${ep}","nextPollSeconds":`), { status: res.status, headers: h }) + } + const seen: Array = [] + const l = start(w, dir, { floorUrls: ["https://conwip-floor-2.example/"] }, { fetch: f, onEndpoint: (u) => seen.push(u) }) + await until("ignored", () => l.has("endpoint-ignored")) + ep = "https://conwip-floor-2.example/" + await until("accepted", () => l.has("endpoint-accepted")) + expect(seen).toEqual(["https://conwip-floor-2.example/"]) + await l.drain() + }) +}) diff --git a/pkgs/substrate-link/src/apps/link/test/fake-ax.ts b/pkgs/substrate-link/src/apps/link/test/fake-ax.ts new file mode 100644 index 000000000..5cc096e80 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/fake-ax.ts @@ -0,0 +1,84 @@ +// Test double for ax-server, with stock v0.3.0 semantics where they matter to the link (MEASURED in ax d8ed0fe: +// internal/server/server.go UpdateTask is a blind upsert that also overwrites status; DeleteTask is two-phase; +// FIELD-MAP D3: a save re-runs the reconcile and resumes the actor). `p1` switches on the carried patch P1: +// a terminal phase is written by the controller and GetTaskResult exists. ListTasks pages like ax's Redis store +// (ZRevRange: newest first, limit 0 means 50; MEASURED upstream server.go:91-104, store.go:209-212). GetGateway +// answers from `gateways`; the default `halogen` Gateway allows one host, as ax-fleet's does. +import { createHash } from "node:crypto" +import { Effect } from "effect" +import { AxError } from "../src/ax.ts" +import type { AxApi, AxGateway, AxObserved, AxResult } from "../src/ax.ts" +import type { AxTask } from "../src/contract.ts" + +const NOT_FOUND = 5, UNAVAILABLE = 14 +type T = { task: AxTask; phase: string; conditions: Array<{ type: string; status: string; reason: string; message: string }>; result?: string; usage?: { promptTokens: number; completionTokens: number; toolCalls: number }; exited?: boolean } + +export class FakeAx implements AxApi { + readonly tasks = new Map() + readonly updates = new Map() // UpdateTask calls per name + upsertsOnExisting = 0 // the hazard: an UpdateTask on a name that already existed (re-runs the actor on v0.3.0) + resumedAfterExit = 0 // how many times that hazard restarted a finished agent + deletes: Array = [] + events: Array = [] // update: and delete:, in call order + up = true + p1 = true + blockCreate = false // createTask never returns (a crash window before the Task exists) + hangAfterCreate = false // createTask writes the Task, then never returns (a crash window after it exists) + holdPending = new Set() // names the controller never starts + gateways = new Map([["halogen", { hasAllowlist: true, hosts: [{ host: "worker", port: 8731 }] }]]) + listCalls = 0 + + private guard = (): Effect.Effect => this.up ? Effect.void : Effect.fail(new AxError(UNAVAILABLE, "connection refused")) + private obs = (name: string, t: T): AxObserved => ({ name, phase: t.phase, spec: t.task.spec, conditions: t.conditions, usage: t.usage ?? null }) + + /** The controller's work between two resyncs: start Pending Tasks, finish two-phase deletes. */ + tick() { + for (const [name, t] of this.tasks) { + if (t.phase === "Terminating") this.tasks.delete(name) + else if (t.phase === "Pending" && !this.holdPending.has(name)) t.phase = "Running" + } + } + /** The agent exits. With P1 the controller writes the terminal phase; without it the Task stays Running. */ + finish(name: string, exitCode: number, result?: unknown, usage = { promptTokens: 11, completionTokens: 7, toolCalls: 3 }) { + const t = this.tasks.get(name) + if (!t) throw new Error(`no task ${name}`) + t.exited = true + if (!this.p1) return + t.phase = exitCode === 0 ? "Completed" : "Failed" + t.conditions = [{ type: "Ready", status: "False", reason: "CommandExited", message: `ExitCode=${exitCode}` }] + if (result !== undefined) t.result = JSON.stringify(result) + t.usage = usage + } + put(task: AxTask, phase = "Running") { this.tasks.set(task.metadata.name, { task, phase, conditions: [] }) } + live(name: string) { const t = this.tasks.get(name); return t !== undefined && t.phase !== "Terminating" } + + getTask = (name: string) => this.guard().pipe(Effect.map(() => { const t = this.tasks.get(name); return t ? this.obs(name, t) : undefined })) + createTask = (task: AxTask) => this.guard().pipe(Effect.andThen(() => { + if (this.blockCreate) return Effect.never + const name = task.metadata.name + this.updates.set(name, (this.updates.get(name) ?? 0) + 1) + this.events.push(`update:${name}`) + const prior = this.tasks.get(name) + if (prior) { this.upsertsOnExisting++; if (prior.exited) this.resumedAfterExit++ } // v0.3.0: status reset, actor resumed + this.tasks.set(name, { task, phase: "Pending", conditions: [] }) + return this.hangAfterCreate ? Effect.never : Effect.void + })) + listTasks = (limit: number, offset: number) => this.guard().pipe(Effect.map(() => { + if (offset === 0) { this.tick(); this.listCalls++ } + const n = limit > 0 ? limit : 50 + return [...this.tasks].reverse().slice(offset, offset + n).map(([name, t]) => this.obs(name, t)) + })) + getGateway = (name: string) => this.guard().pipe(Effect.map(() => this.gateways.get(name))) + deleteTask = (name: string) => this.guard().pipe(Effect.map(() => { + const t = this.tasks.get(name) + if (t) { t.phase = "Terminating"; this.deletes.push(name); this.events.push(`delete:${name}`) } + })) + getTaskResult = (name: string) => this.guard().pipe(Effect.andThen((): Effect.Effect => { + if (!this.p1) return Effect.succeed("unimplemented" as const) + const t = this.tasks.get(name) + if (!t) return Effect.succeed(undefined) + const content = t.result ?? "null" + return Effect.succeed({ content, sha256: createHash("sha256").update(content).digest("hex"), digestOk: true }) + })) + static notFound = NOT_FOUND +} diff --git a/pkgs/substrate-link/src/apps/link/test/fake-floor.ts b/pkgs/substrate-link/src/apps/link/test/fake-floor.ts new file mode 100644 index 000000000..179354ea9 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/fake-floor.ts @@ -0,0 +1,304 @@ +// Test double for the Cloudflare floor: the FloorLink group served by Effect's stock HTTP RPC server through a web +// handler (the same `RpcServer.layerHttp` + `HttpRouter.toWebHandler` pair the prototype DO runs), called in-process +// through the link's `fetch`. State is in memory. It implements the floor rules of LINK-DESIGN.md section 4 as amended +// by "Critique applied", section H (B16): +// rule 1 states queued, leased, orphaned, done; WIP = leased + orphaned +// rule 2 route by runs-on against the holder's labels, cap per label, an admission stub, refuse at enqueue a job +// whose runs-on is empty or does not name exactly one seat: label +// rule 2b a replayed requestKey returns the same grants (kept for the life of the double, at least lease + grace) +// rule 4 stale-attempt, unknown-lease, duplicate; rule 4b: a verdict for attempt n while n+1 is still queued is +// accepted and n+1 withdrawn +// round 1 rule 4c: the verdict applied is recorded per leaseId, and a repeat for the same (leaseId, attempt) answers +// duplicate: true before rule 4b is considered (a lost reply or a second sender never spends a second +// release); rule 4b also accepts n's verdict while n+1 is leased to the SAME holder, which does not create +// n+1 while n's verdict is undelivered (link.ts dispatch) +// round 2 rule 4b names the grant it withdrew in the answer (`withdrew`); the link never infers a withdrawal. Rule +// 4c's duplicate answer replays the original answer's `withdrew`, so a lost reply is answered the same way +// round 3 rule 4b also names the generation withdrawn (`withdrewTransitions`: the leaseTransitions a grant of it +// carried or would have carried); `sendWithdrewGen: false` is a floor that omits it +// rule 5 a timer (the DO alarm): leased past renew + lease becomes orphaned, even with a cancel pending +// rule 6 the holder's heartbeat renews and re-adopts; lost for anything else; cancelRequested +// rule 7 an omitted orphan is released, so is one past the grace; supersedes lists EVERY earlier attempt +// round 4 rule 6b: a Heartbeat's `pendingRequestKey` vouches for every lease granted under that key (renewed and +// re-adopted, never released as omitted) while the holder has not seen that Lease answered +// rule 8 two budgets: infrastructure releases and agent retries, each final reason naming its budget +// rule 9 a per-lease token, minted only for guest-mode holders (A3), single use +// rule 10 at the deadline the floor sets cancel with reason deadline +// rule 11 holder = f(bearer): a payload holderIdentity that differs answers 403; Complete only for that holder's +// leases; one live session per holder (409); pause per holder +// It is NOT the production floor. +import { randomUUID } from "node:crypto" +import { Effect, Layer } from "effect" +import { HttpRouter } from "effect/unstable/http" +import { RpcSerialization, RpcServer } from "effect/unstable/rpc" +import { FloorLink, FloorLinkClient } from "../src/contract.ts" +import type { AgentJob, FailureOutput } from "../src/contract.ts" + +type Job = { + name: string; job: AgentJob; state: "queued" | "leased" | "orphaned" | "done"; attempt: number + leaseId?: string; holder?: string; acquire?: number; renew?: number; orphanedAt?: number; transitions: number + cancel?: string; result?: string; output?: unknown; usage?: unknown; supersedes: Array; token?: string; deadline?: number + history: Array; infraSpent: number; agentSpent: number; holders: Record +} +export type TokenBinding = string | { holder: string; labels?: Array; guest?: boolean } +export interface FloorConfig { + cap: number; leaseSeconds: number; graceSeconds: number; pollSeconds: number; heartbeatSeconds: number + maxAttempts: number // rule 8: agent retries (actor-crashed, result-unreadable) + infraAttempts?: number // rule 8: infrastructure releases (default 5) + secondMs: number + tokens: Record + labelCaps?: Record // rule 2: cap per label + timer?: boolean // rule 5: the DO alarm as a real timer (default true) + overGrant?: number // a broken floor that grants this many more than asked (B11's test) + skewedEncode?: boolean // round 1: a floor whose schema drifted from the link's; grants go out without validation + sendWithdrewGen?: boolean // round 3: Complete names the generation withdrawn (default true) +} +const DEFAULT_LABELS = ["seat:halogen", "runtime:gvisor"] +const INFRA = new Set(["omitted", "grace", "infra/task-lost", "pre-start/pending-timeout", "pre-start/ax-unavailable", "pre-start/resource-exhausted", "pre-start/superseded-verdict-pending", "infra/link-defect"]) +const AGENT = new Set(["infra/actor-crashed", "infra/result-unreadable"]) + +export class FakeFloor { + readonly jobs = new Map() + readonly grants = new Map>() + readonly sessions = new Map() + readonly calls: Array<{ rpc: string; holder?: string; capacity?: number; granted?: number; leaseIds?: ReadonlyArray; status?: number }> = [] + readonly paused = new Set() + readonly verdicts = new Map() // rule 4c + admit: (job: AgentJob) => boolean = () => true // rule 3 stub (seat readings) + seq = 0 + down = false + redeliver: Array = [] // at-least-once: the next Lease hands these leases out again + private readonly timer: ReturnType | undefined + constructor(readonly o: FloorConfig) { + if (o.timer !== false) { this.timer = setInterval(() => this.sweep(), o.secondMs); this.timer.unref?.() } + } + close() { if (this.timer) clearInterval(this.timer) } + private ms = (s: number) => s * this.o.secondMs + private binding(token: string) { + const b = this.o.tokens[token] + if (b === undefined) return undefined + return typeof b === "string" ? { holder: b, labels: DEFAULT_LABELS, guest: false } : { holder: b.holder, labels: b.labels ?? DEFAULT_LABELS, guest: b.guest ?? false } + } + private bindingOf(holder: string) { + for (const t of Object.keys(this.o.tokens)) { const b = this.binding(t)!; if (b.holder === holder) return b } + return undefined + } + + /** Rule 2 (amended): refuses a job whose runs-on is empty or does not name exactly one seat: label. */ + enqueue(job: AgentJob): boolean { + const seats = job.spec["runs-on"].filter((l) => l.startsWith("seat:")) + if (job.spec["runs-on"].length === 0 || seats.length !== 1) return false + this.enqueueUnchecked(job) + return true + } + /** A floor without rule 2's enqueue check, to test the link's own defence (B7). */ + enqueueUnchecked(job: AgentJob) { + if (!this.jobs.has(job.metadata.name)) this.jobs.set(job.metadata.name, { name: job.metadata.name, job, state: "queued", attempt: 1, transitions: 0, supersedes: [], history: [], infraSpent: 0, agentSpent: 0, holders: {} }) + } + cancel(name: string, why = "tom") { + const j = this.jobs.get(name)! + if (j.state === "queued") this.finish(j, "cancelled", { reason: why }) + else if (j.state !== "done") j.cancel = why + } + wip(label?: string) { return [...this.jobs.values()].filter((j) => (j.state === "leased" || j.state === "orphaned") && (label === undefined || j.job.spec["runs-on"].includes(label))).length } + stats() { return { seq: ++this.seq, cap: this.o.cap, wip: this.wip(), queued: [...this.jobs.values()].filter((j) => j.state === "queued").length, done: [...this.jobs.values()].filter((j) => j.state === "done").length } } + private finish(j: Job, result: string, output?: unknown, usage?: unknown) { + Object.assign(j, { state: "done", result, output, usage, token: undefined }); j.history.push(`done:${result}`) + } + /** An orphan the fleet side no longer vouches for: a cancelled job ends cancelled, any other goes back to the queue. */ + private release(j: Job, why: string) { + if (j.cancel) { j.history.push(`released:${why}`); return this.finish(j, "cancelled", { reason: j.cancel }) } + this.requeue(j, why) + } + private requeue(j: Job, why: string) { + j.history.push(`requeue:${why}`) + const infra = INFRA.has(why) || !AGENT.has(why) + if (infra) j.infraSpent++; else j.agentSpent++ + if (infra && j.infraSpent >= (this.o.infraAttempts ?? 5)) return this.finish(j, "failure", { reason: "infra/retry-budget-spent", message: why }) + if (!infra && j.agentSpent >= this.o.maxAttempts) return this.finish(j, "failure", { reason: "agent/retry-budget-spent", message: why }) + const old = j.leaseId! + Object.assign(j, { state: "queued", attempt: j.attempt + 1, transitions: j.transitions + 1, supersedes: [...j.supersedes, old], leaseId: undefined, holder: undefined, token: undefined, orphanedAt: undefined }) + } + /** The DO storage alarm's work (rules 5, 7 b and 10). Runs on the timer and on every request. */ + sweep(now = Date.now()) { + for (const j of this.jobs.values()) { + if (j.state === "leased" && j.deadline !== undefined && now >= j.deadline && !j.cancel) j.cancel = "deadline" + if (j.state === "leased" && j.renew! + this.ms(this.o.leaseSeconds) <= now) { // L3: expiry never frees a live slot, + j.state = "orphaned"; j.orphanedAt = now; j.history.push("orphaned") // even when a cancel is pending + } + if (j.state === "orphaned" && j.orphanedAt! + this.ms(this.o.graceSeconds) <= now) this.release(j, "grace") // L3 b + } + } + private grantable(j: Job, labels: Array, extra: Map) { + if (j.state !== "queued") return false + if (!j.job.spec["runs-on"].every((l) => labels.includes(l))) return false // rule 2: routed by label + for (const l of j.job.spec["runs-on"]) { + const cap = this.o.labelCaps?.[l] + if (cap !== undefined && this.wip(l) + (extra.get(l) ?? 0) >= cap) return false + } + return this.admit(j.job) + } + lease(p: { holderIdentity: string; capacity: number; requestKey: string }) { + this.sweep() + const b = this.bindingOf(p.holderIdentity) + let rows: Array + const prior = this.grants.get(p.requestKey) + if (prior) rows = prior.map((id) => [...this.jobs.values()].find((j) => j.leaseId === id && j.state !== "done")).filter((j): j is Job => j !== undefined) + else if (this.paused.has(p.holderIdentity) || b === undefined) rows = [] + else { + const free = Math.max(0, Math.min(p.capacity, this.o.cap - this.wip())) + (this.o.overGrant ?? 0) + const extra = new Map() + rows = [] + for (const j of this.jobs.values()) { + if (rows.length >= free) break + if (!this.grantable(j, b.labels, extra)) continue + rows.push(j) + for (const l of j.job.spec["runs-on"]) extra.set(l, (extra.get(l) ?? 0) + 1) + } + const now = Date.now() + for (const j of rows) { + const tm = j.job.spec["timeout-minutes"] + const leaseId = `${j.name}-a${j.attempt}` + Object.assign(j, { state: "leased", leaseId, holder: p.holderIdentity, acquire: now, renew: now, token: b.guest ? randomUUID() : undefined, + deadline: tm === undefined ? undefined : now + this.ms(tm * 60) }) + j.holders[leaseId] = p.holderIdentity + j.history.push(`leased:a${j.attempt}`) + } + this.grants.set(p.requestKey, rows.map((j) => j.leaseId!)) + } + if (this.redeliver.length) { + rows = [...rows, ...this.redeliver.map((id) => [...this.jobs.values()].find((j) => j.leaseId === id)).filter((j): j is Job => j !== undefined)] + this.redeliver = [] + } + this.calls.push({ rpc: "Lease", holder: p.holderIdentity, capacity: p.capacity, granted: rows.length }) + return { + grants: rows.map((j) => ({ + leaseId: j.leaseId!, attempt: j.attempt, job: j.job, + lease: { holderIdentity: j.holder!, leaseDurationSeconds: this.o.leaseSeconds, acquireTime: j.acquire!, renewTime: j.renew!, leaseTransitions: j.transitions }, + ...(j.supersedes.length ? { supersedes: [...j.supersedes] } : {}), ...(j.token ? { leaseToken: j.token } : {}), ...(j.deadline !== undefined ? { deadline: j.deadline } : {}) + })), + stats: this.stats(), nextPollSeconds: this.paused.has(p.holderIdentity) ? this.o.pollSeconds * 10 : this.o.pollSeconds, heartbeatSeconds: this.o.heartbeatSeconds + } + } + vouchByKey = true // round 4 rule 6b; false is an L1-only floor that ignores pendingRequestKey + heartbeat(p: { holderIdentity: string; leaseIds: ReadonlyArray; pendingRequestKey?: string }) { + this.sweep() + this.calls.push({ rpc: "Heartbeat", holder: p.holderIdentity, leaseIds: p.leaseIds }) + const byKey = new Set(this.vouchByKey && p.pendingRequestKey !== undefined ? this.grants.get(p.pendingRequestKey) ?? [] : []) + for (const id of byKey) { // rule 6b: vouched for by key, not listed (the holder never saw these ids) + if (p.leaseIds.includes(id)) continue + const j = [...this.jobs.values()].find((x) => x.leaseId === id && (x.state === "leased" || x.state === "orphaned") && x.holder === p.holderIdentity) + if (!j) continue + if (j.state === "orphaned") { j.state = "leased"; j.history.push("re-adopted") } + j.renew = Date.now() + } + const renewed: Array = [], lost: Array = [], cancelRequested: Array = [] + for (const id of p.leaseIds) { + const j = [...this.jobs.values()].find((x) => x.leaseId === id && (x.state === "leased" || x.state === "orphaned") && x.holder === p.holderIdentity) + if (!j) { lost.push(id); continue } + if (j.state === "orphaned") { j.state = "leased"; j.history.push("re-adopted") } // the holder vouches again + j.renew = Date.now(); renewed.push(id) + if (j.cancel) cancelRequested.push(id) + } + for (const j of this.jobs.values()) // L3 a: an orphaned lease its holder no longer lists is gone on the fleet side + if (j.state === "orphaned" && j.holder === p.holderIdentity && !p.leaseIds.includes(j.leaseId!) && !byKey.has(j.leaseId!)) this.release(j, "omitted") + return { renewed, lost, cancelRequested, stats: this.stats() } + } + /** `holder` is the bearer's holder (rule 11); undefined only for the guest path, which its lease token authorises. */ + complete(p: { leaseId: string; attempt: number; result: string; output?: unknown; usage?: unknown }, holder?: string) { + this.sweep() + this.calls.push({ rpc: "Complete", leaseIds: [p.leaseId] }) + const seen = this.verdicts.get(p.leaseId) // rule 4c: at most one verdict per (leaseId, attempt) + if (seen !== undefined && seen.attempt === p.attempt) return { duplicate: true, stats: this.stats(), ...this.w(seen.withdrew, seen.withdrewTransitions) } + const name = p.leaseId.replace(/-a\d+$/, "") + let j = [...this.jobs.values()].find((x) => x.leaseId === p.leaseId) + let withdrew: string | undefined, withdrewTransitions: number | undefined + if (!j) { + const q = this.jobs.get(name) + if (!q) return { code: "unknown-lease" as const } + // rule 4b: attempt n's verdict while n+1 is still queued: accept it and withdraw n+1 + // rule 4b (round 1): n+1 queued, or leased to the holder of n (that holder has not created it) + const nHolder = q.holders[p.leaseId] + const withdrawable = q.state === "queued" || (q.state === "leased" && q.holder === nHolder) + if (withdrawable && q.attempt === p.attempt + 1 && q.supersedes.includes(p.leaseId) && (holder === undefined || nHolder === holder)) { + q.history.push(`withdrawn:a${q.attempt}`) + withdrew = `${q.name}-a${q.attempt}`; withdrewTransitions = q.transitions + Object.assign(q, { state: "leased", attempt: p.attempt, leaseId: p.leaseId, holder: nHolder, token: undefined, renew: Date.now(), supersedes: q.supersedes.filter((x) => x !== p.leaseId) }) + j = q + } else return { code: "stale-attempt" as const } + } + if (holder !== undefined && j.holders[p.leaseId] !== holder) return { code: "stale-attempt" as const } // rule 11 + if (j.state === "done") return { duplicate: true, stats: this.stats() } + if ((j.state !== "leased" && j.state !== "orphaned") || j.attempt !== p.attempt) return { code: "stale-attempt" as const } + const reason = (p.output as FailureOutput | undefined)?.reason + this.verdicts.set(p.leaseId, { attempt: p.attempt, result: p.result, ...(reason !== undefined ? { reason } : {}), ...(withdrew !== undefined ? { withdrew, withdrewTransitions } : {}) }) + const w = this.w(withdrew, withdrewTransitions) + if (p.result === "failure" && reason !== undefined && (INFRA.has(reason) || AGENT.has(reason))) { this.requeue(j, reason); return { duplicate: false, stats: this.stats(), ...w } } + this.finish(j, p.result, p.output ?? (j.cancel ? { reason: j.cancel } : null), p.usage ?? null) + return { duplicate: false, stats: this.stats(), ...w } + } + private w(withdrew: string | undefined, t: number | undefined) { + if (withdrew === undefined) return {} + return this.o.sendWithdrewGen === false || t === undefined ? { withdrew } : { withdrew, withdrewTransitions: t } + } + /** L7: the guest's own Complete, authorised by its per-lease token alone (single use, bound to one lease). */ + guestComplete(leaseId: string, token: string, result: string, output?: unknown) { + const j = [...this.jobs.values()].find((x) => x.leaseId === leaseId) + if (!j || j.token === undefined || j.token !== token) return { refused: "token" as const } + j.token = undefined // single use + return this.complete({ leaseId, attempt: j.attempt, result, output }) + } + + private handlerCache: ((r: Request) => Promise) | undefined + get handler(): (r: Request) => Promise { return this.handlerCache ??= this.skewed() ? this.lenientHandler() : this.strictHandler() } + private skewed() { return this.o.skewedEncode === true } + /** Round 1: the same handlers served through the link's lenient view of Lease, so a grant is sent as the floor + * holds it, without the strict encode (a floor whose contract drifted). */ + private lenientHandler() { + return HttpRouter.toWebHandler( + RpcServer.layerHttp({ group: FloorLinkClient, path: "/rpc", protocol: "http" }).pipe( + Layer.provide(FloorLinkClient.toLayer({ + Lease: (p) => Effect.sync(() => this.lease(p)), + Heartbeat: (p) => Effect.sync(() => this.heartbeat(p)), + Complete: (p, o) => Effect.suspend(() => { + const r = this.complete(p, String((o.headers as Record)["x-link-holder"] ?? "")) + return r.code !== undefined ? Effect.fail({ code: r.code }) : Effect.succeed({ duplicate: r.duplicate!, stats: r.stats!, ...(r.withdrew !== undefined ? { withdrew: r.withdrew } : {}), ...(r.withdrewTransitions !== undefined ? { withdrewTransitions: r.withdrewTransitions } : {}) }) + }) + })), + Layer.provide(RpcSerialization.layerJson)), { disableLogger: true }).handler as (r: Request) => Promise + } + private strictHandler() { return HttpRouter.toWebHandler( + RpcServer.layerHttp({ group: FloorLink, path: "/rpc", protocol: "http" }).pipe( + Layer.provide(FloorLink.toLayer({ + Lease: (p) => Effect.sync(() => this.lease(p)), + Heartbeat: (p) => Effect.sync(() => this.heartbeat(p)), + Complete: (p, o) => Effect.suspend(() => { + const r = this.complete(p, String((o.headers as Record)["x-link-holder"] ?? "")) + return r.code !== undefined ? Effect.fail({ code: r.code }) : Effect.succeed({ duplicate: r.duplicate!, stats: r.stats!, ...(r.withdrew !== undefined ? { withdrew: r.withdrew } : {}), ...(r.withdrewTransitions !== undefined ? { withdrewTransitions: r.withdrewTransitions } : {}) }) + }) + })), + Layer.provide(RpcSerialization.layerJson)), { disableLogger: true }).handler as (r: Request) => Promise } + + /** The Worker entry: network, bearer (per-link token bound to a holder), holder binding (403), then one session per + * holder (409, L5). The resolved holder reaches the handlers as `x-link-holder`; a client's own copy is dropped. */ + readonly fetch: typeof globalThis.fetch = async (input, init) => { + if (this.down) throw new TypeError("fetch failed") + const req = new Request(input as string | URL | Request, init) + const b = this.binding((req.headers.get("authorization") ?? "").replace(/^Bearer /, "")) + const answer = (status: number, body: string) => { this.calls.push({ rpc: "entry", status }); return new Response(body, { status }) } + if (b === undefined) return answer(401, "unauthorized") + const body = await req.clone().text() + let msgs: Array = [] + try { const m = JSON.parse(body); msgs = Array.isArray(m) ? m : [m] } catch { /* not JSON: the RPC server answers */ } + if (msgs.some((m) => m?.payload && typeof m.payload.holderIdentity === "string" && m.payload.holderIdentity !== b.holder)) return answer(403, "holder mismatch") + const sid = req.headers.get("x-link-session") ?? "" + const s = this.sessions.get(b.holder), now = Date.now() + if (s && s.id !== sid && s.until > now) return answer(409, "session held") + this.sessions.set(b.holder, { id: sid, until: now + this.ms(3 * this.o.heartbeatSeconds) }) + const u = new URL(req.url) + if (u.pathname === "/rpc/") u.pathname = "/rpc" // Effect's client posts to "/" (PROTO.md section 4) + const h = new Headers(req.headers) + h.set("x-link-holder", b.holder) + return this.handler(new Request(u, { method: req.method, headers: h, body })) + } +} diff --git a/pkgs/substrate-link/src/apps/link/test/final-pass.test.ts b/pkgs/substrate-link/src/apps/link/test/final-pass.test.ts new file mode 100644 index 000000000..2530486bc --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/final-pass.test.ts @@ -0,0 +1,70 @@ +// Final pass (2026-09-23): regression tests for the two link-side findings of wave/VERIFY-LINK.md that were still open at +// 3b036da, both reproduced there by the chaos harness against the prototype floor under wrangler dev --local +// (link/final-verify/chaos-fresh-3b036da: f1 ran as a1 and a2, OVERLAP 8 s). +// FP-1 (R3-1): a heartbeat sleep begun at the 30 s default while nothing was held outlived a short lease granted +// during it; the floor released the orphan and the job ran twice. +// FP-2 (V1): the link fenced only what Grant.supersedes named; a floor that omits it got attempt n+1 created beside +// a still Running attempt n. +import { expect, it } from "vitest" +import { bodyOf, job, sleep, start, until, world } from "./review-r4-harness.ts" + +it("FP-1 a short lease granted during the first long heartbeat sleep is renewed in time", async () => { + // Lease 6 s and grace 2 s at SEC = 20 ms; the link starts with the production default heartbeat (30 s = 600 ms). + const w = world({ leaseSeconds: 6, graceSeconds: 2, heartbeatSeconds: 2 }) + // Every Lease reply is held 30 ms, as a real HTTPS round trip is (the chaos run: wrangler dev --local), so the heartbeat + // fiber sizes its first sleep before the first reply has set the server's interval. With an in-process reply the fiber + // happened to start after it, which hid the defect from every earlier test. + const slowLease: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + if (body.includes('"Lease"')) await sleep(30) + return w.floor.fetch(input as any, { ...init, body }) + } + const l = start(w, undefined, { initialHeartbeatSeconds: 30 }, slowLease) + await until("first Lease answered", () => w.floor.jobs.size === 0 && l.logs.some((x) => x.ev === "ax-up"), 2000).catch(() => {}) + await sleep(60) // the first heartbeat sleep has begun, sized from the 30 s default + w.floor.enqueue(job("1")) + const J = () => w.floor.jobs.get("wf-test-1")! + await until("a1 running", () => w.ax.tasks.get("wf-test-1-a1") !== undefined, 3000) + await sleep(1200) // twice the default heartbeat: without the fix the lease expires and the grace runs out first + const out = { history: J().history, attempt: J().attempt, a2: w.ax.updates.get("wf-test-1-a2") ?? 0 } + console.log(JSON.stringify(out)) + await l.drain(); w.floor.close() + expect(out.attempt).toBe(1) + expect(out.a2).toBe(0) + expect(out.history).not.toContain("requeue:omitted") + expect(out.history.filter((h: string) => h.startsWith("requeue")).length).toBe(0) +}) + +it("FP-2 attempt n+1 is never created beside a live attempt n, even when the floor omits supersedes", async () => { + const w = world({ leaseSeconds: 6, graceSeconds: 2, heartbeatSeconds: 2 }) + let failHeartbeat = false + const strip = (s: string) => s.replace(/,"supersedes":\[[^\]]*\]/g, "").replace(/"supersedes":\[[^\]]*\],/g, "") + const fetch: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + if (failHeartbeat && body.includes('"Heartbeat"')) throw new TypeError("fetch failed") + const res = await w.floor.fetch(input as any, { ...init, body }) + const text = await res.text() + return new Response(strip(text), { status: res.status, headers: res.headers }) + } + let a1LiveAtA2Create: boolean | undefined + const create = w.ax.createTask + ;(w.ax as any).createTask = (task: any) => { + if (task.metadata.name === "wf-test-1-a2" && a1LiveAtA2Create === undefined) a1LiveAtA2Create = w.ax.live("wf-test-1-a1") + return create(task) + } + w.floor.enqueue(job("1")) + const l = start(w, undefined, {}, fetch) + const J = () => w.floor.jobs.get("wf-test-1")! + await until("a1 running", () => w.ax.live("wf-test-1-a1") && l.has("created", "wf-test-1-a1"), 3000) + failHeartbeat = true // a1's lease expires and the grace runs out; the floor requeues attempt 2 and grants it on Lease + await until("floor requeued attempt 2", () => J().attempt === 2, 4000) + await until("a2 leased", () => l.has("leased", "wf-test-1-a2"), 4000) + failHeartbeat = false + await until("a2 created", () => (w.ax.updates.get("wf-test-1-a2") ?? 0) > 0, 4000) + const out = { a1LiveAtA2Create, extended: l.has("fence-set-extended", "wf-test-1-a2"), events: (w.ax as any).events?.filter((e: string) => e.includes("wf-test-1")) } + console.log(JSON.stringify(out)) + await l.drain(); w.floor.close() + expect(out.extended).toBe(true) + expect(out.a1LiveAtA2Create).toBe(false) + expect(w.ax.upsertsOnExisting).toBe(0) +}) diff --git a/pkgs/substrate-link/src/apps/link/test/fixtures/ax-p1.proto b/pkgs/substrate-link/src/apps/link/test/fixtures/ax-p1.proto new file mode 100644 index 000000000..a18e4a43a --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/fixtures/ax-p1.proto @@ -0,0 +1,405 @@ +// TEST FIXTURE: the vendored ax.proto (v0.3.0 d8ed0fe) plus the three hunks of carried patch P1 that the link reads: +// rpc GetTaskResult, messages GetTaskResultRequest and TaskResult, UsageStats.tool_calls. Copied from dotfiles branch +// ax/fleet-bringup pkgs/ax/patches/p1-completion.patch (lines 3240-3300) on 2026-09-23. Used only by the workerd +// integration test; production vendors the P1 proto when ax-fleet ships it (LINK-DESIGN section 9). +// Copyright 2026 Google LLC +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +syntax = "proto3"; + +package ax.v1alpha1; + +import "google/protobuf/struct.proto"; +import "google/protobuf/timestamp.proto"; + +option go_package = "github.com/google/ax/pkg/apis/v1alpha1"; + +// AX defines the core control plane gRPC service for autonomous agent orchestration. +service AX { + // Manifests are parsed by the client (see `ax apply`) and submitted through the + // typed Update* RPCs below; the server never receives raw YAML. + + // Tasks + rpc GetTask(GetTaskRequest) returns (Task); + rpc ListTasks(ListTasksRequest) returns (ListTasksResponse); + rpc UpdateTask(UpdateTaskRequest) returns (Task); + rpc DeleteTask(DeleteTaskRequest) returns (DeleteTaskResponse); + rpc SuspendTask(SuspendTaskRequest) returns (Task); + rpc ResumeTask(ResumeTaskRequest) returns (Task); + rpc WatchTask(WatchTaskRequest) returns (stream WatchTaskResponse); + // GetTaskResult returns the result file the task's command wrote, as copied + // by the controller when the command exited. NotFound until then. + rpc GetTaskResult(GetTaskResultRequest) returns (TaskResult); + + // Gateways + rpc GetGateway(GetGatewayRequest) returns (Gateway); + rpc ListGateways(ListGatewaysRequest) returns (ListGatewaysResponse); + rpc UpdateGateway(UpdateGatewayRequest) returns (Gateway); + rpc DeleteGateway(DeleteGatewayRequest) returns (DeleteGatewayResponse); + + // Workspaces + rpc GetWorkspace(GetWorkspaceRequest) returns (Workspace); + rpc ListWorkspaces(ListWorkspacesRequest) returns (ListWorkspacesResponse); + rpc UpdateWorkspace(UpdateWorkspaceRequest) returns (Workspace); + rpc DeleteWorkspace(DeleteWorkspaceRequest) returns (DeleteWorkspaceResponse); + + // Models + rpc GetModel(GetModelRequest) returns (Model); + rpc ListModels(ListModelsRequest) returns (ListModelsResponse); + rpc UpdateModel(UpdateModelRequest) returns (Model); + rpc DeleteModel(DeleteModelRequest) returns (DeleteModelResponse); + +} + +// ObjectMeta is metadata for AX resources. +message ObjectMeta { + string name = 1; + string atespace = 2; + google.protobuf.Timestamp creation_timestamp = 3; +} + +// --- Task --- + +message Task { + string api_version = 1; + string kind = 2; + ObjectMeta metadata = 3; + TaskSpec spec = 4; + TaskStatus status = 5; +} + +message TaskSpec { + // Field 1 was `goal`, removed; the workspace goal lives on WorkspaceRef. + reserved 1; + reserved "goal"; + bool suspend = 2; + string image = 3; + repeated string command = 4; + repeated EnvVar env = 5; + ResourceReqs resources = 6; + // workspaces binds one or more Workspaces, each mounted at its own path + // under /workspace. The first entry is the task command's working directory. + repeated WorkspaceRef workspaces = 7; + GatewayRef gateway = 8; + // Field 9 was `policies` (budget and approval config), removed for now. + reserved 9; + reserved "policies"; + // debug enables the in-container guest services (process execution and file + // access) that back `ax ssh`. Off by default. + bool debug = 10; +} + +message EnvVar { + string name = 1; + string value = 2; +} + +message ResourceReqs { + ResourceList requests = 1; + ResourceList limits = 2; +} + +message ResourceList { + string cpu = 1; + string memory = 2; +} + +message WorkspaceRef { + string name = 1; + string path = 2; + string goal = 3; +} + +message GatewayRef { + string name = 1; +} + +message TaskStatus { + string phase = 1; + string id = 2; + string actor = 3; + // json_name keeps the established `workerIP` spelling in YAML and JSON. + string worker_ip = 4 [json_name = "workerIP"]; + PendingApproval pending_approval = 5; + UsageStats usage = 6; + repeated Condition conditions = 7; +} + +message PendingApproval { + string id = 1; + string action = 2; + google.protobuf.Timestamp requested_at = 3; +} + +message UsageStats { + int32 prompt_tokens = 1; + int32 completion_tokens = 2; + int32 tool_calls = 3; +} + +message Condition { + string type = 1; + string status = 2; + google.protobuf.Timestamp last_transition_time = 3; + string reason = 4; + string message = 5; +} + +// --- Gateway --- + +message Gateway { + string api_version = 1; + string kind = 2; + ObjectMeta metadata = 3; + GatewaySpec spec = 4; +} + +message GatewaySpec { + repeated Listener listeners = 1; + EgressConfig egress = 2; +} + +message Listener { + string name = 1; + int32 port = 2; + string protocol = 3; +} + +message EgressConfig { + EgressAllowlist allowlist = 1; +} + +message EgressAllowlist { + repeated HostRule hosts = 1; +} + +message HostRule { + string host = 1; + int32 port = 2; +} + +// --- Workspace --- + +message Workspace { + string api_version = 1; + string kind = 2; + ObjectMeta metadata = 3; + WorkspaceSpec spec = 4; +} + +message WorkspaceSpec { + repeated GitRepo git = 1; + MCPConfig mcp = 2; + SkillsConfig skills = 3; +} + +message GitRepo { + string name = 1; + string repo = 2; + string branch = 3; + string dir = 4; + int32 depth = 5; +} + +message MCPConfig { + repeated MCPRegistry registries = 1; + repeated MCPServer servers = 2; +} + +message MCPRegistry { + string provider = 1; + string project = 2; + string query = 3; + repeated MCPServer servers = 4; +} + +message MCPServer { + string name = 1; + string endpoint = 2; + string command = 3; + repeated string args = 4; +} + +message SkillsConfig { + repeated SkillRegistry registries = 1; + string path = 2; +} + +message SkillRegistry { + string provider = 1; + string project = 2; + string query = 3; +} + +// --- Model --- + +message Model { + string api_version = 1; + string kind = 2; + ObjectMeta metadata = 3; + ModelSpec spec = 4; +} + +message ModelSpec { + string provider = 1; + string model = 2; + // Fields 3-5 were typed temperature, max_tokens, and system_instruction; + // provider settings now live in the free-form `parameters` map. + reserved 3, 4, 5; + reserved "temperature", "max_tokens", "system_instruction"; + SecretKeyRef secret_key = 6; + // parameters are provider-specific generation settings passed through to the + // model API as-is, for example temperature or maxOutputTokens for Gemini. + google.protobuf.Struct parameters = 7; +} + +message SecretKeyRef { + string name = 1; + string key = 2; +} + +// --- RPC Request & Response Messages --- + +// Tasks +message GetTaskRequest { + string atespace = 1; + string name = 2; +} + +message GetTaskResultRequest { + string atespace = 1; + string name = 2; +} + +message TaskResult { + bytes content = 1; + string sha256 = 2; +} + +message ListTasksRequest { + string atespace = 1; + int64 limit = 2; + int64 offset = 3; +} + +message ListTasksResponse { + repeated Task tasks = 1; +} + +message UpdateTaskRequest { + Task task = 1; +} + +message DeleteTaskRequest { + string atespace = 1; + string name = 2; +} + +message DeleteTaskResponse {} + +message SuspendTaskRequest { + string atespace = 1; + string name = 2; +} + +message ResumeTaskRequest { + string atespace = 1; + string name = 2; +} + +message WatchTaskRequest { + string atespace = 1; + string name = 2; +} + +message WatchTaskResponse { + Task task = 1; + string action = 2; +} + +// Gateways +message GetGatewayRequest { + string atespace = 1; + string name = 2; +} + +message ListGatewaysRequest { + string atespace = 1; +} + +message ListGatewaysResponse { + repeated Gateway gateways = 1; +} + +message UpdateGatewayRequest { + Gateway gateway = 1; +} + +message DeleteGatewayRequest { + string atespace = 1; + string name = 2; +} + +message DeleteGatewayResponse {} + +// Workspaces +message GetWorkspaceRequest { + string atespace = 1; + string name = 2; +} + +message ListWorkspacesRequest { + string atespace = 1; +} + +message ListWorkspacesResponse { + repeated Workspace workspaces = 1; +} + +message UpdateWorkspaceRequest { + Workspace workspace = 1; +} + +message DeleteWorkspaceRequest { + string atespace = 1; + string name = 2; +} + +message DeleteWorkspaceResponse {} + +// Models +message GetModelRequest { + string atespace = 1; + string name = 2; +} + +message ListModelsRequest { + string atespace = 1; +} + +message ListModelsResponse { + repeated Model models = 1; +} + +message UpdateModelRequest { + Model model = 1; +} + +message DeleteModelRequest { + string atespace = 1; + string name = 2; +} + +message DeleteModelResponse {} + diff --git a/pkgs/substrate-link/src/apps/link/test/link.test.ts b/pkgs/substrate-link/src/apps/link/test/link.test.ts new file mode 100644 index 000000000..d095d6fbb --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/link.test.ts @@ -0,0 +1,357 @@ +// The failure matrix of LINK-DESIGN.md section 8, one test per row, against the two doubles (fake ax with v0.3.0 +// upsert semantics and an optional P1; fake floor served through Effect's HTTP RPC server). Every server-set +// interval is scaled so one "second" is SEC ms. A crash is a fiber interrupt: nothing after it runs, and only what the +// journal fsynced survives, which is what kill -9 leaves. +import { mkdtempSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import { Deferred, Effect, Exit, Fiber } from "effect" +import { afterEach, beforeEach, describe, expect, it } from "vitest" +import type { AgentJob } from "../src/contract.ts" +import { rpcFloor } from "../src/floor.ts" +import { Journal } from "../src/journal.ts" +import { runLink } from "../src/link.ts" +import type { LinkConfig } from "../src/link.ts" +import { FakeAx } from "./fake-ax.ts" +import { FakeFloor } from "./fake-floor.ts" +import type { FloorConfig } from "./fake-floor.ts" + +const SEC = 20 +const A = "ultracode.mecattaf.dev/" +const JK = `${"cd".repeat(32)}:1` // round 4: FIELD-MAP 5a journal key (jobs.ts refuses a grant without it) +const job = (n: string, spec: Partial = {}, ann: Record = {}, w: Partial = {}): AgentJob => ({ + apiVersion: "ultracode.mecattaf.dev/v1alpha1", kind: "AgentJob", + metadata: { + name: `wf-test-${n}`, + labels: { [A + "run-id"]: "wf-test", [A + "workflow"]: "link-test", [A + "phase-index"]: "1" }, + annotations: { [A + "run-id-raw"]: "wf_test", [A + "label"]: `probe:${n}`, [A + "item-key"]: `wf_test#${n}`, [A + "journal-key"]: JK, [A + "phase-title"]: "Probe", ...ann } + }, + spec: { + "runs-on": ["seat:halogen", "runtime:gvisor"], + with: { prompt: `say ${n}`, prompt_ref: { sha256: "ab".repeat(32), bytes: 5, uri: `journal://wf_test/${n}/prompt.md` }, model: "halogen-qwen3.8-flash-next", ...w }, + ...spec + } +}) +const world = (o: Partial = {}) => ({ + floor: new FakeFloor({ cap: 2, leaseSeconds: 6, graceSeconds: 15, pollSeconds: 1, heartbeatSeconds: 2, maxAttempts: 3, secondMs: SEC, tokens: { "tok-nas": "nas-link-1" }, ...o }), + ax: new FakeAx() +}) +type World = ReturnType + +const GUEST = { completion: "guest" as const, shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"], completeUrl: "http://link.fleet.internal/guest" } } +const guestWorld = () => { const w = world({ tokens: { "tok-nas": { holder: "nas-link-1", guest: true } } }); w.ax.p1 = false; return w } + +function start(w: World, dir: string, over: Partial = {}, session = `s:${dir}`) { + const logs: Array> = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { + holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 1500, deleteAfterMs: 0, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, ...over + } + const fiber = Effect.runFork(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "tok-nas", sessionId: session, fetch: w.floor.fetch }) + return yield* runLink(cfg, { ax: w.ax, floor, journal: Journal.open(dir), log: (ev, f) => logs.push({ ev, ...f }), stop }) + }))) + return { + logs, + has: (ev: string, leaseId?: string) => logs.some((l) => l.ev === ev && (leaseId === undefined || l.leaseId === leaseId)), + crash: () => Effect.runPromise(Fiber.interrupt(fiber)), + drain: () => { Effect.runSync(Deferred.succeed(stop, undefined)); return Effect.runPromise(Fiber.await(fiber)) }, + exit: () => Effect.runPromise(Fiber.await(fiber)) + } +} +const until = async (what: string, pred: () => boolean, ms = 5000) => { + const t0 = Date.now() + while (!pred()) { if (Date.now() - t0 > ms) throw new Error(`timeout waiting for: ${what}`); await new Promise((r) => setTimeout(r, 5)) } +} +const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const J = (w: World, n = "1") => w.floor.jobs.get(`wf-test-${n}`)! + +describe("conwip link failure matrix", () => { + let dir: string + beforeEach(() => { dir = mkdtempSync(join(tmpdir(), "conwip-link-")) }) + afterEach(() => rmSync(dir, { recursive: true, force: true })) + + it("F0 happy path on P1: one create, typed verdict with usage, Task deleted after the floor acknowledged", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + const env = Object.fromEntries(w.ax.tasks.get("wf-test-1-a1")!.task.spec.env.map((e) => [e.name, e.value])) + expect(env.AX_CONWIP_ITEM_KEY).toBe("wf_test#1"); expect(env.AX_CONWIP_ATTEMPT).toBe("1"); expect(env.AX_CONWIP_PROMPT).toBe("say 1") + expect(env.AX_CONWIP_LEASE_TOKEN).toBeUndefined() // P1 present: the guest carries no token at all + expect(w.ax.tasks.get("wf-test-1-a1")!.task.apiVersion).toBe("ax.io/v1alpha1") + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).output, J(w).usage]).toEqual(["success", { answer: 42 }, { prompt_tokens: 11, completion_tokens: 7, tool_calls: 3 }]) + await until("janitor deleted the Task", () => !w.ax.tasks.has("wf-test-1-a1")) + expect(w.ax.updates.get("wf-test-1-a1")).toBe(1); expect(w.ax.upsertsOnExisting).toBe(0) + await l.drain() + }) + + it("F1 duplicate delivery: a redelivered grant creates nothing twice", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.redeliver = ["wf-test-1-a1"] + await until("duplicate seen", () => l.has("duplicate-grant", "wf-test-1-a1")) + w.ax.finish("wf-test-1-a1", 0, { ok: true }) + await until("floor done", () => J(w).state === "done") + expect(w.ax.updates.get("wf-test-1-a1")).toBe(1); expect(w.ax.upsertsOnExisting).toBe(0) + await l.drain() + }) + + it("F2 crash after Lease, before UpdateTask: the restart creates exactly once", async () => { + const w = world(); w.floor.enqueue(job("1")); w.ax.blockCreate = true + const l1 = start(w, dir) + await until("grant journaled", () => l1.has("leased", "wf-test-1-a1")) + await l1.crash() + w.ax.blockCreate = false + const l2 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.finish("wf-test-1-a1", 0, { ok: 1 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).attempt, w.ax.updates.get("wf-test-1-a1")]).toEqual(["success", 1, 1]) + await l2.drain() + }) + + it("F3 crash after UpdateTask, before the journal said created: the restart adopts, no second UpdateTask", async () => { + const w = world(); w.floor.enqueue(job("1")); w.ax.hangAfterCreate = true + const l1 = start(w, dir) + await until("task exists", () => w.ax.tasks.has("wf-test-1-a1")) + await l1.crash() + w.ax.hangAfterCreate = false + const l2 = start(w, dir) + await until("adopted", () => l2.has("adopted", "wf-test-1-a1")) + w.ax.finish("wf-test-1-a1", 0, { ok: 1 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, w.ax.updates.get("wf-test-1-a1"), w.ax.upsertsOnExisting]).toEqual(["success", 1, 0]) + await l2.drain() + }) + + it("F4 crash while running; the agent finishes while the link is down; the restart reports once", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l1 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + await l1.crash() + w.ax.finish("wf-test-1-a1", 0, { late: true }) + const l2 = start(w, dir) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).output, J(w).attempt]).toEqual(["success", { late: true }, 1]) + expect(w.floor.calls.filter((c) => c.rpc === "Complete").length).toBe(1) + await l2.drain() + }) + + it("F5 floor unreachable: the verdict waits in the outbox across a crash, the lease is orphaned not requeued, then lands", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l1 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + w.ax.finish("wf-test-1-a1", 0, { v: 1 }) + await until("outbox keeps", () => l1.has("outbox-keep", "wf-test-1-a1")) + await l1.crash() + await sleep(7 * SEC) // past the 6 s lease, well inside the 15 s grace + const l2 = start(w, dir) + await sleep(3 * SEC) + w.floor.down = false + w.floor.sweep() // the DO alarm fires + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).attempt]).toEqual(["success", 1]) + expect(J(w).history.some((h) => h.startsWith("requeue"))).toBe(false) + await l2.drain() + }) + + it("F6 ax unreachable: capacity 0 on every poll, held leases still renewed, work resumes when ax returns", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.up = false + await until("link saw ax down", () => l.has("ax-down")) + const mark = w.floor.calls.length + w.floor.enqueue(job("2")) + await sleep(10 * SEC) // longer than the 6 s lease + if (process.env.LINK_DEBUG) console.log(JSON.stringify({ history: J(w).history, calls: w.floor.calls.slice(mark).map((c) => [c.rpc, c.capacity, c.leaseIds]), logs: l.logs.slice(-15) })) + const during = w.floor.calls.slice(mark).filter((c) => c.rpc === "Lease") + expect(during.length).toBeGreaterThan(0) + expect(during.every((c) => c.capacity === 0 && c.granted === 0)).toBe(true) + expect([J(w).state, J(w, "2").state]).toEqual(["leased", "queued"]) + w.ax.up = true + await until("second job created", () => w.ax.tasks.has("wf-test-2-a1")) + await l.drain() + }) + + it("F7 lease expiry, link back inside the grace: the orphaned lease is re-adopted, same attempt, no second Task", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l1 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + await l1.crash() + await sleep(7 * SEC); w.floor.sweep() + expect([J(w).state, w.floor.wip()]).toEqual(["orphaned", 1]) // L3: never reclaim a live slot + const l2 = start(w, dir) + await until("re-adopted", () => J(w).history.includes("re-adopted")) + w.ax.finish("wf-test-1-a1", 0, { ok: 1 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).attempt, w.ax.updates.get("wf-test-1-a1")]).toEqual(["success", 1, 1]) + await l2.drain() + }) + + it("F8 journal lost: the holder's complete heartbeat omits the orphan, attempt 2 supersedes it, old Task deleted first", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l1 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + await l1.crash() + rmSync(dir, { recursive: true, force: true }) + await sleep(7 * SEC); w.floor.sweep() + const l2 = start(w, dir, {}, "s:fresh-state") + await until("attempt 2 created", () => w.ax.tasks.has("wf-test-1-a2")) + expect(J(w).history).toContain("requeue:omitted") + expect(w.ax.events.indexOf("delete:wf-test-1-a1")).toBeGreaterThanOrEqual(0) + expect(w.ax.events.indexOf("delete:wf-test-1-a1")).toBeLessThan(w.ax.events.indexOf("update:wf-test-1-a2")) + w.ax.tick(); w.ax.finish("wf-test-1-a2", 0, { ok: 2 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).attempt]).toEqual(["success", 2]) + await l2.drain() + }) + + it("F9 link dead past the grace: the alarm requeues; the returning link drops attempt 1 (lost) and runs attempt 2", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l1 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + await l1.crash() + await sleep(7 * SEC); w.floor.sweep() + await sleep(16 * SEC); w.floor.sweep() + expect(J(w).history).toContain("requeue:grace") + const l2 = start(w, dir) + await until("attempt 2 created", () => w.ax.tasks.has("wf-test-1-a2")) + expect(w.ax.deletes).toContain("wf-test-1-a1") + w.ax.tick(); w.ax.finish("wf-test-1-a2", 0, { ok: 2 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).attempt]).toEqual(["success", 2]) + // the late verdict of attempt 1 can never be recorded: the floor fences on the attempt + expect(w.floor.complete({ leaseId: "wf-test-1-a1", attempt: 1, result: "success" })).toEqual({ code: "stale-attempt" }) + await l2.drain() + }) + + it("F10 cancel: carried in the heartbeat reply, DeleteTask, then Complete(cancelled) once the Task is gone", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.cancel("wf-test-1") + await until("floor done", () => J(w).state === "done") + if (process.env.LINK_DEBUG) console.log(JSON.stringify({ history: J(w).history, events: w.ax.events, phase: w.ax.tasks.get("wf-test-1-a1")?.phase, logs: l.logs })) + expect([J(w).result, J(w).attempt, w.ax.tasks.has("wf-test-1-a1")]).toEqual(["cancelled", 1, false]) + await l.drain() + }) + + it("F11 stock v0.3.0 (no P1), guest mode configured: the Task stays Running after exit; the guest completes with its lease token; the link deletes on lost", async () => { + const w = guestWorld(); w.floor.enqueue(job("1")) + const l = start(w, dir, GUEST) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + const env = Object.fromEntries(w.ax.tasks.get("wf-test-1-a1")!.task.spec.env.map((e) => [e.name, e.value])) + expect(env.AX_CONWIP_COMPLETE_URL).toBe("http://link.fleet.internal/guest") + expect(Object.values(env).some((v) => v.includes("tok-nas"))).toBe(false) // the fleet bearer never enters the sandbox + w.ax.finish("wf-test-1-a1", 0) + await sleep(5 * SEC) + expect(w.ax.tasks.get("wf-test-1-a1")!.phase).toBe("Running") // FIELD-MAP D3: no terminal phase without P1 + expect(w.floor.guestComplete("wf-test-1-a1", env.AX_CONWIP_LEASE_TOKEN!, "success", { guest: true })).toMatchObject({ duplicate: false }) + expect(w.floor.guestComplete("wf-test-1-a1", env.AX_CONWIP_LEASE_TOKEN!, "success", { guest: true })).toEqual({ refused: "token" }) + await until("link deleted the still-Running Task", () => !w.ax.tasks.has("wf-test-1-a1")) + expect([J(w).result, J(w).output]).toEqual(["success", { guest: true }]) + await l.drain() + }) + + it("F12 stuck Running (no P1, the guest never reports): the floor's deadline cancels it and ax frees the worker", async () => { + const w = guestWorld(); w.floor.enqueue(job("1", { "timeout-minutes": 1 })) // 60 s = 1200 ms here + const l = start(w, dir, GUEST) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.finish("wf-test-1-a1", 0) // exited, invisible to ax without P1 + await until("floor done", () => J(w).state === "done", 8000) + expect([J(w).result, J(w).output]).toEqual(["cancelled", { reason: "deadline" }]) + await until("Task gone", () => !w.ax.tasks.has("wf-test-1-a1")) + await l.drain() + }) + + it("F13 deadline passes while the floor is unreachable: the local backstop deletes the Task and queues the verdict", async () => { + const w = guestWorld(); w.floor.enqueue(job("1", { "timeout-minutes": 1 })) + const l = start(w, dir, GUEST) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + await until("backstop verdict", () => l.logs.some((x) => x.ev === "verdict" && x.reason === "deadline-exceeded"), 8000) + await until("Task deleted", () => !w.ax.tasks.has("wf-test-1-a1") || w.ax.tasks.get("wf-test-1-a1")!.phase === "Terminating") + await l.crash() + }) + + it("F14 stuck Pending: pre-start failure at the timeout, retried by the floor as attempt 2 superseding attempt 1", async () => { + const w = world(); w.floor.enqueue(job("1")); w.ax.holdPending.add("wf-test-1-a1") + const l = start(w, dir, { pendingTimeoutMs: 200 }) + await until("attempt 2 created", () => w.ax.tasks.has("wf-test-1-a2"), 8000) + expect(J(w).history).toContain("requeue:pre-start/pending-timeout") + expect(w.ax.deletes).toContain("wf-test-1-a1") + w.ax.tick(); w.ax.finish("wf-test-1-a2", 0, { ok: 2 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).attempt]).toEqual(["success", 2]) + await l.drain() + }) + + it("F15 name conflict: a foreign Task with the lease's name is never overwritten (create-only, L4)", async () => { + const w = world(); w.floor.enqueue(job("1")) + w.ax.put({ apiVersion: "ax.io/v1alpha1", kind: "Task", metadata: { name: "wf-test-1-a1", atespace: "fleet" }, spec: { image: "someone-else", command: ["sleep"], env: [] } }) + const l = start(w, dir) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, (J(w).output as { reason: string }).reason]).toEqual(["failure", "pre-start/name-conflict"]) + expect([w.ax.updates.get("wf-test-1-a1"), w.ax.tasks.get("wf-test-1-a1")?.task.spec.image]).toEqual([undefined, "someone-else"]) + await l.drain() + }) + + it("F16 deterministic refusals are final and touch nothing in ax: env over budget, unresolved prompt, prompt too big to inline (B12)", async () => { + const w = world({ cap: 3 }) + w.floor.enqueue(job("1", {}, {}, { schema: { description: "x".repeat(20000) } })) + w.floor.enqueue(job("2", {}, { [A + "prompt-unresolved"]: "no agent() call carries label" })) + w.floor.enqueue(job("3", {}, {}, { prompt: "p".repeat(17000) })) + const l = start(w, dir, { maxInFlight: 3 }) + await until("all done", () => J(w).state === "done" && J(w, "2").state === "done" && J(w, "3").state === "done") + expect([(J(w, "3").output as { reason: string }).reason, J(w, "3").attempt]).toEqual(["pre-start/prompt-file-route-missing", 1]) + expect([(J(w).output as { reason: string }).reason, J(w).attempt]).toEqual(["pre-start/env-over-budget", 1]) + expect([(J(w, "2").output as { reason: string }).reason, J(w, "2").attempt]).toEqual(["pre-start/prompt-unresolved", 1]) + expect(w.ax.updates.size).toBe(0) + await l.drain() + }) + + it("F17 a second replica with the same identity is refused (409) and exits; the first keeps its session", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l1 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + const dir2 = mkdtempSync(join(tmpdir(), "conwip-link-b-")) + const l2 = start(w, dir2, {}, "s:second-replica") + const exit = await l2.exit() + rmSync(dir2, { recursive: true, force: true }) + expect(Exit.isFailure(exit)).toBe(true) + expect(JSON.stringify(exit)).toContain("session-conflict") + w.ax.finish("wf-test-1-a1", 0, { ok: 1 }) + await until("floor done", () => J(w).state === "done") + await l1.drain() + }) + + it("F19 ax lost a Task it had created (Redis loss): infra/task-lost, the floor retries as attempt 2", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.tasks.delete("wf-test-1-a1") // gone without a DeleteTask: the store forgot it + await until("attempt 2 created", () => w.ax.tasks.has("wf-test-1-a2")) + expect(J(w).history).toContain("requeue:infra/task-lost") + w.ax.tick(); w.ax.finish("wf-test-1-a2", 0, { ok: 2 }) + await until("floor done", () => J(w).state === "done") + expect([J(w).result, J(w).attempt]).toEqual(["success", 2]) + await l.drain() + }) + + it("F18 drain on SIGTERM: stop leasing, flush the outbox, leave running Tasks to be re-adopted", async () => { + const w = world(); w.floor.enqueue(job("1")) + const l = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + const exit = await l.drain() + expect(Exit.isSuccess(exit)).toBe(true) + expect([w.ax.tasks.get("wf-test-1-a1")?.phase, J(w).state]).toEqual(["Running", "leased"]) + }) +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r1.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r1.test.ts new file mode 100644 index 000000000..5e49eeb77 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r1.test.ts @@ -0,0 +1,391 @@ +// Review round 1 (2026-09-23): the ten findings, each as a regression test that failed before its fix. The scratch +// repros are in /home/tom/today/evals-2026-09-23/link/scratch-r1-{0,1,2}; the log is link/REVIEW-LOG.md. +import { mkdtempSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import { Deferred, Effect, Fiber } from "effect" +import { afterEach, beforeEach, describe, expect, it } from "vitest" +import { AxError } from "../src/ax.ts" +import type { AgentJob, AxTask } from "../src/contract.ts" +import { rpcFloor } from "../src/floor.ts" +import type { FloorOptions } from "../src/floor.ts" +import { Journal } from "../src/journal.ts" +import { runLink } from "../src/link.ts" +import type { LinkConfig } from "../src/link.ts" +import { FakeAx } from "./fake-ax.ts" +import { FakeFloor } from "./fake-floor.ts" +import type { FloorConfig } from "./fake-floor.ts" +import { ConfigInvalid, readLinkEnv } from "../src/config.ts" +import type { FloorApi } from "../src/floor.ts" +import { axTaskFromGrant } from "../src/jobs.ts" + +const SEC = 20 +const UNAVAILABLE = 14 +const A = "ultracode.mecattaf.dev/" +const JK = `${"cd".repeat(32)}:1` // round 4: FIELD-MAP 5a journal key (jobs.ts refuses a grant without it) +const job = (n: string, spec: Partial = {}): AgentJob => ({ + apiVersion: "ultracode.mecattaf.dev/v1alpha1", kind: "AgentJob", + metadata: { + name: `wf-test-${n}`, + labels: { [A + "run-id"]: "wf-test", [A + "workflow"]: "link-test", [A + "phase-index"]: "1" }, + annotations: { [A + "run-id-raw"]: "wf_test", [A + "label"]: `probe:${n}`, [A + "item-key"]: `wf_test#${n}`, [A + "journal-key"]: JK, [A + "phase-title"]: "Probe" } + }, + spec: { + "runs-on": ["seat:halogen", "runtime:gvisor"], + with: { prompt: `say ${n}`, prompt_ref: { sha256: "ab".repeat(32), bytes: 5, uri: `journal://wf_test/${n}/prompt.md` }, model: "halogen-qwen3.8-flash-next" }, + ...spec + } +}) +const world = (o: Partial = {}) => ({ + floor: new FakeFloor({ cap: 2, leaseSeconds: 6, graceSeconds: 15, pollSeconds: 1, heartbeatSeconds: 2, maxAttempts: 3, secondMs: SEC, tokens: { "tok-nas": "nas-link-1" }, ...o }), + ax: new FakeAx() +}) +type World = ReturnType +const foreign = (name: string): AxTask => ({ apiVersion: "ax.io/v1alpha1", kind: "Task", metadata: { name, atespace: "fleet" }, spec: { image: "someone-else", command: ["true"], env: [] } }) + +interface Extra { fetch?: typeof globalThis.fetch; timeoutsMs?: FloorOptions["timeoutsMs"]; onEndpoint?: (u: string) => void } +function start(w: World, dir: string, over: Partial = {}, x: Extra = {}) { + const logs: Array> = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { + holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 1500, deleteAfterMs: 0, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, fenceTimeoutMs: 300, resultReadTries: 3, ...over + } + const fiber = Effect.runFork(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "tok-nas", sessionId: `s:${dir}`, fetch: x.fetch ?? w.floor.fetch, ...(x.timeoutsMs ? { timeoutsMs: x.timeoutsMs } : {}) }) + return yield* runLink(cfg, { ax: w.ax, floor, journal: Journal.open(dir), log: (ev, f) => logs.push({ ev, ...f }), stop, floorUrl: "http://floor.test", ...(x.onEndpoint ? { onEndpoint: x.onEndpoint } : {}) }) + }))) + return { + logs, + has: (ev: string, leaseId?: string) => logs.some((l) => l.ev === ev && (leaseId === undefined || l.leaseId === leaseId)), + verdict: (reason: string) => logs.some((l) => l.ev === "verdict" && l.reason === reason), + crash: () => Effect.runPromise(Fiber.interrupt(fiber)), + drain: () => { Effect.runSync(Deferred.succeed(stop, undefined)); return Promise.race([Effect.runPromise(Fiber.await(fiber)), sleep(1500)]) } + } +} +const until = async (what: string, pred: () => boolean, ms = 5000) => { + const t0 = Date.now() + while (!pred()) { if (Date.now() - t0 > ms) throw new Error(`timeout waiting for: ${what}`); await new Promise((r) => setTimeout(r, 5)) } +} +const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const J = (w: World, n = "1") => w.floor.jobs.get(`wf-test-${n}`)! +const liveOwn = (w: World) => [...w.ax.tasks.keys()].filter((k) => k.startsWith("wf-test-") && w.ax.live(k) && !["Completed", "Failed"].includes(w.ax.tasks.get(k)!.phase)).length + + +const bodyOf = async (input: unknown, init?: RequestInit) => { + const b = init?.body + if (typeof b === "string") return b + if (b instanceof Uint8Array) return new TextDecoder().decode(b) + if (b) return await new Response(b as ConstructorParameters[0]).text() + return (input instanceof Request) ? await input.clone().text() : "" +} +const isComplete = async (input: unknown, init?: RequestInit) => (await bodyOf(input, init)).includes('"Complete"') +const stats = { seq: 1, cap: 2, wip: 0, queued: 0, done: 0 } +const grantOf = (n: string, extra: Record = {}) => ({ + leaseId: `wf-test-${n}-a1`, attempt: 1, job: job(n), + lease: { holderIdentity: "nas-link-1", leaseDurationSeconds: 30, acquireTime: Date.now(), renewTime: Date.now(), leaseTransitions: 0 }, ...extra +}) as any +/** A floor that hands out `grants` once and renews everything: for link-side properties the double cannot stage. */ +const stubFloor = (grants: Array) => { + let sent = false + const completes: Array = [] + const api: FloorApi = { + lease: (p) => Effect.sync(() => { const g = !sent && p.capacity > 0 ? grants : []; if (g.length) sent = true; return { grants: g, stats, nextPollSeconds: 1, heartbeatSeconds: 2 } }), + heartbeat: (p) => Effect.succeed({ renewed: [...p.leaseIds], lost: [], cancelRequested: [], stats }), + complete: (p) => Effect.sync(() => { completes.push(p); return { duplicate: false, stats } }) + } + return { api, completes } +} +const cfgOf = (over: Partial = {}): LinkConfig => ({ + holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 60_000, deleteAfterMs: 60_000, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, fenceTimeoutMs: 300, resultReadTries: 3, ...over +}) +const runStub = async (cfg: LinkConfig, ax: FakeAx, floor: FloorApi, ms: number) => { + const d = mkdtempSync(join(tmpdir(), "conwip-link-r1s-")) + const logs: Array> = [] + const stop = Effect.runSync(Deferred.make()) + const f = Effect.runFork(runLink(cfg, { ax, floor, journal: Journal.open(d), log: (ev, x) => logs.push({ ev, ...x }), stop })) + await sleep(ms) + Effect.runSync(Deferred.succeed(stop, undefined)); await Effect.runPromise(Fiber.await(f)); rmSync(d, { recursive: true, force: true }) + return logs +} + +describe("review round 1: fixed", () => { + let dir: string + let worlds: Array = [] + const W = (o: Partial = {}) => { const w = world(o); worlds.push(w); return w } + beforeEach(() => { dir = mkdtempSync(join(tmpdir(), "conwip-link-r1-")) }) + afterEach(() => { for (const w of worlds) w.floor.close(); worlds = []; rmSync(dir, { recursive: true, force: true }) }) + + // ------------------------------------------------------------------ findings 1 and 2: one verdict, one application + it("R1-1 no fault: every verdict costs exactly one Complete (single-flight outbox)", async () => { + const w = W({ cap: 8 }) + for (let i = 1; i <= 8; i++) w.floor.enqueue(job(String(i))) + const l = start(w, dir, { maxInFlight: 8, resyncMs: 5 }) + await until("all running", () => [1, 2, 3, 4, 5, 6, 7, 8].every((i) => w.ax.tasks.get(`wf-test-${i}-a1`)?.phase === "Running")) + for (let i = 1; i <= 8; i++) w.ax.finish(`wf-test-${i}-a1`, 0, { i }) + await until("all done", () => [1, 2, 3, 4, 5, 6, 7, 8].every((i) => J(w, String(i)).state === "done")) + await sleep(100) + const per = new Map() + for (const c of w.floor.calls.filter((c) => c.rpc === "Complete")) per.set(c.leaseIds![0]!, (per.get(c.leaseIds![0]!) ?? 0) + 1) + expect([...per.values()]).toEqual([1, 1, 1, 1, 1, 1, 1, 1]) + expect(l.logs.filter((x) => x.ev === "complete").length).toBe(8) + await l.drain() + }) + + it("R1-1 (D5) one retryable failure with a budget of two releases leases attempt 2", async () => { + const w = W({ infraAttempts: 2 }); w.floor.enqueue(job("1")); w.ax.holdPending.add("wf-test-1-a1") + const l = start(w, dir, { pendingTimeoutMs: 150 }) + await until("attempt 2 leased", () => J(w).state === "leased" && J(w).attempt === 2, 4000) + expect(J(w).infraSpent).toBe(1) + expect(J(w).history).not.toContain("withdrawn:a2") + await l.drain() + }) + + it("R1-2 (D2) a Complete whose reply is lost after the floor committed is resent and answered duplicate, not applied twice", async () => { + const w = W({ infraAttempts: 2 }); w.floor.enqueue(job("1")); w.ax.holdPending.add("wf-test-1-a1") + let dropped = 0 + const fetch: typeof globalThis.fetch = async (input, init) => { + const c = await isComplete(input, init) + const res = await w.floor.fetch(input, init) + if (c && dropped === 0) { dropped++; throw new TypeError("fetch failed: reply lost after the floor committed") } + return res + } + const l = start(w, dir, { pendingTimeoutMs: 150 }, { fetch }) + await until("attempt 2 leased", () => J(w).state === "leased" && J(w).attempt === 2, 4000) + await until("the resend is acknowledged", () => l.logs.some((x) => x.ev === "complete" && x.leaseId === "wf-test-1-a1")) + expect(dropped).toBe(1) + expect(J(w).infraSpent).toBe(1) + expect(l.logs.find((x) => x.ev === "complete" && x.leaseId === "wf-test-1-a1")!.duplicate).toBe(true) + await l.drain() + }) + + // ------------------------------------------------------------------ finding 3: the create window is write-ahead + it("R1-3 (D1) crash after UpdateTask landed, before `created`, then a cancel: the Task is deleted and NotFound precedes `cancelled`", async () => { + const w = W(); w.floor.enqueue(job("1")); w.ax.hangAfterCreate = true + const l1 = start(w, dir) + await until("Task written to ax", () => w.ax.tasks.has("wf-test-1-a1")) + await l1.crash() + w.ax.hangAfterCreate = false + w.floor.cancel("wf-test-1") + const l2 = start(w, dir) + await until("floor done", () => J(w).state === "done") + expect(J(w).result).toBe("cancelled") + expect(w.ax.deletes).toContain("wf-test-1-a1") + expect(w.ax.tasks.has("wf-test-1-a1")).toBe(false) // NotFound before the verdict (reconcile reports on gone) + const evs = l2.logs.filter((x) => x.leaseId === "wf-test-1-a1" || x.task === "wf-test-1-a1").map((x) => x.ev) + expect(evs.indexOf("delete")).toBeLessThan(evs.indexOf("verdict")) + await l2.drain() + expect(Journal.open(dir).recs.has("wf-test-1-a1")).toBe(false) // compacted only once ax is clean + }) + + it("R1-3 (D1 lost variant) crash in the create window, the floor gives up on the job: `lost` deletes the maybe-created Task", async () => { + const w = W({ infraAttempts: 1 }); w.floor.enqueue(job("1")); w.ax.hangAfterCreate = true + const l1 = start(w, dir) + await until("Task written to ax", () => w.ax.tasks.has("wf-test-1-a1")) + await l1.crash() + w.ax.hangAfterCreate = false + await sleep(7 * SEC); w.floor.sweep(); await sleep(16 * SEC); w.floor.sweep() + expect(J(w).state).toBe("done") // budget spent at the first release: no re-grant + const l2 = start(w, dir) + await until("lost", () => l2.has("lost", "wf-test-1-a1")) + await until("Task gone", () => !w.ax.tasks.has("wf-test-1-a1")) + expect(w.ax.deletes).toContain("wf-test-1-a1") + await l2.drain() + }) + + // ------------------------------------------------------------------ findings 4 and 8: B15 + it("R1-8 (B15 deterministic) the first Complete after the floor returns is lost in transit: no Lease runs before it lands", async () => { + const w = W(); w.floor.enqueue(job("1")) + let failCompletes = 0 + const fetch: typeof globalThis.fetch = async (input, init) => { + if (failCompletes > 0 && await isComplete(input, init)) { failCompletes--; throw new TypeError("fetch failed") } + return w.floor.fetch(input, init) + } + const l = start(w, dir, {}, { fetch }) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + w.ax.finish("wf-test-1-a1", 0, { v: 1 }) + await until("outbox keeps", () => l.has("outbox-keep", "wf-test-1-a1")) + await until("requeued", () => J(w).history.includes("requeue:grace"), 3000) + failCompletes = 1 + const mark = w.floor.calls.length + w.floor.down = false + await until("floor done", () => J(w).state === "done", 4000) + expect([J(w).result, J(w).output, J(w).attempt]).toEqual(["success", { v: 1 }, 1]) + // round 3 (finding 12): the gate is asserted on the floor's side, not on the log line + const after = w.floor.calls.slice(mark), landed = after.findIndex((c) => c.rpc === "Complete") + expect(after.slice(0, landed).filter((c) => c.rpc === "Lease").map((c) => c.capacity).filter((c) => c !== 0)).toEqual([]) + expect(J(w).history).not.toContain("leased:a2") + expect(w.ax.updates.has("wf-test-1-a2")).toBe(false) + expect(l.has("complete-dropped", "wf-test-1-a1")).toBe(false) + await l.drain() + }) + + it("R1-4 (D3) Complete answers 503 while Lease answers: the gate asks for nothing until the success lands", async () => { + const w = W(); w.floor.enqueue(job("1")) + let failComplete = false + const fetch: typeof globalThis.fetch = async (input, init) => { + if (failComplete && await isComplete(input, init)) return new Response("upstream timeout", { status: 503 }) + return w.floor.fetch(input, init) + } + const l = start(w, dir, {}, { fetch }) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + await until("verdict written", () => l.has("verdict", "wf-test-1-a1")) + await until("floor requeued attempt 2", () => J(w).state === "queued" && J(w).attempt === 2, 4000) + failComplete = true + const mark = w.floor.calls.length + w.floor.down = false + await sleep(400) // several polls while only Complete fails + expect(l.has("lease-gated-by-outbox")).toBe(true) + // round 3 (finding 12): every Lease while the success is undelivered asks for nothing, and nothing is leased + const leases = w.floor.calls.slice(mark).filter((c) => c.rpc === "Lease") + expect([leases.length > 0, leases.filter((c) => c.capacity !== 0).length, J(w).history.includes("leased:a2")]).toEqual([true, 0, false]) + expect(w.ax.tasks.has("wf-test-1-a2")).toBe(false) + // a1 may be deleted as `lost` (its verdict is already journaled); it is never fenced as `superseded` + expect(l.logs.some((x) => x.ev === "delete" && x.task === "wf-test-1-a1" && x.why === "superseded")).toBe(false) + failComplete = false + await until("floor done", () => J(w).state === "done", 4000) + expect([J(w).result, J(w).output, J(w).attempt]).toEqual(["success", { answer: 42 }, 1]) + await l.drain() + }) + + it("R1-4 (D3 at dispatch) a grant that supersedes an undelivered success is withdrawn, never fenced or created", async () => { + const w = W(); w.floor.enqueue(job("1")) + let failComplete = false + const fetch: typeof globalThis.fetch = async (input, init) => { + if (failComplete && await isComplete(input, init)) return new Response("upstream timeout", { status: 503 }) + return w.floor.fetch(input, init) + } + const l = start(w, dir, {}, { fetch, }) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + await until("verdict written", () => l.has("verdict", "wf-test-1-a1")) + await until("floor requeued attempt 2", () => J(w).state === "queued" && J(w).attempt === 2, 4000) + failComplete = true + w.floor.o.overGrant = 1 // a floor that grants on a capacity-0 Lease: the link's own defence must hold + w.floor.down = false + await until("a2 granted", () => l.has("leased", "wf-test-1-a2"), 4000) + w.floor.o.overGrant = 0 + await until("dispatch waits for the verdict", () => l.has("supersede-waits-for-verdict", "wf-test-1-a2")) + failComplete = false + await until("floor done", () => J(w).state === "done", 4000) + expect([J(w).result, J(w).output, J(w).attempt]).toEqual(["success", { answer: 42 }, 1]) + expect(J(w).history).toContain("withdrawn:a2") + await until("a2 withdrawn locally", () => l.has("withdrawn", "wf-test-1-a2")) + expect(w.ax.updates.has("wf-test-1-a2")).toBe(false) + // the finished Task was not fenced before its verdict (a `lost` delete is fine: the verdict is already journaled) + expect(l.logs.some((x) => x.ev === "delete" && x.task === "wf-test-1-a1" && x.why === "superseded")).toBe(false) + await l.drain() + }) + + // ------------------------------------------------------------------ finding 5: supersedes is scoped to own Tasks + it("R1-5 supersedes naming a Task this link never created: no DeleteTask, the grant is refused", async () => { + const ax = new FakeAx() + ax.put(foreign("someone-elses-task"), "Running") + const f = stubFloor([grantOf("1", { attempt: 2, leaseId: "wf-test-1-a2", supersedes: ["someone-elses-task"] })]) + const logs = await runStub(cfgOf(), ax, f.api, 600) + expect(ax.deletes).toEqual([]) + expect(ax.live("someone-elses-task")).toBe(true) + expect(ax.tasks.has("wf-test-1-a2")).toBe(false) + expect(f.completes.map((c) => [c.leaseId, c.output?.reason])).toEqual([["wf-test-1-a2", "pre-start/supersedes-foreign"]]) + expect(logs.some((x) => x.ev === "supersedes-foreign")).toBe(true) + }) + + it("R1-5 an earlier attempt of the same job that carries another holder's mark is not deleted either", async () => { + const ax = new FakeAx() + const built = axTaskFromGrant(grantOf("1"), { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"], holder: "other-link" }) + if (built._tag !== "ok") throw new Error("build") + ax.put(built.task, "Running") + const f = stubFloor([grantOf("1", { attempt: 2, leaseId: "wf-test-1-a2", supersedes: ["wf-test-1-a1"] })]) + await runStub(cfgOf(), ax, f.api, 600) + expect(ax.deletes).toEqual([]) + expect(f.completes.map((c) => c.output?.reason)).toEqual(["pre-start/supersedes-foreign"]) + }) + + // ------------------------------------------------------------------ finding 6: the gate is checked per UpdateTask + it("R1-6 the Gateway disappears during the UpdateTask retry window: nothing is created while it is missing", async () => { + const ax = new FakeAx() + let failed = false + const orig = ax.createTask + ax.createTask = (task) => { if (!failed) { failed = true; ax.gateways.delete("halogen"); return Effect.fail(new AxError(UNAVAILABLE, "transient")) } return orig(task) } + const f = stubFloor([grantOf("3")]) + const logs = await runStub(cfgOf(), ax, f.api, 1000) + expect(logs.some((x) => x.ev === "gateway-missing")).toBe(true) + expect(ax.tasks.has("wf-test-3-a1")).toBe(false) + expect(logs.some((x) => x.ev === "gate-closed-before-create")).toBe(true) + }) + + it("R1-6 ax returns without P1 during the retry window: nothing is created in p1 mode", async () => { + const ax = new FakeAx() + let failed = false + const orig = ax.createTask + ax.createTask = (task) => { + if (!failed) { failed = true; ax.up = false; setTimeout(() => { ax.p1 = false; ax.up = true }, 60); return Effect.fail(new AxError(UNAVAILABLE, "transient")) } + return orig(task) + } + const f = stubFloor([grantOf("4")]) + const logs = await runStub(cfgOf(), ax, f.api, 1200) + expect(logs.some((x) => x.ev === "p1-missing")).toBe(true) + expect(ax.tasks.has("wf-test-4-a1")).toBe(false) + }) + + // ------------------------------------------------------------------ finding 7: guest mode fails closed + it("R1-7 guest mode with no leaseToken in the grant: refused before UpdateTask, logged once", async () => { + const ax = new FakeAx(); ax.p1 = false + const f = stubFloor([grantOf("2")]) + const logs = await runStub(cfgOf({ completion: "guest", shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"], completeUrl: "http://link.fleet.internal/guest" } }), ax, f.api, 600) + expect(ax.tasks.has("wf-test-2-a1")).toBe(false) + expect(f.completes.map((c) => c.output?.reason)).toEqual(["pre-start/lease-token-missing"]) + expect(logs.filter((x) => x.ev === "guest-token-missing").length).toBe(1) + }) + + // ------------------------------------------------------------------ finding 9: one bad grant is refused alone + it("R1-9 one undecodable grant next to a good one: the good one runs, the bad one is refused, no crash loop", async () => { + const w = W({ skewedEncode: true }) // the floor sends what the link's contract calls malformed + w.floor.enqueue(job("good")) + w.floor.enqueue(job("bad", { "timeout-minutes": 1.5 } as any)) // a float where the contract says Schema.Int + const l = start(w, dir) + await until("good created", () => w.ax.tasks.has("wf-test-good-a1")) + await until("bad refused", () => w.floor.jobs.get("wf-test-bad")!.state === "done") + expect((w.floor.jobs.get("wf-test-bad")!.output as any).reason).toBe("pre-start/invalid-spec") + expect(l.has("grant-invalid", "wf-test-bad-a1")).toBe(true) + await l.drain() + const j = Journal.open(dir) + expect(j.pendingLeaseKey).toBeUndefined() + expect(j.invalid.size).toBe(0) // told and compacted + }) + + it("R1-9 a Lease reply the floor cannot even encode is a typed transient error: the link stays up and heartbeats", async () => { + const w = W() // strict floor: its own encode of the float fails + w.floor.enqueue(job("bad", { "timeout-minutes": 1.5 } as any)) + const l = start(w, dir) + await until("lease errors", () => l.logs.filter((x) => x.ev === "lease-error").length >= 2) + await until("still heartbeating", () => w.floor.calls.filter((c) => c.rpc === "Heartbeat").length >= 2) + expect(l.logs.some((x) => String(x.ev).endsWith("-died"))).toBe(false) + expect(await l.drain()).toBeDefined() // exits on the stop signal, not on a defect + }) + + // ------------------------------------------------------------------ finding 10: configuration fails closed + it("R1-10 numeric and enum configuration is validated: a typo is ConfigInvalid naming the key, never a NaN interval", () => { + const bad = (env: Record) => { try { readLinkEnv(env); return undefined } catch (e) { return e instanceof ConfigInvalid ? e.key : "other" } } + expect(bad({ LINK_RESYNC_SECONDS: "15s" })).toBe("LINK_RESYNC_SECONDS") + expect(bad({ LINK_MAX_IN_FLIGHT: "0" })).toBe("LINK_MAX_IN_FLIGHT") + expect(bad({ LINK_CREATE_ATTEMPTS: "NaN" })).toBe("LINK_CREATE_ATTEMPTS") + expect(bad({ LINK_PENDING_TIMEOUT_SECONDS: "1.5" })).toBe("LINK_PENDING_TIMEOUT_SECONDS") + expect(bad({ LINK_COMPLETION: "gues" })).toBe("LINK_COMPLETION") + expect(bad({ LINK_SEAT_COMMANDS: "{halogen:[ax-agent]}" })).toBe("LINK_SEAT_COMMANDS") + expect(bad({ LINK_SEAT_COMMANDS: '{"halogen":[]}' })).toBe("LINK_SEAT_COMMANDS") + expect(bad({ LINK_OUTBOX_BACKOFF_MIN_SECONDS: "900", LINK_OUTBOX_BACKOFF_MAX_SECONDS: "60" })).toBe("LINK_OUTBOX_BACKOFF_MAX_SECONDS") + const ok = readLinkEnv({}) + expect([ok.resyncSeconds, ok.maxInFlight, ok.createAttempts, ok.completion, ok.seatCommands]).toEqual([15, 2, 7, "auto", { halogen: ["ax-agent", "pi"] }]) + try { readLinkEnv({ LINK_RESYNC_SECONDS: "secret-looking-value" }) } catch (e) { expect(String((e as Error).message)).not.toContain("secret-looking-value") } + }) +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r2.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r2.test.ts new file mode 100644 index 000000000..710d332c5 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r2.test.ts @@ -0,0 +1,419 @@ +// Review round 2 (2026-09-23): the twelve findings, each as a regression test. The scratch repros they restate are in +// /home/tom/today/evals-2026-09-23/link/scratch-r2-{0,1,2}; they failed against c5ed453. The log is link/REVIEW-LOG.md. +import { mkdtempSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import { Deferred, Effect, Fiber } from "effect" +import { afterEach, beforeEach, describe, expect, it } from "vitest" +import { cidrIsOpen, gatewayAllowsAll } from "../src/ax.ts" +import type { AgentJob } from "../src/contract.ts" +import { rpcFloor } from "../src/floor.ts" +import { Journal } from "../src/journal.ts" +import { isFleetInternalUrl, runLink } from "../src/link.ts" +import type { LinkConfig } from "../src/link.ts" +import { FakeAx } from "./fake-ax.ts" +import { FakeFloor } from "./fake-floor.ts" +import type { FloorConfig } from "./fake-floor.ts" + +const SEC = 20 +const A = "ultracode.mecattaf.dev/" +const JK = `${"cd".repeat(32)}:1` // round 4: FIELD-MAP 5a journal key (jobs.ts refuses a grant without it) +const job = (n: string): AgentJob => ({ + apiVersion: "ultracode.mecattaf.dev/v1alpha1", kind: "AgentJob", + metadata: { name: `wf-test-${n}`, labels: { [A + "run-id"]: "wf-test", [A + "workflow"]: "link-test", [A + "phase-index"]: "1" }, + annotations: { [A + "run-id-raw"]: "wf_test", [A + "label"]: `probe:${n}`, [A + "item-key"]: `wf_test#${n}`, [A + "journal-key"]: JK, [A + "phase-title"]: "Probe" } }, + spec: { "runs-on": ["seat:halogen", "runtime:gvisor"], + with: { prompt: `say ${n}`, prompt_ref: { sha256: "ab".repeat(32), bytes: 5, uri: `journal://wf_test/${n}/prompt.md` }, model: "halogen-qwen3.8-flash-next" } } +}) +const world = (o: Partial = {}) => ({ + floor: new FakeFloor({ cap: 2, leaseSeconds: 6, graceSeconds: 15, pollSeconds: 1, heartbeatSeconds: 2, maxAttempts: 3, secondMs: SEC, tokens: { "tok-nas": "nas-link-1" }, ...o }), + ax: new FakeAx() +}) +type World = ReturnType +function start(w: World, dir: string, over: Partial = {}, fetch?: typeof globalThis.fetch) { + const logs: Array> = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { + holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 1500, deleteAfterMs: 0, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, fenceTimeoutMs: 300, resultReadTries: 3, ...over + } + const fiber = Effect.runFork(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "tok-nas", sessionId: `s:${dir}`, fetch: fetch ?? w.floor.fetch }) + return yield* runLink(cfg, { ax: w.ax, floor, journal: Journal.open(dir), log: (ev, f) => logs.push({ t: Date.now(), ev, ...f }), stop, floorUrl: "http://floor.test" }) + }))) + return { + logs, fiber, + has: (ev: string, leaseId?: string) => logs.some((l) => l.ev === ev && (leaseId === undefined || l.leaseId === leaseId)), + trace: () => logs.filter((x) => typeof x.leaseId === "string" || typeof x.task === "string").map((x) => `${x.ev}:${x.leaseId ?? x.task}${x.why ? ":" + x.why : ""}`), + crash: () => Effect.runPromise(Fiber.interrupt(fiber)), + drain: () => { Effect.runSync(Deferred.succeed(stop, undefined)); return Promise.race([Effect.runPromise(Fiber.await(fiber)), sleep(1500)]) } + } +} +const until = async (what: string, pred: () => boolean, ms = 5000) => { + const t0 = Date.now() + while (!pred()) { if (Date.now() - t0 > ms) throw new Error(`timeout waiting for: ${what}`); await new Promise((r) => setTimeout(r, 5)) } +} +const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const J = (w: World, n = "1") => w.floor.jobs.get(`wf-test-${n}`)! +const bodyOf = async (input: unknown, init?: RequestInit) => { + const b = init?.body + if (typeof b === "string") return b + if (b instanceof Uint8Array) return new TextDecoder().decode(b) + if (b) return await new Response(b as ConstructorParameters[0]).text() + return (input instanceof Request) ? await input.clone().text() : "" +} +const rpcOf = async (input: unknown, init?: RequestInit) => { const b = await bodyOf(input, init); return b.includes('"Complete"') ? "Complete" : b.includes('"Lease"') ? "Lease" : b.includes('"Heartbeat"') ? "Heartbeat" : "?" } +/** A capacity race: Failed with ResourceExhausted, which the link maps to retryable pre-start/resource-exhausted. */ +const failRetryable = (w: World, name: string) => { + const t = w.ax.tasks.get(name)! + t.exited = true; t.phase = "Failed" + t.conditions = [{ type: "Ready", status: "False", reason: "ResourceExhausted", message: "ResourceExhausted: no free worker" }] +} +const why = (w: World, l: ReturnType) => JSON.stringify({ floor: J(w).history, link: l.trace() }) + +// A floor stub for the fail-closed tests (scratch-r2-1/repro.mts): hands out its grants once, renews everything. +const stats = { seq: 1, cap: 2, wip: 0, queued: 0, done: 0 } +const grantOf = (n: string, extra: Record = {}) => ({ leaseId: `wf-test-${n}-a1`, attempt: 1, job: job(n), + lease: { holderIdentity: "nas-link-1", leaseDurationSeconds: 30, acquireTime: Date.now(), renewTime: Date.now(), leaseTransitions: 0 }, ...extra }) +const stubFloor = (grants: Array, gate: () => boolean = () => true) => { + let sent = false + return { + lease: (p: any) => Effect.sync(() => { const g = !sent && p.capacity > 0 && gate() ? grants : []; if (g.length) sent = true; return { grants: g, invalid: [], stats, nextPollSeconds: 1, heartbeatSeconds: 2 } }), + heartbeat: (p: any) => Effect.succeed({ renewed: [...p.leaseIds], lost: [], cancelRequested: [], stats }), + complete: (_p: any) => Effect.succeed({ duplicate: false, stats }) + } +} +const runStub = async (over: Partial, ax: FakeAx, floor: any, ms: number) => { + const dir = mkdtempSync(join(tmpdir(), "conwip-link-r2s-")) + const logs: Array> = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 60_000, deleteAfterMs: 60_000, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, ...over } + const f = Effect.runFork(runLink(cfg, { ax, floor, journal: Journal.open(dir), log: (ev, x) => logs.push({ t: Date.now(), ev, ...x }), stop })) + await sleep(ms) + Effect.runSync(Deferred.succeed(stop, undefined)); await Effect.runPromise(Fiber.await(f)); rmSync(dir, { recursive: true, force: true }) + return logs +} + +describe("review round 2: fixed", () => { + let dir: string + let worlds: Array = [] + const W = (o: Partial = {}) => { const w = world(o); worlds.push(w); return w } + beforeEach(() => { dir = mkdtempSync(join(tmpdir(), "conwip-link-r2-")) }) + afterEach(() => { for (const w of worlds) w.floor.close(); worlds = []; rmSync(dir, { recursive: true, force: true }) }) + + // ---------------------------------------------------------------- R2-1: heartbeat `lost` and the janitor read first + it("R2-1a kill -9, the agent finishes, the floor requeues past grace, restart while ax is down: the success is delivered, a2 never runs", async () => { + const w = W(); w.floor.enqueue(job("1")) + const l1 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + await l1.crash() + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + await until("requeued by the alarm", () => J(w).history.includes("requeue:grace"), 3000) + w.ax.up = false + const l2 = start(w, dir) + await until("heartbeat says lost", () => l2.has("lost", "wf-test-1-a1")) + w.ax.up = true + await until("floor settles", () => J(w).state === "done" || w.ax.updates.has("wf-test-1-a2"), 4000) + await sleep(200) + expect(w.ax.updates.has("wf-test-1-a2"), why(w, l2)).toBe(false) + expect([J(w).result, J(w).output, J(w).attempt]).toEqual(["success", { answer: 42 }, 1]) + expect(l2.logs.findIndex((x) => x.ev === "verdict")).toBeLessThan(l2.logs.findIndex((x) => x.ev === "delete" && x.task === "wf-test-1-a1")) + await l2.drain() + }) + + it("R2-1b no restart: the agent finishes as the partition heals and the heartbeat's lost beats the resync: the success is delivered", async () => { + const w = W(); w.floor.enqueue(job("1")) + const l = start(w, dir, { resyncMs: 300 }) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + await until("requeued by the alarm", () => J(w).history.includes("requeue:grace"), 3000) + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + w.floor.down = false + await until("floor settles", () => J(w).state === "done" || w.ax.updates.has("wf-test-1-a2"), 4000) + await sleep(200) + expect(w.ax.updates.has("wf-test-1-a2"), why(w, l)).toBe(false) + expect([J(w).result, J(w).output, J(w).attempt]).toEqual(["success", { answer: 42 }, 1]) + await l.drain() + }) + + // ---------------------------------------------------------------- R2-2: the supersede fence reads first + it("R2-2 the Lease wins the race (heartbeats failing): a2's fence reads a1's Completed result, the floor withdraws a2", async () => { + const w = W(); w.floor.enqueue(job("1")) + let failHeartbeat = false + const fetch: typeof globalThis.fetch = async (input, init) => { + if (failHeartbeat && await rpcOf(input, init) === "Heartbeat") throw new TypeError("fetch failed") + return w.floor.fetch(input, init) + } + const l = start(w, dir, { resyncMs: 300 }, fetch) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + await until("requeued by the alarm", () => J(w).history.includes("requeue:grace"), 3000) + failHeartbeat = true + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + w.floor.down = false + await until("floor settles", () => J(w).state === "done" || w.ax.updates.has("wf-test-1-a2"), 4000) + failHeartbeat = false + await sleep(200) + expect(w.ax.updates.has("wf-test-1-a2"), why(w, l)).toBe(false) + expect([J(w).result, J(w).output, J(w).attempt]).toEqual(["success", { answer: 42 }, 1]) + await l.drain() + }) + + // ---------------------------------------------------------------- R2-3: one verdict never wedges the link + it("R2-3a a result over the link's output cap is refused agent/output-too-large; later jobs are leased", async () => { + const w = W({ cap: 4 }); w.floor.enqueue(job("1")) + const l = start(w, dir, { maxInFlight: 2, outboxBackoffMs: [20, 60] }) + await until("a1 running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.finish("wf-test-1-a1", 0, { report: "x".repeat(2 * 1024 * 1024 + 1000) }) + await until("verdict written", () => l.has("verdict", "wf-test-1-a1")) + w.floor.enqueue(job("2")); w.floor.enqueue(job("3")) + await until("job 2 leased", () => J(w, "2").state !== "queued", 3000) + expect([J(w).result, (J(w).output as any)?.reason]).toEqual(["failure", "agent/output-too-large"]) + await l.drain() + }) + + it("R2-3b a verdict the floor answers 500 every time is replaced by infra/verdict-undeliverable, and never gates Lease", async () => { + const w = W({ cap: 4 }); w.floor.enqueue(job("1")) + let refused = 0 + const fetch: typeof globalThis.fetch = async (input, init) => { + const b = await bodyOf(input, init) + if (b.includes('"Complete"') && b.includes('"success"')) { refused++; return new Response("internal error", { status: 500 }) } + return w.floor.fetch(input, init) + } + const l = start(w, dir, { outboxBackoffMs: [20, 60], verdictAttempts: 4 }, fetch) + await until("a1 running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + await until("verdict written", () => l.has("verdict", "wf-test-1-a1")) + w.floor.enqueue(job("2")) + await until("job 2 leased", () => J(w, "2").state !== "queued", 3000) + await until("job 1 closed", () => J(w).state === "done", 3000) + expect([J(w).result, (J(w).output as any)?.reason, refused]).toEqual(["failure", "infra/verdict-undeliverable", 4]) + expect(l.has("verdict-replaced", "wf-test-1-a1")).toBe(true) + await l.drain() + }) + + it("R2-3c an RPC Defect reply (SQLITE_TOOBIG) is a server-error, not an encode defect", async () => { + const fetch: typeof globalThis.fetch = async (input, init) => { + const req = JSON.parse(await bodyOf(input, init)), id = (Array.isArray(req) ? req[0] : req).id + return new Response(JSON.stringify([{ _tag: "Exit", requestId: id, exit: { _tag: "Failure", cause: [{ _tag: "Die", defect: { message: "string or blob too big: SQLITE_TOOBIG" } }] } }]), { status: 200, headers: { "content-type": "application/json" } }) + } + const e = await Effect.runPromise(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "t", sessionId: "s", fetch }) + return yield* floor.complete({ leaseId: "x-a1", attempt: 1, result: "success", output: {} }).pipe(Effect.flip) + }))) + expect([e._tag, (e as any).kind, /SQLITE_TOOBIG/.test((e as any).message ?? "")]).toEqual(["FloorError", "server-error", true]) + }) + + it("R2-3d Complete classification: 413 and 422 drop the verdict, 404 is fatal misrouted (verdict kept), 500 is server-error, 503 transient", async () => { + const out: Record = {} + for (const status of [413, 422, 404, 500, 503, 429]) { + const e = await Effect.runPromise(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "t", sessionId: "s", fetch: async () => new Response("x", { status, headers: { "retry-after": "7" } }) }) + return yield* floor.complete({ leaseId: "x-a1", attempt: 1, result: "success" }).pipe(Effect.flip) + }))) + out[status] = e._tag === "CompleteRefused" ? `refused:${e.code}` : `${e.kind}:${e.fatal}${e.kind === "rate-limited" ? `:${e.retryAfterMs}` : ""}` + } + // round 3 (finding 13): 429 is rate-limited with its Retry-After, the verdict kept (not `rejected`, which drops it) + expect(out).toEqual({ 413: "refused:http-413", 422: "refused:http-422", 404: "misrouted:true", 500: "server-error:false", 503: "transient:false", 429: "rate-limited:false:7000" }) + }) + + // ---------------------------------------------------------------- R2-4: withdrawal is explicit on the wire + it("R2-4a a Lease crosses a1's retryable verdict, whose reply is lost; the resend answers duplicate: a2 is created", async () => { + const w = W(); w.floor.enqueue(job("1")) + let holdLease = false, loseCompleteReply = false, failComplete = false + const fetch: typeof globalThis.fetch = async (input, init) => { + const rpc = await rpcOf(input, init) + if (rpc === "Lease") while (holdLease) await sleep(2) + if (rpc === "Complete" && failComplete) return new Response("upstream timeout", { status: 503 }) + if (rpc === "Complete" && loseCompleteReply) { await w.floor.fetch(input, init); loseCompleteReply = false; failComplete = true; throw new TypeError("fetch failed") } + return w.floor.fetch(input, init) + } + const l = start(w, dir, {}, fetch) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + holdLease = true + await sleep(60) + loseCompleteReply = true + failRetryable(w, "wf-test-1-a1") + await until("floor applied a1 and requeued a2", () => J(w).state === "queued" && J(w).attempt === 2, 4000) + holdLease = false + await until("a2 granted", () => l.has("leased", "wf-test-1-a2"), 4000) + await until("dispatch waits for the verdict", () => l.has("supersede-waits-for-verdict", "wf-test-1-a2")) + failComplete = false + await until("a2 created or withdrawn", () => w.ax.updates.has("wf-test-1-a2") || l.has("withdrawn", "wf-test-1-a2"), 4000) + await sleep(300) + expect(l.has("withdrawn", "wf-test-1-a2"), why(w, l)).toBe(false) + expect(w.ax.updates.get("wf-test-1-a2")).toBe(1) + expect(J(w).history).not.toContain("requeue:omitted") + await l.drain() + }) + + it("R2-4b the same race with an infrastructure budget of 2: the retry runs once, the job is not ended retry-budget-spent", async () => { + const w = W({ infraAttempts: 2 }); w.floor.enqueue(job("1")) + let armed = false, dropped = 0, committed!: () => void + const done = new Promise((r) => { committed = r }) + const fetch: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + if (armed && body.includes('"Lease"') && /"capacity":[1-9]/.test(body)) { armed = false; await done } + if (body.includes('"Complete"') && body.includes("wf-test-1-a1") && dropped > 0) await sleep(200) + const res = await w.floor.fetch(input, init) + if (body.includes('"Complete"') && body.includes("wf-test-1-a1") && dropped === 0) { dropped++; committed(); throw new TypeError("fetch failed: reply lost") } + return res + } + const l = start(w, dir, { outboxBackoffMs: [150, 150], fenceTimeoutMs: 3000 }, fetch) + await until("a1 running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + armed = true + await until("a Lease with capacity is held", () => !armed, 3000) + failRetryable(w, "wf-test-1-a1") + await until("a2 leased", () => l.has("leased", "wf-test-1-a2"), 4000) + await until("a2 created", () => w.ax.updates.has("wf-test-1-a2"), 4000).catch(() => {}) + expect(w.ax.updates.get("wf-test-1-a2") ?? 0, why(w, l)).toBe(1) + expect(l.has("withdrawn", "wf-test-1-a2")).toBe(false) + expect((J(w).output as any)?.reason).not.toBe("infra/retry-budget-spent") + await l.drain() + }) + + // ---------------------------------------------------------------- R2-5: a regrant of a withdrawn leaseId is new work + it("R2-5a rule 4b withdraws a2 for a1's retryable verdict and the floor grants a2 again: the new generation is created", async () => { + const w = W(); w.floor.enqueue(job("1")) + let holdLease = false, failHeartbeat = false, failComplete = false + const fetch: typeof globalThis.fetch = async (input, init) => { + const rpc = await rpcOf(input, init) + if (rpc === "Heartbeat" && failHeartbeat) throw new TypeError("fetch failed") + if (rpc === "Complete" && failComplete) return new Response("upstream timeout", { status: 503 }) + if (rpc === "Lease") while (holdLease) await sleep(2) + return w.floor.fetch(input, init) + } + const l = start(w, dir, {}, fetch) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + await until("floor requeued attempt 2", () => J(w).state === "queued" && J(w).attempt === 2, 4000) + failHeartbeat = true; holdLease = true; failComplete = true + w.floor.down = false + await sleep(80) + failRetryable(w, "wf-test-1-a1") + await until("verdict written", () => l.has("verdict", "wf-test-1-a1")) + holdLease = false + await until("a2 received", () => l.has("leased", "wf-test-1-a2"), 4000) + await until("dispatch waits for the verdict", () => l.has("supersede-waits-for-verdict", "wf-test-1-a2")) + failComplete = false; failHeartbeat = false + await until("a2 created, or the floor gives up on a2", () => w.ax.updates.has("wf-test-1-a2") || J(w).attempt >= 3 || J(w).state === "done", 4000).catch(() => {}) + expect(w.ax.updates.get("wf-test-1-a2") ?? 0, why(w, l)).toBe(1) + expect(l.has("duplicate-grant", "wf-test-1-a2")).toBe(false) + await l.drain() + }) + + it("R2-5b the Lease reply carrying a2 is slow; a1's retryable verdict lands first and withdraws it; the regrant is created", async () => { + const w = W(); w.floor.enqueue(job("1")) + let holdLeaseReply = false, failHeartbeat = false + const fetch: typeof globalThis.fetch = async (input, init) => { + const rpc = await rpcOf(input, init) + if (rpc === "Heartbeat" && failHeartbeat) throw new TypeError("fetch failed") + if (rpc === "Lease" && holdLeaseReply) { const r = await w.floor.fetch(input, init); while (holdLeaseReply) await sleep(2); return r } + return w.floor.fetch(input, init) + } + const l = start(w, dir, {}, fetch) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + await until("floor requeued attempt 2", () => J(w).state === "queued" && J(w).attempt === 2, 4000) + failHeartbeat = true; holdLeaseReply = true + w.floor.down = false + await until("floor leased a2 to this link", () => J(w).state === "leased" && J(w).attempt === 2, 4000) + failRetryable(w, "wf-test-1-a1") + await until("verdict written", () => l.has("verdict", "wf-test-1-a1")) + holdLeaseReply = false + await until("a2 received", () => l.has("leased", "wf-test-1-a2"), 4000) + failHeartbeat = false + await until("a2 created, or the floor gives up", () => w.ax.updates.has("wf-test-1-a2") || J(w).attempt >= 3 || J(w).state === "done", 4000).catch(() => {}) + expect(w.ax.updates.get("wf-test-1-a2") ?? 0, why(w, l)).toBe(1) + await l.drain() + }) + + // ---------------------------------------------------------------- R2-6: redirects are never followed + it("R2-6 the floor answers 307 to another origin: nothing is leased or created and the link stops fatal (redirect)", async () => { + const w = W(); w.floor.enqueue(job("1")) + let rogue = 0 + const fetch: typeof globalThis.fetch = async (input, init) => { + if ((init as RequestInit | undefined)?.redirect !== "manual") { rogue++; return w.floor.fetch(input, init) } // a follower would land on the rogue origin + return new Response(null, { status: 307, headers: { location: "http://127.0.0.1:9/rpc" } }) + } + const l = start(w, dir, {}, fetch) + const exit = await Promise.race([Effect.runPromise(Fiber.await(l.fiber)), sleep(1500).then(() => undefined)]) + expect([rogue, w.ax.tasks.size, l.has("leased")]).toEqual([0, 0, false]) + expect(exit?._tag).toBe("Failure") + expect(JSON.stringify(exit)).toContain("redirect") + }) + + // ---------------------------------------------------------------- R2-7, R2-8: fail-closed checks + it("R2-7 isFleetInternalUrl matches private ranges on IP literals only", () => { + for (const u of ["https://10.evil.example/guest", "https://127.0.0.1.nip.io/guest", "https://192.168.0.1.attacker.com/c", "https://link/guest", "https://x.workers.dev/", "https://10.internal.example.com/"]) + expect([u, isFleetInternalUrl(u)]).toEqual([u, false]) + for (const u of ["http://10.201.0.80:8080/c", "http://127.0.0.1/c", "http://100.64.0.7/c", "http://[fd7a:115c:a1e0::1]/c", "http://[::1]/c", "http://link.fleet.internal/guest", "http://nas.lan/c", "http://localhost:1/c", "http://0x7f.1/c"]) + expect([u, isFleetInternalUrl(u)]).toEqual([u, true]) + expect(isFleetInternalUrl("http://link/guest", ["link"])).toBe(true) + }) + + it("R2-7b guest mode with a public name that starts with a private prefix: no Task, no token handed out", async () => { + const ax = new FakeAx(); ax.p1 = false + const logs = await runStub({ completion: "guest", shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"], completeUrl: "https://10.evil.example/guest" } }, + ax, stubFloor([grantOf("g", { leaseToken: "tok" })]), 400) + expect([logs.some((l) => l.ev === "guest-url-invalid"), ax.tasks.has("wf-test-g-a1")]).toEqual([true, false]) + }) + + it("R2-8 a Gateway open by CIDR arithmetic is refused: ::/0, 0.0.0.0/1 + 128.0.0.0/1, and unparseable CIDRs", async () => { + const gw = (hosts: Array) => ({ hasAllowlist: true, hosts: hosts.map((host) => ({ host, port: 443 })) }) + expect(gatewayAllowsAll(gw(["worker", "::/0"]))).toBe(true) + expect(gatewayAllowsAll(gw(["0.0.0.0/1", "128.0.0.0/1"]))).toBe(true) + expect(gatewayAllowsAll(gw([" * "]))).toBe(true) + expect(gatewayAllowsAll(gw(["10.0.0.0/x"]))).toBe(true) + expect(gatewayAllowsAll(gw(["worker", "10.201.0.0/24", "fd7a:115c:a1e0::/48"]))).toBe(false) + expect([cidrIsOpen("8.0.0.0/8"), cidrIsOpen("10.1.0.0/16"), cidrIsOpen("2000::/3"), cidrIsOpen("fd00::/64")]).toEqual([true, false, true, false]) + const ax = new FakeAx() + ax.gateways.set("halogen", gw(["worker", "::/0"])) + const logs = await runStub({}, ax, stubFloor([grantOf("w")]), 400) + expect([logs.some((l) => l.ev === "gateway-missing"), ax.tasks.has("wf-test-w-a1")]).toEqual([true, false]) + }) + + // ---------------------------------------------------------------- R2-9: nothing from before an ax outage opens the gate + it("R2-9 guest mode: ax returns without the Gateway and GetGateway is slow; nothing is created on pre-outage state", async () => { + const ax = new FakeAx(); ax.p1 = false + const origGw = ax.getGateway + let outageOver = false, release = false + ax.getGateway = (name: string) => outageOver ? Effect.sleep(300).pipe(Effect.andThen(origGw(name))) : origGw(name) + setTimeout(() => { ax.up = false }, 150) + setTimeout(() => { release = true }, 170) + setTimeout(() => { ax.gateways.delete("halogen"); outageOver = true; ax.up = true }, 400) + const logs = await runStub({ completion: "guest", shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"], completeUrl: "http://link.fleet.internal/guest" } }, + ax, stubFloor([grantOf("u", { leaseToken: "tok" })], () => release), 1200) + expect(ax.tasks.has("wf-test-u-a1"), JSON.stringify(logs.map((l) => l.ev))).toBe(false) + expect(logs.some((l) => l.ev === "gateway-missing")).toBe(true) + }) + + // ---------------------------------------------------------------- R2-11: a permanent Complete refusal is dropped + for (const variant of ["http-413", "unknown-error-code"] as const) + it(`R2-11 one Complete refused for good (${variant}) is dropped and job 2 is leased`, async () => { + const w = W(); w.floor.enqueue(job("1")) + const fetch: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + if (body.includes('"Complete"') && body.includes("wf-test-1-a1")) { + if (variant === "http-413") return new Response("payload too large", { status: 413 }) + const req = JSON.parse(body), id = (Array.isArray(req) ? req[0] : req).id + return new Response(JSON.stringify([{ _tag: "Exit", requestId: id, exit: { _tag: "Failure", cause: [{ _tag: "Fail", error: { code: "output-too-large" } }] } }]), { status: 200, headers: { "content-type": "application/json" } }) + } + return w.floor.fetch(input, init) + } + const l = start(w, dir, { outboxBackoffMs: [20, 40] }, fetch) + await until("a1 running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.finish("wf-test-1-a1", 0, { big: "x" }) + await until("verdict", () => l.has("verdict", "wf-test-1-a1")) + w.floor.enqueue(job("2")) + await until("job 2 leased", () => J(w, "2").state !== "queued", 3000) + const d = l.logs.find((x) => x.ev === "complete-dropped" && x.leaseId === "wf-test-1-a1") + expect(d?.code).toBe(variant === "http-413" ? "http-413" : "output-too-large") + await l.drain() + }) +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r3.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r3.test.ts new file mode 100644 index 000000000..057b6faa2 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r3.test.ts @@ -0,0 +1,484 @@ +// Review round 3 (2026-09-23): the thirteen findings, each as a regression test. The scratch repros they restate are +// in /home/tom/today/evals-2026-09-23/link/scratch-r3-{0,1,2}; they failed against d582ead. The log is +// link/REVIEW-LOG.md. Test adequacy findings 11 to 13 also strengthen R1-4, R1-8 and R2-3d in place. +import { appendFileSync, mkdtempSync, rmSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import { Deferred, Effect, Fiber } from "effect" +import { afterEach, beforeEach, describe, expect, it } from "vitest" +import { AxError } from "../src/ax.ts" +import type { AgentJob } from "../src/contract.ts" +import { rpcFloor } from "../src/floor.ts" +import { Journal } from "../src/journal.ts" +import { boundedSeconds, isFleetInternalUrl, runLink } from "../src/link.ts" +import type { LinkConfig } from "../src/link.ts" +import { FakeAx } from "./fake-ax.ts" +import { FakeFloor } from "./fake-floor.ts" +import type { FloorConfig } from "./fake-floor.ts" + +const SEC = 20 +const A = "ultracode.mecattaf.dev/" +const JK = `${"cd".repeat(32)}:1` // round 4: FIELD-MAP 5a journal key (jobs.ts refuses a grant without it) +const job = (n: string, spec: Record = {}): AgentJob => ({ + apiVersion: "ultracode.mecattaf.dev/v1alpha1", kind: "AgentJob", + metadata: { name: `wf-test-${n}`, labels: { [A + "run-id"]: "wf-test", [A + "workflow"]: "link-test", [A + "phase-index"]: "1" }, + annotations: { [A + "run-id-raw"]: "wf_test", [A + "label"]: `probe:${n}`, [A + "item-key"]: `wf_test#${n}`, [A + "journal-key"]: JK, [A + "phase-title"]: "Probe" } }, + spec: { "runs-on": ["seat:halogen", "runtime:gvisor"], + with: { prompt: `say ${n}`, prompt_ref: { sha256: "ab".repeat(32), bytes: 5, uri: `journal://wf_test/${n}/prompt.md` }, model: "halogen-qwen3.8-flash-next" }, ...spec } as AgentJob["spec"] +}) +const world = (o: Partial = {}) => ({ + floor: new FakeFloor({ cap: 2, leaseSeconds: 6, graceSeconds: 15, pollSeconds: 1, heartbeatSeconds: 2, maxAttempts: 3, secondMs: SEC, tokens: { "tok-nas": "nas-link-1" }, ...o }), + ax: new FakeAx() +}) +type World = ReturnType +function start(w: World, dir: string, over: Partial = {}, fetch?: typeof globalThis.fetch) { + const logs: Array> = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { + holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 1500, deleteAfterMs: 0, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, fenceTimeoutMs: 300, resultReadTries: 3, ...over + } + const fiber = Effect.runFork(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "tok-nas", sessionId: `s:${dir}`, fetch: fetch ?? w.floor.fetch }) + return yield* runLink(cfg, { ax: w.ax, floor, journal: Journal.open(dir), log: (ev, f) => logs.push({ t: Date.now(), ev, ...f }), stop, floorUrl: "http://floor.test" }) + }))) + return { + logs, fiber, + has: (ev: string, leaseId?: string) => logs.some((l) => l.ev === ev && (leaseId === undefined || l.leaseId === leaseId)), + trace: () => logs.filter((x) => typeof x.leaseId === "string" || typeof x.task === "string").map((x) => `${x.ev}:${x.leaseId ?? x.task}${x.why ? ":" + x.why : ""}`), + crash: () => Effect.runPromise(Fiber.interrupt(fiber)), + drain: () => { Effect.runSync(Deferred.succeed(stop, undefined)); return Promise.race([Effect.runPromise(Fiber.await(fiber)), sleep(1500)]) } + } +} +const until = async (what: string, pred: () => boolean, ms = 5000) => { + const t0 = Date.now() + while (!pred()) { if (Date.now() - t0 > ms) throw new Error(`timeout waiting for: ${what}`); await new Promise((r) => setTimeout(r, 5)) } +} +const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms)) +const J = (w: World, n = "1") => w.floor.jobs.get(`wf-test-${n}`)! +const bodyOf = async (input: unknown, init?: RequestInit) => { + const b = init?.body + if (typeof b === "string") return b + if (b instanceof Uint8Array) return new TextDecoder().decode(b) + if (b) return await new Response(b as ConstructorParameters[0]).text() + return (input instanceof Request) ? await input.clone().text() : "" +} +const rpcOf = async (input: unknown, init?: RequestInit) => { const b = await bodyOf(input, init); return b.includes('"Complete"') ? "Complete" : b.includes('"Lease"') ? "Lease" : b.includes('"Heartbeat"') ? "Heartbeat" : "?" } +const failRetryable = (w: World, name: string) => { + const t = w.ax.tasks.get(name)! + t.exited = true; t.phase = "Failed" + t.conditions = [{ type: "Ready", status: "False", reason: "ResourceExhausted", message: "ResourceExhausted: no free worker" }] +} +const why = (w: World, l: ReturnType) => JSON.stringify({ floor: J(w).history, link: l.trace() }) +/** Rewrite every JSON reply of one RPC with `patch` (a floor whose wire drifted from the link's). */ +const rewriting = (w: World, rpc: string, patch: (x: any) => any): typeof globalThis.fetch => async (input, init) => { + const b = await bodyOf(input, init) + const res = await w.floor.fetch(input, init) + if (!b.includes(`"${rpc}"`)) return res + const txt = await res.text() + let j: unknown; try { j = JSON.parse(txt) } catch { return new Response(txt, { status: res.status, headers: res.headers }) } + return new Response(JSON.stringify(patch(j)), { status: res.status, headers: { "content-type": "application/json" } }) +} +const deep = (f: (o: Record) => void) => { const walk = (x: any): any => { + if (Array.isArray(x)) return x.map(walk) + if (x && typeof x === "object") { const o: Record = {}; for (const [k, v] of Object.entries(x)) o[k] = walk(v); f(o); return o } + return x +}; return walk } + +// A floor stub that grants one lease named after a Task someone else owns (scratch-r3-1). +const stats = { seq: 1, cap: 2, wip: 0, queued: 0, done: 0 } +const foreignTask = (name: string) => ({ apiVersion: "ax.io/v1alpha1" as const, kind: "Task" as const, metadata: { name, atespace: "fleet" }, + spec: { image: "someone-else", command: ["long-job"], env: [{ name: "AX_CONWIP_HOLDER", value: "nas-link-2" }], gateway: { name: "halogen" } } }) +const foreignGrant = (name: string) => ({ leaseId: name, attempt: 1, job: job("x"), + lease: { holderIdentity: "nas-link-1", leaseDurationSeconds: 30, acquireTime: Date.now(), renewTime: Date.now(), leaseTransitions: 0 } }) +const runStub = async (dir: string, ax: FakeAx, floor: any, ms: number, over: Partial = {}) => { + const logs: Array> = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 60_000, deleteAfterMs: 600_000, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, ...over } + const f = Effect.runFork(runLink(cfg, { ax, floor, journal: Journal.open(dir), log: (ev, x) => logs.push({ t: Date.now(), ev, ...x }), stop })) + await sleep(ms) + Effect.runSync(Deferred.succeed(stop, undefined)); await Effect.runPromise(Fiber.await(f)) + return logs +} + +describe("review round 3", () => { + let dir: string + let worlds: Array = [] + const W = (o: Partial = {}) => { const w = world(o); worlds.push(w); return w } + beforeEach(() => { dir = mkdtempSync(join(tmpdir(), "conwip-link-r3-")) }) + afterEach(() => { for (const w of worlds) w.floor.close(); worlds = []; rmSync(dir, { recursive: true, force: true }) }) + + // ---------------------------------------------------------------- R3-1: a withdrawal names a generation + for (const sendWithdrewGen of [true, false]) { + it(`R3-1a (E) rule 4b withdraws a QUEUED a2 for a1's retryable verdict; the requeued a2 is created (floor names the generation: ${sendWithdrewGen})`, async () => { + const w = W({ sendWithdrewGen }); w.floor.enqueue(job("1")) + // Deterministic E: no Lease reaches the floor between its return and a1's verdict landing, so a2 gen 1 is withdrawn + // while still QUEUED and this link never receives it (a Lease retried across the return could carry a capacity + // computed before the verdict and receive gen 1 first: that is R2-5a's interleaving, not this one). + let gate = false, landed = false + const fetch: typeof globalThis.fetch = async (input, init) => { + const rpc = await rpcOf(input, init) + if (rpc === "Lease" && gate && !landed) throw new TypeError("fetch failed") + const r = await w.floor.fetch(input, init) + if (rpc === "Complete" && gate) landed = true + return r + } + const l = start(w, dir, {}, fetch) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + gate = true + w.floor.down = true + await until("floor requeued attempt 2", () => J(w).state === "queued" && J(w).attempt === 2, 4000) + failRetryable(w, "wf-test-1-a1") + await until("verdict written", () => l.has("verdict", "wf-test-1-a1")) + w.floor.down = false + await until("a2 created", () => w.ax.updates.has("wf-test-1-a2"), 4000).catch(() => { throw new Error(why(w, l)) }) + expect(J(w).history).toContain("withdrawn:a2") + expect(l.logs.some((x) => x.ev === "leased" && x.leaseId === "wf-test-1-a2" && l.logs.indexOf(x) < l.logs.findIndex((y) => y.ev === "complete"))).toBe(false) + w.ax.finish("wf-test-1-a2", 0, { ok: 2 }) + await until("floor done", () => J(w).state === "done", 4000) + expect([J(w).result, J(w).attempt, w.ax.updates.get("wf-test-1-a2")]).toEqual(["success", 2, 1]) + await l.drain() + }) + + it(`R3-1b (C) kill -9 between the withdrawal and the regrant: the regrant of a2 is created after restart (floor names the generation: ${sendWithdrewGen})`, async () => { + const w = W({ sendWithdrewGen }); w.floor.enqueue(job("1")) + let holdLease = false, failHeartbeat = false, failComplete = false, holdAfterComplete = false + const fetch: typeof globalThis.fetch = async (input, init) => { + const rpc = await rpcOf(input, init) + if (rpc === "Heartbeat" && failHeartbeat) throw new TypeError("fetch failed") + if (rpc === "Complete" && failComplete) return new Response("upstream timeout", { status: 503 }) + if (rpc === "Lease") while (holdLease) await sleep(2) + const r = await w.floor.fetch(input, init) + if (rpc === "Complete" && holdAfterComplete) holdLease = true // the regrant never reaches this process + return r + } + const over = { deleteAfterMs: 60_000 } // a1's record survives the restart's compaction; a2 gen 1 does not + const l1 = start(w, dir, over, fetch) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + await until("floor requeued attempt 2", () => J(w).state === "queued" && J(w).attempt === 2, 4000) + failHeartbeat = true; holdLease = true; failComplete = true + w.floor.down = false + await sleep(80) + failRetryable(w, "wf-test-1-a1") + await until("verdict written", () => l1.has("verdict", "wf-test-1-a1")) + holdLease = false + await until("a2 received", () => l1.has("leased", "wf-test-1-a2"), 4000) + await until("dispatch waits for the verdict", () => l1.has("supersede-waits-for-verdict", "wf-test-1-a2")) + holdAfterComplete = true; failComplete = false; failHeartbeat = false + await until("floor withdrew a2 and the link journaled it", () => l1.has("withdrawn", "wf-test-1-a2"), 4000) + await l1.crash() + const l2 = start(w, dir, over) + await until("a2 created", () => w.ax.updates.has("wf-test-1-a2"), 4000).catch(() => { throw new Error(why(w, l2)) }) + expect(w.ax.updates.get("wf-test-1-a2")).toBe(1) + expect(J(w).history.filter((h) => h === "leased:a2").length).toBe(2) // gen 1 withdrawn, gen 2 run + await l2.drain() + }) + } + + // ---------------------------------------------------------------- R3-2: the cancel intent is journaled first + it("R3-2 (A) a cancel whose DeleteTask lands but whose reply is lost is reported cancelled; no attempt 2 is started", async () => { + const w = W({ leaseSeconds: 20, heartbeatSeconds: 6, graceSeconds: 40 }); w.floor.enqueue(job("1")) + const l = start(w, dir, { initialHeartbeatSeconds: 6 }) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + const real = w.ax.deleteTask + let lostReplies = 0 + w.ax.deleteTask = (name: string) => real(name).pipe(Effect.andThen(() => lostReplies++ === 0 ? Effect.fail(new AxError(4, "deadline exceeded")) : Effect.void)) + w.floor.cancel("wf-test-1") + await until("floor done", () => J(w).state === "done", 4000).catch(() => { throw new Error(why(w, l)) }) + await sleep(200) + const v = l.logs.find((x) => x.ev === "verdict" && x.leaseId === "wf-test-1-a1")! + expect([v.result, J(w).result, J(w).attempt, J(w).infraSpent]).toEqual(["cancelled", "cancelled", 1, 0]) + expect(w.ax.updates.has("wf-test-1-a2")).toBe(false) + expect(l.logs.some((x) => x.ev === "delete-deferred" && x.task === "wf-test-1-a1")).toBe(true) + await l.drain() + }) + + it("R3-2b a journaled DeleteTask that did not land is sent again by the resync", async () => { + const w = W({ leaseSeconds: 20, heartbeatSeconds: 6, graceSeconds: 40 }); w.floor.enqueue(job("1")) + const l = start(w, dir, { initialHeartbeatSeconds: 6 }) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + const real = w.ax.deleteTask + let refused = 0 + w.ax.deleteTask = (name: string) => refused++ === 0 ? Effect.fail(new AxError(14, "unavailable")) : real(name) + w.floor.cancel("wf-test-1") + await until("floor done", () => J(w).state === "done", 4000).catch(() => { throw new Error(why(w, l)) }) + expect([J(w).result, w.ax.deletes.filter((d) => d === "wf-test-1-a1").length]).toEqual(["cancelled", 1]) + await l.drain() + }) + + // ---------------------------------------------------------------- R3-3: finished leases are tombstoned + for (const restart of [false, true]) it(`R3-3 (D) a redelivered grant of a finished lease is never run again (restart=${restart})`, async () => { + const w = W(); w.floor.enqueue(job("1")) + const l1 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.finish("wf-test-1-a1", 0, { answer: 42 }) + await until("done", () => J(w).state === "done") + await until("janitor deleted a1", () => !w.ax.tasks.has("wf-test-1-a1"), 3000) + await sleep(100) + let l = l1 + if (restart) { await l1.crash(); l = start(w, dir); await sleep(100) } + w.floor.redeliver = ["wf-test-1-a1"] + await until("duplicate-grant", () => l.has("duplicate-grant", "wf-test-1-a1"), 2000) + await sleep(200) + expect(w.ax.updates.get("wf-test-1-a1")).toBe(1) + await l.drain() + }) + + it("R3-3b the tombstone survives compaction and expires after TOMB_MS; a later generation of the same leaseId is not refused", () => { + const j = Journal.open(dir) + const g = { leaseId: "x-a1", attempt: 1, job: job("x"), lease: { holderIdentity: "h", leaseDurationSeconds: 30, acquireTime: 1000, renewTime: 1000, leaseTransitions: 0 } } + j.append({ ev: "grant", leaseId: "x-a1", grant: g }) + j.append({ ev: "report", leaseId: "x-a1", attempt: 1, result: "success" }) + j.append({ ev: "reported", leaseId: "x-a1", duplicate: false }) + j.compact() + expect(j.recs.has("x-a1")).toBe(false) + expect(Journal.open(dir).tombs.get("x-a1")?.gen).toEqual({ leaseTransitions: 0, acquireTime: 1000 }) + const keep = Journal.TOMB_MS + try { Journal.TOMB_MS = -1; Journal.open(dir).compact(); expect(Journal.open(dir).tombs.has("x-a1")).toBe(false) } finally { Journal.TOMB_MS = keep } + }) + + // ---------------------------------------------------------------- R3-4: renewal does not wait for the first Lease + it("R3-4 (B) a restart whose first Lease is rate-limited keeps heartbeating; the running a1 is not requeued", async () => { + const w = W(); w.floor.enqueue(job("1")) + const l1 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + await l1.crash() + let leases = 0 + const hbAt: Array = [] + const f: typeof globalThis.fetch = async (input, init) => { + const rpc = await rpcOf(input, init) + if (rpc === "Heartbeat") hbAt.push(Date.now()) + if (rpc === "Lease" && leases++ === 0) return new Response("slow down", { status: 429, headers: { "retry-after": "2" } }) + return w.floor.fetch(input, init) + } + const t0 = Date.now() + const l2 = start(w, dir, {}, f) + await sleep(2300) + expect(hbAt.filter((t) => t - t0 > 200 && t - t0 < 2000).length).toBeGreaterThan(5) + expect(J(w).history).not.toContain("requeue:grace") + expect([J(w).attempt, w.ax.tasks.get("wf-test-1-a1")?.phase]).toEqual([1, "Running"]) + await l2.drain() + }) + + // ---------------------------------------------------------------- R3-5: a Task the link did not write is never touched + for (const variant of ["janitor", "lost", "cancel"] as const) for (const phase of ["Running", "Completed"] as const) + it(`R3-5a a grant named after a foreign ${phase} Task: refused name-conflict, never deleted or read (${variant})`, async () => { + const ax = new FakeAx() + ax.put(foreignTask("other-team-build"), "Running") + if (phase === "Completed") ax.finish("other-team-build", 0, { secret: "other tenant's output" }) + let sent = false + const completes: Array = [] + const floor = { + lease: (p: any) => Effect.sync(() => { const g = !sent && p.capacity > 0 ? [foreignGrant("other-team-build")] : []; if (g.length) sent = true; return { grants: g, invalid: [], stats, nextPollSeconds: 1, heartbeatSeconds: 2 } }), + heartbeat: (p: any) => Effect.sync(() => { + const lost = variant === "lost" ? p.leaseIds.filter((x: string) => x === "other-team-build") : [] + const cancel = variant === "cancel" ? p.leaseIds.filter((x: string) => x === "other-team-build") : [] + return { renewed: p.leaseIds.filter((x: string) => !lost.includes(x)), lost, cancelRequested: cancel, stats } + }), + complete: (p: any) => Effect.sync(() => { completes.push(p); return { duplicate: false, stats } }) + } + const logs = await runStub(dir, ax, floor, 800) + expect(ax.deletes).toEqual([]) + expect(ax.tasks.get("other-team-build")?.phase).toBe(phase) + expect(completes.every((c) => c.result !== "success" && JSON.stringify(c).includes("secret") === false)).toBe(true) + expect(logs.some((x) => x.ev === "not-mine" && x.leaseId === "other-team-build")).toBe(true) + }) + + it("R3-5b (salvage) a crash in the create window, then `lost`: a foreign finished Task's result is never reported, the Task never deleted", async () => { + const g = foreignGrant("other-team-build") + const t = Date.now() + appendFileSync(join(dir, "journal.jsonl"), JSON.stringify({ t, ev: "grant", leaseId: g.leaseId, grant: g }) + "\n" + JSON.stringify({ t, ev: "creating", leaseId: g.leaseId }) + "\n") + const ax = new FakeAx() + ax.put(foreignTask("other-team-build"), "Running") + ax.finish("other-team-build", 0, { secret: "other tenant's output" }) + const completes: Array = [] + const floor = { + lease: (_p: any) => Effect.succeed({ grants: [], invalid: [], stats, nextPollSeconds: 1, heartbeatSeconds: 2 }), + heartbeat: (p: any) => Effect.succeed({ renewed: [], lost: p.leaseIds.filter((x: string) => x === "other-team-build"), cancelRequested: [], stats }), + complete: (p: any) => Effect.sync(() => { completes.push(p); return { duplicate: false, stats } }) + } + const logs = await runStub(dir, ax, floor, 600) + expect([completes, ax.deletes]).toEqual([[], []]) + expect(logs.some((x) => x.ev === "not-mine")).toBe(true) + }) + + it("R3-5c the link's own Task under a `creating`-only record (crash before `created`) is still salvaged and deleted", async () => { + const w = W(); w.floor.enqueue(job("1")) + w.ax.hangAfterCreate = true + const l1 = start(w, dir) + await until("created in ax", () => w.ax.tasks.has("wf-test-1-a1")) + await l1.crash() + w.ax.hangAfterCreate = false + w.ax.tick() + w.ax.finish("wf-test-1-a1", 0, { mine: true }) + const l2 = start(w, dir) + await until("floor done", () => J(w).state === "done", 4000).catch(() => { throw new Error(why(w, l2)) }) + expect([J(w).result, J(w).output]).toEqual(["success", { mine: true }]) + await until("deleted", () => !w.ax.tasks.has("wf-test-1-a1"), 3000) + expect(l2.has("not-mine")).toBe(false) + await l2.drain() + }) + + // ---------------------------------------------------------------- R3-6 and R3-7: server-driven intervals are bounded + for (const v of [0, -5]) it(`R3-6 a floor answering nextPollSeconds=${v} and heartbeatSeconds=${v} gets a bounded call rate (production secondMs)`, async () => { + const w = W({ pollSeconds: v, heartbeatSeconds: v, secondMs: 1000, leaseSeconds: 60, graceSeconds: 60 }) + w.floor.enqueue(job("1")) + const l = start(w, dir, { secondMs: 1000, resyncMs: 1000, initialPollSeconds: 15, initialHeartbeatSeconds: 30 }) + await sleep(1500) + const n = (rpc: string) => w.floor.calls.filter((c) => c.rpc === rpc).length + expect(n("Lease")).toBeLessThanOrEqual(4) + expect(n("Heartbeat")).toBeLessThanOrEqual(4) + expect(l.has("interval-clamped")).toBe(true) + await l.drain() + }) + + it("R3-7 boundedSeconds: default for missing, non-finite and not positive; clamped to [1, hi]", () => { + expect([boundedSeconds(undefined, 15, 300), boundedSeconds(0, 15, 300), boundedSeconds(-5, 15, 300), boundedSeconds(Number.NaN, 15, 300), + boundedSeconds(0.2, 15, 300), boundedSeconds(10 / 3, 15, 300), boundedSeconds(1e9, 15, 300), boundedSeconds(60, 30, 10)]).toEqual([15, 15, 15, 15, 1, 10 / 3, 300, 10]) + }) + + // ---------------------------------------------------------------- R3-8: the Lease envelope decodes leniently + it("R3-8 a Lease reply with heartbeatSeconds 3.333, endpoint null and no stats is used; the job runs", async () => { + const w = W(); w.floor.enqueue(job("1")) + const fetch = rewriting(w, "Lease", deep((o) => { if ("heartbeatSeconds" in o) { o.heartbeatSeconds = 10 / 3; o.endpoint = null; delete o.stats } })) + const l = start(w, dir, {}, fetch) + await until("created", () => w.ax.updates.has("wf-test-1-a1"), 4000).catch(() => { throw new Error(why(w, l)) }) + w.ax.finish("wf-test-1-a1", 0, { ok: 1 }) + await until("floor done", () => J(w).state === "done", 4000) + expect([J(w).result, J(w).attempt, l.has("lease-error")]).toEqual(["success", 1, false]) + await l.drain() + }) + + it("R3-8b a Lease key whose replies never decode is abandoned after leaseKeyAttempts, not replayed for ever; the job then runs", async () => { + const w = W(); w.floor.enqueue(job("1")) + let broken = true + const keys: Array = [] + const inner = rewriting(w, "Lease", deep((o) => { if (broken && "grants" in o && "nextPollSeconds" in o) o.grants = "not-an-array" })) + const fetch: typeof globalThis.fetch = async (input, init) => { + const b = await bodyOf(input, init) + const k = /"requestKey":"([^"]+)"/.exec(b)?.[1] + if (k && b.includes('"Lease"') && broken) keys.push(k) + return inner(input, init) + } + const l = start(w, dir, { leaseKeyAttempts: 3 }, fetch) + await until("key abandoned twice", () => l.logs.filter((x) => x.ev === "lease-key-abandoned").length >= 2, 4000).catch(() => { throw new Error(why(w, l)) }) + broken = false + expect(new Set(keys).size).toBeGreaterThan(1) // before round 3: one key, `transient`, for as long as the floor answers so + expect(l.logs.filter((x) => x.ev === "lease-error").every((x) => x.kind === "server-error")).toBe(true) + await until("created", () => w.ax.updates.size > 0, 4000).catch(() => { throw new Error(why(w, l)) }) + await l.drain() + }) + + it("R3-8c (R1-9b scenario) a floor that cannot encode one job: the link stays up and a job enqueued later runs", async () => { + // The good job co-granted with the bad one is only ever sent inside replies the strict floor cannot encode, so no + // link-side change can deliver it (REFUTED for the link; the floor port must encode per grant, round 1 H). + const w = W() + w.floor.enqueue(job("bad", { "timeout-minutes": 1.5 })) + w.floor.enqueue(job("good")) + const l = start(w, dir) + await sleep(500) + w.floor.enqueue(job("later")) + await until("later created", () => w.ax.updates.has("wf-test-later-a1"), 8000).catch(() => { throw new Error(why(w, l)) }) + expect(l.logs.filter((x) => x.ev === "lease-error").every((x) => x.kind === "server-error")).toBe(true) + await l.drain() + }) + + // ---------------------------------------------------------------- R3-9: Complete's success decodes leniently + it("R3-9 a Complete success reply with `withdrew: null` is reported once; the next jobs run", async () => { + const w = W(); w.floor.enqueue(job("1")); w.floor.enqueue(job("2")) + let completes = 0 + const patch = deep((o) => { if ("duplicate" in o && "stats" in o && !("withdrew" in o)) o.withdrew = null }) + const inner = rewriting(w, "Complete", patch) + const fetch: typeof globalThis.fetch = async (input, init) => { if (await rpcOf(input, init) === "Complete") completes++; return inner(input, init) } + const l = start(w, dir, { maxInFlight: 1 }, fetch) + await until("a1", () => w.ax.tasks.has("wf-test-1-a1")) + w.ax.finish("wf-test-1-a1", 0, { ok: true }) + await until("job 2 created", () => w.ax.updates.has("wf-test-2-a1"), 4000).catch(() => { throw new Error(why(w, l)) }) + expect([J(w).state, J(w).result, completes, l.has("outbox-keep")]).toEqual(["done", "success", 1, false]) + await l.drain() + }) + + it("R3-9b a Complete success reply that is not the protocol at all is a server-error: bounded by verdictAttempts, never gates Lease", async () => { + const w = W(); w.floor.enqueue(job("1")); w.floor.enqueue(job("2")) + // the floor applies the verdict; its reply's success value is replaced by a number + const fetch = rewriting(w, "Complete", deep((o) => { if (o.value && typeof o.value === "object" && "duplicate" in (o.value as object)) o.value = 7 })) + const l = start(w, dir, { maxInFlight: 1, verdictAttempts: 2 }, fetch) + await until("a1", () => w.ax.tasks.has("wf-test-1-a1")) + w.ax.finish("wf-test-1-a1", 0, { ok: true }) + await until("job 2 created", () => w.ax.updates.has("wf-test-2-a1"), 4000).catch(() => { throw new Error(why(w, l)) }) + expect(l.logs.filter((x) => x.ev === "outbox-keep").every((x) => x.kind === "server-error")).toBe(true) + await l.drain() + }) + + // ---------------------------------------------------------------- R3-10: B2 (n+1 only after NotFound for n) + it("R3-10 (F8 shape) attempt n+1 is written only once attempt n is absent from ax; while n stays Terminating it is never written", async () => { + const w = W() + const create = w.ax.createTask + const seenAtCreate: Array = [] + w.ax.createTask = (t) => { if (t.metadata.name === "wf-test-1-a2") seenAtCreate.push(...[...w.ax.tasks].map(([n, x]) => `${n}:${x.phase}`)); return create(t) } + w.floor.enqueue(job("1")) + const l1 = start(w, dir) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + await l1.crash() + rmSync(dir, { recursive: true, force: true }) // the journal is lost: a2 is fenced on the Task's own marks + await sleep(7 * SEC); w.floor.sweep() + const tick = w.ax.tick.bind(w.ax) + let hold = true + w.ax.tick = () => { if (hold) { for (const t of w.ax.tasks.values()) if (t.phase === "Pending") t.phase = "Running" } else tick() } + const l2 = start(w, mkdtempSync(join(tmpdir(), "conwip-link-r3-")), { fenceTimeoutMs: 4000 }) + await until("a1 Terminating", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Terminating", 4000).catch(() => { throw new Error(why(w, l2)) }) + await sleep(250) + expect(w.ax.updates.has("wf-test-1-a2")).toBe(false) // the delete was asked, NotFound not yet seen + hold = false + await until("a2 written", () => w.ax.updates.has("wf-test-1-a2"), 6000).catch(() => { throw new Error(why(w, l2)) }) + expect(seenAtCreate.filter((x) => x.startsWith("wf-test-1-a1"))).toEqual([]) + await l2.drain() + }) + + // ---------------------------------------------------------------- R3-11: 429, Retry-After, endpoint and address rows + it("R3-11 a Complete answered 429 Retry-After: 1 keeps the verdict and waits at least Retry-After before the next try", async () => { + const w = W(); w.floor.enqueue(job("1")) + const at: Array = [] + const fetch: typeof globalThis.fetch = async (input, init) => { + if (await rpcOf(input, init) !== "Complete") return w.floor.fetch(input, init) + at.push(Date.now()) + if (at.length === 1) return new Response("slow down", { status: 429, headers: { "retry-after": "1" } }) + return w.floor.fetch(input, init) + } + const l = start(w, dir, {}, fetch) + await until("a1", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.ax.finish("wf-test-1-a1", 0, { v: 1 }) + await until("floor done", () => J(w).state === "done", 5000).catch(() => { throw new Error(why(w, l)) }) + expect([J(w).result, J(w).attempt, l.has("complete-dropped")]).toEqual(["success", 1, false]) + expect(l.logs.some((x) => x.ev === "outbox-keep" && x.kind === "rate-limited")).toBe(true) + expect(at[1]! - at[0]!).toBeGreaterThanOrEqual(950) + await l.drain() + }) + + it("R3-11b an http:// endpoint is ignored even when it is in the declared list; 100.x outside 100.64/10 is not fleet-internal", async () => { + const w = W() + const fetch = rewriting(w, "Lease", deep((o) => { if ("nextPollSeconds" in o) o.endpoint = "http://floor2.test" })) + const switched: Array = [] + const logs: Array> = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 1500, deleteAfterMs: 0, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, floorUrls: ["http://floor.test", "http://floor2.test"] } + const fiber = Effect.runFork(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "tok-nas", sessionId: "s", fetch }) + return yield* runLink(cfg, { ax: w.ax, floor, journal: Journal.open(dir), log: (ev, f) => logs.push({ ev, ...f }), stop, floorUrl: "http://floor.test", onEndpoint: (u) => switched.push(u) }) + }))) + await until("endpoint seen", () => logs.some((x) => x.ev === "endpoint-ignored" || x.ev === "endpoint-accepted"), 3000) + Effect.runSync(Deferred.succeed(stop, undefined)); await Promise.race([Effect.runPromise(Fiber.await(fiber)), sleep(1500)]) + expect([switched, logs.some((x) => x.ev === "endpoint-ignored")]).toEqual([[], true]) + expect(["http://100.1.2.3/c", "http://100.128.0.1/c", "http://100.63.255.255/c", "http://100.64.0.1/c", "http://100.127.255.254/c"].map((u) => isFleetInternalUrl(u))) + .toEqual([false, false, false, true, true]) + }) +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r4-expired.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r4-expired.test.ts new file mode 100644 index 000000000..c2b499f56 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r4-expired.test.ts @@ -0,0 +1,86 @@ +// Round 4 regression test, ported from /home/tom/today/evals-2026-09-23/link/scratch-r4-1/expired.repro.ts (assertions kept; additions marked). +// Round 4, lens fail-closed and capacity gate: a grant that waits at the slot gate (B11: gate closed, e.g. the Gateway +// check failing) is created whenever the gate reopens, with no check that its lease is still live. The floor, which +// the link could not reach (heartbeats failing), has already expired it, requeued it, and may grant attempt 2. +// Buildkite's pattern the design adopts (row 8, "reserve with expiry, finish -1 before start") would refuse it. +import { expect, it } from "vitest" +import { bodyOf, job, sleep, start, until, world } from "./review-r4-harness.ts" + +it("a grant whose lease expired while it waited at the gate is still created", async () => { + const w = world({ leaseSeconds: 6, graceSeconds: 15 }) // at secondMs 20: lease 120 ms, grace 300 ms + w.floor.enqueue(job("1")) + const gw = w.ax.gateways.get("halogen")! + let holdOnce = true + let l: ReturnType + const fetch: typeof globalThis.fetch = async (input, init) => { + const b = await bodyOf(input, init) + const res = await w.floor.fetch(input, init) + if (holdOnce && b.includes('"Lease"')) { + const txt = await res.clone().text() + if (txt.includes("wf-test-1-a1")) { + holdOnce = false + // the Gateway is edited away while the reply is in flight; the next resync closes the gate (B10) + w.ax.gateways.delete("halogen") + await until("gateway-missing", () => l.has("gateway-missing"), 3000) + } + } + return res + } + l = start(w, undefined, { maxInFlight: 1 }, fetch) + await until("waiting-for-slot", () => l.has("waiting-for-slot", "wf-test-1-a1"), 3000) + // the NAS loses its route to Cloudflare; ax stays reachable + w.floor.down = true + const J = w.floor.jobs.get("wf-test-1")! + await until("floor requeued a1 after lease + grace", () => J.state === "queued" && J.attempt === 2, 3000) + const floorStateAtReopen = { state: J.state, attempt: J.attempt, history: [...J.history] } + w.ax.gateways.set("halogen", gw) // the Gateway is back; the floor is still unreachable + await until("a1 created", () => (w.ax.updates.get("wf-test-1-a1") ?? 0) > 0, 3000).catch(() => {}) + const createdA1WhileExpired = (w.ax.updates.get("wf-test-1-a1") ?? 0) > 0 + await sleep(200) + const a1BeforeFloorReturned = w.ax.updates.get("wf-test-1-a1") ?? 0 // round 4 addition + w.floor.down = false + await until("a2 created", () => w.ax.tasks.has("wf-test-1-a2"), 3000).catch(() => {}) + await sleep(300) + const out = { + floorStateAtReopen, createdA1WhileExpired, axEvents: w.ax.events, + linkTrace: l!.logs.filter((x) => typeof x.leaseId === "string" || typeof x.task === "string").map((x) => `${x.ev}:${x.leaseId ?? x.task}${x.why ? ":" + x.why : ""}`), + heartbeatErrors: l!.logs.filter((x) => x.ev === "heartbeat-error").length, floorHistory: J.history + } + console.log(JSON.stringify(out)) + await l!.drain(); w.floor.close() + expect(createdA1WhileExpired).toBe(false) // fail closed: a lease known to be past renewTime + duration is not started + expect(a1BeforeFloorReturned).toBe(0) // round 4 addition: nothing of a1 before the floor returns + expect([...w.ax.updates.keys()]).toEqual(["wf-test-1-a2"]) // exactly one created attempt, a2, afterwards + expect(l!.has("lease-unconfirmed", "wf-test-1-a1")).toBe(true) +}) + +// Round 4 addition (finding 5, deadline half): a grant whose deadline has passed is refused with no create. The gate +// closes while the grant is in flight and the Heartbeats fail, so the floor cannot cancel at its deadline (the lease is +// orphaned, grace is long); only the link's own check can end it. +it("R4-5b: a grant past its deadline is failed deadline-exceeded before any UpdateTask", async () => { + const w = world({ graceSeconds: 400 }) + w.floor.enqueue(job("d", { "timeout-minutes": 1 })) // 60 s x 20 ms = 1.2 s deadline + const gw = w.ax.gateways.get("halogen")! + let hbDown = false + let l: ReturnType + const fetch: typeof globalThis.fetch = async (input, init) => { + const b = await bodyOf(input, init) + if (hbDown && b.includes('"Heartbeat"')) throw new TypeError("fetch failed") + const res = await w.floor.fetch(input, init) + if (!hbDown && b.includes('"Lease"') && (await res.clone().text()).includes("wf-test-d-a1")) { + hbDown = true + w.ax.gateways.delete("halogen") + await until("gateway-missing", () => l.has("gateway-missing"), 3000) + } + return res + } + l = start(w, undefined, { maxInFlight: 1 }, fetch) + await until("waiting", () => l.has("waiting-for-slot", "wf-test-d-a1") || l.has("lease-unconfirmed", "wf-test-d-a1"), 3000) + await sleep(1400) + w.ax.gateways.set("halogen", gw) + await until("verdict", () => l.has("verdict", "wf-test-d-a1"), 3000) + const v = l.logs.find((x) => x.ev === "verdict" && x.leaseId === "wf-test-d-a1")! + await l.drain(); w.floor.close() + expect(v.reason).toBe("deadline-exceeded") + expect(w.ax.updates.get("wf-test-d-a1") ?? 0).toBe(0) +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r4-fiber-defect.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r4-fiber-defect.test.ts new file mode 100644 index 000000000..f68fcaa37 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r4-fiber-defect.test.ts @@ -0,0 +1,96 @@ +// Round 4 regression test, ported from /home/tom/today/evals-2026-09-23/link/scratch-r4-2/fiber-defect.repro.ts (assertions kept; additions marked). +// Effect idiom: dispatch ends in `Effect.catchCause(... log("job-fiber-died"))`, so a defect in one job fiber is logged +// and swallowed while the process lives on. Here one journal write fails once (ENOSPC, then the disk recovers). The +// grant stays in `held()`, every Heartbeat renews it on the floor, nothing re-dispatches it, and nothing reports it: +// the job never runs and never ends, and it holds a floor WIP slot and a local slot until the process restarts. +import { mkdtempSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import { Deferred, Effect, Fiber } from "effect" +import { expect, it } from "vitest" +import { rpcFloor } from "../src/floor.ts" +import { Journal } from "../src/journal.ts" +import { runLink } from "../src/link.ts" +import type { LinkConfig } from "../src/link.ts" +import { job, sleep, until, world, SEC } from "./review-r4-harness.ts" + +it("one failed journal write in a dispatch fiber strands the lease for the life of the process", async () => { + const w = world({ cap: 1 }) + w.floor.enqueue(job("1")) + w.floor.enqueue(job("2")) + const dir = mkdtempSync(join(tmpdir(), "r4-2-defect-")) + const j = Journal.open(dir) + const orig = j.append.bind(j) + let failed = 0 + ;(j as any).append = (e: any, t?: number) => { + if (e.ev === "creating" && failed === 0) { failed++; throw Object.assign(new Error("ENOSPC: no space left on device, write"), { code: "ENOSPC" }) } + return orig(e, t) + } + const logs: Array = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { + holder: "nas-link-1", maxInFlight: 1, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 1500, deleteAfterMs: 0, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, fenceTimeoutMs: 300, resultReadTries: 3 + } + const fiber = Effect.runFork(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "tok-nas", sessionId: "s:defect", fetch: w.floor.fetch }) + return yield* runLink(cfg, { ax: w.ax, floor, journal: j, log: (ev, f) => logs.push({ ev, ...f }), stop, floorUrl: "http://floor.test" }) + }))) + try { + await until("fiber died", () => logs.some((x) => x.ev === "job-fiber-died")) + await sleep(3000) // 25 lease durations (6 s x 20 ms); the disk is fine again + const J1 = w.floor.jobs.get("wf-test-1")!, J2 = w.floor.jobs.get("wf-test-2")! + const out = { + linkAlive: fiber.pollUnsafe() === undefined, failedWrites: failed, + job1: { state: J1.state, attempt: J1.attempt, history: J1.history, task: w.ax.tasks.get("wf-test-1-a1")?.phase ?? "absent" }, + job2: { state: J2.state, history: J2.history }, + held: j.held(), heartbeats: w.floor.calls.filter((c) => c.rpc === "Heartbeat").length, + completes: w.floor.calls.filter((c) => c.rpc === "Complete").length, + events: logs.map((x) => x.ev).filter((e, i, a) => a.indexOf(e) === i) + } + console.log(JSON.stringify(out)) + // Expected: the lease is failed back (retryable) or the process crashes and systemd restarts it; either frees job 1. + expect(J1.state === "done" || J1.attempt > 1 || w.ax.tasks.has("wf-test-1-a1") || !out.linkAlive).toBe(true) + } finally { + Effect.runSync(Deferred.succeed(stop, undefined)) + await Promise.race([Effect.runPromise(Fiber.await(fiber)), sleep(1500)]) + w.floor.close() + } +}) + +// Round 4 addition (finding 7): when even the fail-back cannot be journaled, the process dies (exit 1 under systemd) +// instead of renewing a lease nothing will ever run. +it("R4-7b: a dispatch defect whose fail-back also fails ends the link", async () => { + const w = world({ cap: 1 }) + w.floor.enqueue(job("1")) + const dir = mkdtempSync(join(tmpdir(), "r4-fix-defect-")) + const j = Journal.open(dir) + const orig = j.append.bind(j) + ;(j as any).append = (e: any, t?: number) => { + if (e.ev === "creating" || e.ev === "report") throw Object.assign(new Error("ENOSPC: no space left on device, write"), { code: "ENOSPC" }) + return orig(e, t) + } + const logs: Array = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { + holder: "nas-link-1", maxInFlight: 1, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 1500, deleteAfterMs: 0, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, fenceTimeoutMs: 300, resultReadTries: 3 + } + const fiber = Effect.runFork(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "tok-nas", sessionId: "s:defect2", fetch: w.floor.fetch }) + return yield* runLink(cfg, { ax: w.ax, floor, journal: j, log: (ev, f) => logs.push({ ev, ...f }), stop, floorUrl: "http://floor.test" }) + }))) + try { + const exit = await Promise.race([Effect.runPromise(Fiber.await(fiber)), sleep(3000).then(() => undefined)]) + expect(exit).toBeDefined() + expect(exit!._tag).toBe("Failure") + expect(logs.some((x) => x.ev === "link-defect-fatal")).toBe(true) + } finally { + Effect.runSync(Deferred.succeed(stop, undefined)) + w.floor.close() + } +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r4-harness.ts b/pkgs/substrate-link/src/apps/link/test/review-r4-harness.ts new file mode 100644 index 000000000..94e35fc0b --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r4-harness.ts @@ -0,0 +1,59 @@ +// Round 4 fix pass: the reviewers' scratch harness (scratch-r4-0/1/2, identical but for the temp prefix), moved into +// the worktree. The only change: the job carries the FIELD-MAP 5a journal-key annotation jobs.ts now requires. +import { mkdtempSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import { Deferred, Effect, Fiber } from "effect" +import type { AgentJob } from "../src/contract.ts" +import { rpcFloor } from "../src/floor.ts" +import { Journal } from "../src/journal.ts" +import { runLink } from "../src/link.ts" +import type { LinkConfig } from "../src/link.ts" +import { FakeAx } from "./fake-ax.ts" +import { FakeFloor } from "./fake-floor.ts" +import type { FloorConfig } from "./fake-floor.ts" +export const SEC = 20 +const A = "ultracode.mecattaf.dev/" +export const job = (n: string, spec: Record = {}): AgentJob => ({ + apiVersion: "ultracode.mecattaf.dev/v1alpha1", kind: "AgentJob", + metadata: { name: `wf-test-${n}`, labels: { [A + "run-id"]: "wf-test", [A + "workflow"]: "link-test", [A + "phase-index"]: "1" }, + annotations: { [A + "run-id-raw"]: "wf_test", [A + "label"]: `probe:${n}`, [A + "item-key"]: `wf_test#${n}`, [A + "journal-key"]: `${"cd".repeat(32)}:1`, [A + "phase-title"]: "Probe" } }, + spec: { "runs-on": ["seat:halogen", "runtime:gvisor"], + with: { prompt: `say ${n}`, prompt_ref: { sha256: "ab".repeat(32), bytes: 5, uri: `journal://wf_test/${n}/prompt.md` }, model: "halogen-qwen3.8-flash-next" }, ...spec } as any +}) +export const world = (o: Partial = {}) => ({ + floor: new FakeFloor({ cap: 2, leaseSeconds: 6, graceSeconds: 15, pollSeconds: 1, heartbeatSeconds: 2, maxAttempts: 3, secondMs: SEC, tokens: { "tok-nas": "nas-link-1" }, ...o }), + ax: new FakeAx() +}) +export type World = ReturnType +export function start(w: World, dir = mkdtempSync(join(tmpdir(), "r4-fix-")), over: Partial = {}, fetch?: typeof globalThis.fetch) { + const logs: Array> = [] + const stop = Effect.runSync(Deferred.make()) + const cfg: LinkConfig = { + holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "auto", secondMs: SEC, resyncMs: 30, pendingTimeoutMs: 1500, deleteAfterMs: 0, deadlineBackstopMs: 300, + createAttempts: 3, outboxBackoffMs: [20, 100], initialPollSeconds: 1, initialHeartbeatSeconds: 2, fenceTimeoutMs: 300, resultReadTries: 3, ...over + } + const fiber = Effect.runFork(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: "http://floor.test", token: "tok-nas", sessionId: `s:${dir}`, fetch: fetch ?? w.floor.fetch }) + return yield* runLink(cfg, { ax: w.ax, floor, journal: Journal.open(dir), log: (ev, f) => logs.push({ t: Date.now(), ev, ...f }), stop, floorUrl: "http://floor.test" }) + }))) + return { + dir, logs, fiber, + has: (ev: string, leaseId?: string) => logs.some((l) => l.ev === ev && (leaseId === undefined || l.leaseId === leaseId)), + drain: () => { Effect.runSync(Deferred.succeed(stop, undefined)); return Promise.race([Effect.runPromise(Fiber.await(fiber)), sleep(1500)]) } + } +} +export const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms)) +export const until = async (what: string, pred: () => boolean, ms = 5000) => { + const t0 = Date.now() + while (!pred()) { if (Date.now() - t0 > ms) throw new Error(`timeout waiting for: ${what}`); await new Promise((r) => setTimeout(r, 5)) } +} +export const bodyOf = async (input: unknown, init?: RequestInit) => { + const b = init?.body + if (typeof b === "string") return b + if (b instanceof Uint8Array) return new TextDecoder().decode(b) + if (b) return await new Response(b as any).text() + return (input instanceof Request) ? await input.clone().text() : "" +} diff --git a/pkgs/substrate-link/src/apps/link/test/review-r4-heartbeat-envelope.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r4-heartbeat-envelope.test.ts new file mode 100644 index 000000000..81b5de396 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r4-heartbeat-envelope.test.ts @@ -0,0 +1,134 @@ +// Round 4 regression test, ported from /home/tom/today/evals-2026-09-23/link/scratch-r4-2/heartbeat-envelope.repro.ts (assertions kept; additions marked). +// Schema compatibility: the round-3 fix made the Lease and Complete success envelopes lenient ("a changed stats shape +// would otherwise fail the decode"), but Heartbeat still decodes with the strict L1 schema. The same drift on the +// Heartbeat reply (here `stats: null`, which Lease tolerates) makes every heartbeat a transient error: held leases are +// never renewed, the floor requeues a healthy running attempt as a2, and the link fences a1 and runs the job again. +import { expect, it } from "vitest" +import { job, start, until, world, bodyOf, sleep } from "./review-r4-harness.ts" + +const drift = (rpc: string, floorFetch: typeof globalThis.fetch): typeof globalThis.fetch => async (input, init) => { + const body = await bodyOf(input, init) + const res = await floorFetch(input, init) + if (!body.includes(`"${rpc}"`)) return res + const text = await res.text() + let m: any + try { m = JSON.parse(text) } catch { return new Response(text, { status: res.status, headers: res.headers }) } + const walk = (x: any) => { if (x && typeof x === "object") { if ("stats" in x && "seq" in (x.stats ?? {})) x.stats = null; for (const v of Object.values(x)) walk(v) } } + walk(m) + return new Response(JSON.stringify(m), { status: res.status, headers: res.headers }) +} + +for (const rpc of ["Lease", "Heartbeat"]) { + it(`stats: null on every ${rpc} reply`, async () => { + const w = world() + w.floor.enqueue(job("1")) + const l = start(w, undefined, {}, drift(rpc, w.floor.fetch)) + try { + await until("a1 running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + await sleep(1500) // about 12 lease durations (6 s x 20 ms) with a1 healthy and running + const J = w.floor.jobs.get("wf-test-1")! + const hbErr = l.logs.filter((x) => x.ev === "heartbeat-error").length + const out = { rpc, floorState: J.state, attempt: J.attempt, history: J.history, heartbeatErrors: hbErr, + a1: w.ax.tasks.get("wf-test-1-a1")?.phase ?? "deleted", a2: w.ax.tasks.get("wf-test-1-a2")?.phase ?? "absent", + events: l.logs.map((x) => x.ev).filter((e, i, a) => a.indexOf(e) === i) } + console.log(JSON.stringify(out)) + // The expected behaviour, as for the Lease drift: a healthy a1 keeps its lease and nothing reruns. + expect(J.attempt).toBe(1) + expect(w.ax.tasks.get("wf-test-1-a2")).toBeUndefined() + } finally { await l.drain(); w.floor.close() } + }) +} + +// The consequence: the floor renews on its side, but the link never reads `cancelRequested`, `lost` or `renewed`. +for (const rpc of ["Lease", "Heartbeat"]) { + it(`cancel with stats: null on every ${rpc} reply`, async () => { + const w = world() + w.floor.enqueue(job("1")) + const l = start(w, undefined, {}, drift(rpc, w.floor.fetch)) + try { + await until("a1 running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.cancel("wf-test-1") + await sleep(1500) + const J = w.floor.jobs.get("wf-test-1")! + const out = { rpc, floorState: J.state, result: J.result ?? null, cancel: J.cancel ?? null, history: J.history, + a1: w.ax.tasks.get("wf-test-1-a1")?.phase ?? "deleted", cancelLogged: l.has("cancel") } + console.log(JSON.stringify(out)) + expect(J.state).toBe("done") + expect(J.result).toBe("cancelled") + } finally { await l.drain(); w.floor.close() } + }) +} + +// B13: a grant journaled by a previous process and never created is resumed only once a Heartbeat reply renews it. +for (const rpc of ["Lease", "Heartbeat"]) { + it(`restart resume with stats: null on every ${rpc} reply`, async () => { + const { Effect } = await import("effect") + const w = world() + w.floor.enqueue(job("1")) + const orig = w.ax.createTask + ;(w.ax as any).createTask = () => Effect.never // first process: UpdateTask never answers, then the process stops + const l1 = start(w) + const dir = l1.dir + await until("a1 creating", () => l1.has("leased")) + await sleep(100) + await l1.drain() + ;(w.ax as any).createTask = orig + const J = w.floor.jobs.get("wf-test-1")! + const l2 = start(w, dir, {}, drift(rpc, w.floor.fetch)) + try { + await sleep(2000) + const out = { rpc, floorState: J.state, attempt: J.attempt, history: J.history, result: J.result ?? null, + a1: w.ax.tasks.get("wf-test-1-a1")?.phase ?? "absent", resumed: l2.has("resume"), + heartbeatErrors: l2.logs.filter((x) => x.ev === "heartbeat-error").length, + events: l2.logs.map((x) => x.ev).filter((e, i, a) => a.indexOf(e) === i) } + console.log(JSON.stringify(out)) + expect(J.state === "done" || w.ax.tasks.get("wf-test-1-a1") !== undefined).toBe(true) + } finally { await l2.drain(); w.floor.close() } + }) +} + +// Round 4 addition (finding 6): a Heartbeat reply that omits `cancelRequested` (or has it null) is read for the rest. +const omit = (key: string, value: "omit" | null, floorFetch: typeof globalThis.fetch): typeof globalThis.fetch => async (input, init) => { + const body = await bodyOf(input, init) + const res = await floorFetch(input, init) + if (!body.includes('"Heartbeat"')) return res + const text = await res.text() + let m: any + try { m = JSON.parse(text) } catch { return new Response(text, { status: res.status, headers: res.headers }) } + const walk = (x: any) => { if (x && typeof x === "object") { if (key in x && "renewed" in x) { if (value === "omit") delete x[key]; else x[key] = value } for (const v of Object.values(x)) walk(v) } } + walk(m) + return new Response(JSON.stringify(m), { status: res.status, headers: res.headers }) +} +for (const value of ["omit", null] as const) { + it(`R4-6b: restart resume with cancelRequested ${value === "omit" ? "omitted" : "null"} on every Heartbeat reply`, async () => { + const { Effect } = await import("effect") + const w = world() + w.floor.enqueue(job("1")) + const orig = w.ax.createTask + ;(w.ax as any).createTask = () => Effect.never + const l1 = start(w) + await until("a1 creating", () => l1.has("leased")) + await sleep(100) + await l1.drain() + ;(w.ax as any).createTask = orig + const l2 = start(w, l1.dir, {}, omit("cancelRequested", value, w.floor.fetch)) + try { + await until("a1 resumed and created", () => w.ax.tasks.get("wf-test-1-a1") !== undefined, 3000) + expect(l2.has("resume", "wf-test-1-a1")).toBe(true) + expect(l2.logs.filter((x) => x.ev === "heartbeat-error").length).toBe(0) + } finally { await l2.drain(); w.floor.close() } + }) +} +it("R4-6c: a lost lease is still acted on when the Heartbeat reply omits stats and cancelRequested", async () => { + const w = world() + w.floor.enqueue(job("1")) + const both: typeof globalThis.fetch = omit("cancelRequested", "omit", drift("Heartbeat", w.floor.fetch)) + const l = start(w, undefined, {}, both) + try { + await until("a1 running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + const J = w.floor.jobs.get("wf-test-1")! + J.holder = "someone-else" // the floor answers lost:[a1] from now on + await until("lost", () => l.has("lost", "wf-test-1-a1"), 2000) + await until("a1 deleted", () => w.ax.deletes.includes("wf-test-1-a1"), 2000) + } finally { await l.drain(); w.floor.close() } +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r4-journal-key.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r4-journal-key.test.ts new file mode 100644 index 000000000..4f39aed63 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r4-journal-key.test.ts @@ -0,0 +1,48 @@ +// Round 4 regression test for finding 8 (repro: /home/tom/today/evals-2026-09-23/link/scratch-r4-2/journal-key.repro.ts). +// AX_CONWIP_JOURNAL_KEY must be the FIELD-MAP 5a journal key ("sha256 of canonical [prompt, opts minus label] plus the +// occurrence"), carried on the AgentJob as the `ultracode.mecattaf.dev/journal-key` annotation, never the prompt +// digest. The expected values are pinned from the ax-conwip taskspec (journalKey in +// /home/tom/mecattaf/ax-conwip-wt-integration/src/taskspec.ts at b42b310, run 2026-09-23), so this test does not +// import another worktree. +import { createHash } from "node:crypto" +import { expect, it } from "vitest" +import { axTaskFromGrant } from "../src/jobs.ts" +import { job } from "./review-r4-harness.ts" + +const TASKSPEC = { + halogen1: "29a9c6cb8956c07073345109161d0e7eaa08708f4c2039f4e4af7a09d212ee07:1", // journalKey("same prompt", {model: halogen}, 1) + opus1: "bd381a2d66f7a1d93c7df296954218de4c9749a0feb9d94719b9c484f8c06718:1", // journalKey("same prompt", {model: opus}, 1) + halogen2: "29a9c6cb8956c07073345109161d0e7eaa08708f4c2039f4e4af7a09d212ee07:2" // the same call, second occurrence +} +const sha = (s: string) => createHash("sha256").update(s).digest("hex") +const A = "ultracode.mecattaf.dev/" +const grant = (n: string, model: string, key: string | undefined) => { + const j: any = job(n) + j.spec.with.prompt = "same prompt" + j.spec.with.prompt_ref = { sha256: sha("same prompt"), bytes: 11, uri: `journal://wf_test/${n}/prompt.md` } + j.spec.with.model = model + if (key === undefined) delete j.metadata.annotations[A + "journal-key"] + else j.metadata.annotations[A + "journal-key"] = key + return { leaseId: `wf-test-${n}-a1`, attempt: 1, job: j, lease: { holderIdentity: "h", leaseDurationSeconds: 90, acquireTime: 1, renewTime: 1, leaseTransitions: 0 } } +} +const shape = { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] } +const env = (g: any) => { const b: any = axTaskFromGrant(g, shape); expect(b._tag).toBe("ok"); return Object.fromEntries(b.task.spec.env.map((e: any) => [e.name, e.value])) } + +it("R4-8a: AX_CONWIP_JOURNAL_KEY is the taskspec journal key for two calls that share a prompt", () => { + const a = env(grant("1", "halogen-qwen3.8-flash-next", TASKSPEC.halogen1)) + const b = env(grant("2", "claude-opus-5-5", TASKSPEC.opus1)) + const c = env(grant("3", "halogen-qwen3.8-flash-next", TASKSPEC.halogen2)) + expect(a.AX_CONWIP_JOURNAL_KEY).toBe(TASKSPEC.halogen1) + expect(b.AX_CONWIP_JOURNAL_KEY).toBe(TASKSPEC.opus1) + expect(c.AX_CONWIP_JOURNAL_KEY).toBe(TASKSPEC.halogen2) + expect(a.AX_CONWIP_JOURNAL_KEY).not.toBe(a.AX_CONWIP_PROMPT_SHA256) + expect(a.AX_CONWIP_PROMPT_SHA256).toBe(b.AX_CONWIP_PROMPT_SHA256) // the prompt digest is shared; the journal key is not +}) + +it("R4-8b: a grant without the annotation, or with a malformed one, is refused pre-start/invalid-spec", () => { + for (const key of [undefined, "", sha("same prompt"), `${sha("x")}:`, `${sha("x").toUpperCase()}:1`]) { + const b = axTaskFromGrant(grant("4", "halogen-qwen3.8-flash-next", key) as any, shape) + expect(b._tag).toBe("refused") + if (b._tag === "refused") expect(b.reason).toBe("pre-start/invalid-spec") + } +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r4-overtake.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r4-overtake.test.ts new file mode 100644 index 000000000..e1274ca32 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r4-overtake.test.ts @@ -0,0 +1,78 @@ +// Round 4 regression test, ported from /home/tom/today/evals-2026-09-23/link/scratch-r4-0/overtake.repro.ts (assertions kept; additions marked). +// Reply ordering: the regrant of a2 (gen 2) overtakes the Complete reply that withdrew a2 (gen 1). Set-up is R2-5a +// (review-r2.test.ts): a1 fails retryable while a2 gen 1 is leased to this link and waits for a1's verdict. a1's first +// Complete answers 500 once (server-error), which by design stops gating Lease (round 2), so a Lease with capacity 1 is +// in flight while the retried Complete is applied. The floor withdraws gen 1, requeues gen 2 and grants it on that +// Lease; the Lease reply lands before the Complete reply. The link still holds gen 1 unreleased, so gen 2 is dropped as +// `duplicate-grant`; the Complete reply then releases gen 1. Nothing of a2 is ever created, although the floor has +// gen 2 leased to this link: the job waits for lease expiry and is released `omitted` (an infra release). +import { expect, it } from "vitest" +import { bodyOf, job, sleep, start, until, world } from "./review-r4-harness.ts" +import type { World } from "./review-r4-harness.ts" + +const rpcOf = async (input: unknown, init?: RequestInit) => { const b = await bodyOf(input, init); return b.includes('"Complete"') ? "Complete" : b.includes('"Lease"') ? "Lease" : b.includes('"Heartbeat"') ? "Heartbeat" : "?" } +const failRetryable = (w: World, name: string) => { + const t = w.ax.tasks.get(name)! + t.exited = true; t.phase = "Failed" + t.conditions = [{ type: "Ready", status: "False", reason: "ResourceExhausted", message: "ResourceExhausted: no free worker" }] +} + +// Round 4 addition: also against a floor that omits withdrewTransitions (the link then uses the generation of a2 it +// held when it SENT a1's verdict, not the regrant it holds when the reply lands). +for (const sendWithdrewGen of [true, false]) +it(`the regrant of a2 is created even when its Lease reply lands before the Complete reply that withdrew gen 1 (withdrewTransitions ${sendWithdrewGen ? "sent" : "omitted"})`, async () => { + const w = world({ sendWithdrewGen }); w.floor.enqueue(job("1")) + const J = () => w.floor.jobs.get("wf-test-1")! + let holdLease = false, failHeartbeat = false, failComplete = false, fiveHundredOnce = false + let l: ReturnType + let armed = false, leaseHeld = false + const fetch: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + const rpc = await rpcOf(input, init) + if (rpc === "Heartbeat" && (failHeartbeat || armed)) throw new TypeError("fetch failed") // no Heartbeat in the window + if (rpc === "Complete" && failComplete) return new Response("upstream timeout", { status: 503 }) + if (rpc === "Complete" && fiveHundredOnce) { fiveHundredOnce = false; armed = true; return new Response("internal error", { status: 500 }) } + if (rpc === "Lease") while (holdLease) await sleep(2) + if (rpc === "Lease" && armed && /"capacity":[1-9]/.test(body) && !leaseHeld) { + leaseHeld = true // a Lease sent before the retried Complete, delivered after it (network reordering) + await until("the retried Complete was applied", () => J().history.includes("withdrawn:a2"), 3000).catch(() => {}) + // Round 4 addition: longer than the dispatch wait period (at most 2 x SEC), so the gen-1 dispatch is inside, or + // waiting on, the outbox lock when the Complete reply lands, and reaches its withdrawn branch deterministically + await sleep(80) + } + const res = await w.floor.fetch(input as any, { ...init, body }) + if (rpc === "Complete" && armed) { + await until("the Lease reply reached the link", () => l.has("duplicate-grant", "wf-test-1-a2") || l.has("regrant", "wf-test-1-a2"), 1500).catch(() => {}) + armed = false + } + return res + } + l = start(w, undefined, {}, fetch) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + await until("floor requeued attempt 2", () => J().state === "queued" && J().attempt === 2, 4000) + failHeartbeat = true; holdLease = true; failComplete = true + w.floor.down = false + await sleep(80) + failRetryable(w, "wf-test-1-a1") + await until("verdict written", () => l.has("verdict", "wf-test-1-a1")) + holdLease = false + await until("a2 received", () => l.has("leased", "wf-test-1-a2"), 4000) + await until("dispatch waits for the verdict", () => l.has("supersede-waits-for-verdict", "wf-test-1-a2")) + fiveHundredOnce = true + failComplete = false; failHeartbeat = false + await until("a2 created, or 1.5 s", () => w.ax.updates.has("wf-test-1-a2"), 1500).catch(() => {}) + await sleep(300) + const trace = l.logs.filter((x) => x.leaseId === "wf-test-1-a2" || x.task === "wf-test-1-a2" || x.ev === "complete" || x.ev === "outbox-keep" || x.ev === "lease-gated-by-outbox").map((x) => `${x.ev}:${x.leaseId ?? x.task}${x.why ? ":" + x.why : ""}${x.withdrew ? ":withdrew=" + x.withdrew : ""}`) + const out = { leaseHeld, floor: J().history, state: J().state, attempt: J().attempt, infraSpent: J().infraSpent, a2Updates: w.ax.updates.get("wf-test-1-a2") ?? 0, trace } + console.log(JSON.stringify(out)) + if (out.a2Updates !== 1 || out.floor.includes("requeue:omitted")) console.log("FULL", JSON.stringify(l.logs.map((x) => ({ ...x, cause: undefined }))), JSON.stringify(w.floor.calls.slice(-40))) + await l.drain(); w.floor.close() + expect(out.a2Updates).toBe(1) + expect(out.floor).not.toContain("requeue:omitted") // round 4 addition: no infra release spent on the stall + expect(out.leaseHeld).toBe(true) + // Round 4 addition: the gen-1 dispatch never releases the regrant (the withdrawn branch is bound to its own grant) + const a2 = l.logs.filter((x) => x.leaseId === "wf-test-1-a2").map((x) => String(x.ev)) + expect(a2.slice(a2.indexOf("regrant"))).not.toContain("withdrawn") + expect(w.ax.deletes).not.toContain("wf-test-1-a2") +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r4-p1stale.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r4-p1stale.test.ts new file mode 100644 index 000000000..17e776007 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r4-p1stale.test.ts @@ -0,0 +1,74 @@ +// Round 4 regression test, ported from /home/tom/today/evals-2026-09-23/link/scratch-r4-1/p1stale.repro.ts (assertions kept; additions marked). +// Round 4, lens capacity gate: canCreate() needs "P1 proven", but the probe runs only on an ax-up transition +// (link.ts resync: `if (wasDown) serverP1 = undefined`). ax-server is a replicas:1 Deployment with the default +// RollingUpdate strategy (upstream deploy/ax-server.yaml), so a rollout to a server without P1 happens with no failed +// ListTasks, and the link keeps creating Tasks it can never read a verdict from. +import { Effect } from "effect" +import { expect, it } from "vitest" +import { job, sleep, start, until, world } from "./review-r4-harness.ts" + +it("A: ax rolled to stock v0.3.0 (server and controller) with no ListTasks failure", async () => { + const w = world() + w.floor.enqueue(job("1")) + const l = start(w) + await until("a1 running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + const probesBefore = l.logs.filter((x) => x.ev === "completion-probe").length + w.ax.p1 = false // the rollout: GetTaskResult is now UNIMPLEMENTED and no terminal phase is ever written + await until("p1-missing logged", () => l.has("p1-missing"), 1000) // round 4 fix: the probe runs on every resync + const leasesAtFlip = w.floor.calls.filter((c) => c.rpc === "Lease").length + await sleep(100) // several resyncs (30 ms) and Lease polls pass; none fails + w.floor.enqueue(job("2")) + await sleep(1500) // round 4: job 2 is never created on the stock server (the repro waited for its create here) + const J2 = w.floor.jobs.get("wf-test-2")! + const out = { + probesBefore, probesAfter: l.logs.filter((x) => x.ev === "completion-probe").length, + p1MissingLogged: l.has("p1-missing"), + job2: { state: J2.state, attempt: J2.attempt, history: J2.history }, + task2Phase: w.ax.tasks.get("wf-test-2-a1")?.phase, task2Exited: (w.ax.tasks.get("wf-test-2-a1") as any)?.exited, + verdictJob2: l.has("verdict", "wf-test-2-a1"), + leaseCapacitiesAfterRollout: [...new Set(w.floor.calls.filter((c) => c.rpc === "Lease").slice(-10).map((c) => c.capacity))] + } + console.log("A", JSON.stringify(out)) + const leasesAfter = w.floor.calls.filter((c) => c.rpc === "Lease").slice(leasesAtFlip + 1) // round 4 addition + await l.drain(); w.floor.close() + // Fail-closed expectation: after the P1 route disappears the link leases and creates nothing new. + expect(w.ax.updates.get("wf-test-2-a1") ?? 0).toBe(0) + expect(out.p1MissingLogged).toBe(true) + expect(leasesAfter.length).toBeGreaterThan(0) + expect(leasesAfter.every((c) => c.capacity === 0)).toBe(true) + expect(J2.state).toBe("queued") +}) + +it("B: ax-server rolled to stock, the P1 controller still writes Completed", async () => { + const w = world() + w.floor.enqueue(job("1")) + const l = start(w) + await until("a1 running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + let stockServer = false + const orig = w.ax.getTaskResult + ;(w.ax as any).getTaskResult = (n: string) => stockServer ? Effect.succeed("unimplemented" as const) : orig(n) + stockServer = true + w.ax.finish("wf-test-1-a1", 0, { answer: 1 }) // Completed, result stored by the controller, unreadable via ax-server + for (let i = 0; i < 40; i++) { // every attempt the floor grants finishes with a success the link cannot read + for (const [name, t] of w.ax.tasks) if (t.phase === "Running" && !t.exited) w.ax.finish(name, 0, { answer: name }) + await sleep(50) + } + const J = w.floor.jobs.get("wf-test-1")! + const out = { + probes: l.logs.filter((x) => x.ev === "completion-probe").length, p1MissingLogged: l.has("p1-missing"), + creates: [...w.ax.updates.keys()], job1: { state: J.state, result: (J as any).result, output: (J as any).output, history: J.history }, + verdicts: l.logs.filter((x) => x.ev === "verdict").map((x) => `${x.leaseId}:${x.result}:${x.reason ?? ""}`) + } + console.log("B", JSON.stringify(out)) + // Round 4 addition: the Task is kept, uncharged, and read once P1 is back: the success is reported, not thrown away + stockServer = false + await until("job 1 done", () => J.state === "done", 3000) + const leasesUnimpl = w.floor.calls.filter((c) => c.rpc === "Lease") + await l.drain(); w.floor.close() + expect(out.creates.length).toBe(1) // fail closed: nothing created once the server answered UNIMPLEMENTED + expect(out.p1MissingLogged).toBe(true) + expect(out.verdicts.some((v) => v.includes("result-unreadable"))).toBe(false) + expect((J as any).result).toBe("success") + expect((J as any).output).toEqual({ answer: 1 }) + expect(leasesUnimpl.length).toBeGreaterThan(0) +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r4-restart-replay.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r4-restart-replay.test.ts new file mode 100644 index 000000000..9307a75a5 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r4-restart-replay.test.ts @@ -0,0 +1,143 @@ +// Round 4 regression test, ported from /home/tom/today/evals-2026-09-23/link/scratch-r4-0/restart-replay.repro.ts (assertions kept; additions marked). +// Kill -9 mid-Lease (B8): the floor applied the Lease and granted a1, the link journaled the requestKey but died +// before the reply was journaled. The restart takes longer than the lease (orphaned) and less than lease + grace. +// B8 says the journaled key is replayed so the grant is recovered. At startup the link sends its first Heartbeat +// (held set without a1) BEFORE it replays the key, so the floor releases the omitted orphan (rule 7 / L3 a), and the +// replay then returns nothing: one infra release is spent and the job reruns as a2. +import { readFileSync } from "node:fs" +import { Effect, Fiber } from "effect" +import { expect, it } from "vitest" +import { job, SEC, sleep, start, until, world, bodyOf } from "./review-r4-harness.ts" + +it("a journaled requestKey whose grant outlived the lease is recovered on restart", async () => { + const w = world() + w.floor.enqueue(job("r1")) + // link #1: the first capacity>0 Lease reaches the floor, then the process "dies" before the reply lands + let leased = false + const dying: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + if (/"Lease"/.test(body) && /"capacity":[1-9]/.test(body)) { + await w.floor.fetch(input as any, { ...init, body }) + leased = true + return new Promise(() => {}) // the reply never reaches the journal + } + return w.floor.fetch(input as any, { ...init, body }) + } + const l1 = start(w, undefined, {}, dying) + await until("floor granted a1", () => leased) + Effect.runFork(Fiber.interrupt(l1.fiber)) // kill -9: nothing after this point runs in link #1 + await sleep(50) + const journal1 = readFileSync(l1.dir + "/journal.jsonl", "utf8") + const keyJournaled = /"ev":"lease-key"/.test(journal1) && !/"ev":"lease-replied"/.test(journal1) + // downtime: past the lease (6 s * SEC = 120 ms) and well inside lease + grace (21 s * SEC = 420 ms) + await sleep(8 * SEC) + const stateAtRestart = w.floor.jobs.get("wf-test-r1")!.state + const l2 = start(w, l1.dir) + await until("a Task exists", () => w.ax.tasks.size > 0, 3000).catch(() => {}) + await sleep(200) + const j = w.floor.jobs.get("wf-test-r1")! + const out = { keyJournaled, stateAtRestart, history: j.history, infraSpent: j.infraSpent, tasks: [...w.ax.tasks.keys()], + l2: l2.logs.filter((l) => ["leased", "duplicate-grant", "created", "resume"].includes(String(l.ev))).map((l) => `${l.ev}:${l.leaseId}`) } + console.log(JSON.stringify(out)) + await l2.drain() + w.floor.close() + expect(keyJournaled).toBe(true) + expect(stateAtRestart).toBe("orphaned") + expect(j.history).not.toContain("requeue:omitted") // the grant under the journaled key was released by the first heartbeat + expect(out.tasks).toContain("wf-test-r1-a1") + expect(j.infraSpent).toBe(0) // round 4 addition + expect(out.tasks).not.toContain("wf-test-r1-a2") +}) + +it("control: the same kill with a restart shorter than the lease recovers a1", async () => { + const w = world() + w.floor.enqueue(job("r2")) + let leased = false + const dying: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + if (/"Lease"/.test(body) && /"capacity":[1-9]/.test(body)) { await w.floor.fetch(input as any, { ...init, body }); leased = true; return new Promise(() => {}) } + return w.floor.fetch(input as any, { ...init, body }) + } + const l1 = start(w, undefined, {}, dying) + await until("floor granted a1", () => leased) + Effect.runFork(Fiber.interrupt(l1.fiber)) + await sleep(2 * SEC) + const l2 = start(w, l1.dir) + await until("a Task exists", () => w.ax.tasks.size > 0, 3000).catch(() => {}) + const j = w.floor.jobs.get("wf-test-r2")! + console.log(JSON.stringify({ control: true, history: j.history, tasks: [...w.ax.tasks.keys()] })) + await l2.drain(); w.floor.close() + expect([...w.ax.tasks.keys()]).toContain("wf-test-r2-a1") +}) + +it("running process: the reply is lost and Lease stays unanswered past the lease while Heartbeat works", async () => { + const w = world() + w.floor.enqueue(job("r3")) + let first = true, blockUntil = 0 + const f: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + if (/"Lease"/.test(body)) { + if (first && /"capacity":[1-9]/.test(body)) { first = false; await w.floor.fetch(input as any, { ...init, body }); blockUntil = Date.now() + 10 * SEC; throw new TypeError("fetch failed") } + if (Date.now() < blockUntil) throw new TypeError("fetch failed") + } + return w.floor.fetch(input as any, { ...init, body }) + } + const l = start(w, undefined, {}, f) + await until("a Task exists", () => w.ax.tasks.size > 0, 4000).catch(() => {}) + await sleep(100) + const j = w.floor.jobs.get("wf-test-r3")! + console.log(JSON.stringify({ running: true, history: j.history, infraSpent: j.infraSpent, tasks: [...w.ax.tasks.keys()] })) + await l.drain(); w.floor.close() + expect(j.history).not.toContain("requeue:omitted") + expect([...w.ax.tasks.keys()]).toEqual(["wf-test-r3-a1"]) // round 4 addition +}) + +// Round 4 addition: the running-process half needs the floor's rule 6b. Against an L1-only floor (vouchByKey false) +// the residual is still there; this pins that the link-side fix alone does not claim it. +it("control: an L1-only floor that ignores pendingRequestKey still releases the omitted orphan", async () => { + const w = world() + w.floor.vouchByKey = false + w.floor.enqueue(job("r4")) + let first = true, blockUntil = 0 + const f: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + if (/"Lease"/.test(body)) { + if (first && /"capacity":[1-9]/.test(body)) { first = false; await w.floor.fetch(input as any, { ...init, body }); blockUntil = Date.now() + 10 * SEC; throw new TypeError("fetch failed") } + if (Date.now() < blockUntil) throw new TypeError("fetch failed") + } + return w.floor.fetch(input as any, { ...init, body }) + } + const l = start(w, undefined, {}, f) + await until("a Task exists", () => w.ax.tasks.size > 0, 4000).catch(() => {}) + const j = w.floor.jobs.get("wf-test-r4")! + await l.drain(); w.floor.close() + expect(j.history).toContain("requeue:omitted") +}) + +// Round 4 addition: the startup half does not depend on the floor. Against an L1-only floor (no rule 6b) the journaled +// key is still answered before the first Heartbeat, so the orphan granted under it is listed and re-adopted. +it("restart past the lease against an L1-only floor: the startup replay recovers a1", async () => { + const w = world() + w.floor.vouchByKey = false + w.floor.enqueue(job("r5")) + let leased = false + const dying: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + if (/"Lease"/.test(body) && /"capacity":[1-9]/.test(body)) { await w.floor.fetch(input as any, { ...init, body }); leased = true; return new Promise(() => {}) } + return w.floor.fetch(input as any, { ...init, body }) + } + const l1 = start(w, undefined, {}, dying) + await until("floor granted a1", () => leased) + Effect.runFork(Fiber.interrupt(l1.fiber)) + await sleep(8 * SEC) + const stateAtRestart = w.floor.jobs.get("wf-test-r5")!.state + const l2 = start(w, l1.dir) + await until("a Task exists", () => w.ax.tasks.size > 0, 3000).catch(() => {}) + await sleep(100) + const j = w.floor.jobs.get("wf-test-r5")! + await l2.drain(); w.floor.close() + expect(stateAtRestart).toBe("orphaned") + expect(l2.has("lease-key-replayed")).toBe(true) + expect(j.history).not.toContain("requeue:omitted") + expect([...w.ax.tasks.keys()]).toEqual(["wf-test-r5-a1"]) +}) diff --git a/pkgs/substrate-link/src/apps/link/test/review-r4-stale-lost.test.ts b/pkgs/substrate-link/src/apps/link/test/review-r4-stale-lost.test.ts new file mode 100644 index 000000000..a999b9fd9 --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/review-r4-stale-lost.test.ts @@ -0,0 +1,67 @@ +// Round 4 regression test, ported from /home/tom/today/evals-2026-09-23/link/scratch-r4-0/stale-lost.repro.ts (assertions kept; additions marked). +// Duplicate/stale delivery: a Heartbeat reply is applied to whatever record holds that leaseId when the reply lands, +// not to the generation the Heartbeat vouched for. Set-up is R2-5a (review-r2.test.ts): a1 fails retryable while a2 +// gen 1 is leased to this link and waits for a1's verdict. One Heartbeat carrying [a2] (gen 1) reaches the floor +// after rule 4b withdrew a2 gen 1 and requeued a2 gen 2, so the floor answers lost:[a2]. Its reply lands after the +// link took the regrant of a2 (gen 2). The link then releases gen 2 as `lost` and never creates it, although the floor +// has gen 2 leased to it: the job is stranded until the lease expires, then released `omitted` (an infra release). +import { Effect } from "effect" +import { expect, it } from "vitest" +import { bodyOf, job, sleep, start, until, world } from "./review-r4-harness.ts" +import type { World } from "./review-r4-harness.ts" + +const rpcOf = async (input: unknown, init?: RequestInit) => { const b = await bodyOf(input, init); return b.includes('"Complete"') ? "Complete" : b.includes('"Lease"') ? "Lease" : b.includes('"Heartbeat"') ? "Heartbeat" : "?" } +const failRetryable = (w: World, name: string) => { + const t = w.ax.tasks.get(name)! + t.exited = true; t.phase = "Failed" + t.conditions = [{ type: "Ready", status: "False", reason: "ResourceExhausted", message: "ResourceExhausted: no free worker" }] +} + +it("a Heartbeat reply computed for a2 gen 1 does not release a2 gen 2", async () => { + const w = world(); w.floor.enqueue(job("1")) + const J = () => w.floor.jobs.get("wf-test-1")! + let holdLease = false, failHeartbeat = false, failComplete = false + let armed = false, hbForwarded = false, hbIntercepted = false + let l: ReturnType + const fetch: typeof globalThis.fetch = async (input, init) => { + const body = await bodyOf(input, init) + const rpc = await rpcOf(input, init) + if (rpc === "Heartbeat" && failHeartbeat) throw new TypeError("fetch failed") + if (rpc === "Complete" && failComplete) return new Response("upstream timeout", { status: 503 }) + if (rpc === "Complete" && armed) await until("a Heartbeat vouching for a2 gen 1 is in flight", () => hbIntercepted, 4000) + if (rpc === "Lease") { while (holdLease) await sleep(2); while (armed && !hbForwarded) await sleep(2) } + if (rpc === "Heartbeat" && armed && !hbIntercepted && body.includes("wf-test-1-a2")) { + hbIntercepted = true + await until("floor withdrew a2 gen 1 and requeued a2 gen 2", () => J().history.includes("withdrawn:a2") && J().state === "queued", 4000) + const res = await w.floor.fetch(input as any, { ...init, body }) + hbForwarded = true // Lease may go now: the floor regrants a2 (gen 2) + await until("link took the regrant", () => l.has("regrant", "wf-test-1-a2"), 4000).catch(() => {}) + return res // the reply (lost:[a2]) lands after the regrant + } + return w.floor.fetch(input as any, { ...init, body }) + } + l = start(w, undefined, {}, fetch) + await until("running", () => w.ax.tasks.get("wf-test-1-a1")?.phase === "Running") + w.floor.down = true + await until("floor requeued attempt 2", () => J().state === "queued" && J().attempt === 2, 4000) + failHeartbeat = true; holdLease = true; failComplete = true + w.floor.down = false + await sleep(80) + failRetryable(w, "wf-test-1-a1") + await until("verdict written", () => l.has("verdict", "wf-test-1-a1")) + holdLease = false + await until("a2 received", () => l.has("leased", "wf-test-1-a2"), 4000) + await until("dispatch waits for the verdict", () => l.has("supersede-waits-for-verdict", "wf-test-1-a2")) + armed = true + failComplete = false; failHeartbeat = false + await until("a2 created, or 1 s", () => w.ax.updates.has("wf-test-1-a2"), 1000).catch(() => {}) + await sleep(300) + const trace = l.logs.filter((x) => x.leaseId === "wf-test-1-a2" || x.task === "wf-test-1-a2").map((x) => `${x.ev}${x.why ? ":" + x.why : ""}`) + const out = { hbIntercepted, floor: J().history, state: J().state, attempt: J().attempt, infraSpent: J().infraSpent, a2Updates: w.ax.updates.get("wf-test-1-a2") ?? 0, a2trace: trace } + console.log(JSON.stringify(out)) + await l.drain(); w.floor.close() + expect(hbIntercepted).toBe(true) + expect(out.a2Updates).toBe(1) + expect(out.floor).not.toContain("requeue:omitted") // round 4 addition + expect(out.a2trace).toContain("stale-heartbeat-reply") +}) diff --git a/pkgs/substrate-link/src/apps/link/test/workerd.integration.test.ts b/pkgs/substrate-link/src/apps/link/test/workerd.integration.test.ts new file mode 100644 index 000000000..866ba309e --- /dev/null +++ b/pkgs/substrate-link/src/apps/link/test/workerd.integration.test.ts @@ -0,0 +1,190 @@ +// Integration: the link against a REAL floor process and a REAL gRPC ax wire. +// floor the option-A prototype Durable Object (../arc/proto/floor, Effect's RpcServer.layerHttp inside a DO, DO +// SQLite, a storage alarm) served by `wrangler dev --local` on loopback (workerd). Copied to a temp dir first, +// so nothing is written into the prototype's own tree. It is NOT the floor port (FL1 to FL8 are not in it). +// ax a gRPC server on loopback built from the vendored ax.proto plus P1's GetTaskResult (fixtures/ax-p1.proto), +// with ax's own ListTasks paging (50 rows, newest first) and a Gateway. The link uses its production grpcAx. +// Opt-in: LINK_WORKERD=1, run inside ~/.local/bin/runtime-test (it starts workerd). Skipped otherwise. +import { spawn } from "node:child_process" +import type { ChildProcess } from "node:child_process" +import { createHash } from "node:crypto" +import { cpSync, mkdtempSync, rmSync, symlinkSync } from "node:fs" +import { tmpdir } from "node:os" +import { join } from "node:path" +import * as grpc from "@grpc/grpc-js" +import * as protoLoader from "@grpc/proto-loader" +import { Deferred, Effect, Exit, Fiber } from "effect" +import { afterAll, beforeAll, describe, expect, it } from "vitest" +import { grpcAx } from "../src/ax.ts" +import type { AgentJob } from "../src/contract.ts" +import { rpcFloor } from "../src/floor.ts" +import { Journal } from "../src/journal.ts" +import { runLink } from "../src/link.ts" +import type { LinkConfig } from "../src/link.ts" + +const RUN = process.env.LINK_WORKERD === "1" +const PROTO_DIR = "/home/tom/today/evals-2026-09-23/arc/proto" +const P1_PROTO = join(import.meta.dirname, "fixtures", "ax-p1.proto") +const TOKEN = "local-dev-only-not-a-secret" // the prototype's wrangler.jsonc var, a public test value +const PORT = 18000 + Math.floor(Math.random() * 900) +const F = `http://127.0.0.1:${PORT}` +const A = "ultracode.mecattaf.dev/" + +const job = (n: string): AgentJob => ({ + apiVersion: "ultracode.mecattaf.dev/v1alpha1", kind: "AgentJob", + metadata: { name: `wf-int-${n}`, labels: { [A + "workflow"]: "link-int", [A + "phase-index"]: "1" }, annotations: { [A + "run-id-raw"]: "wf_int", [A + "label"]: `int:${n}`, [A + "item-key"]: `wf_int#${n}`, [A + "journal-key"]: `${"cd".repeat(32)}:1`, [A + "phase-title"]: "Integration" } }, + spec: { "runs-on": ["seat:halogen", "runtime:gvisor"], with: { prompt: `say ${n}`, prompt_ref: { sha256: "cd".repeat(32), bytes: 5, uri: `journal://wf_int/${n}/prompt.md` }, model: "halogen-qwen3.8-flash-next" } } +}) + +// ---------------------------------------------------------------- the ax gRPC server (P1 wire) +const tasks = new Map() +const results = new Map() +let axUnavailable = false +let axServer: grpc.Server +let axPort = 0 +const unavailable = { code: grpc.status.UNAVAILABLE, details: "ax-server unavailable" } +const nf = (what: string) => ({ code: grpc.status.NOT_FOUND, details: `${what} not found` }) +const guard = (cb: grpc.sendUnaryData, f: () => void) => axUnavailable ? cb(unavailable) : f() + +async function startAx() { + const def = protoLoader.loadSync(P1_PROTO, { keepCase: false, longs: String, enums: String, defaults: true, oneofs: true }) + const svc = (grpc.loadPackageDefinition(def) as any).ax.v1alpha1.AX.service + axServer = new grpc.Server() + axServer.addService(svc, { + GetTask: (c: any, cb: any) => guard(cb, () => { const t = tasks.get(c.request.name); t ? cb(null, t) : cb(nf(c.request.name)) }), + UpdateTask: (c: any, cb: any) => guard(cb, () => { tasks.set(c.request.task.metadata.name, { ...c.request.task, status: { phase: "Pending" } }); cb(null, c.request.task) }), + ListTasks: (c: any, cb: any) => guard(cb, () => { const n = Number(c.request.limit) > 0 ? Number(c.request.limit) : 50, o = Number(c.request.offset) || 0; cb(null, { tasks: [...tasks.values()].reverse().slice(o, o + n) }) }), + DeleteTask: (c: any, cb: any) => guard(cb, () => { tasks.delete(c.request.name); results.delete(c.request.name); cb(null, {}) }), + GetGateway: (c: any, cb: any) => guard(cb, () => c.request.name === "halogen" + ? cb(null, { apiVersion: "ax.io/v1alpha1", kind: "Gateway", metadata: { name: "halogen", atespace: "fleet" }, spec: { egress: { allowlist: { hosts: [{ host: "worker", port: 8731 }] } } } }) + : cb(nf("gateway"))), + GetTaskResult: (c: any, cb: any) => guard(cb, () => { const b = results.get(c.request.name); b ? cb(null, { content: b, sha256: createHash("sha256").update(b).digest("hex") }) : cb(nf("result")) }) + }) + axPort = await new Promise((res, rej) => axServer.bindAsync("127.0.0.1:0", grpc.ServerCredentials.createInsecure(), (e, p) => e ? rej(e) : res(p))) +} +const run = (name: string) => { tasks.get(name).status = { phase: "Running" } } +const finish = (name: string, out: unknown) => { + results.set(name, Buffer.from(JSON.stringify(out))) + tasks.get(name).status = { phase: "Completed", usage: { promptTokens: 5, completionTokens: 3, toolCalls: 1 }, conditions: [{ type: "Ready", status: "False", reason: "CommandExited", message: "ExitCode=0" }] } +} + +// ---------------------------------------------------------------- the floor under wrangler dev --local +let work = "", wrangler: ChildProcess | undefined +const floorLog: Array = [] +async function floorUp() { + wrangler = spawn("wrangler", ["dev", "--local", "--ip", "127.0.0.1", "--port", String(PORT), "--persist-to", join(work, "state"), + "--var", "LEASE_SECONDS:30", "--show-interactive-dev-session=false"], + { cwd: join(work, "floor"), env: { ...process.env, WRANGLER_SEND_METRICS: "false", CI: "1" }, detached: true, stdio: ["ignore", "pipe", "pipe"] }) + wrangler.stdout!.on("data", (d) => floorLog.push(String(d))) + wrangler.stderr!.on("data", (d) => floorLog.push(String(d))) + await until("floor up", async () => (await admin("GET", "/admin/state").catch(() => undefined)) !== undefined, 60_000) +} +async function floorDown() { + if (wrangler?.pid) { try { process.kill(-wrangler.pid, "SIGTERM") } catch { /* gone */ } } + await until("floor down", async () => (await admin("GET", "/admin/state").catch(() => undefined)) === undefined, 20_000) +} +async function admin(method: string, path: string, body?: unknown): Promise { + const r = await fetch(F + path, { method, headers: { authorization: `Bearer ${TOKEN}`, "content-type": "application/json" }, ...(body ? { body: JSON.stringify(body) } : {}), signal: AbortSignal.timeout(3000) }) + if (!r.ok) throw new Error(`admin ${path}: ${r.status}`) + return r.json() +} +const jobRow = async (n: string) => ((await admin("GET", "/admin/state")).jobs as Array).find((j) => j.name === `wf-int-${n}`) + +async function until(what: string, pred: () => boolean | Promise, ms = 30_000) { + const t0 = Date.now() + while (!(await pred())) { if (Date.now() - t0 > ms) throw new Error(`timeout waiting for: ${what}`); await new Promise((r) => setTimeout(r, 100)) } +} +const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms)) + +function startLink(token = TOKEN) { + const logs: Array> = [] + const dir = mkdtempSync(join(tmpdir(), "conwip-link-int-")) + const stop = Effect.runSync(Deferred.make()) + const ax = grpcAx(`127.0.0.1:${axPort}`, "fleet", 5_000, P1_PROTO) + const cfg: LinkConfig = { + holder: "nas-link-1", maxInFlight: 2, servedLabels: ["seat:halogen", "runtime:gvisor"], + shape: { atespace: "fleet", image: "ax-agent", gateway: "halogen", command: () => ["ax-agent", "pi"] }, + completion: "p1", secondMs: 1000, resyncMs: 500, pendingTimeoutMs: 60_000, deleteAfterMs: 0, deadlineBackstopMs: 60_000, + createAttempts: 3, outboxBackoffMs: [200, 1000], fenceTimeoutMs: 5_000, initialPollSeconds: 1, initialHeartbeatSeconds: 2 + } + const fiber = Effect.runFork(Effect.scoped(Effect.gen(function*() { + const floor = yield* rpcFloor({ url: F, token, sessionId: `s:${dir}`, timeoutsMs: { lease: 5_000, heartbeat: 5_000, complete: 10_000 } }) + return yield* runLink(cfg, { ax, floor, journal: Journal.open(dir), log: (ev, f) => logs.push({ ev, ...f }), stop, floorUrl: F }) + }))) + return { + logs, + has: (ev: string) => logs.some((l) => l.ev === ev), + exit: () => Effect.runPromise(Fiber.await(fiber)), + drain: async () => { Effect.runSync(Deferred.succeed(stop, undefined)); const e = await Effect.runPromise(Fiber.await(fiber)); ax.close(); rmSync(dir, { recursive: true, force: true }); return e } + } +} + +describe.skipIf(!RUN)("link against the prototype floor under wrangler dev --local and a P1 gRPC ax", () => { + beforeAll(async () => { + work = mkdtempSync(join(tmpdir(), "conwip-floor-int-")) + cpSync(join(PROTO_DIR, "floor"), join(work, "floor"), { recursive: true, filter: (s) => !s.includes(".wrangler") }) + cpSync(join(PROTO_DIR, "contract.ts"), join(work, "contract.ts")) + symlinkSync(join(PROTO_DIR, "node_modules"), join(work, "node_modules")) + await startAx() + await floorUp() + }, 90_000) + afterAll(async () => { + await floorDown().catch(() => undefined) + axServer?.forceShutdown() + if (process.env.LINK_DEBUG) console.log(floorLog.join("").slice(-4000)) + if (work) rmSync(work, { recursive: true, force: true }) + }, 30_000) + + it("I1 happy path over real HTTP RPC and real gRPC: lease, create-only, P1 result with sha256, Complete, janitor delete", async () => { + await admin("POST", "/admin/enqueue", job("1")) + const l = startLink() + await until("created", () => tasks.has("wf-int-1-a1")) + expect(tasks.get("wf-int-1-a1").spec.env.find((e: any) => e.name === "AX_CONWIP_PROMPT").value).toBe("say 1") + run("wf-int-1-a1"); finish("wf-int-1-a1", { answer: 42 }) + await until("floor done", async () => (await jobRow("1"))?.state === "done") + const r = await jobRow("1") + expect([r.result, JSON.parse(r.output), JSON.parse(r.usage), r.attempt]).toEqual(["success", { answer: 42 }, { prompt_tokens: 5, completion_tokens: 3, tool_calls: 1 }, 1]) + await until("janitor deleted the Task", () => !tasks.has("wf-int-1-a1")) + expect(l.has("completion-probe") && l.has("gateway-ok")).toBe(true) + expect(Exit.isSuccess(await l.drain())).toBe(true) + }, 60_000) + + it("I2 floor process restarted while a verdict is pending: the outbox keeps it and it lands once, attempt 1", async () => { + await admin("POST", "/admin/enqueue", job("2")) + const l = startLink() + await until("created", () => tasks.has("wf-int-2-a1")) + run("wf-int-2-a1") + await floorDown() + finish("wf-int-2-a1", { restarted: true }) + await until("outbox keeps", () => l.has("outbox-keep")) + await floorUp() // same --persist-to: the DO's SQLite and alarm survive + await until("floor done", async () => (await jobRow("2"))?.state === "done", 45_000) + const r = await jobRow("2") + expect([r.result, JSON.parse(r.output), r.attempt, r.transitions]).toEqual(["success", { restarted: true }, 1, 0]) + expect(Exit.isSuccess(await l.drain())).toBe(true) + }, 120_000) + + it("I3 ax answers UNAVAILABLE: every Lease asks for 0, the job stays queued; ax returns and the job runs", async () => { + axUnavailable = true + const l = startLink() + await sleep(1_500) // the start-up resync fails; nothing is probed, capacity stays 0 + expect(l.has("completion-probe")).toBe(false) + await admin("POST", "/admin/enqueue", job("3")) + await sleep(6_000) // three polls at the floor's 2 s + expect((await jobRow("3")).state).toBe("queued") + axUnavailable = false + await until("created", () => tasks.has("wf-int-3-a1")) + run("wf-int-3-a1"); finish("wf-int-3-a1", { back: true }) + await until("floor done", async () => (await jobRow("3"))?.state === "done") + expect((await jobRow("3")).result).toBe("success") + expect(Exit.isSuccess(await l.drain())).toBe(true) + }, 60_000) + + it("I4 the real Worker entry answers 401 to a wrong bearer: the link stops with kind auth (exit 78 in main)", async () => { + const l = startLink("not-the-token") + const exit = await l.exit() + expect(Exit.isFailure(exit)).toBe(true) + expect(JSON.stringify(exit)).toContain("\"kind\":\"auth\"") + await l.drain() + }, 30_000) +}) diff --git a/pkgs/substrate-link/src/package.json b/pkgs/substrate-link/src/package.json new file mode 100644 index 000000000..2d2b8d8b0 --- /dev/null +++ b/pkgs/substrate-link/src/package.json @@ -0,0 +1,22 @@ +{ + "name": "ax-conwip", + "version": "0.1.0", + "private": true, + "type": "module", + "description": "CONWIP scheduler that turns Claude ultracode workflow run records into ax Tasks under a fixed work-in-progress cap.", + "scripts": { + "test": "vitest run", + "conwip": "tsx src/cli.ts" + }, + "dependencies": { + "@grpc/grpc-js": "^1.12.6", + "@grpc/proto-loader": "^0.7.13", + "effect": "4.0.0-rc.112" + }, + "devDependencies": { + "@types/node": "^22.10.2", + "tsx": "^4.19.2", + "typescript": "^5.9.3", + "vitest": "^3.2.4" + } +} diff --git a/pkgs/substrate-link/src/pnpm-lock.yaml b/pkgs/substrate-link/src/pnpm-lock.yaml new file mode 100644 index 000000000..5fc4eb8b8 --- /dev/null +++ b/pkgs/substrate-link/src/pnpm-lock.yaml @@ -0,0 +1,1401 @@ +lockfileVersion: '9.0' + +settings: + autoInstallPeers: true + excludeLinksFromLockfile: false + +importers: + + .: + dependencies: + '@grpc/grpc-js': + specifier: ^1.12.6 + version: 1.14.5 + '@grpc/proto-loader': + specifier: ^0.7.13 + version: 0.7.15 + effect: + specifier: 4.0.0-rc.112 + version: 4.0.0-rc.112 + devDependencies: + '@types/node': + specifier: ^22.10.2 + version: 22.20.4 + tsx: + specifier: ^4.19.2 + version: 4.23.15 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + vitest: + specifier: ^3.2.4 + version: 3.2.7(@types/node@22.20.4)(tsx@4.23.15) + +packages: + + '@esbuild/aix-ppc64@0.28.2': + resolution: {integrity: sha512-XExcO+dvLKvVtNTibSTBej1NCAbaGhWn9Ww1ZPx80qsahhPFe/8jgWP0IchNe0F3HwkU7n8ejhH8bjonqht8mQ==} + engines: {node: '>=18'} + cpu: [ppc64] + os: [aix] + + '@esbuild/android-arm64@0.28.2': + resolution: {integrity: sha512-5YfKeeI8qWfBZIX+u2xZC3Zlb3Os/gLS2sbEKM+I4ZOcsWmHS2WLysCcQZDAFRslDUU5Oiq44gf6PYN1vGwG5A==} + engines: {node: '>=18'} + cpu: [arm64] + os: [android] + + '@esbuild/android-arm@0.28.2': + resolution: {integrity: sha512-kXXoiPVVGQcnIYGOeaovwOURpniDBpSq4A03qkQ+BMQqtGG6HYap3xne9C1O1yo4TR3qxlCX5IqqmX6fFo2Lqg==} + engines: {node: '>=18'} + cpu: [arm] + os: [android] + + '@esbuild/android-x64@0.28.2': + resolution: {integrity: sha512-O387ite7SzUyCcy3JQX4P4bLtEA7bLLkx+esve5JHnyYfNTxcVpXZo9jhdB0lTKN44gztELTdU7nS8Nr16Fs1Q==} + engines: {node: '>=18'} + cpu: [x64] + os: [android] + + '@esbuild/darwin-arm64@0.28.2': + resolution: {integrity: sha512-n4KqkOQrraxHJcgjM1RvwbigfQKIKJVpM7xp+KsxiyUSrRdIXnt73VhrPAx0fV44hgfmIVKjxMN9J1t5jySVkw==} + engines: {node: '>=18'} + cpu: [arm64] + os: [darwin] + + '@esbuild/darwin-x64@0.28.2': + resolution: {integrity: sha512-uq6suIWYP37qzGddBKPw5QEQPi6HiLGsO7UmkpfyaYNQ3D+rN6w6WfwH+nuqcGXWvawGwxOEroO4YGnFh95azw==} + engines: {node: '>=18'} + cpu: [x64] + os: [darwin] + + '@esbuild/freebsd-arm64@0.28.2': + resolution: {integrity: sha512-n+I0BTSRIoy+d6RPKnEVwql5UwBJolytvY4mAOIEJorKlqgPII8ix6slVVrfZ5Tnj7glIZvloylbB/EJPMWEXw==} + engines: {node: '>=18'} + cpu: [arm64] + os: [freebsd] + + '@esbuild/freebsd-x64@0.28.2': + resolution: {integrity: sha512-78XJTJkvPs0kz2w61301PJjXl4g7q3JqiYMZ/M/yVI73EHBrCRTgkhu9oqG7vPqq+a/yadEW8aD+agKlk5xrmg==} + engines: {node: '>=18'} + cpu: [x64] + os: [freebsd] + + '@esbuild/linux-arm64@0.28.2': + resolution: {integrity: sha512-pW4AC0P3it8c7do9MVM4p51FzHzdM/TZrerurgRcHJ2WTa1VQ1CIq18xncfpBJw4ojkiZZrKW2yIBWBP92j6Ug==} + engines: {node: '>=18'} + cpu: [arm64] + os: [linux] + + '@esbuild/linux-arm@0.28.2': + resolution: {integrity: sha512-XlDnu2q5yoqems+xay6wSAcg9DDD7K9RLKZEBOMZm3ckNpJBvOX20tSfby8KfrrhINDyv9V2YVZKY/SpoGJI8w==} + engines: {node: '>=18'} + cpu: [arm] + os: [linux] + + '@esbuild/linux-ia32@0.28.2': + resolution: {integrity: sha512-CYbnj78HsIeA+DhgUKgFCfvNsTHFhMMrinUrMZpDXJXKN8T3XViTZ/+wtHeVxEWY8ewSzTFN+nRmSwO2tZaLUQ==} + engines: {node: '>=18'} + cpu: [ia32] + os: [linux] + + '@esbuild/linux-loong64@0.28.2': + resolution: {integrity: sha512-buwkd8nsph4R+ajRvw0qM5Hja/TXQow3ptzWO2EbG/cqcIkHloRrdlBtQlshyYGTNFvfkfJ5tpPLVkY4DtsPfQ==} + engines: {node: '>=18'} + cpu: [loong64] + os: [linux] + + '@esbuild/linux-mips64el@0.28.2': + resolution: {integrity: sha512-ZVykbDyk7519VwiNb9Lcj9m8XM6v5V9uKPvrEMkkEedVewf+0itkhahp4HDpgERXhwLRpWFypsGbG/J8s0QjJA==} + engines: {node: '>=18'} + cpu: [mips64el] + os: [linux] + + '@esbuild/linux-ppc64@0.28.2': + resolution: {integrity: sha512-CAXl+Dtd9UUuJd8pKKdwh6MLm3MUMiqMPmhZ3tTSXPqfyQ3vDl6R5hZdZ/kYojK4ofXtdfSv1tFq8XzWx3heNQ==} + engines: {node: '>=18'} + cpu: [ppc64] + os: [linux] + + '@esbuild/linux-riscv64@0.28.2': + resolution: {integrity: sha512-GeXCej4IQtU1B+QlDV8W/RRvbzI3O/Stss+/bCXv4lZls5WGRtu2a+3JkA3i4qIUlMXpcHebWpF8AkJhATowuA==} + engines: {node: '>=18'} + cpu: [riscv64] + os: [linux] + + '@esbuild/linux-s390x@0.28.2': + resolution: {integrity: sha512-3H1weTYZPxt/WOhByszQZybS9w5lKzUn1FDMsgEChbHWQwHYQQRfBxgCcZvPhjHfKyJjIievvMmEUawJrdY9Dg==} + engines: {node: '>=18'} + cpu: [s390x] + os: [linux] + + '@esbuild/linux-x64@0.28.2': + resolution: {integrity: sha512-4xTZr1FUmSoQW4XIWmit3tzQrUTZM+N3P0XV8xROKYF50XfI7xeO90+1bZvNwxIufQ9hDQVRJH5YhgPVF8A/HQ==} + engines: {node: '>=18'} + cpu: [x64] + os: [linux] + + '@esbuild/netbsd-arm64@0.28.2': + resolution: {integrity: sha512-sSATRjPeDBg3pdgHoQfoYBob11Kk1FGa9lui5RIHZCoCkJa9QKlvl3/vKz2usCmYYjs7ymJR/2Nnsqe+Hjt5nw==} + engines: {node: '>=18'} + cpu: [arm64] + os: [netbsd] + + '@esbuild/netbsd-x64@0.28.2': + resolution: {integrity: sha512-lqnzCV+mM0gIADaKihiCg6ifgfU2L3h5E33rNQBN1Y4MaVGnzryzmvvf7UHxprpQdE8hpqLolJ9Rl+SkIRDpyw==} + engines: {node: '>=18'} + cpu: [x64] + os: [netbsd] + + '@esbuild/openbsd-arm64@0.28.2': + resolution: {integrity: sha512-AL2qJILH7lNjrDmCQDvdxMfAUIv8KMNZOvrwAQ8i8//ntL9FflhOyMJ8OZSMBb8/AWXe3/5v5S20y3zCoZWKoQ==} + engines: {node: '>=18'} + cpu: [arm64] + os: [openbsd] + + '@esbuild/openbsd-x64@0.28.2': + resolution: {integrity: sha512-QtiuPytchRyC4rwUKhexJdQKvDuZ6hWloi3igqPQNUJCS1/v9EiO3UTOXR6A3FoMo4fnAKbWJdqaIwhOzh8qEw==} + engines: {node: '>=18'} + cpu: [x64] + os: [openbsd] + + '@esbuild/openharmony-arm64@0.28.2': + resolution: {integrity: sha512-WkhYDmpTjLvGlScA1rwjRUmhl4k8oXR3cIbtqWmELgU/dFeHHlEllxDvdWcNJV9rbzCexB5vz8gtNewWLgCT7Q==} + engines: {node: '>=18'} + cpu: [arm64] + os: [openharmony] + + '@esbuild/sunos-x64@0.28.2': + resolution: {integrity: sha512-GPMSkTOtMnv2U2F8gxe4Io6qmVs+YKyp832Etqqxr0hFngmXQ3rzwytelm3GIn7T4VviRUlf3sOgBOiTdvaf7g==} + engines: {node: '>=18'} + cpu: [x64] + os: [sunos] + + '@esbuild/win32-arm64@0.28.2': + resolution: {integrity: sha512-PIhhEkE9uPBleRBrQEJpUn7MBnibZzbGzYWPmY3x+YoVg/95zbjB4CxPPOQ8l5tYYM4mMaCthF8/1DIfBQQyWQ==} + engines: {node: '>=18'} + cpu: [arm64] + os: [win32] + + '@esbuild/win32-ia32@0.28.2': + resolution: {integrity: sha512-YmJbfTlvU7Sdn9BB+4PRES4oB6pxgS37MAONj+hBr/cpXS1aBPKXxNnDbu+QCWPj0o9dgyxeq79g6c5P8KeuYA==} + engines: {node: '>=18'} + cpu: [ia32] + os: [win32] + + '@esbuild/win32-x64@0.28.2': + resolution: {integrity: sha512-5ebpxr3nWMzrL/rnUI755Jkuee0bHL/Gq0WTF9lvcpv73wAp5eu8MfBUgWK9bhWvZjj7yX8etf/8tI8Ney695g==} + engines: {node: '>=18'} + cpu: [x64] + os: [win32] + + '@grpc/grpc-js@1.14.5': + resolution: {integrity: sha512-7VZM+SVdEcUUqSQeNI3zM8Qs/BhQKZndPo2h5VkYkAM8Iz0wJIa8mKV5ekQGqG8UUsnkQ0NMxIxwkIHYvj0qOw==} + engines: {node: '>=12.10.0'} + + '@grpc/proto-loader@0.7.15': + resolution: {integrity: sha512-tMXdRCfYVixjuFK+Hk0Q1s38gV9zDiDJfWL3h1rv4Qc39oILCu1TRTDt7+fGUI8K4G1Fj125Hx/ru3azECWTyQ==} + engines: {node: '>=6'} + hasBin: true + + '@grpc/proto-loader@0.8.1': + resolution: {integrity: sha512-wtF6h+DY6M3YaDBPAmvuuA6jV8Sif9MjtOI5euKFWRgCDl5PeDpPsHR9u2l6St5ceY8AZgoNDww5+HvEsXFsGg==} + engines: {node: '>=6'} + hasBin: true + + '@jridgewell/sourcemap-codec@1.6.0': + resolution: {integrity: sha512-T7jf+5zgsZHwNJ4lvQ7/aezbyk0nNX+zJVWpmHA7VYsEx7a7qr5Rg5IbtJFqkgze5Y2sruq1RUY8Q837Od7iFw==} + + '@js-sdsl/ordered-map@4.4.2': + resolution: {integrity: sha512-iUKgm52T8HOE/makSxjqoWhe95ZJA1/G1sYsGev2JDKUSS14KAgg1LHb+Ba+IPow0xflbnSkOsZcO08C7w1gYw==} + + '@msgpackr-extract/msgpackr-extract-darwin-arm64@3.0.4': + resolution: {integrity: sha512-LCkGo6JDfaBhgST7UpPWgNgLINpcpabaHfyz5OBx75nUYxBsaEPxjnyNjWpeb/xBup/682QnBfRBy2/LvPutZQ==} + cpu: [arm64] + os: [darwin] + + '@msgpackr-extract/msgpackr-extract-darwin-x64@3.0.4': + resolution: {integrity: sha512-zExlW9zUJKZH/tOtVMttwjKa4Xm/3KcNjnE3dPN92uCktwavMxpgCA3MoJK/DOnTWsQgo224OaST27/mPNAf+w==} + cpu: [x64] + os: [darwin] + + '@msgpackr-extract/msgpackr-extract-linux-arm64@3.0.4': + resolution: {integrity: sha512-dgX0P/9wGPJeHFBG+ZmhgE6bmtMt7NP5CRBGyyktpopdk/mW4POnrpQsSLtKI1dwpc+pPLuXHDh6vvskyQE/sw==} + cpu: [arm64] + os: [linux] + + '@msgpackr-extract/msgpackr-extract-linux-arm@3.0.4': + resolution: {integrity: sha512-Tg3yX65f5GbtXLkrYEHE5oibZG9epyYWas7FogTTEJeDEF9JlXJzKgXaNhT3UXlTOeA+AfZpYZYZ0uPj7Cfquw==} + cpu: [arm] + os: [linux] + + '@msgpackr-extract/msgpackr-extract-linux-x64@3.0.4': + resolution: {integrity: sha512-8TNXMEjJc3QEy7R/x1INhgiU+XakDAFUzBhaz7+Rbrs8NH5UQeHQxxmzsSBJGyV6I1jW79undiQm8tOI+D+8FQ==} + cpu: [x64] + os: [linux] + + '@msgpackr-extract/msgpackr-extract-win32-x64@3.0.4': + resolution: {integrity: sha512-CmCXPQrkbwExx3j946/PtHWHbYJiCRBRDl4BlkRQcJB/YOwQxJRTpoo7aTsortjgoJ1x7opzTSxn7C+ASSLVjQ==} + cpu: [x64] + os: [win32] + + '@napi-rs/lzma-linux-x64-gnu@1.5.1': + resolution: {integrity: sha512-oTXEIha4SsuXdTA4Iyskj0kpdx2yVXdhd75c2v3xGrHFfVMsbhTPZU/nMPL4sWKo4pBHm3aucLaqGlF696dTyQ==} + engines: {node: ^22.20 || ^24.12 || >=25} + cpu: [x64] + os: [linux] + libc: [glibc] + + '@protobufjs/aspromise@1.1.2': + resolution: {integrity: sha512-j+gKExEuLmKwvz3OgROXtrJ2UG2x8Ch2YZUxahh+s1F2HZ+wAceUNLkvy6zKCPVRkU++ZWQrdxsUeQXmcg4uoQ==} + + '@protobufjs/base64@1.1.2': + resolution: {integrity: sha512-AZkcAA5vnN/v4PDqKyMR5lx7hZttPDgClv83E//FMNhR2TMcLUhfRUBHCmSl0oi9zMgDDqRUJkSxO3wm85+XLg==} + + '@protobufjs/codegen@2.0.5': + resolution: {integrity: sha512-zgXFLzW3Ap33e6d0Wlj4MGIm6Ce8O89n/apUaGNB/jx+hw+ruWEp7EwGUshdLKVRCxZW12fp9r40E1mQrf/34g==} + + '@protobufjs/eventemitter@1.1.1': + resolution: {integrity: sha512-vW1GmwMZNnL+gMRaovlh9yZX74kc+TTU3FObkkurpMaRtBfLP3ldjS9KQWlwZgraRE0+dheEEoAxdzcJQ8eXZg==} + + '@protobufjs/fetch@1.1.1': + resolution: {integrity: sha512-GpptLrs57adMSuHi3VNj0mAF8dwh36LMaYF6XyJ6JMWlVsc+t42tm1HSEDmOs3A8fC9yyeisgLhsTVQokOZ0zw==} + + '@protobufjs/float@1.0.2': + resolution: {integrity: sha512-Ddb+kVXlXst9d+R9PfTIxh1EdNkgoRe5tOX6t01f1lYWOvJnSPDBlG241QLzcyPdoNTsblLUdujGSE4RzrTZGQ==} + + '@protobufjs/path@1.1.2': + resolution: {integrity: sha512-6JOcJ5Tm08dOHAbdR3GrvP+yUUfkjG5ePsHYczMFLq3ZmMkAD98cDgcT2iA1lJ9NVwFd4tH/iSSoe44YWkltEA==} + + '@protobufjs/pool@1.1.0': + resolution: {integrity: sha512-0kELaGSIDBKvcgS4zkjz1PeddatrjYcmMWOlAuAPwAeccUrPHdUqo/J6LiymHHEiJT5NrF1UVwxY14f+fy4WQw==} + + '@protobufjs/utf8@1.1.2': + resolution: {integrity: sha512-b1UQwcEZ4yCnMCD8DAL1VlbvBJE9/IX4FTIp7BG1xYpf29SLazLSrqUkj4w7Y5y7cCVP6E5tcqqcI0xemPkHug==} + + '@rollup/rollup-android-arm-eabi@4.63.4': + resolution: {integrity: sha512-I+BSHzTAhKN2n7ZwGZsegGcZjDpLqFOMAtJz/u6uFGe0pUFbq56dEHjqJV/ZUdRJtNXNxA+hREUatZBvMR3Oiw==} + cpu: [arm] + os: [android] + + '@rollup/rollup-android-arm64@4.63.4': + resolution: {integrity: sha512-pu3BdjS2LtEzRu2elmGzS3fIeWSZy4BMDIaLNwjorO76+k2d0LMluijhsDx3KQyQBQ/lLUZCQA9/s6csvUfuhw==} + cpu: [arm64] + os: [android] + + '@rollup/rollup-darwin-arm64@4.63.4': + resolution: {integrity: sha512-xfSrj9MHnWK9GaSqT9U0ImHtH/N8WZlHLx4cZHiuLcqs640hvZ3hLPd5UR2AZS57FaE8HrRUSpltbZdWRxHiDA==} + cpu: [arm64] + os: [darwin] + + '@rollup/rollup-darwin-x64@4.63.4': + resolution: {integrity: sha512-bqU99PLJb/dqb3S0GIMdeuyAEETSUgZBoqXYd3Sd+WCsV+MmPhnN6JrotWyir31+QgH7EvvE5/mwGJlEoci8Fw==} + cpu: [x64] + os: [darwin] + + '@rollup/rollup-freebsd-arm64@4.63.4': + resolution: {integrity: sha512-JinsFZ5G40oXQb+sUuiA5x689vhr6dDYK0H0NL+rwKdL6CqnmYN8PE4ZwfRSoIjrCxqTQG/SLfTtSvHeGxoVlw==} + cpu: [arm64] + os: [freebsd] + + '@rollup/rollup-freebsd-x64@4.63.4': + resolution: {integrity: sha512-GAdA4UxpiNm27cLHr2GqXBpAD0x9FqwYBY7/YSP0Ss0/PNi4k8gbviqpIpYbVSRBaS2ZcegXEzgTQMbRNCwxCw==} + cpu: [x64] + os: [freebsd] + + '@rollup/rollup-linux-arm-gnueabihf@4.63.4': + resolution: {integrity: sha512-qDd6NoA1znaLjp4jR5U/KWCdLAKDJNB8W9ChbbDaKbo0xA+Atln5HK6LFCZ4oJQpemtRZA288DCirFRjrspptw==} + cpu: [arm] + os: [linux] + libc: [glibc] + + '@rollup/rollup-linux-arm-musleabihf@4.63.4': + resolution: {integrity: sha512-WtB5Tz5KTNINb8ZA+8sQ7bmjuS1JrRT7YverYIhUGdWWDlpzVWmIwuZE+jidkEXUn1l0zrEkaIMa8dHF3NGcsA==} + cpu: [arm] + os: [linux] + libc: [musl] + + '@rollup/rollup-linux-arm64-gnu@4.63.4': + resolution: {integrity: sha512-VcQ3L1tjnkKzWjryAVaFhHEWcqOfICX9uxVVoDzm2t0DpgKRHd2zOpVrJc0xsWeBZcBFyYROCIBdyR/fS174pg==} + cpu: [arm64] + os: [linux] + libc: [glibc] + + '@rollup/rollup-linux-arm64-musl@4.63.4': + resolution: {integrity: sha512-6+ZQX6P5s0cMDN2Ypb8Lbm2+/sZYmZjdaYny992ujUU9UKi/4CWoJWsl1pNvjWJHNHGK51m+jKGLlh1ylb2ifQ==} + cpu: [arm64] + os: [linux] + libc: [musl] + + '@rollup/rollup-linux-loong64-gnu@4.63.4': + resolution: {integrity: sha512-D72ZnvkFkBXOfzMMQLcwfPLyGkKb7HZ9/mf97B7v6/P5Lbv4oFOtSY/uHbS8lH6uKUOxoKiuokdb50XZSzzbJw==} + cpu: [loong64] + os: [linux] + libc: [glibc] + + '@rollup/rollup-linux-loong64-musl@4.63.4': + resolution: {integrity: sha512-piU6BxeqA3O9KSu3kRCIQQtNqFFaTu21SEV4FwaRZowpnj3bLaWPZHw+xFqCs0XlJ+aOH3PTRWGoglH+mKA/OA==} + cpu: [loong64] + os: [linux] + libc: [musl] + + '@rollup/rollup-linux-ppc64-gnu@4.63.4': + resolution: {integrity: sha512-/5PGpHwqt2EEEOUs1XwzubE/ucr0dWDQ+to3zqi4Ds7EWpwtQ79wXc4JBoxqj/OwpawTsKWzJxHfSuBOq3DrWA==} + cpu: [ppc64] + os: [linux] + libc: [glibc] + + '@rollup/rollup-linux-ppc64-musl@4.63.4': + resolution: {integrity: sha512-cX3beZDLWt7G2oJF+nhChiT+qtaihs+S2xi7ziGmVB+2pwPng6D0Ed0HmElQOgv2UsUmSJJLGwpBao/3TDx3VA==} + cpu: [ppc64] + os: [linux] + libc: [musl] + + '@rollup/rollup-linux-riscv64-gnu@4.63.4': + resolution: {integrity: sha512-1uz2mGWHyptR7DgHHrlbdRAjXK7v7elGZ9lMja910/RP+ZYbX6xAmCiU9UZSX4hqmgtHMv6lr5l3kq1HIOpcag==} + cpu: [riscv64] + os: [linux] + libc: [glibc] + + '@rollup/rollup-linux-riscv64-musl@4.63.4': + resolution: {integrity: sha512-nLS8topojxyz7SRpKR2IODRpQ0XPZ+xaOXvT3+hqK/Uy8Lo5HFgkkIBiIrCu5tL5YqzTvgovGw55PwpahTAGig==} + cpu: [riscv64] + os: [linux] + libc: [musl] + + '@rollup/rollup-linux-s390x-gnu@4.63.4': + resolution: {integrity: sha512-gs7DRKotr3l3q+jGPQBjH0ng1FjlEDm5ueQrkw5JtQvtLyEIcLASqAEaor56BhkKRzk+IcQzrcanBdb/bBQn8g==} + cpu: [s390x] + os: [linux] + libc: [glibc] + + '@rollup/rollup-linux-x64-gnu@4.63.4': + resolution: {integrity: sha512-791ET7W17NnScOZM7h4dX5hYspxE28htPFsb1awY/NRR8+PRNkS53e475rDdxXXDrP+kwnCcNWg9CX5ztn/Aqw==} + cpu: [x64] + os: [linux] + libc: [glibc] + + '@rollup/rollup-linux-x64-musl@4.63.4': + resolution: {integrity: sha512-iwZQRcmj7g88g3tzefIrQY7qvmuA/cfYwhrDtTBhsmukO4U2huVO5W+86XacUMRvdSFVAc6kZUZy21JaRwiB9w==} + cpu: [x64] + os: [linux] + libc: [musl] + + '@rollup/rollup-openbsd-x64@4.63.4': + resolution: {integrity: sha512-dVHFp9gRWrdTpnqQuGfCwd7hOQDatK1VCP2iWhLY/cGrOQs/ucFzJ6A5SRqbXX12ZDI8EUuejSM5kwg+ja7Png==} + cpu: [x64] + os: [openbsd] + + '@rollup/rollup-openharmony-arm64@4.63.4': + resolution: {integrity: sha512-t3NlauOW6gxZVVFcBEnO62Cb4wbyDFL416gTg1uFI/2tgqYQlf69FbSE115Ajre9I+c26Lk4mcmdFUsS/DGifQ==} + cpu: [arm64] + os: [openharmony] + + '@rollup/rollup-win32-arm64-msvc@4.63.4': + resolution: {integrity: sha512-xWuIaSye5FWZF8+UYtVEcHtRJDN5kN9Kfgxx3Kq8XIov9KSKbc1fiqQCm90SKrgQbUXZelbnUhnlUJmfSE7P9A==} + cpu: [arm64] + os: [win32] + + '@rollup/rollup-win32-ia32-msvc@4.63.4': + resolution: {integrity: sha512-9ALJJUOg/ZflMJepVo2PlgsGxSaxN7SQ4Z8GoZfVlarWr6r3rkHUNsd/zAio7p4YMtChSMXPionxej4Hkf6CXQ==} + cpu: [ia32] + os: [win32] + + '@rollup/rollup-win32-x64-gnu@4.63.4': + resolution: {integrity: sha512-blj9z5qx/Pv4WU0W1NMFDB97e0JH5ed+aZGywW8WCvp/NhWX/4PFAq5uu6Q0AebNn+Vo6KzUYDT++JzTT5ojlQ==} + cpu: [x64] + os: [win32] + + '@rollup/rollup-win32-x64-msvc@4.63.4': + resolution: {integrity: sha512-Erx822VRBwLa124shbj+wNXe//BOgMEctDV0m1aqTQdNO1S69DgNUCFKC1RCeZfixs1J31l6igk1ziyXErbigQ==} + cpu: [x64] + os: [win32] + + '@types/chai@5.2.3': + resolution: {integrity: sha512-Mw558oeA9fFbv65/y4mHtXDs9bPnFMZAL/jxdPFUpOHHIXX91mcgEHbS5Lahr+pwZFR8A7GQleRWeI6cGFC2UA==} + + '@types/deep-eql@4.0.2': + resolution: {integrity: sha512-c9h9dVVMigMPc4bwTvC5dxqtqJZwQPePsWjPlpSOnojbor6pGqdk541lfA7AqFQr5pB1BRdq0juY9db81BwyFw==} + + '@types/estree@1.0.9': + resolution: {integrity: sha512-GhdPgy1el4/ImP05X05Uw4cw2/M93BCUmnEvWZNStlCzEKME4Fkk+YpoA5OiHNQmoS7Cafb8Xa3Pya8m1Qrzeg==} + + '@types/node@22.20.4': + resolution: {integrity: sha512-zJRE40jpHtKqE/C4fgHrAKQLJuSpzEnP9ff9Y7YtoR3Wd2pwqzlekDeEuUQXjRd+QCYnVnNwuJYmhdk9XV8gvA==} + + '@vitest/expect@3.2.7': + resolution: {integrity: sha512-E8eBXaKibuvH2pSZErOjdVb5vF4PbKYcrnluBTYxEk1l/VhhwZg1kZQsdtjq+CsF5CFydf2Rdkz7jDHKSisi3w==} + + '@vitest/mocker@3.2.7': + resolution: {integrity: sha512-Trr0hYO9CM3Wj6ksWHRhK9IZpIY6wTMO5u/MqXurMxT57sWBaOPEtP3Oq60ihZuh5JsiagKfz95OcxdEP6dBrA==} + peerDependencies: + msw: ^2.4.9 + vite: ^5.0.0 || ^6.0.0 || ^7.0.0-0 + peerDependenciesMeta: + msw: + optional: true + vite: + optional: true + + '@vitest/pretty-format@3.2.7': + resolution: {integrity: sha512-KUHlwqVu0sRlhCdyPdQ/wBoTfRahjUky1MubOmYw9fWfIZy1gNoHpuaaQBPAaMaVYdQYHJLurzj8ECCj5OwTqA==} + + '@vitest/runner@3.2.7': + resolution: {integrity: sha512-sB9y4ovltoQP+WaUPwmSxO9WIg9Ig694Di5PalVPsYHklAdE027mehpWF2SQSVq+k6sFgaivbTjTJwZLSHbedA==} + + '@vitest/snapshot@3.2.7': + resolution: {integrity: sha512-7C+MwShwtBSI5Buwoyg3s/iY1eHL9PKAf+O1wVh/TdnjXUtkoL/9YQtre90i4MtNXM6edP1wJ2zOBpfCyhIS7g==} + + '@vitest/spy@3.2.7': + resolution: {integrity: sha512-Q2eQGI6d2L/hBtZ0qNuKcAGid68XK6cv1xsoaIma6PaJhHPoqcEJhYpXZ/5myCMqkNgtP6UKuBhbc0nHKnrkuQ==} + + '@vitest/utils@3.2.7': + resolution: {integrity: sha512-x6BDOd7dyo3PFLY3I9/HJ25X/6OurhGXk2/B9gOZNPF7XDVjeBK4k01lQE5uvDpbuheErh91qYuE1E2OEjK3Rw==} + + ansi-regex@5.0.1: + resolution: {integrity: sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ==} + engines: {node: '>=8'} + + ansi-styles@4.3.0: + resolution: {integrity: sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg==} + engines: {node: '>=8'} + + assertion-error@2.0.1: + resolution: {integrity: sha512-Izi8RQcffqCeNVgFigKli1ssklIbpHnCYc6AknXGYoB6grJqyeby7jv12JUQgmTAnIDnbck1uxksT4dzN3PWBA==} + engines: {node: '>=12'} + + cac@6.7.14: + resolution: {integrity: sha512-b6Ilus+c3RrdDk+JhLKUAQfzzgLEPy6wcXqS7f/xe1EETvsDP6GORG7SFuOs6cID5YkqchW/LXZbX5bc8j7ZcQ==} + engines: {node: '>=8'} + + chai@5.3.3: + resolution: {integrity: sha512-4zNhdJD/iOjSH0A05ea+Ke6MU5mmpQcbQsSOkgdaUMJ9zTlDTD/GYlwohmIE2u0gaxHYiVHEn1Fw9mZ/ktJWgw==} + engines: {node: '>=18'} + + check-error@2.1.3: + resolution: {integrity: sha512-PAJdDJusoxnwm1VwW07VWwUN1sl7smmC3OKggvndJFadxxDRyFJBX/ggnu/KE4kQAB7a3Dp8f/YXC1FlUprWmA==} + engines: {node: '>= 16'} + + cliui@8.0.1: + resolution: {integrity: sha512-BSeNnyus75C4//NQ9gQt1/csTXyo/8Sb+afLAkzAptFuMsod9HFokGNudZpi/oQV73hnVK+sR+5PVRMd+Dr7YQ==} + engines: {node: '>=12'} + + color-convert@2.0.1: + resolution: {integrity: sha512-RRECPsj7iu/xb5oKYcsFHSppFNnsj/52OVTRKb4zP5onXwVF3zVmmToNcOfGC+CRDpfK/U584fMg38ZHCaElKQ==} + engines: {node: '>=7.0.0'} + + color-name@1.1.4: + resolution: {integrity: sha512-dOy+3AuW3a2wNbZHIuMZpTcgjGuLU/uBL/ubcZF9OXbDo8ff4O8yVp5Bf0efS8uEoYo5q4Fx7dY9OgQGXgAsQA==} + + debug@4.4.3: + resolution: {integrity: sha512-RGwwWnwQvkVfavKVt22FGLw+xYSdzARwm0ru6DhTVA3umU5hZc28V3kO4stgYryrTlLpuvgI9GiijltAjNbcqA==} + engines: {node: '>=6.0'} + peerDependencies: + supports-color: '*' + peerDependenciesMeta: + supports-color: + optional: true + + deep-eql@5.0.2: + resolution: {integrity: sha512-h5k/5U50IJJFpzfL6nO9jaaumfjO/f2NjK/oYB2Djzm4p9L+3T9qWpZqZ2hAbLPuuYq9wrU08WQyBTL5GbPk5Q==} + engines: {node: '>=6'} + + detect-libc@2.1.2: + resolution: {integrity: sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ==} + engines: {node: '>=8'} + + effect@4.0.0-rc.112: + resolution: {integrity: sha512-wXxwuh1Ywnv4cPRM3Wfa0vDwuOHnZ1TsTgHJkG9XgzND6inhBH9n1vBxhg3iIXOia/OrpmvVmd3lrD4vq6bF3A==} + + emoji-regex@8.0.0: + resolution: {integrity: sha512-MSjYzcWNOA0ewAHpz0MxpYFvwg6yjy1NG3xteoqz644VCo/RPgnr1/GGt+ic3iJTzQ8Eu3TdM14SawnVUmGE6A==} + + es-module-lexer@1.7.0: + resolution: {integrity: sha512-jEQoCwk8hyb2AZziIOLhDqpm5+2ww5uIE6lkO/6jcOCusfk6LhMHpXXfBLXTZ7Ydyt0j4VoUQv6uGNYbdW+kBA==} + + esbuild@0.28.2: + resolution: {integrity: sha512-HKVLS8dvII+xoKW9kmqxbRKrnWEXfJJr/FZhhJmiqIB0e053QNYFqOBouTMO/k5sID4MvCiUCvv8b9M4h32wIA==} + engines: {node: '>=18'} + hasBin: true + + escalade@3.2.0: + resolution: {integrity: sha512-WUj2qlxaQtO4g6Pq5c29GTcWGDyd8itL8zTlipgECz3JesAiiOKotd8JU6otB3PACgG6xkJUyVhboMS+bje/jA==} + engines: {node: '>=6'} + + estree-walker@3.0.3: + resolution: {integrity: sha512-7RUKfXgSMMkzt6ZuXmqapOurLGPPfgj6l9uRZ7lRGolvk0y2yocc35LdcxKC5PQZdn2DMqioAQ2NoWcrTKmm6g==} + + expect-type@1.4.0: + resolution: {integrity: sha512-KfYbmpRm0VbLjEvVa9yGwCi9GI34xvi7A/HXYWQO65CSD2u3MczUJSuwXKFIxlGsgBQizV9q5J9NHj4VG0n+pA==} + engines: {node: '>=12.0.0'} + + fast-check@4.10.2: + resolution: {integrity: sha512-iK2f+YrcmoeGqk6fA0ea2bptcu/itMIm4NfEozq6N25+aG6h7s5HZbB/k1aV7b5w5sFLMCbbtRUsTVR+BgC3xw==} + engines: {node: '>=12.17.0'} + + fdir@6.5.0: + resolution: {integrity: sha512-tIbYtZbucOs0BRGqPJkshJUYdL+SDH7dVM8gjy+ERp3WAUjLEFJE+02kanyHtwjWOnwrKYBiwAmM0p4kLJAnXg==} + engines: {node: '>=12.0.0'} + peerDependencies: + picomatch: ^3 || ^4 + peerDependenciesMeta: + picomatch: + optional: true + + fsevents@2.3.3: + resolution: {integrity: sha512-5xoDfX+fL7faATnagmWPpbFtwh/R77WmMMqqHGS65C3vvB0YHrgF+B1YmZ3441tMj5n63k0212XNoJwzlhffQw==} + engines: {node: ^8.16.0 || ^10.6.0 || >=11.0.0} + os: [darwin] + + get-caller-file@2.0.5: + resolution: {integrity: sha512-DyFP3BM/3YHTQOCUL/w0OZHR0lpKeGrxotcHWcqNEdnltqFwXVfhEBQ94eIo34AfQpo0rGki4cyIiftY06h2Fg==} + engines: {node: 6.* || 8.* || >= 10.*} + + is-fullwidth-code-point@3.0.0: + resolution: {integrity: sha512-zymm5+u+sCsSWyD9qNaejV3DFvhCKclKdizYaJUuHA83RLjb7nSuGnddCHGv0hk+KY7BMAlsWeK4Ueg6EV6XQg==} + engines: {node: '>=8'} + + js-tokens@9.0.1: + resolution: {integrity: sha512-mxa9E9ITFOt0ban3j6L5MpjwegGz6lBQmM1IJkWeBZGcMxto50+eWdjC/52xDbS2vy0k7vIMK0Fe2wfL9OQSpQ==} + + lodash.camelcase@4.3.0: + resolution: {integrity: sha512-TwuEnCnxbc3rAvhf/LbG7tJUDzhqXyFnv3dtzLOPgCG/hODL7WFnsbwktkD7yUV0RrreP/l1PALq/YSg6VvjlA==} + + long@5.3.2: + resolution: {integrity: sha512-mNAgZ1GmyNhD7AuqnTG3/VQ26o760+ZYBPKjPvugO8+nLbYfX6TVpJPseBvopbdY+qpZ/lKUnmEc1LeZYS3QAA==} + + loupe@3.2.1: + resolution: {integrity: sha512-CdzqowRJCeLU72bHvWqwRBBlLcMEtIvGrlvef74kMnV2AolS9Y8xUv1I0U/MNAWMhBlKIoyuEgoJ0t/bbwHbLQ==} + + magic-string@0.30.21: + resolution: {integrity: sha512-vd2F4YUyEXKGcLHoq+TEyCjxueSeHnFxyyjNp80yg0XV4vUhnDer/lvvlqM/arB5bXQN5K2/3oinyCRyx8T2CQ==} + + ms@2.1.3: + resolution: {integrity: sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==} + + msgpackr-extract@3.0.4: + resolution: {integrity: sha512-4kmO/MdyUIkLIvTPr8VHLil4AtoKIoniWPIEk5+CDy0xnWC84azhSFmuJ7PxZdsYtiP5kEeQsORAVIeMgxT+Hw==} + hasBin: true + + msgpackr@2.1.0: + resolution: {integrity: sha512-p/pBCVO63CsvvpkomUnNNag6+n38rULuDA6HHe70o2gtC8ODI52foF/4ko2qQcp6OiErJXTmrZeXmsGGHsIQNQ==} + + nanoid@3.3.19: + resolution: {integrity: sha512-Y2tUNy4ouw6tq5oDSKeQYGOyhkUBhNOcGV/02KC+6kd9eDGqdZd++mjMiIDilrBYvjEnCYvVtsuHCuP+okSfug==} + engines: {node: ^10 || ^12 || ^13.7 || ^14 || >=15.0.1} + hasBin: true + + node-gyp-build-optional-packages@5.2.2: + resolution: {integrity: sha512-s+w+rBWnpTMwSFbaE0UXsRlg7hU4FjekKU4eyAih5T8nJuNZT1nNsskXpxmeqSK9UzkBl6UgRlnKc8hz8IEqOw==} + hasBin: true + + pathe@2.0.3: + resolution: {integrity: sha512-WUjGcAqP1gQacoQe+OBJsFA7Ld4DyXuUIjZ5cc75cLHvJ7dtNsTugphxIADwspS+AraAUePCKrSVtPLFj/F88w==} + + pathval@2.0.1: + resolution: {integrity: sha512-//nshmD55c46FuFw26xV/xFAaB5HF9Xdap7HJBBnrKdAd6/GxDBaNA1870O79+9ueg61cZLSVc+OaFlfmObYVQ==} + engines: {node: '>= 14.16'} + + picocolors@1.1.1: + resolution: {integrity: sha512-xceH2snhtb5M9liqDsmEw56le376mTZkEX/jEb/RxNFyegNul7eNslCXP9FDj/Lcu0X8KEyMceP2ntpaHrDEVA==} + + picomatch@4.0.7: + resolution: {integrity: sha512-qcJu88Q2IWqJsDD529JKMdwGm/dvInW4HvQnRwiH9JtihJvzGOscDtHE3x1pBKeUOTysQ8kVmLnJ2kJu7yhcGA==} + engines: {node: '>=12'} + + postcss@8.5.28: + resolution: {integrity: sha512-RRuzqDtt5Y9h3quz5hWhK+TPnsmVs6WwSU6LkJMeY4HstUEDuYTG8UJSdawMRzmzAtV+KEoG8N3Qg2qLy5vM/A==} + engines: {node: ^10 || ^12 || >=14} + + protobufjs@7.6.6: + resolution: {integrity: sha512-dYDWdjSl5RNb7SgPxGQcRU+GtvP7s2fpkrY0r432PcOIaZ0/rBcxEZnQN67iJhFuQiVw754JDoPruPCNdGsbjg==} + engines: {node: '>=12.0.0'} + + pure-rand@8.4.2: + resolution: {integrity: sha512-vvuOGgcuPJAirlHvuQw1TrOiw7ptaIXXmIbNuiNOY6lNGJJH49PQ1Kj4nd783nPdQhQdicgOjVI2yI/9BD6/Ng==} + + require-directory@2.1.1: + resolution: {integrity: sha512-fGxEI7+wsG9xrvdjsrlmL22OMTTiHRwAMroiEeMgq8gzoLC/PQr7RsRDSTLUg/bZAZtF+TVIkHc6/4RIKrui+Q==} + engines: {node: '>=0.10.0'} + + rollup@4.63.4: + resolution: {integrity: sha512-4U0liVayNIoLp3GFl1FcI8561WepLnZ1rqfraGh7S9B3Ur5F9S283y8Futii7RUU2C/97tOBmBy7nYvhoiOpbQ==} + engines: {node: '>=18.0.0', npm: '>=8.0.0'} + hasBin: true + + siginfo@2.0.0: + resolution: {integrity: sha512-ybx0WO1/8bSBLEWXZvEd7gMW3Sn3JFlW3TvX1nREbDLRNQNaeNN8WK0meBwPdAaOI7TtRRRJn/Es1zhrrCHu7g==} + + source-map-js@1.2.1: + resolution: {integrity: sha512-UXWMKhLOwVKb728IUtQPXxfYU+usdybtUrK/8uGE8CQMvrhOpwvzDBwj0QhSL7MQc7vIsISBG8VQ8+IDQxpfQA==} + engines: {node: '>=0.10.0'} + + stackback@0.0.2: + resolution: {integrity: sha512-1XMJE5fQo1jGH6Y/7ebnwPOBEkIEnT4QF32d5R1+VXdXveM0IBMJt8zfaxX1P3QhVwrYe+576+jkANtSS2mBbw==} + + std-env@3.10.0: + resolution: {integrity: sha512-5GS12FdOZNliM5mAOxFRg7Ir0pWz8MdpYm6AY6VPkGpbA7ZzmbzNcBJQ0GPvvyWgcY7QAhCgf9Uy89I03faLkg==} + + string-width@4.2.3: + resolution: {integrity: sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g==} + engines: {node: '>=8'} + + strip-ansi@6.0.1: + resolution: {integrity: sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A==} + engines: {node: '>=8'} + + strip-literal@3.1.0: + resolution: {integrity: sha512-8r3mkIM/2+PpjHoOtiAW8Rg3jJLHaV7xPwG+YRGrv6FP0wwk/toTpATxWYOW0BKdWwl82VT2tFYi5DlROa0Mxg==} + + tinybench@2.9.0: + resolution: {integrity: sha512-0+DUvqWMValLmha6lr4kD8iAMK1HzV0/aKnCtWb9v9641TnP/MFb7Pc2bxoxQjTXAErryXVgUOfv2YqNllqGeg==} + + tinyexec@0.3.2: + resolution: {integrity: sha512-KQQR9yN7R5+OSwaK0XQoj22pwHoTlgYqmUscPYoknOoWCWfj/5/ABTMRi69FrKU5ffPVh5QcFikpWJI/P1ocHA==} + + tinyglobby@0.2.17: + resolution: {integrity: sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g==} + engines: {node: '>=12.0.0'} + + tinypool@1.1.1: + resolution: {integrity: sha512-Zba82s87IFq9A9XmjiX5uZA/ARWDrB03OHlq+Vw1fSdt0I+4/Kutwy8BP4Y/y/aORMo61FQ0vIb5j44vSo5Pkg==} + engines: {node: ^18.0.0 || >=20.0.0} + + tinyrainbow@2.0.0: + resolution: {integrity: sha512-op4nsTR47R6p0vMUUoYl/a+ljLFVtlfaXkLQmqfLR1qHma1h/ysYk4hEXZ880bf2CYgTskvTa/e196Vd5dDQXw==} + engines: {node: '>=14.0.0'} + + tinyspy@4.0.6: + resolution: {integrity: sha512-u8KszXvGfU68hVcZpRHKG28T0krMuv2G5nDhiHaMLen/gIuFEgIJhaJuO69qjnXg5paSrbPMFfx3brNuN8eVSg==} + engines: {node: '>=14.0.0'} + + tsx@4.23.15: + resolution: {integrity: sha512-Yiex1Ovn8z2xPpOWckIiysV1SSyRMY9BkLF++q0yKiDxCqRhosKfMg3janKkiLBwZ5c/YryloKwGZcrEmtwxKw==} + engines: {node: '>=18.0.0'} + hasBin: true + + typescript@5.9.3: + resolution: {integrity: sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==} + engines: {node: '>=14.17'} + hasBin: true + + undici-types@6.21.0: + resolution: {integrity: sha512-iwDZqg0QAGrg9Rav5H4n0M64c3mkR59cJ6wQp+7C4nI0gsmExaedaYLNO44eT4AtBBwjbTiGPMlt2Md0T9H9JQ==} + + vite-node@3.2.4: + resolution: {integrity: sha512-EbKSKh+bh1E1IFxeO0pg1n4dvoOTt0UDiXMd/qn++r98+jPO1xtJilvXldeuQ8giIB5IkpjCgMleHMNEsGH6pg==} + engines: {node: ^18.0.0 || ^20.0.0 || >=22.0.0} + hasBin: true + + vite@7.3.6: + resolution: {integrity: sha512-4XP60spRGjSZFf1qYH+dJIkK2znL3zQfl9KkOV9MkkRR/3Dls0dxaBsQPTloEc5BLXWPL9vsOxopxyKoMmDueg==} + engines: {node: ^20.19.0 || >=22.12.0} + hasBin: true + peerDependencies: + '@types/node': ^20.19.0 || >=22.12.0 + jiti: '>=1.21.0' + less: ^4.0.0 + lightningcss: ^1.21.0 + sass: ^1.70.0 + sass-embedded: ^1.70.0 + stylus: '>=0.54.8' + sugarss: ^5.0.0 + terser: ^5.16.0 + tsx: ^4.8.1 + yaml: ^2.4.2 + peerDependenciesMeta: + '@types/node': + optional: true + jiti: + optional: true + less: + optional: true + lightningcss: + optional: true + sass: + optional: true + sass-embedded: + optional: true + stylus: + optional: true + sugarss: + optional: true + terser: + optional: true + tsx: + optional: true + yaml: + optional: true + + vitest@3.2.7: + resolution: {integrity: sha512-KrxIJ62Fd89gfysR4WotlgZABiz2dqFPgqGzX7s+CwsqLFomRH7777ZcrOD6+WVAh7khPQP41A+BKbpcJFrdEg==} + engines: {node: ^18.0.0 || ^20.0.0 || >=22.0.0} + hasBin: true + peerDependencies: + '@edge-runtime/vm': '*' + '@types/debug': ^4.1.12 + '@types/node': ^18.0.0 || ^20.0.0 || >=22.0.0 + '@vitest/browser': 3.2.7 + '@vitest/ui': 3.2.7 + happy-dom: '*' + jsdom: '*' + peerDependenciesMeta: + '@edge-runtime/vm': + optional: true + '@types/debug': + optional: true + '@types/node': + optional: true + '@vitest/browser': + optional: true + '@vitest/ui': + optional: true + happy-dom: + optional: true + jsdom: + optional: true + + why-is-node-running@2.3.0: + resolution: {integrity: sha512-hUrmaWBdVDcxvYqnyh09zunKzROWjbZTiNy8dBEjkS7ehEDQibXJ7XvlmtbwuTclUiIyN+CyXQD4Vmko8fNm8w==} + engines: {node: '>=8'} + hasBin: true + + wrap-ansi@7.0.0: + resolution: {integrity: sha512-YVGIj2kamLSTxw6NsZjoBxfSwsn0ycdesmc4p+Q21c5zPuZ1pl+NfxVdxPtdHvmNVOQ6XSYG4AUtyt/Fi7D16Q==} + engines: {node: '>=10'} + + y18n@5.0.8: + resolution: {integrity: sha512-0pfFzegeDWJHJIAmTLRP2DwHjdF5s7jo9tuztdQxAhINCdvS+3nGINqPd00AphqJR/0LhANUS6/+7SCb98YOfA==} + engines: {node: '>=10'} + + yargs-parser@21.1.1: + resolution: {integrity: sha512-tVpsJW7DdjecAiFpbIB1e3qxIQsE6NoPc5/eTdrbbIC4h0LVsWhnoa3g+m2HclBIujHzsxZ4VJVA+GUuc2/LBw==} + engines: {node: '>=12'} + + yargs@17.7.3: + resolution: {integrity: sha512-GZtjxm/J/4TSxuL3FNYjCmLktBTnIw/rVmKSIyKeYAZpmJB2ig9VauCC5xsa82GNKVKDAqpOn3KVzNt0zmrU0g==} + engines: {node: '>=12'} + +snapshots: + + '@esbuild/aix-ppc64@0.28.2': + optional: true + + '@esbuild/android-arm64@0.28.2': + optional: true + + '@esbuild/android-arm@0.28.2': + optional: true + + '@esbuild/android-x64@0.28.2': + optional: true + + '@esbuild/darwin-arm64@0.28.2': + optional: true + + '@esbuild/darwin-x64@0.28.2': + optional: true + + '@esbuild/freebsd-arm64@0.28.2': + optional: true + + '@esbuild/freebsd-x64@0.28.2': + optional: true + + '@esbuild/linux-arm64@0.28.2': + optional: true + + '@esbuild/linux-arm@0.28.2': + optional: true + + '@esbuild/linux-ia32@0.28.2': + optional: true + + '@esbuild/linux-loong64@0.28.2': + optional: true + + '@esbuild/linux-mips64el@0.28.2': + optional: true + + '@esbuild/linux-ppc64@0.28.2': + optional: true + + '@esbuild/linux-riscv64@0.28.2': + optional: true + + '@esbuild/linux-s390x@0.28.2': + optional: true + + '@esbuild/linux-x64@0.28.2': + optional: true + + '@esbuild/netbsd-arm64@0.28.2': + optional: true + + '@esbuild/netbsd-x64@0.28.2': + optional: true + + '@esbuild/openbsd-arm64@0.28.2': + optional: true + + '@esbuild/openbsd-x64@0.28.2': + optional: true + + '@esbuild/openharmony-arm64@0.28.2': + optional: true + + '@esbuild/sunos-x64@0.28.2': + optional: true + + '@esbuild/win32-arm64@0.28.2': + optional: true + + '@esbuild/win32-ia32@0.28.2': + optional: true + + '@esbuild/win32-x64@0.28.2': + optional: true + + '@grpc/grpc-js@1.14.5': + dependencies: + '@grpc/proto-loader': 0.8.1 + '@js-sdsl/ordered-map': 4.4.2 + + '@grpc/proto-loader@0.7.15': + dependencies: + lodash.camelcase: 4.3.0 + long: 5.3.2 + protobufjs: 7.6.6 + yargs: 17.7.3 + + '@grpc/proto-loader@0.8.1': + dependencies: + lodash.camelcase: 4.3.0 + long: 5.3.2 + protobufjs: 7.6.6 + yargs: 17.7.3 + + '@jridgewell/sourcemap-codec@1.6.0': {} + + '@js-sdsl/ordered-map@4.4.2': {} + + '@msgpackr-extract/msgpackr-extract-darwin-arm64@3.0.4': + optional: true + + '@msgpackr-extract/msgpackr-extract-darwin-x64@3.0.4': + optional: true + + '@msgpackr-extract/msgpackr-extract-linux-arm64@3.0.4': + optional: true + + '@msgpackr-extract/msgpackr-extract-linux-arm@3.0.4': + optional: true + + '@msgpackr-extract/msgpackr-extract-linux-x64@3.0.4': + optional: true + + '@msgpackr-extract/msgpackr-extract-win32-x64@3.0.4': + optional: true + + '@napi-rs/lzma-linux-x64-gnu@1.5.1': + optional: true + + '@protobufjs/aspromise@1.1.2': {} + + '@protobufjs/base64@1.1.2': {} + + '@protobufjs/codegen@2.0.5': {} + + '@protobufjs/eventemitter@1.1.1': {} + + '@protobufjs/fetch@1.1.1': + dependencies: + '@protobufjs/aspromise': 1.1.2 + + '@protobufjs/float@1.0.2': {} + + '@protobufjs/path@1.1.2': {} + + '@protobufjs/pool@1.1.0': {} + + '@protobufjs/utf8@1.1.2': {} + + '@rollup/rollup-android-arm-eabi@4.63.4': + optional: true + + '@rollup/rollup-android-arm64@4.63.4': + optional: true + + '@rollup/rollup-darwin-arm64@4.63.4': + optional: true + + '@rollup/rollup-darwin-x64@4.63.4': + optional: true + + '@rollup/rollup-freebsd-arm64@4.63.4': + optional: true + + '@rollup/rollup-freebsd-x64@4.63.4': + optional: true + + '@rollup/rollup-linux-arm-gnueabihf@4.63.4': + optional: true + + '@rollup/rollup-linux-arm-musleabihf@4.63.4': + optional: true + + '@rollup/rollup-linux-arm64-gnu@4.63.4': + optional: true + + '@rollup/rollup-linux-arm64-musl@4.63.4': + optional: true + + '@rollup/rollup-linux-loong64-gnu@4.63.4': + optional: true + + '@rollup/rollup-linux-loong64-musl@4.63.4': + optional: true + + '@rollup/rollup-linux-ppc64-gnu@4.63.4': + optional: true + + '@rollup/rollup-linux-ppc64-musl@4.63.4': + optional: true + + '@rollup/rollup-linux-riscv64-gnu@4.63.4': + optional: true + + '@rollup/rollup-linux-riscv64-musl@4.63.4': + optional: true + + '@rollup/rollup-linux-s390x-gnu@4.63.4': + optional: true + + '@rollup/rollup-linux-x64-gnu@4.63.4': + optional: true + + '@rollup/rollup-linux-x64-musl@4.63.4': + optional: true + + '@rollup/rollup-openbsd-x64@4.63.4': + optional: true + + '@rollup/rollup-openharmony-arm64@4.63.4': + optional: true + + '@rollup/rollup-win32-arm64-msvc@4.63.4': + optional: true + + '@rollup/rollup-win32-ia32-msvc@4.63.4': + optional: true + + '@rollup/rollup-win32-x64-gnu@4.63.4': + optional: true + + '@rollup/rollup-win32-x64-msvc@4.63.4': + optional: true + + '@types/chai@5.2.3': + dependencies: + '@types/deep-eql': 4.0.2 + assertion-error: 2.0.1 + + '@types/deep-eql@4.0.2': {} + + '@types/estree@1.0.9': {} + + '@types/node@22.20.4': + dependencies: + undici-types: 6.21.0 + + '@vitest/expect@3.2.7': + dependencies: + '@types/chai': 5.2.3 + '@vitest/spy': 3.2.7 + '@vitest/utils': 3.2.7 + chai: 5.3.3 + tinyrainbow: 2.0.0 + + '@vitest/mocker@3.2.7(vite@7.3.6(@types/node@22.20.4)(tsx@4.23.15))': + dependencies: + '@vitest/spy': 3.2.7 + estree-walker: 3.0.3 + magic-string: 0.30.21 + optionalDependencies: + vite: 7.3.6(@types/node@22.20.4)(tsx@4.23.15) + + '@vitest/pretty-format@3.2.7': + dependencies: + tinyrainbow: 2.0.0 + + '@vitest/runner@3.2.7': + dependencies: + '@vitest/utils': 3.2.7 + pathe: 2.0.3 + strip-literal: 3.1.0 + + '@vitest/snapshot@3.2.7': + dependencies: + '@vitest/pretty-format': 3.2.7 + magic-string: 0.30.21 + pathe: 2.0.3 + + '@vitest/spy@3.2.7': + dependencies: + tinyspy: 4.0.6 + + '@vitest/utils@3.2.7': + dependencies: + '@vitest/pretty-format': 3.2.7 + loupe: 3.2.1 + tinyrainbow: 2.0.0 + + ansi-regex@5.0.1: {} + + ansi-styles@4.3.0: + dependencies: + color-convert: 2.0.1 + + assertion-error@2.0.1: {} + + cac@6.7.14: {} + + chai@5.3.3: + dependencies: + assertion-error: 2.0.1 + check-error: 2.1.3 + deep-eql: 5.0.2 + loupe: 3.2.1 + pathval: 2.0.1 + + check-error@2.1.3: {} + + cliui@8.0.1: + dependencies: + string-width: 4.2.3 + strip-ansi: 6.0.1 + wrap-ansi: 7.0.0 + + color-convert@2.0.1: + dependencies: + color-name: 1.1.4 + + color-name@1.1.4: {} + + debug@4.4.3: + dependencies: + ms: 2.1.3 + + deep-eql@5.0.2: {} + + detect-libc@2.1.2: + optional: true + + effect@4.0.0-rc.112: + dependencies: + fast-check: 4.10.2 + msgpackr: 2.1.0 + + emoji-regex@8.0.0: {} + + es-module-lexer@1.7.0: {} + + esbuild@0.28.2: + optionalDependencies: + '@esbuild/aix-ppc64': 0.28.2 + '@esbuild/android-arm': 0.28.2 + '@esbuild/android-arm64': 0.28.2 + '@esbuild/android-x64': 0.28.2 + '@esbuild/darwin-arm64': 0.28.2 + '@esbuild/darwin-x64': 0.28.2 + '@esbuild/freebsd-arm64': 0.28.2 + '@esbuild/freebsd-x64': 0.28.2 + '@esbuild/linux-arm': 0.28.2 + '@esbuild/linux-arm64': 0.28.2 + '@esbuild/linux-ia32': 0.28.2 + '@esbuild/linux-loong64': 0.28.2 + '@esbuild/linux-mips64el': 0.28.2 + '@esbuild/linux-ppc64': 0.28.2 + '@esbuild/linux-riscv64': 0.28.2 + '@esbuild/linux-s390x': 0.28.2 + '@esbuild/linux-x64': 0.28.2 + '@esbuild/netbsd-arm64': 0.28.2 + '@esbuild/netbsd-x64': 0.28.2 + '@esbuild/openbsd-arm64': 0.28.2 + '@esbuild/openbsd-x64': 0.28.2 + '@esbuild/openharmony-arm64': 0.28.2 + '@esbuild/sunos-x64': 0.28.2 + '@esbuild/win32-arm64': 0.28.2 + '@esbuild/win32-ia32': 0.28.2 + '@esbuild/win32-x64': 0.28.2 + + escalade@3.2.0: {} + + estree-walker@3.0.3: + dependencies: + '@types/estree': 1.0.9 + + expect-type@1.4.0: {} + + fast-check@4.10.2: + dependencies: + pure-rand: 8.4.2 + + fdir@6.5.0(picomatch@4.0.7): + optionalDependencies: + picomatch: 4.0.7 + + fsevents@2.3.3: + optional: true + + get-caller-file@2.0.5: {} + + is-fullwidth-code-point@3.0.0: {} + + js-tokens@9.0.1: {} + + lodash.camelcase@4.3.0: {} + + long@5.3.2: {} + + loupe@3.2.1: {} + + magic-string@0.30.21: + dependencies: + '@jridgewell/sourcemap-codec': 1.6.0 + + ms@2.1.3: {} + + msgpackr-extract@3.0.4: + dependencies: + node-gyp-build-optional-packages: 5.2.2 + optionalDependencies: + '@msgpackr-extract/msgpackr-extract-darwin-arm64': 3.0.4 + '@msgpackr-extract/msgpackr-extract-darwin-x64': 3.0.4 + '@msgpackr-extract/msgpackr-extract-linux-arm': 3.0.4 + '@msgpackr-extract/msgpackr-extract-linux-arm64': 3.0.4 + '@msgpackr-extract/msgpackr-extract-linux-x64': 3.0.4 + '@msgpackr-extract/msgpackr-extract-win32-x64': 3.0.4 + optional: true + + msgpackr@2.1.0: + optionalDependencies: + msgpackr-extract: 3.0.4 + + nanoid@3.3.19: {} + + node-gyp-build-optional-packages@5.2.2: + dependencies: + detect-libc: 2.1.2 + optional: true + + pathe@2.0.3: {} + + pathval@2.0.1: {} + + picocolors@1.1.1: {} + + picomatch@4.0.7: {} + + postcss@8.5.28: + dependencies: + nanoid: 3.3.19 + picocolors: 1.1.1 + source-map-js: 1.2.1 + + protobufjs@7.6.6: + dependencies: + '@protobufjs/aspromise': 1.1.2 + '@protobufjs/base64': 1.1.2 + '@protobufjs/codegen': 2.0.5 + '@protobufjs/eventemitter': 1.1.1 + '@protobufjs/fetch': 1.1.1 + '@protobufjs/float': 1.0.2 + '@protobufjs/path': 1.1.2 + '@protobufjs/pool': 1.1.0 + '@protobufjs/utf8': 1.1.2 + '@types/node': 22.20.4 + long: 5.3.2 + + pure-rand@8.4.2: {} + + require-directory@2.1.1: {} + + rollup@4.63.4: + dependencies: + '@types/estree': 1.0.9 + optionalDependencies: + '@napi-rs/lzma-linux-x64-gnu': 1.5.1 + '@rollup/rollup-android-arm-eabi': 4.63.4 + '@rollup/rollup-android-arm64': 4.63.4 + '@rollup/rollup-darwin-arm64': 4.63.4 + '@rollup/rollup-darwin-x64': 4.63.4 + '@rollup/rollup-freebsd-arm64': 4.63.4 + '@rollup/rollup-freebsd-x64': 4.63.4 + '@rollup/rollup-linux-arm-gnueabihf': 4.63.4 + '@rollup/rollup-linux-arm-musleabihf': 4.63.4 + '@rollup/rollup-linux-arm64-gnu': 4.63.4 + '@rollup/rollup-linux-arm64-musl': 4.63.4 + '@rollup/rollup-linux-loong64-gnu': 4.63.4 + '@rollup/rollup-linux-loong64-musl': 4.63.4 + '@rollup/rollup-linux-ppc64-gnu': 4.63.4 + '@rollup/rollup-linux-ppc64-musl': 4.63.4 + '@rollup/rollup-linux-riscv64-gnu': 4.63.4 + '@rollup/rollup-linux-riscv64-musl': 4.63.4 + '@rollup/rollup-linux-s390x-gnu': 4.63.4 + '@rollup/rollup-linux-x64-gnu': 4.63.4 + '@rollup/rollup-linux-x64-musl': 4.63.4 + '@rollup/rollup-openbsd-x64': 4.63.4 + '@rollup/rollup-openharmony-arm64': 4.63.4 + '@rollup/rollup-win32-arm64-msvc': 4.63.4 + '@rollup/rollup-win32-ia32-msvc': 4.63.4 + '@rollup/rollup-win32-x64-gnu': 4.63.4 + '@rollup/rollup-win32-x64-msvc': 4.63.4 + fsevents: 2.3.3 + + siginfo@2.0.0: {} + + source-map-js@1.2.1: {} + + stackback@0.0.2: {} + + std-env@3.10.0: {} + + string-width@4.2.3: + dependencies: + emoji-regex: 8.0.0 + is-fullwidth-code-point: 3.0.0 + strip-ansi: 6.0.1 + + strip-ansi@6.0.1: + dependencies: + ansi-regex: 5.0.1 + + strip-literal@3.1.0: + dependencies: + js-tokens: 9.0.1 + + tinybench@2.9.0: {} + + tinyexec@0.3.2: {} + + tinyglobby@0.2.17: + dependencies: + fdir: 6.5.0(picomatch@4.0.7) + picomatch: 4.0.7 + + tinypool@1.1.1: {} + + tinyrainbow@2.0.0: {} + + tinyspy@4.0.6: {} + + tsx@4.23.15: + dependencies: + esbuild: 0.28.2 + optionalDependencies: + fsevents: 2.3.3 + + typescript@5.9.3: {} + + undici-types@6.21.0: {} + + vite-node@3.2.4(@types/node@22.20.4)(tsx@4.23.15): + dependencies: + cac: 6.7.14 + debug: 4.4.3 + es-module-lexer: 1.7.0 + pathe: 2.0.3 + vite: 7.3.6(@types/node@22.20.4)(tsx@4.23.15) + transitivePeerDependencies: + - '@types/node' + - jiti + - less + - lightningcss + - sass + - sass-embedded + - stylus + - sugarss + - supports-color + - terser + - tsx + - yaml + + vite@7.3.6(@types/node@22.20.4)(tsx@4.23.15): + dependencies: + esbuild: 0.28.2 + fdir: 6.5.0(picomatch@4.0.7) + picomatch: 4.0.7 + postcss: 8.5.28 + rollup: 4.63.4 + tinyglobby: 0.2.17 + optionalDependencies: + '@types/node': 22.20.4 + fsevents: 2.3.3 + tsx: 4.23.15 + + vitest@3.2.7(@types/node@22.20.4)(tsx@4.23.15): + dependencies: + '@types/chai': 5.2.3 + '@vitest/expect': 3.2.7 + '@vitest/mocker': 3.2.7(vite@7.3.6(@types/node@22.20.4)(tsx@4.23.15)) + '@vitest/pretty-format': 3.2.7 + '@vitest/runner': 3.2.7 + '@vitest/snapshot': 3.2.7 + '@vitest/spy': 3.2.7 + '@vitest/utils': 3.2.7 + chai: 5.3.3 + debug: 4.4.3 + expect-type: 1.4.0 + magic-string: 0.30.21 + pathe: 2.0.3 + picomatch: 4.0.7 + std-env: 3.10.0 + tinybench: 2.9.0 + tinyexec: 0.3.2 + tinyglobby: 0.2.17 + tinypool: 1.1.1 + tinyrainbow: 2.0.0 + vite: 7.3.6(@types/node@22.20.4)(tsx@4.23.15) + vite-node: 3.2.4(@types/node@22.20.4)(tsx@4.23.15) + why-is-node-running: 2.3.0 + optionalDependencies: + '@types/node': 22.20.4 + transitivePeerDependencies: + - jiti + - less + - lightningcss + - msw + - sass + - sass-embedded + - stylus + - sugarss + - supports-color + - terser + - tsx + - yaml + + why-is-node-running@2.3.0: + dependencies: + siginfo: 2.0.0 + stackback: 0.0.2 + + wrap-ansi@7.0.0: + dependencies: + ansi-styles: 4.3.0 + string-width: 4.2.3 + strip-ansi: 6.0.1 + + y18n@5.0.8: {} + + yargs-parser@21.1.1: {} + + yargs@17.7.3: + dependencies: + cliui: 8.0.1 + escalade: 3.2.0 + get-caller-file: 2.0.5 + require-directory: 2.1.1 + string-width: 4.2.3 + y18n: 5.0.8 + yargs-parser: 21.1.1 diff --git a/pkgs/substrate-link/src/pnpm-workspace.yaml b/pkgs/substrate-link/src/pnpm-workspace.yaml new file mode 100644 index 000000000..269d4cb25 --- /dev/null +++ b/pkgs/substrate-link/src/pnpm-workspace.yaml @@ -0,0 +1,4 @@ +allowBuilds: + esbuild: true + msgpackr-extract: false + protobufjs: false diff --git a/pkgs/substrate-link/src/tsconfig.json b/pkgs/substrate-link/src/tsconfig.json new file mode 100644 index 000000000..ef31c9521 --- /dev/null +++ b/pkgs/substrate-link/src/tsconfig.json @@ -0,0 +1,16 @@ +{ + "compilerOptions": { + "target": "ES2022", + "module": "NodeNext", + "moduleResolution": "NodeNext", + "lib": ["ES2022"], + "strict": true, + "exactOptionalPropertyTypes": false, + "noUncheckedIndexedAccess": true, + "allowImportingTsExtensions": true, + "noEmit": true, + "skipLibCheck": true, + "types": ["node"] + }, + "include": ["src", "test", "apps"] +} From 716dfe5e75ee0a5bf36c8963729dd4bf08950b90 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 22:24:06 +0200 Subject: [PATCH 34/37] substrate-link: NAS module, gate OFF (from conwip-link-nixos-on-ax-fleet-80658978.patch) Applied by hand: the patch no longer applies at f1bba47e (secrets.nix:98). - hosts/nas/substrate-link.nix: myNas.substrateLink, enable = false by default, LoadCredential floor-link-token, RestartPreventExitStatus 78, outbound only, no port. package defaults to pkgs/substrate-link (the "package is null" assertion is gone with the vendored source). - completion defaults to "guest": ax carries no P1, so "p1" and "auto" would lease nothing (B6); guest needs a fleet-internal Complete relay that does not exist yet, so arming fails its assertion until then. - hosts/nas/default.nix imports it after modules/ax-fleet. - secrets.nix: secrets/floor-link-token.age for editors ++ nasOnly (the .age file is not minted; Tom's hand). Co-Authored-By: Claude Opus 5.5 (1M context) --- hosts/nas/default.nix | 1 + hosts/nas/substrate-link.nix | 261 +++++++++++++++++++++++++++++++++++ secrets.nix | 3 + 3 files changed, 265 insertions(+) create mode 100644 hosts/nas/substrate-link.nix diff --git a/hosts/nas/default.nix b/hosts/nas/default.nix index 5ff4591d3..34895451a 100644 --- a/hosts/nas/default.nix +++ b/hosts/nas/default.nix @@ -51,6 +51,7 @@ # "hypervisor on NAS" (Tom, 2026-09-23). The CIDR-overlap assertions in # that module evaluate whether or not the switch below is on. ../../modules/ax-fleet + ./substrate-link.nix # 2026-09-23: Cloudflare floor (Substrate) -> ax link, outbound only; gate OFF (LINK-DESIGN.md) ../../modules/adguardhome.nix inputs.nixos-hardware.nixosModules.common-cpu-amd inputs.nixos-hardware.nixosModules.common-pc diff --git a/hosts/nas/substrate-link.nix b/hosts/nas/substrate-link.nix new file mode 100644 index 000000000..7d3692137 --- /dev/null +++ b/hosts/nas/substrate-link.nix @@ -0,0 +1,261 @@ +{ + config, + options, + lib, + pkgs, + ... +}: +# substrate-link: the NAS side of the Cloudflare floor (Substrate) to ax link, declared and OFF. +# +# DESIGN: ~/today/evals-2026-09-23/link/LINK-DESIGN.md (2026-09-23), BUILD.md beside it. CODE: vendored +# in pkgs/substrate-link/src from ax-conwip eval/2026-09-23-link a021003 apps/link (pkgs/substrate-link/SYNC.md). +# Formerly conwip-link; renamed with the ax-conwip project (now "substrate", Tom 2026-09-23 E10). DATE: 2026-09-23. +# Critique items carried here: N1 (axServer from ax-fleet), N2 (Node 24, pkgs), N3 (guest URL), N4 (floorUrls). +# +# WHAT IT IS. One long-running process that dials OUT to the floor (a Durable Object behind a +# Cloudflare Worker), leases AgentJobs under the floor's CONWIP cap, creates one ax Task per lease +# (GetTask, then UpdateTask only on NotFound: ax has no CreateTask and UpdateTask is a blind upsert), +# watches the Tasks by ListTasks, and reports each verdict through a write-ahead outbox. Nothing +# dials in: no tunnel, no listener, no port opened in this file. +# +# WHY A SYSTEM UNIT ON THE NAS HOST AND NOT A POD. The link needs no Kubernetes API (ax is gRPC), it +# must keep reporting "ax is down" to the floor while k3s is down, its journal must outlive the +# cluster, and its bearer must never become a Kubernetes Secret: ax-controller's ClusterRole reads +# secrets cluster-wide (ax-fleet DESIGN section 9). ax-server's ClusterIP answers host processes on +# a k3s node through kube-proxy, the same path the coordinator's socket proxy uses (DESIGN D13). +# +# THE TOKEN IS A PATH, NEVER A VALUE. agenix decrypts floor-link-token.age for root only (0400); +# LoadCredential hands the unit a private copy under $CREDENTIALS_DIRECTORY. The program reads it +# once and never logs it. Minting the token (a Worker secret and this age file) is Tom's hand. +# +# THE GATE. `enable` defaults to false and nothing in this repository sets it. Arming needs, in +# order: the ax-fleet PR switched on the NAS, the sealed token, the floor URL, a guest Complete +# relay (completion = guest on zero-patch ax), and Tom's ack. Exit 78 (bad token or config) stops the +# restart loop; exit 75 (a second replica holds the session, or an endpoint switch to a URL in +# floorUrls was persisted) retries after RestartSec. +let + cfg = config.myNas.substrateLink; + inherit (lib) mkEnableOption mkIf mkOption types; + # N1: one source for ax-server's address. When modules/ax-fleet is imported its pinned ClusterIP wins. + axFleetIP = if options ? myAxFleet && options.myAxFleet ? axServerClusterIP then config.myAxFleet.axServerClusterIP else null; + # N3 (critique A5): a guest must post to a fleet-internal relay, never to a public Workers host. + publicWorkersHost = u: lib.hasInfix ".workers.dev" u || lib.hasInfix ".pages.dev" u; + # Review round 2: mirrors isFleetInternalUrl (apps/link/src/link.ts). Private ranges match only IPv4 literals + # (10.evil.example and 127.0.0.1.nip.io are names), names only under .internal/.lan/.local/.home.arpa, IPv6 only + # ::1 and ULA, a single label only when listed in internalHosts. Stricter than the program where they differ. + hostOf = u: let m = builtins.match "[a-z]+://([^@/]*@)?(\\[[^]]*]|[^/:?#]+).*" (lib.toLower u); in if m == null then "" else builtins.elemAt m 1; + ipv4Of = h: builtins.match "([0-9]{1,3})\\.([0-9]{1,3})\\.([0-9]{1,3})\\.([0-9]{1,3})" h; + privateV4 = o: let a = lib.toInt (builtins.elemAt o 0); b = lib.toInt (builtins.elemAt o 1); in + a == 127 || a == 10 || (a == 192 && b == 168) || (a == 172 && b >= 16 && b <= 31) || (a == 100 && b >= 64 && b <= 127); + fleetInternalUrl = u: let h = hostOf u; o = ipv4Of h; in + if o != null then privateV4 o + else if lib.hasPrefix "[" h then h == "[::1]" || builtins.match "\\[f[cd][0-9a-f]{2}:.*" h != null + else h == "localhost" || lib.elem h cfg.internalHosts || builtins.match "([a-z0-9-]+\\.)+(internal|lan|local|home\\.arpa)" h != null; +in +{ + options.myNas.substrateLink = { + enable = mkEnableOption "the Substrate floor-to-ax link (outbound only). OFF; see this file's header before flipping it"; + + package = mkOption { + type = types.package; + default = pkgs.callPackage ../../pkgs/substrate-link { }; + defaultText = lib.literalExpression "pkgs.callPackage ../../pkgs/substrate-link { }"; + description = "The link program, built from the source vendored in pkgs/substrate-link/src."; + }; + + floorUrl = mkOption { + type = types.str; + default = ""; + example = "https://substrate.example.dev"; + description = "The floor Worker's base URL. HTTPS only; the link appends /rpc."; + }; + + floorUrls = mkOption { + type = types.listOf types.str; + default = [ cfg.floorUrl ]; + defaultText = lib.literalExpression "[ config.myNas.substrateLink.floorUrl ]"; + description = "N4 (B18): the only floor URLs a Lease `endpoint` may move this link to, exact https matches. The switch is persisted in the state directory and the unit restarts onto it (exit 75)."; + }; + + holder = mkOption { + type = types.str; + default = "nas-link-1"; + description = "holderIdentity. The floor binds it to this link's token; one session per identity."; + }; + + axServer = mkOption { + type = types.str; + default = if axFleetIP != null then "${axFleetIP}:8080" else "10.201.0.80:8080"; + defaultText = lib.literalExpression ''"''${config.myAxFleet.axServerClusterIP}:8080" when modules/ax-fleet is imported, else "10.201.0.80:8080"''; + description = "ax-server as host:port (gRPC h2c, no auth upstream, #376). N1: read from ax-fleet's axServerClusterIP option (D13)."; + }; + + atespace = mkOption { + type = types.str; + default = "fleet"; + }; + + image = mkOption { + type = types.str; + default = "ax-agent"; + example = "localhost:5000/ax-agent@sha256:0000"; + description = "Task image; set it to the digest ax-fleet-image-ref prints."; + }; + + gateway = mkOption { + type = types.str; + default = "halogen"; + description = "The egress Gateway every Task names. A missing Gateway means open egress on v0.3.0, so the link leases nothing while GetGateway answers NotFound or the Gateway allows * (B10)."; + }; + + maxInFlight = mkOption { + type = types.ints.positive; + default = 2; + description = "Sandbox cap this link fills: the ateom-gvisor WorkerPool replicas (ax-fleet section 9)."; + }; + + servedLabels = mkOption { + type = types.listOf types.str; + default = [ + "seat:halogen" + "runtime:gvisor" + ]; + description = "runs-on labels this link accepts. Claude seats stay on the coordinator until Tom rules (W6)."; + }; + + seatCommands = mkOption { + type = types.attrsOf (types.listOf types.str); + default = { + halogen = [ + "ax-agent" + "pi" + ]; + }; + description = "Task command per seat (the ax-agent adapter's modes, ax-fleet section 10.2)."; + }; + + completion = mkOption { + type = types.enum [ + "auto" + "p1" + "guest" + ]; + # The fleet runs stock ax with no P1 (pkgs/ax: patches = [ ]), so "p1" and "auto" lease + # nothing (B6). "guest" is the only mode that completes on this fleet; it needs guestCompleteUrl, + # a fleet-internal relay that does not exist yet, so arming fails its assertion until then. + default = "guest"; + description = "p1: the controller writes the outcome (carried patch P1); the link probes GetTaskResult at start and whenever ax comes back, and leases nothing on a server without P1 (B6). auto: the same, never guest. guest: the L7 reserve, pre-P1, the guest completes with a per-lease token through a fleet-internal relay."; + }; + + guestCompleteUrl = mkOption { + type = types.nullOr types.str; + default = null; + description = "Only for completion = guest: the fleet-internal URL the guest posts its per-lease Complete to (N3: never a workers.dev host)."; + }; + + internalHosts = mkOption { + type = types.listOf types.str; + default = [ ]; + description = "Review round 2: single-label host names a guestCompleteUrl may use (fleet DNS names without a dot)."; + }; + + tokenAgeFile = mkOption { + type = types.path; + default = ../../secrets/floor-link-token.age; + defaultText = lib.literalExpression "../../secrets/floor-link-token.age"; + description = "The sealed per-link bearer (secrets.nix: editors ++ nasOnly). Evaluated only when enabled."; + }; + }; + + config = mkIf cfg.enable { + assertions = [ + { + assertion = lib.hasPrefix "https://" cfg.floorUrl; + message = "myNas.substrateLink.floorUrl must be an https:// URL (the bearer never travels in clear)."; + } + { + assertion = config.networking.hostName == "nas"; + message = "substrate-link runs on the hypervisor host only (placement ruling: NAS = k3s server, Substrate, ax-server)."; + } + { + assertion = cfg.completion != "guest" || (cfg.guestCompleteUrl != null && !(publicWorkersHost cfg.guestCompleteUrl) && fleetInternalUrl cfg.guestCompleteUrl); + message = "completion = guest needs a fleet-internal guestCompleteUrl: a private IPv4 literal, ::1 or ULA, or a name under .internal/.lan/.local/.home.arpa, never a workers.dev or pages.dev host (critique A5, review round 2)."; + } + { + assertion = lib.all (u: lib.hasPrefix "https://" u) cfg.floorUrls && lib.elem cfg.floorUrl cfg.floorUrls; + message = "myNas.substrateLink.floorUrls must be https:// URLs and include floorUrl."; + } + ]; + + age.secrets.floor-link-token = { + file = cfg.tokenAgeFile; + mode = "0400"; + }; + + systemd.services.substrate-link = { + description = "substrate-link: Cloudflare floor to ax, outbound only"; + wantedBy = [ "multi-user.target" ]; + wants = [ "network-online.target" ]; + after = [ + "network-online.target" + "k3s.service" + ]; + environment = { + LINK_FLOOR_URL = cfg.floorUrl; + LINK_FLOOR_URLS = lib.concatStringsSep "," cfg.floorUrls; + LINK_HOLDER = cfg.holder; + AX_SERVER = cfg.axServer; + AX_ATESPACE = cfg.atespace; + LINK_IMAGE = cfg.image; + LINK_GATEWAY = cfg.gateway; + LINK_MAX_IN_FLIGHT = toString cfg.maxInFlight; + LINK_SERVED_LABELS = lib.concatStringsSep "," cfg.servedLabels; + LINK_SEAT_COMMANDS = builtins.toJSON cfg.seatCommands; + LINK_COMPLETION = cfg.completion; + LINK_INTERNAL_HOSTS = lib.concatStringsSep "," cfg.internalHosts; + } + // lib.optionalAttrs (cfg.guestCompleteUrl != null) { LINK_GUEST_COMPLETE_URL = cfg.guestCompleteUrl; }; + serviceConfig = { + ExecStart = lib.getExe cfg.package; + DynamicUser = true; + StateDirectory = "substrate-link"; # journal.jsonl, session-id, floor-endpoint; a few fsynced lines per job + StateDirectoryMode = "0700"; + LoadCredential = [ "floor-link-token:${config.age.secrets.floor-link-token.path}" ]; + Restart = "on-failure"; + RestartSec = "10s"; + RestartPreventExitStatus = [ 78 ]; + TimeoutStopSec = "30s"; # SIGTERM drains: no new leases, outbox flushed; ax Tasks keep running + UMask = "0077"; + MemoryMax = "512M"; + NoNewPrivileges = true; + ProtectSystem = "strict"; + ProtectHome = true; + PrivateTmp = true; + PrivateDevices = true; + ProtectKernelTunables = true; + ProtectKernelModules = true; + ProtectKernelLogs = true; + ProtectControlGroups = true; + ProtectClock = true; + ProtectHostname = true; + LockPersonality = true; + RestrictRealtime = true; + RestrictSUIDSGID = true; + RestrictNamespaces = true; + SystemCallArchitectures = "native"; + SystemCallFilter = [ + "@system-service" + "~@privileged" + ]; + CapabilityBoundingSet = ""; + AmbientCapabilities = ""; + RestrictAddressFamilies = [ + "AF_INET" + "AF_INET6" + "AF_UNIX" + ]; + # MemoryDenyWriteExecute stays off: V8's JIT needs writable and executable pages. + }; + }; + }; +} diff --git a/secrets.nix b/secrets.nix index 05b09be96..974c837ac 100644 --- a/secrets.nix +++ b/secrets.nix @@ -103,6 +103,9 @@ in # Rotate with: nix develop -c agenix -e secrets/.age "secrets/k3s-token.age".publicKeys = editors ++ nasOnly; "secrets/k3s-agent-token.age".publicKeys = editors ++ coordinatorOnly ++ nasOnly; + # substrate-link bearer (hosts/nas/substrate-link.nix): one token per link identity, minted by Tom as a + # Worker secret on the floor and sealed here. NAS only: the link runs on the NAS host. + "secrets/floor-link-token.age".publicKeys = editors ++ nasOnly; # --- wifi PSK tier: the coordinator, whose Freebox uplink # (wlp192s0) is now declarative too (migrated from an imperative profile on # flash night — refs #37). Rekey after this change: nix develop -c agenix -r From aaf7433de4b6d5efad464d361f87c0eefc9ee659 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 22:30:56 +0200 Subject: [PATCH 35/37] ax-fleet: phase 33 loop variable no longer shadows phase 30's typed gw The phases are concatenated into one test script; the test driver's type check rejected rebinding gw (dict[str, Any] in 30-nop1) to an optional gateway name. Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/ax-fleet/phases/33-gateway-default.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/ax-fleet/phases/33-gateway-default.py b/tests/ax-fleet/phases/33-gateway-default.py index c659d64db..cec273e88 100644 --- a/tests/ax-fleet/phases/33-gateway-default.py +++ b/tests/ax-fleet/phases/33-gateway-default.py @@ -33,11 +33,11 @@ def late_curl_body(url: str, wait: int) -> str: worker.wait_for_unit("public-8000.service") nas.wait_until_succeeds(f"curl -sf --max-time 10 {PUBLIC} | grep -q public-reached", timeout=120) results = {} - for label, gw in (("none", None), ("missing", "no-such-gateway")): + for label, missing_gw in (("none", None), ("missing", "no-such-gateway")): name = f"gwdef-{label}" n = f"{name}-a1" t0 = time.monotonic() - fleet_task(n, late_curl_body(PUBLIC, 30), gateway=gw) + fleet_task(n, late_curl_body(PUBLIC, 30), gateway=missing_gw) coordinator.wait_until_succeeds( f"{AX} get task {n} | grep -A1 -E '^\\s+gateway:' | grep -qE 'name:\\s*\"?default\"?'", timeout=120 ) From a6965fededa71bff0d73c2f0d52ab38054c637e6 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 22:59:25 +0200 Subject: [PATCH 36/37] ax-fleet: re-apply, never delete, a Task whose resume failed on a stale connection r1 run 2 (MEASURED): after the LAN flap, after-flap-a1 failed ActorResumeFailed (Unavailable, stale ateapi->atelet connection). fleet_run deleted it; Substrate's atelet Terminate then failed on every try with 'failed to read sandbox record during terminate', because the aborted resume never wrote the record (cmd/atelet/main.go:1281). The actor stayed in ACTOR_STATE_DELETING and ax delete timed out after 300 s. Stock ax reconciles on every SaveTask and re-runs ResumeActor with no phase guard, so the recovery with no ax patch is to apply the same Task again. fleet_run and phase 33 now re-apply up to REAPPLY_RESUME times before falling back to delete-and-recreate. Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/ax-fleet/phases/32-fleet.py | 28 ++++++++++++++++++++- tests/ax-fleet/phases/33-gateway-default.py | 7 ++++++ 2 files changed, 34 insertions(+), 1 deletion(-) diff --git a/tests/ax-fleet/phases/32-fleet.py b/tests/ax-fleet/phases/32-fleet.py index 77640db76..9785239ef 100644 --- a/tests/ax-fleet/phases/32-fleet.py +++ b/tests/ax-fleet/phases/32-fleet.py @@ -48,6 +48,20 @@ def fleet_task(name: str, body: str, gateway: Any = "halogen", image: Any = None coordinator.succeed(f"echo {b} | base64 -d | {AX} apply -f -") +REAPPLY_RESUME = 3 # test parameter: re-applies of a Task whose resume failed transiently + + +def resume_failed_transient(st: Any) -> bool: + """Failed with ActorResumeFailed for a reason other than a full pool.""" + msg = json.dumps(st) + return ( + st.get("phase") == "Failed" + and (st.get("ready") or {}).get("reason") == "ActorResumeFailed" + and "ResourceExhausted" not in msg + and "no free workers" not in msg + ) + + def fleet_run(name: str, body: str, gateway: Any = "halogen", image: Any = None, attempts: int = MAX_ATTEMPTS) -> Any: """Attempts NAME-a1.. until the floor has a report; returns the decoded result.""" tried: list[Any] = [] @@ -55,7 +69,19 @@ def fleet_run(name: str, body: str, gateway: Any = "halogen", image: Any = None, n = f"{name}-a{attempt}" fleet_task(n, body, gateway, image) reps, before, secs = wait_report(n) - entry: dict[str, Any] = {"task": n, "report_seconds": secs, "before_delete": before} + reapplied: list[Any] = [] + # A resume that failed on a stale gRPC connection (ActorResumeFailed, + # Unavailable) left the actor placed but never restored: atelet wrote + # no sandbox record, and Substrate's Terminate then fails on every + # delete (MEASURED r1 run 2, cmd/atelet/main.go:1281). Deleting that + # attempt strands it in ACTOR_STATE_DELETING. Stock ax reconciles on + # every save and re-runs ResumeActor with no phase guard, so the + # recovery is to apply the same Task again, not to delete it. + while not reps and len(reapplied) < REAPPLY_RESUME and resume_failed_transient(before): + reapplied.append({"state": before, "after_s": secs}) + fleet_task(n, body, gateway, image) + reps, before, secs = wait_report(n) + entry: dict[str, Any] = {"task": n, "report_seconds": secs, "before_delete": before, "reapplied": reapplied} if reps: rep = reps[0]["report"] raw = base64.b64decode(rep["result_b64"]) if rep.get("result_b64") else b"{}" diff --git a/tests/ax-fleet/phases/33-gateway-default.py b/tests/ax-fleet/phases/33-gateway-default.py index cec273e88..e26eb69d1 100644 --- a/tests/ax-fleet/phases/33-gateway-default.py +++ b/tests/ax-fleet/phases/33-gateway-default.py @@ -43,6 +43,13 @@ def late_curl_body(url: str, wait: int) -> str: ) repoint_s = round(time.monotonic() - t0, 1) reps, before, secs = wait_report(n) + tries = 0 + while not reps and tries < REAPPLY_RESUME and resume_failed_transient(before): + # Same recovery as fleet_run (32-fleet): apply again, never delete a + # Task whose resume failed on a stale connection. + tries += 1 + fleet_task(n, late_curl_body(PUBLIC, 30), gateway=missing_gw) + reps, before, secs = wait_report(n) assert reps, f"{n}: no floor report (the default Gateway allows the floor): {before}" rep = reps[0]["report"] res = json.loads(base64.b64decode(rep["result_b64"]) or b"{}") From 40a2ec87546fc4ef00e40251ff163f2e2c2af2b8 Mon Sep 17 00:00:00 2001 From: mecattaf Date: Wed, 23 Sep 2026 23:18:53 +0200 Subject: [PATCH 37/37] ax-fleet: rollback (a) reads probe-coord's pod IP on every try r1 run 3 (MEASURED): after the in-ax-on teardown and the k3s restart, kubectl wait Ready passed in 0.24 s on the pre-teardown status, and the test read status.podIP once (10.200.1.5, the old sandbox) and curled it for 300 s (curl rc 7). The teardown recreated every sandbox (four new veths on cni0). The wait now re-reads the IP on each try, records the stale and the final reads, and on a timeout records the pod path (neighbours, forwarding, flannel FDB, listener) before re-raising. Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/ax-fleet/phases/90-rollback.py | 25 +++++++++++++++++++++++-- 1 file changed, 23 insertions(+), 2 deletions(-) diff --git a/tests/ax-fleet/phases/90-rollback.py b/tests/ax-fleet/phases/90-rollback.py index f6d52a1ea..55a045755 100644 --- a/tests/ax-fleet/phases/90-rollback.py +++ b/tests/ax-fleet/phases/90-rollback.py @@ -61,8 +61,29 @@ def guards(machine): # Discriminating: the NAS's pod reaches a coordinator pod over VXLAN, and # the coordinator's sshd listens on its flannel.1 address, yet the pod # gets no SSH banner from it. - coord_pod = jsonpath("pod probe-coord", "{.status.podIP}") - nas.wait_until_succeeds(f"k3s kubectl exec probe-nas -- curl -sf --max-time 5 http://{coord_pod}:8000/ | grep -x pod-ok", timeout=300) + # The teardown killed the sandboxes; `kubectl wait` above can pass on the + # pre-teardown status (MEASURED r1 run 3: 0.24 s), so podIP may still name + # the old sandbox for a while. Read it on every try, never once. + record("probe_coord_ip_stale_read", jsonpath("pod probe-coord", "{.status.podIP}")) + try: + nas.wait_until_succeeds( + "ip=$(k3s kubectl get pod probe-coord -o jsonpath='{.status.podIP}') && " + "k3s kubectl exec probe-nas -- curl -sf --max-time 5 http://$ip:8000/ | grep -x pod-ok", + timeout=300, + ) + except Exception: + record("rollback_a_pod_path_diag", { + "pod": nas.execute("k3s kubectl get pod probe-coord -o wide 2>&1")[1], + "curl": nas.execute("ip=$(k3s kubectl get pod probe-coord -o jsonpath='{.status.podIP}'); " + "k3s kubectl exec probe-nas -- curl -sv --max-time 5 http://$ip:8000/ 2>&1 | tail -5")[1], + "coord_cni0": coordinator.execute("ip -4 -o addr show dev cni0; ip neigh show dev cni0 2>&1")[1], + "coord_forward": coordinator.execute("sysctl -n net.ipv4.ip_forward; iptables -S FORWARD 2>&1 | head -30")[1], + "coord_flannel": coordinator.execute("ip -d link show flannel.1 2>&1 | head -3; ip route 2>&1")[1], + "nas_fdb": nas.execute("bridge fdb show dev flannel.1 2>&1; ip neigh show dev flannel.1 2>&1")[1], + "coord_listen": coordinator.execute("ss -ltnp 2>&1 | grep -w 8000 || true")[1], + }) + raise + record("probe_coord_ip_after_k3s_restart", jsonpath("pod probe-coord", "{.status.podIP}")) fl = coordinator.succeed("ip -4 -o addr show dev flannel.1 | awk '{print $4}' | cut -d/ -f1").strip() coordinator.succeed(f"timeout 10 bash -c 'exec 3<>/dev/tcp/{fl}/22'") kubectl(f"exec probe-nas -- sh -c '! (nc -w 5 {fl} 22 /dev/null | grep -q SSH)'")