From d54864190d3d2b0a55b414c727d902174a06543d Mon Sep 17 00:00:00 2001 From: Dmitry Kireev Date: Tue, 5 May 2026 06:50:48 +0000 Subject: [PATCH] cell serve gains OpenAI Responses API + SSE streaming, configurable system prompts across flags/env/toml, and per-stack user-image tagging that ends per-session image sprawl MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - feat(serve): add `POST /v1/responses` (OpenAI Responses API) — n8n "Message a Model", OpenAI Agents SDK and other newer clients can now target cell serve directly - feat(serve): stream `/v1/chat/completions` and `/v1/responses` over SSE for the claude agent — incremental token deltas, 15s `:keepalive` heartbeats so long agentic turns survive proxy idle timeouts; opencode falls back to buffered - feat(serve): add `--system-prompt` / `--system-prompt-file` flags, `DEVCELL_SYSTEM_PROMPT` / `DEVCELL_SYSTEM_PROMPT_FILE` env vars, and `[llm].system_prompt_file` TOML key — operators can pin a baseline prompt that composes with per-request `instructions`/`system` instead of overriding it - feat(serve): expose `GET /v1/models`, `GET /healthz`, `GET /api/openapi.json`, `GET /swagger/` UI — clients can discover available models and the API is browsable - feat(serve): honor OpenAI `reasoning_effort` / `reasoning.effort` (`low|medium|high`) → `claude --effort` per request; non-spec values silently dropped - feat(serve): parse claude `--output-format=json` envelope into per-response token + cost telemetry (input/output/cache tokens, total_cost_usd) and surface it in OpenAI `usage` shape - feat(serve): opt-in `DEVCELL_LOG_PROMPTS=1` logs full prompt + reply at INFO; off by default since prompts often carry secrets / PII - feat(serve): reuse a fixed `DEVCELL_API_KEY` when set instead of always generating, and stop printing the key on startup unless it was generated — deployments no longer leak the key into stderr/logs - feat(serve): claude is now invoked with `--dangerously-skip-permissions` so tool calls don't block on a TTY-less permission gate; the bearer API key is the auth boundary - feat(runner): default user-image tag is now `devcell-user:[--]` instead of `devcell-user:` — one image per stack/module-set instead of one per tmux session, eliminating ~13 GB-per-session image sprawl - feat(cfg): add `[cell].per_session_image` toml key + `DEVCELL_PER_SESSION_IMAGE` env to opt back into the legacy per-session tagging - feat(runner): `DEVCELL_DOCKER_BUILD_ARGS="KEY=VAL ..."` injects `--build-arg` pairs into image builds — lets users override Dockerfile ARGs without forking - fix(op): `ResolveItems` now collects per-item errors and continues instead of aborting on the first failure — a single missing/locked 1Password item no longer blocks the whole agent launch - fix(config): clamp generated VNC/RDP ports above 65535 back into the valid TCP range — projects with high port prefixes no longer fail to bind - fix(logger): server mode renders plain ASCII logs with timestamps and no ANSI colors — log aggregators (CloudWatch, journald) no longer see escape codes - feat(nixhome/mcp): add `enabled` attribute on MCP server entries — variants can be registered in nix without being staged into Claude/OpenCode/Codex configs (used by new `notion-oauth` opt-in) - feat(nixhome/infra): add `notion-api` local stdio MCP (npx-wrapped @notionhq/notion-mcp-server) using `NOTION_API_KEY` — non-interactive, works for headless agents; `notion-oauth` remote variant kept as opt-in - feat(nixhome/project-management): add `n8n` MCP server (czlonkowski/n8n-mcp) for workflow-automation control via `N8N_API_URL` / `N8N_API_KEY` - feat(nixhome/security): add ghidra, radare2, rizin, binwalk, yara, upx, pev, detect-it-easy, capstone, ropper, foremost, sleuthkit — full PE/ELF/Mach-O reverse-engineering and forensics toolkit available in the security stack - feat(nixhome/base): add 7zz, p7zip, tinyxxd, hexedit — broader archive and hex-editing coverage in every stack - feat(nixhome/go): add go-swag — `swag init` available for generating OpenAPI specs - chore(build): pin `docker buildx bake` output to gzip and disable provenance/sbom attestations — older Docker daemons and registries that choke on zstd or OCI provenance can now pull images - chore(build): regenerate Swagger docs in goreleaser `before:hooks`, `task cell:build`, and the Dockerfile builder stage so `/swagger/` is always in sync with annotations - refactor(runner): split `BuildSystemPrompt` into `ContainerContext` (auto-generated mounts/paths/constraints) + `ResolveSystemPrompt` (7-tier source chain) + `AssembleSystemPrompt` (concatenator) — both `cell claude` and `cell serve` now share the same prompt-resolution logic - refactor(serve): `Executor.Run` now takes an `ExecOpts` struct instead of positional args — no user-facing impact - refactor(serve): trim trailing newlines from agent stdout before placing into `output_text` / `message.content` — clients no longer see a stray `\n` at the end of every reply - chore(deps): add swaggo/swag + http-swagger, promote charmbracelet/bubbles+bubbletea to direct deps, drop go-md2man/blackfriday indirects — no user-facing impact - test(serve): add full coverage for responses, sse_chat, sse_responses, claude_json, claude_stream, exec, exec_stream, plus expanded handler/server/openai compat tests - test(runner): expanded systemprompt tests covering all seven resolution tiers, mutual-exclusion errors, and file-relative path handling - test(cfg): add tests for `per_session_image` resolution and TOML round-trip --- .gitignore | 1 + .goreleaser.dev.yaml | 4 + .goreleaser.yaml | 4 + Taskfile.yml | 9 +- cmd/root.go | 43 +- cmd/serve.go | 203 ++++++++- docker-bake.hcl | 15 +- go.mod | 21 +- go.sum | 43 +- images/Dockerfile | 1 + internal/cfg/cfg.go | 34 +- internal/config/config.go | 22 +- internal/config/config_test.go | 22 + internal/logger/logger.go | 14 +- internal/op/resolve.go | 15 +- internal/runner/runner.go | 56 ++- internal/runner/runner_test.go | 105 ++++- internal/runner/systemprompt.go | 142 +++++- internal/runner/systemprompt_test.go | 272 ++++++++--- internal/serve/claude_json.go | 83 ++++ internal/serve/claude_json_test.go | 71 +++ internal/serve/claude_stream.go | 217 +++++++++ internal/serve/claude_stream_test.go | 112 +++++ internal/serve/exec.go | 102 ++++- internal/serve/exec_stream.go | 58 +++ internal/serve/exec_stream_test.go | 123 +++++ internal/serve/exec_test.go | 282 ++++++++++++ internal/serve/handler.go | 224 ++++++++- internal/serve/handler_test.go | 119 ++++- internal/serve/logging.go | 52 +++ internal/serve/models.go | 36 +- internal/serve/openai_compat_test.go | 6 +- internal/serve/responses.go | 574 ++++++++++++++++++++++++ internal/serve/responses_compat_test.go | 169 +++++++ internal/serve/responses_test.go | 503 +++++++++++++++++++++ internal/serve/server.go | 109 ++++- internal/serve/server_test.go | 111 ++++- internal/serve/sse_chat.go | 238 ++++++++++ internal/serve/sse_chat_test.go | 129 ++++++ internal/serve/sse_responses.go | 314 +++++++++++++ internal/serve/sse_responses_test.go | 118 +++++ nixhome/flake.lock | 6 +- nixhome/modules/go.nix | 1 + nixhome/modules/infra.nix | 34 +- nixhome/modules/llm/claude.nix | 7 +- nixhome/modules/llm/codex.nix | 5 +- nixhome/modules/llm/mcp.nix | 11 +- nixhome/modules/llm/opencode.nix | 5 +- nixhome/modules/project-management.nix | 39 +- nixhome/modules/security.nix | 12 + 50 files changed, 4647 insertions(+), 249 deletions(-) create mode 100644 internal/serve/claude_json.go create mode 100644 internal/serve/claude_json_test.go create mode 100644 internal/serve/claude_stream.go create mode 100644 internal/serve/claude_stream_test.go create mode 100644 internal/serve/exec_stream.go create mode 100644 internal/serve/exec_stream_test.go create mode 100644 internal/serve/exec_test.go create mode 100644 internal/serve/logging.go create mode 100644 internal/serve/responses.go create mode 100644 internal/serve/responses_compat_test.go create mode 100644 internal/serve/responses_test.go create mode 100644 internal/serve/sse_chat.go create mode 100644 internal/serve/sse_chat_test.go create mode 100644 internal/serve/sse_responses.go create mode 100644 internal/serve/sse_responses_test.go diff --git a/.gitignore b/.gitignore index 5198e7dd..af7a1002 100644 --- a/.gitignore +++ b/.gitignore @@ -23,3 +23,4 @@ test/results .devcell.toml .gocache .gomodcache +docs/ diff --git a/.goreleaser.dev.yaml b/.goreleaser.dev.yaml index d553d505..2ae34811 100644 --- a/.goreleaser.dev.yaml +++ b/.goreleaser.dev.yaml @@ -2,6 +2,10 @@ version: 2 project_name: cell +before: + hooks: + - go run github.com/swaggo/swag/cmd/swag@latest init -g cmd/serve.go -o docs --parseDependency --parseInternal + builds: - id: cell main: ./cmd diff --git a/.goreleaser.yaml b/.goreleaser.yaml index 89c63331..b0a8a230 100644 --- a/.goreleaser.yaml +++ b/.goreleaser.yaml @@ -2,6 +2,10 @@ version: 2 project_name: cell +before: + hooks: + - go run github.com/swaggo/swag/cmd/swag@latest init -g cmd/serve.go -o docs --parseDependency --parseInternal + builds: - id: cell main: ./cmd diff --git a/Taskfile.yml b/Taskfile.yml index 934bad18..85412088 100644 --- a/Taskfile.yml +++ b/Taskfile.yml @@ -13,9 +13,16 @@ env: tasks: + swag:generate: + desc: Regenerate Swagger docs from annotations + dir: "{{.TASKFILE_DIR}}" + cmds: + - go run github.com/swaggo/swag/cmd/swag@latest init -g cmd/serve.go -o docs --parseDependency --parseInternal + cell:build: desc: Build cell CLI binary dir: "{{.TASKFILE_DIR}}" + deps: [swag:generate] cmds: - CGO_ENABLED=0 go build -ldflags "{{.CELL_LDFLAGS}}" -o ~/.local/bin/cell ./cmd/ @@ -66,7 +73,7 @@ tasks: desc: Build ci group and push to registry (multi-arch) silent: true cmds: - - GIT_COMMIT={{.GIT_COMMIT_HASH}} docker buildx bake --file {{.TASKFILE_DIR}}/docker-bake.hcl --set '*.output=type=image,push=true,compression=zstd,compression-level=3,force-compression=true' ci {{.CLI_ARGS}} + - GIT_COMMIT={{.GIT_COMMIT_HASH}} docker buildx bake --file {{.TASKFILE_DIR}}/docker-bake.hcl --push ci {{.CLI_ARGS}} nix:validate: desc: Validate all nixhome stacks — syntax check then attr check (no build, no activation) diff --git a/cmd/root.go b/cmd/root.go index e31a903f..33cbab68 100644 --- a/cmd/root.go +++ b/cmd/root.go @@ -243,6 +243,11 @@ func runAgent(binary string, defaultFlags, userArgs []string, extraEnv map[strin cellCfg := cfg.LoadFromOS(c.ConfigDir, c.BaseDir) + // Set stack/modules so UserImageTag() produces stack-based tags. + runner.Stack = cellCfg.Cell.ResolvedStack() + runner.Modules = cellCfg.Cell.Modules + runner.PerSessionImage = cellCfg.Cell.ResolvedPerSessionImage() + // Resolve available GUI ports — probe and bump if already bound if cellCfg.Cell.ResolvedGUI() { c.ResolveAvailablePorts() @@ -356,10 +361,21 @@ func runAgent(binary string, defaultFlags, userArgs []string, extraEnv map[strin imageID = "" } - // Inject system prompt for Claude Code — describes container environment, - // bind mounts, and host path mappings so Claude understands its runtime context. + // Inject system prompt for Claude Code — container context (mounts, + // host paths, constraints) plus the operator/project prompt resolved + // from env vars and devcell.toml. See runner.AssembleSystemPrompt for + // the full source-precedence chain (cell claude doesn't expose flags + // today; cell serve does). if binary == "claude" { - prompt := runner.BuildSystemPrompt(c, cellCfg) + prompt, err := runner.AssembleSystemPrompt(c, cellCfg, runner.ResolveOpts{ + EnvFile: os.Getenv("DEVCELL_SYSTEM_PROMPT_FILE"), + EnvInline: os.Getenv("DEVCELL_SYSTEM_PROMPT"), + CellCfg: cellCfg, + CfgBaseDir: c.BaseDir, + }) + if err != nil { + return fmt.Errorf("system prompt: %w", err) + } defaultFlags = append(defaultFlags, "--append-system-prompt", prompt) } @@ -392,18 +408,17 @@ func runAgent(binary string, defaultFlags, userArgs []string, extraEnv map[strin } ux.Debugf("1Password: resolving %d document(s): %v", len(opDocs), opDocs) if _, err := exec.LookPath("op"); err == nil { - resolved, err := op.ResolveItems(opDocs) - if err != nil { - fmt.Fprintf(os.Stderr, "warning: 1Password: %v\n", err) - } else { - keys := make([]string, 0, len(resolved)) - for k, v := range resolved { - os.Setenv(k, v) - inheritEnv = append(inheritEnv, k) - keys = append(keys, k) - } - ux.Debugf("1Password: resolved %d secret(s): %v", len(keys), keys) + resolved, errs := op.ResolveItems(opDocs) + for _, e := range errs { + fmt.Fprintf(os.Stderr, "warning: 1Password: %v\n", e) + } + keys := make([]string, 0, len(resolved)) + for k, v := range resolved { + os.Setenv(k, v) + inheritEnv = append(inheritEnv, k) + keys = append(keys, k) } + ux.Debugf("1Password: resolved %d secret(s) from %d document(s) (%d failed): %v", len(keys), len(opDocs)-len(errs), len(errs), keys) } else { ux.Debugf("1Password: op CLI not found, skipping secret resolution") } diff --git a/cmd/serve.go b/cmd/serve.go index 39083227..68a42ed9 100644 --- a/cmd/serve.go +++ b/cmd/serve.go @@ -5,46 +5,197 @@ import ( "fmt" "os" "os/signal" + "strconv" "syscall" + _ "github.com/DimmKirr/devcell/docs" // swagger docs (generated by swag init) + "github.com/DimmKirr/devcell/internal/cfg" + "github.com/DimmKirr/devcell/internal/config" + "github.com/DimmKirr/devcell/internal/logger" + "github.com/DimmKirr/devcell/internal/runner" "github.com/DimmKirr/devcell/internal/serve" "github.com/spf13/cobra" ) +// @title DevCell Serve API +// @version 1.0 +// @description DevCell Serve exposes an OpenAI-compatible HTTP API that proxies LLM requests to agent binaries (Claude Code, OpenCode) running inside a DevCell container. +// @description +// @description ## Why use this? +// @description +// @description Any tool that speaks the OpenAI protocol — Cursor, Continue, n8n, custom scripts, CI pipelines, the OpenAI Agents SDK — can target DevCell Serve as its backend. The server routes requests to the appropriate agent binary based on the `model` field — no SDK or CLI wrapper needed. +// @description +// @description ## Endpoints +// @description +// @description - **`POST /v1/chat/completions`** — Chat Completions API (the most widely supported OpenAI surface). Use this for traditional chat clients. +// @description - **`POST /v1/responses`** — Responses API (newer, used by OpenAI Agents SDK and n8n's "Message a Model" node). Same model routing as chat completions; stateless (no `previous_response_id` chain). +// @description - **`GET /v1/models`** — list available models discovered from installed agents. +// @description +// @description ## Quick start +// @description +// @description 1. Start the server: `cell serve` (or `cell serve --port 9090`) +// @description 2. The server prints an API key on startup. Set `DEVCELL_API_KEY` env var to use a fixed key. +// @description 3. Send requests with `Authorization: Bearer ` header. +// @description +// @description ## Model routing +// @description +// @description The `model` field selects which agent handles the prompt (same for both `/v1/chat/completions` and `/v1/responses`): +// @description - `"claude"` or `"anthropic"` → routes to the Claude Code CLI +// @description - `"opencode"` → routes to the OpenCode CLI +// @description - `"anthropic/sonnet"`, `"claude/claude-sonnet-4-5"` → Claude Code with a specific sub-model +// @description +// @description ## Limitations +// @description +// @description - **Streaming is not supported.** Requests with `"stream": true` to `/v1/responses` return 400; `/v1/chat/completions` returns the full response synchronously regardless. +// @description - **No tool calling.** The `tools` field is accepted for compatibility but never invokes a tool — the underlying CLI agents have their own internal tool loop. +// @description - **Stateless.** `previous_response_id` is accepted and ignored; clients must re-send full conversation history each request. +// @description - **Token usage is stubbed at zero** in responses. +// @description +// @description ## Reasoning effort +// @description +// @description Both endpoints honor the OpenAI `reasoning_effort` / `reasoning.effort` field (values: `low`, `medium`, `high`). It maps to the `claude --effort` CLI flag, controlling thinking budget on a per-request basis. Non-spec values (e.g. Claude's `xhigh`/`max`) are silently dropped. +// @description +// @description ## Debug logging +// @description +// @description By default the server logs only request metadata (method, path, status, duration) plus agent metadata at DEBUG level. Prompt and response bodies are **never** logged — they often contain secrets, PII, or large pasted content from upstream tools. +// @description +// @description Set `DEVCELL_LOG_PROMPTS=1` (combined with `LOG_LEVEL=info` or lower) to log the full assembled prompt and the model's reply for every `/v1/chat/completions` and `/v1/responses` request. Use only for debugging client integrations; do not leave on in production. +// @description +// @description ## Example curl +// @description +// @description Chat Completions: +// @description ```bash +// @description curl http://localhost:8484/v1/chat/completions \ +// @description -H "Authorization: Bearer $DEVCELL_API_KEY" \ +// @description -H "Content-Type: application/json" \ +// @description -d '{"model":"anthropic/sonnet","messages":[{"role":"user","content":"explain this repo"}]}' +// @description ``` +// @description +// @description Responses API: +// @description ```bash +// @description curl http://localhost:8484/v1/responses \ +// @description -H "Authorization: Bearer $DEVCELL_API_KEY" \ +// @description -H "Content-Type: application/json" \ +// @description -d '{"model":"anthropic/sonnet","input":"explain this repo"}' +// @description ``` +// @host localhost:8484 +// @BasePath / +// @securityDefinitions.apikey BearerAuth +// @in header +// @name Authorization +// @description Bearer token — set via DEVCELL_API_KEY env var or use the auto-generated key printed on startup. Format: `Bearer dcl-abc123...` + var serveCmd = &cobra.Command{ Use: "serve", Short: "Start HTTP API server for LLM commands", - Long: `Starts an OpenAI-compatible HTTP server that proxies chat completions + Long: `Starts an OpenAI-compatible HTTP server that proxies requests to LLM agent binaries (claude, opencode). Endpoints: - POST /v1/chat/completions — OpenAI chat completions API - GET /api/v1/health — health check + POST /v1/chat/completions — OpenAI Chat Completions API + POST /v1/responses — OpenAI Responses API (newer; n8n, Agents SDK) + GET /v1/models — list available models + GET /healthz — health check (k8s convention) + GET /api/v1/health — health check (REST convention) + GET /api/openapi.json — OpenAPI spec + GET /swagger/ — Swagger UI The model field selects the agent: "claude", "opencode", or -"claude/claude-sonnet-4-5" (agent/submodel). - -Request: - - {"model": "claude", "messages": [{"role": "user", "content": "explain this"}]} +"anthropic/sonnet", "claude/claude-sonnet-4-5" (agent/submodel). + +Chat Completions request: + + {"model": "anthropic/sonnet", + "messages": [{"role": "user", "content": "explain this"}]} + +Responses API request: + + {"model": "anthropic/sonnet", "input": "explain this"} + +Streaming is supported on /v1/chat/completions and /v1/responses for +the claude agent (set "stream": true). Server-Sent Events are emitted +in the standard OpenAI shape: chat.completion.chunk frames terminated +by data: [DONE] for chat, response. frames terminated by +response.completed for responses. Heartbeat ":keepalive" comments are +sent every 15s while idle so long-running agentic turns survive proxy +idle timeouts. Opencode has no streaming surface and falls back to +buffered. previous_response_id and tools are accepted but ignored. + +The claude binary is invoked with --dangerously-skip-permissions (same +default cell claude uses) — without it any tool call would block on the +permission gate, since the served claude has no TTY for stdin. The +operator's auth boundary is the bearer API key (DEVCELL_API_KEY). + +Environment: + + DEVCELL_API_KEY Bearer token (auto-generated if empty) + PORT Listen port (overridden by --port) + LOG_LEVEL debug|info|warn|error (default: warn) + DEVCELL_LOG_PROMPTS=1 Log full prompt + response bodies at INFO level + (off by default; prompts may contain secrets) + DEVCELL_SYSTEM_PROMPT Inline system prompt (overridden by --system-prompt) + DEVCELL_SYSTEM_PROMPT_FILE Path to a file used as the system prompt + (overridden by --system-prompt-file) + +System-prompt resolution order (first match wins): + 1. --system-prompt-file + 2. --system-prompt + 3. DEVCELL_SYSTEM_PROMPT_FILE + 4. DEVCELL_SYSTEM_PROMPT + 5. [llm].system_prompt_file in devcell.toml (path relative to project) + 6. [llm].system_prompt in devcell.toml (inline) + +A container-context preamble (bind mounts, host paths, runtime +constraints) is auto-prepended to whichever prompt resolves above. +Per-request 'instructions' (Responses) / 'system' role (Chat) from +the API body still merge into the user prompt independently. Examples: cell serve - cell serve --port 9090`, + cell serve --port 9090 + cell serve --system-prompt-file ./SYSTEM.md + cell serve --system-prompt "You are a backend code assistant for project X." + DEVCELL_LOG_PROMPTS=1 LOG_LEVEL=info cell serve # debug a client integration`, RunE: runServe, } -var servePort int +var ( + servePort int + serveSystemPrompt string + serveSystemPromptFile string +) func init() { serveCmd.Flags().IntVar(&servePort, "port", serve.DefaultPort, "port to listen on") + serveCmd.Flags().StringVar(&serveSystemPrompt, "system-prompt", "", + "system prompt passed to claude as --append-system-prompt on every request "+ + "(env: DEVCELL_SYSTEM_PROMPT). Composes with per-request `instructions`/`system` from the API body.") + serveCmd.Flags().StringVar(&serveSystemPromptFile, "system-prompt-file", "", + "path to a file whose contents are used as the system prompt "+ + "(env: DEVCELL_SYSTEM_PROMPT_FILE). Mutually exclusive with --system-prompt.") } func runServe(cmd *cobra.Command, args []string) error { + logLevel := os.Getenv("LOG_LEVEL") + if logLevel == "" { + logLevel = "warn" + } + logger.Initialize(logLevel, true) // plain text, no colors for server logs + + // Port priority: --port flag > PORT env > default + if !cmd.Flags().Changed("port") { + if envPort := os.Getenv("PORT"); envPort != "" { + if p, err := strconv.Atoi(envPort); err == nil { + servePort = p + } + } + } + apiKey := os.Getenv("DEVCELL_API_KEY") - if apiKey == "" { + generated := apiKey == "" + if generated { apiKey = serve.GenerateAPIKey() } @@ -56,9 +207,35 @@ func runServe(cmd *cobra.Command, args []string) error { ctx, cancel := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM) defer cancel() + c, err := config.LoadFromOS() + if err != nil { + return fmt.Errorf("load config: %w", err) + } + cellCfg := cfg.LoadFromOS(c.ConfigDir, c.BaseDir) + + systemPrompt, err := runner.AssembleSystemPrompt(c, cellCfg, runner.ResolveOpts{ + FlagFile: serveSystemPromptFile, + FlagInline: serveSystemPrompt, + EnvFile: os.Getenv("DEVCELL_SYSTEM_PROMPT_FILE"), + EnvInline: os.Getenv("DEVCELL_SYSTEM_PROMPT"), + CellCfg: cellCfg, + CfgBaseDir: c.BaseDir, + }) + if err != nil { + return fmt.Errorf("system prompt: %w", err) + } + exec := &serve.ShellExecutor{} srv := serve.NewServer(exec, servePort) srv.SetAPIKey(apiKey) + srv.SetSystemPrompt(systemPrompt) + // Off by default. Setting DEVCELL_LOG_PROMPTS=1 makes /v1/chat/completions + // and /v1/responses log full prompt + response text at INFO level. Useful + // for debugging client integrations; risky for prod logs because prompts + // often carry secrets / PII / large pasted content. + if os.Getenv("DEVCELL_LOG_PROMPTS") == "1" { + srv.SetLogPrompts(true) + } addr, errCh := srv.Start(ctx) if addr == "" { @@ -66,7 +243,9 @@ func runServe(cmd *cobra.Command, args []string) error { } fmt.Fprintf(os.Stderr, "devcell serve listening on %s\n", addr) - fmt.Fprintf(os.Stderr, "API key: %s\n", apiKey) + if generated { + fmt.Fprintf(os.Stderr, "API key: %s\n", apiKey) + } return <-errCh } diff --git a/docker-bake.hcl b/docker-bake.hcl index 0e892d39..58f991f2 100644 --- a/docker-bake.hcl +++ b/docker-bake.hcl @@ -5,8 +5,12 @@ # docker buildx bake # builds default group (ci) # docker buildx bake base # single target # docker buildx bake release # all release variants -# docker buildx bake --push release # build + push (gzip) -# docker buildx bake --set '*.output=type=image,push=true,compression=zstd,compression-level=3,force-compression=true' release # push with zstd +# docker buildx bake --push release # build + push (gzip, no provenance) +# +# Compression and provenance are pinned in `_base-args` (gzip + provenance=false) +# to maximise pull compatibility with older Docker daemons and registries that +# choke on zstd layers or OCI provenance attestations. Override per-invocation +# with `--set '*.output=...'` and `--set '*.attest=...'` if needed. # # Variables can be overridden via env: # VERSION=1.2.3 docker buildx bake release @@ -66,6 +70,13 @@ target "_base-args" { USER_GID = USER_GID GIT_COMMIT = GIT_COMMIT } + attest = [ + "type=provenance,disabled=true", + "type=sbom,disabled=true", + ] + output = [ + "type=image,compression=gzip,force-compression=true", + ] } # ── Stack image targets ────────────────────────────────────────────────────── diff --git a/go.mod b/go.mod index d4ae2969..d9d3b3a3 100644 --- a/go.mod +++ b/go.mod @@ -4,6 +4,8 @@ go 1.24.1 require ( github.com/BurntSushi/toml v1.4.0 + github.com/charmbracelet/bubbles v0.21.1-0.20250623103423-23b8fd6302d7 + github.com/charmbracelet/bubbletea v1.3.10 github.com/charmbracelet/huh v1.0.0 github.com/charmbracelet/lipgloss v1.1.0 github.com/charmbracelet/log v1.0.0 @@ -12,13 +14,17 @@ require ( github.com/ollama/ollama v0.17.6 github.com/openai/openai-go v1.12.0 github.com/spf13/cobra v1.10.2 + github.com/swaggo/http-swagger/v2 v2.0.2 + github.com/swaggo/swag v1.16.6 github.com/testcontainers/testcontainers-go v0.40.0 golang.org/x/mod v0.30.0 + gopkg.in/yaml.v3 v3.0.1 ) require ( dario.cat/mergo v1.0.2 // indirect github.com/Azure/go-ansiterm v0.0.0-20210617225240-d185dfc1b5a1 // indirect + github.com/KyleBanks/depth v1.2.1 // indirect github.com/Microsoft/go-winio v0.6.2 // indirect github.com/atotto/clipboard v0.1.4 // indirect github.com/aymanbagabas/go-osc52/v2 v2.0.1 // indirect @@ -27,8 +33,6 @@ require ( github.com/catppuccin/go v0.3.0 // indirect github.com/cenkalti/backoff/v4 v4.3.0 // indirect github.com/cespare/xxhash/v2 v2.3.0 // indirect - github.com/charmbracelet/bubbles v0.21.1-0.20250623103423-23b8fd6302d7 // indirect - github.com/charmbracelet/bubbletea v1.3.10 // indirect github.com/charmbracelet/colorprofile v0.2.3-0.20250311203215-f60798e515dc // indirect github.com/charmbracelet/x/ansi v0.10.1 // indirect github.com/charmbracelet/x/cellbuf v0.0.13 // indirect @@ -39,7 +43,6 @@ require ( github.com/containerd/log v0.1.0 // indirect github.com/containerd/platforms v0.2.1 // indirect github.com/cpuguy83/dockercfg v0.3.2 // indirect - github.com/cpuguy83/go-md2man/v2 v2.0.6 // indirect github.com/davecgh/go-spew v1.1.1 // indirect github.com/distribution/reference v0.6.0 // indirect github.com/docker/go-connections v0.6.0 // indirect @@ -52,8 +55,13 @@ require ( github.com/go-logr/logr v1.4.3 // indirect github.com/go-logr/stdr v1.2.2 // indirect github.com/go-ole/go-ole v1.2.6 // indirect + github.com/go-openapi/jsonpointer v0.19.5 // indirect + github.com/go-openapi/jsonreference v0.20.0 // indirect + github.com/go-openapi/spec v0.20.6 // indirect + github.com/go-openapi/swag v0.19.15 // indirect github.com/google/uuid v1.6.0 // indirect github.com/inconshreveable/mousetrap v1.1.0 // indirect + github.com/josharian/intern v1.0.0 // indirect github.com/klauspost/compress v1.18.3 // indirect github.com/lucasb-eyer/go-colorful v1.2.0 // indirect github.com/lufia/plan9stats v0.0.0-20211012122336-39d0f177ccd0 // indirect @@ -80,11 +88,11 @@ require ( github.com/pmezard/go-difflib v1.0.0 // indirect github.com/power-devops/perfstat v0.0.0-20210106213030-5aafc221ea8c // indirect github.com/rivo/uniseg v0.4.7 // indirect - github.com/russross/blackfriday/v2 v2.1.0 // indirect github.com/shirou/gopsutil/v4 v4.25.6 // indirect github.com/sirupsen/logrus v1.9.3 // indirect github.com/spf13/pflag v1.0.9 // indirect github.com/stretchr/testify v1.11.1 // indirect + github.com/swaggo/files/v2 v2.0.0 // indirect github.com/tidwall/gjson v1.14.4 // indirect github.com/tidwall/match v1.1.1 // indirect github.com/tidwall/pretty v1.2.1 // indirect @@ -102,11 +110,12 @@ require ( go.opentelemetry.io/otel/sdk v1.40.0 // indirect go.opentelemetry.io/otel/trace v1.40.0 // indirect go.opentelemetry.io/proto/otlp v1.9.0 // indirect - go.yaml.in/yaml/v3 v3.0.4 // indirect golang.org/x/crypto v0.43.0 // indirect golang.org/x/exp v0.0.0-20250218142911-aa4b98e5adaa // indirect + golang.org/x/sync v0.17.0 // indirect golang.org/x/sys v0.40.0 // indirect golang.org/x/text v0.30.0 // indirect + golang.org/x/tools v0.38.0 // indirect google.golang.org/protobuf v1.36.11 // indirect - gopkg.in/yaml.v3 v3.0.1 // indirect + gopkg.in/yaml.v2 v2.4.0 // indirect ) diff --git a/go.sum b/go.sum index 3bcefc6a..baa2fbe9 100644 --- a/go.sum +++ b/go.sum @@ -6,6 +6,8 @@ github.com/Azure/go-ansiterm v0.0.0-20210617225240-d185dfc1b5a1 h1:UQHMgLO+TxOEl github.com/Azure/go-ansiterm v0.0.0-20210617225240-d185dfc1b5a1/go.mod h1:xomTg63KZ2rFqZQzSB4Vz2SUXa1BpHTVz9L5PTmPC4E= github.com/BurntSushi/toml v1.4.0 h1:kuoIxZQy2WRRk1pttg9asf+WVv6tWQuBNVmK8+nqPr0= github.com/BurntSushi/toml v1.4.0/go.mod h1:ukJfTF/6rtPPRCnwkur4qwRxa8vTRFBF0uk2lLoLwho= +github.com/KyleBanks/depth v1.2.1 h1:5h8fQADFrWtarTdtDudMmGsC7GPbOAu6RVB3ffsVFHc= +github.com/KyleBanks/depth v1.2.1/go.mod h1:jzSb9d0L43HxTQfT+oSA1EEp2q+ne2uh6XgeJcm8brE= github.com/MakeNowJust/heredoc v1.0.0 h1:cXCdzVdstXyiTqTvfqk9SDHpKNjxuom+DOlyEeQ4pzQ= github.com/MakeNowJust/heredoc v1.0.0/go.mod h1:mG5amYoWBHf8vpLOuehzbGGw0EHxpZZ6lCpQ4fNJ8LE= github.com/Microsoft/go-winio v0.6.2 h1:F2VQgta7ecxGYO8k3ZZz3RS8fVIXVxONVUPlNERoyfY= @@ -66,8 +68,8 @@ github.com/containerd/platforms v0.2.1 h1:zvwtM3rz2YHPQsF2CHYM8+KtB5dvhISiXh5ZpS github.com/containerd/platforms v0.2.1/go.mod h1:XHCb+2/hzowdiut9rkudds9bE5yJ7npe7dG/wG+uFPw= github.com/cpuguy83/dockercfg v0.3.2 h1:DlJTyZGBDlXqUZ2Dk2Q3xHs/FtnooJJVaad2S9GKorA= github.com/cpuguy83/dockercfg v0.3.2/go.mod h1:sugsbF4//dDlL/i+S+rtpIWp+5h0BHJHfjj5/jFyUJc= -github.com/cpuguy83/go-md2man/v2 v2.0.6 h1:XJtiaUW6dEEqVuZiMTn1ldk455QWwEIsMIJlo5vtkx0= github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g= +github.com/creack/pty v1.1.9/go.mod h1:oKZEueFk5CKHvIhNR5MUki03XCEU+Q6VDXinZuGJ33E= github.com/creack/pty v1.1.24 h1:bJrF4RRfyJnbTJqzRLHzcGaZK1NeM5kTC9jGgovnR1s= github.com/creack/pty v1.1.24/go.mod h1:08sCNb52WyoAwi2QDyzUCTgcvVFhUzewun7wtTfvcwE= github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= @@ -98,6 +100,16 @@ github.com/go-logr/stdr v1.2.2 h1:hSWxHoqTgW2S2qGc0LTAI563KZ5YKYRhT3MFKZMbjag= github.com/go-logr/stdr v1.2.2/go.mod h1:mMo/vtBO5dYbehREoey6XUKy/eSumjCCveDpRre4VKE= github.com/go-ole/go-ole v1.2.6 h1:/Fpf6oFPoeFik9ty7siob0G6Ke8QvQEuVcuChpwXzpY= github.com/go-ole/go-ole v1.2.6/go.mod h1:pprOEPIfldk/42T2oK7lQ4v4JSDwmV0As9GaiUsvbm0= +github.com/go-openapi/jsonpointer v0.19.3/go.mod h1:Pl9vOtqEWErmShwVjC8pYs9cog34VGT37dQOVbmoatg= +github.com/go-openapi/jsonpointer v0.19.5 h1:gZr+CIYByUqjcgeLXnQu2gHYQC9o73G2XUeOFYEICuY= +github.com/go-openapi/jsonpointer v0.19.5/go.mod h1:Pl9vOtqEWErmShwVjC8pYs9cog34VGT37dQOVbmoatg= +github.com/go-openapi/jsonreference v0.20.0 h1:MYlu0sBgChmCfJxxUKZ8g1cPWFOB37YSZqewK7OKeyA= +github.com/go-openapi/jsonreference v0.20.0/go.mod h1:Ag74Ico3lPc+zR+qjn4XBUmXymS4zJbYVCZmcgkasdo= +github.com/go-openapi/spec v0.20.6 h1:ich1RQ3WDbfoeTqTAb+5EIxNmpKVJZWBNah9RAT0jIQ= +github.com/go-openapi/spec v0.20.6/go.mod h1:2OpW+JddWPrpXSCIX8eOx7lZ5iyuWj3RYR6VaaBKcWA= +github.com/go-openapi/swag v0.19.5/go.mod h1:POnQmlKehdgb5mhVOsnJFsivZCEZ/vjK9gh66Z9tfKk= +github.com/go-openapi/swag v0.19.15 h1:D2NRCBzS9/pEY3gP9Nl8aDqGUcPFrwG2p+CNFrLyrCM= +github.com/go-openapi/swag v0.19.15/go.mod h1:QYRuS/SOXUCsnplDa677K7+DxSOj6IPNl/eQntq43wQ= github.com/google/go-cmp v0.5.6/go.mod h1:v8dTdLbMG2kIc/vJvl+f65V22dbkXbowE6jgT/gNBxE= github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= @@ -107,11 +119,15 @@ github.com/grpc-ecosystem/grpc-gateway/v2 v2.27.2 h1:8Tjv8EJ+pM1xP8mK6egEbD1OgnV github.com/grpc-ecosystem/grpc-gateway/v2 v2.27.2/go.mod h1:pkJQ2tZHJ0aFOVEEot6oZmaVEZcRme73eIFmhiVuRWs= github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8= github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw= +github.com/josharian/intern v1.0.0 h1:vlS4z54oSdjm0bgjRigI+G1HpF+tI+9rE5LLzOg8HmY= github.com/josharian/intern v1.0.0/go.mod h1:5DoeVV0s6jJacbCEi61lwdGj/aVlrQvzHFFd8Hwg//Y= github.com/klauspost/compress v1.18.3 h1:9PJRvfbmTabkOX8moIpXPbMMbYN60bWImDDU7L+/6zw= github.com/klauspost/compress v1.18.3/go.mod h1:R0h/fSBs8DE4ENlcrlib3PsXS61voFxhIs2DeRhCvJ4= +github.com/kr/pretty v0.1.0/go.mod h1:dAy3ld7l9f0ibDNOQOHHMYYIIbhfbHSm3C4ZsoJORNo= github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE= github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk= +github.com/kr/pty v1.1.1/go.mod h1:pFQYn66WHrOpPYNljwOMqo10TkYh1fy3cYio2l3bCsQ= +github.com/kr/text v0.1.0/go.mod h1:4Jbv+DJW3UT/LiOwJeYQe1efqtUx/iVham/4vfdArNI= github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE= github.com/lucasb-eyer/go-colorful v1.2.0 h1:1nnpGOrhyZZuNyfu1QjKiUICQ74+3FNCN69Aj6K7nkY= @@ -120,6 +136,9 @@ github.com/lufia/plan9stats v0.0.0-20211012122336-39d0f177ccd0 h1:6E+4a0GO5zZEnZ github.com/lufia/plan9stats v0.0.0-20211012122336-39d0f177ccd0/go.mod h1:zJYVVT2jmtg6P3p1VtQj7WsuWi/y4VnjVBn7F8KPB3I= github.com/magiconair/properties v1.8.10 h1:s31yESBquKXCV9a/ScB3ESkOjUYYv+X0rg8SYxI99mE= github.com/magiconair/properties v1.8.10/go.mod h1:Dhd985XPs7jluiymwWYZ0G4Z61jb3vdS329zhj2hYo0= +github.com/mailru/easyjson v0.0.0-20190614124828-94de47d64c63/go.mod h1:C1wdFJiN94OJF2b5HbByQZoLdCWB1Yqtg26g4irojpc= +github.com/mailru/easyjson v0.0.0-20190626092158-b2ccc519800e/go.mod h1:C1wdFJiN94OJF2b5HbByQZoLdCWB1Yqtg26g4irojpc= +github.com/mailru/easyjson v0.7.6/go.mod h1:xzfreul335JAWq5oZzymOObrkdz5UnU4kGfJJLY9Nlc= github.com/mailru/easyjson v0.7.7 h1:UGYAvKxe3sBsEDzO8ZeWOSlIQfWFlxbzLZe7hwFURr0= github.com/mailru/easyjson v0.7.7/go.mod h1:xzfreul335JAWq5oZzymOObrkdz5UnU4kGfJJLY9Nlc= github.com/mattn/go-isatty v0.0.20 h1:xfD0iDuEKnDkl03q4limB+vH+GxLEtL/jb4xVJSWWEY= @@ -154,6 +173,7 @@ github.com/muesli/cancelreader v0.2.2 h1:3I4Kt4BQjOR54NavqnDogx/MIoWBFa0StPA8ELU github.com/muesli/cancelreader v0.2.2/go.mod h1:3XuTXfFS2VjM+HTLZY9Ak0l6eUKfijIfMUZ4EgX0QYo= github.com/muesli/termenv v0.16.0 h1:S5AlUN9dENB57rsbnkPyfdGuWIlkmzJjbFf0Tf5FWUc= github.com/muesli/termenv v0.16.0/go.mod h1:ZRfOIKPFDYQoDFF4Olj7/QJbW60Ol/kL1pU3VfY/Cnk= +github.com/niemeyer/pretty v0.0.0-20200227124842-a10e7caefd8e/go.mod h1:zD1mROLANZcx1PVRCS0qkT7pwLkGfwJo4zjcN/Tysno= github.com/ollama/ollama v0.17.6 h1:zxInCopQToAMm+OniZSiHFcry03kiL6i1mmcTvpK4Us= github.com/ollama/ollama v0.17.6/go.mod h1:tCX4IMV8DHjl3zY0THxuEkpWDZSOchJpzTuLACpMwFw= github.com/openai/openai-go v1.12.0 h1:NBQCnXzqOTv5wsgNC36PrFEiskGfO5wccfCWDo9S1U0= @@ -173,8 +193,6 @@ github.com/rivo/uniseg v0.4.7 h1:WUdvkW8uEhrYfLC4ZzdpI2ztxP1I582+49Oc5Mq64VQ= github.com/rivo/uniseg v0.4.7/go.mod h1:FN3SvrM+Zdj16jyLfmOkMNblXMcoc8DfTHruCPUcx88= github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ= github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc= -github.com/russross/blackfriday v1.6.0 h1:KqfZb0pUVN2lYqZUYRddxF4OR8ZMURnJIG5Y3VRLtww= -github.com/russross/blackfriday/v2 v2.1.0 h1:JIOH55/0cWyOuilr9/qlrm0BSXldqnqwMsf35Ld67mk= github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM= github.com/shirou/gopsutil/v4 v4.25.6 h1:kLysI2JsKorfaFPcYmcJqbzROzsBWEOAtw6A7dIfqXs= github.com/shirou/gopsutil/v4 v4.25.6/go.mod h1:PfybzyydfZcN+JMMjkF6Zb8Mq1A/VcogFFg7hj50W9c= @@ -187,9 +205,17 @@ github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME= github.com/stretchr/objx v0.5.2 h1:xuMeJ0Sdp5ZMRXx/aWO6RZxdr3beISkG5/G/aIRr3pY= github.com/stretchr/objx v0.5.2/go.mod h1:FRsXN1f5AsAjCGJKqEizvkpNtU+EGNCLh3NxZ/8L+MA= +github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI= +github.com/stretchr/testify v1.6.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U= github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U= +github.com/swaggo/files/v2 v2.0.0 h1:hmAt8Dkynw7Ssz46F6pn8ok6YmGZqHSVLZ+HQM7i0kw= +github.com/swaggo/files/v2 v2.0.0/go.mod h1:24kk2Y9NYEJ5lHuCra6iVwkMjIekMCaFq/0JQj66kyM= +github.com/swaggo/http-swagger/v2 v2.0.2 h1:FKCdLsl+sFCx60KFsyM0rDarwiUSZ8DqbfSyIKC9OBg= +github.com/swaggo/http-swagger/v2 v2.0.2/go.mod h1:r7/GBkAWIfK6E/OLnE8fXnviHiDeAHmgIyooa4xm3AQ= +github.com/swaggo/swag v1.16.6 h1:qBNcx53ZaX+M5dxVyTrgQ0PJ/ACK+NzhwcbieTt+9yI= +github.com/swaggo/swag v1.16.6/go.mod h1:ngP2etMK5a0P3QBizic5MEwpRmluJZPHjXcMoj4Xesg= github.com/testcontainers/testcontainers-go v0.40.0 h1:pSdJYLOVgLE8YdUY2FHQ1Fxu+aMnb6JfVz1mxk7OeMU= github.com/testcontainers/testcontainers-go v0.40.0/go.mod h1:FSXV5KQtX2HAMlm7U3APNyLkkap35zNLxukw9oBi/MY= github.com/tidwall/gjson v1.14.2/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk= @@ -230,7 +256,6 @@ go.opentelemetry.io/otel/trace v1.40.0 h1:WA4etStDttCSYuhwvEa8OP8I5EWu24lkOzp+ZY go.opentelemetry.io/otel/trace v1.40.0/go.mod h1:zeAhriXecNGP/s2SEG3+Y8X9ujcJOTqQ5RgdEJcawiA= go.opentelemetry.io/proto/otlp v1.9.0 h1:l706jCMITVouPOqEnii2fIAuO3IVGBRPV5ICjceRb/A= go.opentelemetry.io/proto/otlp v1.9.0/go.mod h1:xE+Cx5E/eEHw+ISFkwPLwCZefwVjY+pqKg1qcK03+/4= -go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc= go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg= golang.org/x/crypto v0.43.0 h1:dduJYIi3A3KOfdGOHX8AVZ/jGiyPa3IbBozJ5kNuE04= golang.org/x/crypto v0.43.0/go.mod h1:BFbav4mRNlXJL4wNeejLpWxB7wMbc79PdRGhWKncxR0= @@ -240,6 +265,8 @@ golang.org/x/mod v0.30.0 h1:fDEXFVZ/fmCKProc/yAXXUijritrDzahmwwefnjoPFk= golang.org/x/mod v0.30.0/go.mod h1:lAsf5O2EvJeSFMiBxXDki7sCgAxEUcZHXoXMKT4GJKc= golang.org/x/net v0.46.0 h1:giFlY12I07fugqwPuWJi68oOnpfqFnJIJzaIIm2JVV4= golang.org/x/net v0.46.0/go.mod h1:Q9BGdFy1y4nkUwiLvT5qtyhAnEHgnQ/zd8PfU6nc210= +golang.org/x/sync v0.17.0 h1:l60nONMj9l5drqw6jlhIELNv9I0A4OFgRsG9k2oT9Ug= +golang.org/x/sync v0.17.0/go.mod h1:9KTHXmSnoGruLpwFjVSX0lNNA75CykiMECbovNTZqGI= golang.org/x/sys v0.0.0-20190916202348-b4ddaad3f8a3/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= golang.org/x/sys v0.0.0-20201204225414-ed752295db88/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= golang.org/x/sys v0.0.0-20210616094352-59db8d763f22/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= @@ -256,6 +283,8 @@ golang.org/x/text v0.30.0 h1:yznKA/E9zq54KzlzBEAWn1NXSQ8DIp/NYMy88xJjl4k= golang.org/x/text v0.30.0/go.mod h1:yDdHFIX9t+tORqspjENWgzaCVXgk0yYnYuSZ8UzzBVM= golang.org/x/time v0.0.0-20220210224613-90d013bbcef8 h1:vVKdlvoWBphwdxWKrFZEuM0kGgGLxUOYcY4U/2Vjg44= golang.org/x/time v0.0.0-20220210224613-90d013bbcef8/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ= +golang.org/x/tools v0.38.0 h1:Hx2Xv8hISq8Lm16jvBZ2VQf+RLmbd7wVUsALibYI/IQ= +golang.org/x/tools v0.38.0/go.mod h1:yEsQ/d/YK8cjh0L6rZlY8tgtlKiBNTL14pGDJPJpYQs= golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= google.golang.org/genproto/googleapis/api v0.0.0-20250825161204-c5933d9347a5 h1:BIRfGDEjiHRrk0QKZe3Xv2ieMhtgRGeLcZQ0mIVn4EY= google.golang.org/genproto/googleapis/api v0.0.0-20250825161204-c5933d9347a5/go.mod h1:j3QtIyytwqGr1JUDtYXwtMXWPKsEa5LtzIFN1Wn5WvE= @@ -266,9 +295,15 @@ google.golang.org/grpc v1.75.1/go.mod h1:JtPAzKiq4v1xcAB2hydNlWI2RnF85XXcV0mhKXr google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE= google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= +gopkg.in/check.v1 v1.0.0-20180628173108-788fd7840127/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= +gopkg.in/check.v1 v1.0.0-20200227125254-8fa46927fb4f/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q= +gopkg.in/yaml.v2 v2.2.2/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI= +gopkg.in/yaml.v2 v2.4.0 h1:D8xgwECY7CYvx+Y2n4sBz93Jn9JRvxdiyyo8CTfuKaY= +gopkg.in/yaml.v2 v2.4.0/go.mod h1:RDklbk79AGWmwhnvt/jBztapEOGDOx6ZbXqjP6csGnQ= gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= +gopkg.in/yaml.v3 v3.0.0-20200615113413-eeeca48fe776/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= gotest.tools/v3 v3.5.2 h1:7koQfIKdy+I8UTetycgUqXWSDwpgv193Ka+qRsmBY8Q= diff --git a/images/Dockerfile b/images/Dockerfile index e1cba17f..eed2e82d 100644 --- a/images/Dockerfile +++ b/images/Dockerfile @@ -14,6 +14,7 @@ WORKDIR /src COPY go.mod go.sum ./ RUN go mod download COPY . . +RUN go run github.com/swaggo/swag/cmd/swag@latest init -g cmd/serve.go -o docs --parseDependency --parseInternal RUN CGO_ENABLED=0 go build -o /cell ./cmd ############################################################################### diff --git a/internal/cfg/cfg.go b/internal/cfg/cfg.go index 32f5657b..9d6b25ef 100644 --- a/internal/cfg/cfg.go +++ b/internal/cfg/cfg.go @@ -26,7 +26,8 @@ type CellSection struct { Engine string `toml:"engine"` // execution engine: "docker" (default) or "vagrant" VagrantProvider string `toml:"vagrant_provider"` // vagrant provider: "utm" (default) or "libvirt" VagrantBox string `toml:"vagrant_box"` // vagrant box name override (default: "utm/bookworm") - DockerPrivileged bool `toml:"docker_privileged"` // run container with --privileged; default: false + DockerPrivileged bool `toml:"docker_privileged"` // run container with --privileged; default: false + PerSessionImage *bool `toml:"per_session_image"` // tag user image per tmux session instead of per stack; default: false } // ResolvedRegistry returns the effective registry: env > toml > default. @@ -48,6 +49,14 @@ func (c CellSection) ResolvedGUI() bool { return *c.GUI } +// ResolvedPerSessionImage returns true only when explicitly enabled. +func (c CellSection) ResolvedPerSessionImage() bool { + if c.PerSessionImage == nil { + return false + } + return *c.PerSessionImage +} + // ResolvedStack returns Stack if set, else "base". func (c CellSection) ResolvedStack() string { if c.Stack != "" { @@ -80,10 +89,17 @@ type LLMModelsSection struct { } // LLMSection holds [llm] config — all AI agent settings in one place. +// +// SystemPrompt and SystemPromptFile are mutually exclusive — set one or +// neither. The resolver in internal/runner.ResolveSystemPrompt validates +// this and returns an error when both are set, so we don't fail config +// load for projects where the conflict is harmless (e.g. callers that +// don't read system prompts). type LLMSection struct { - SystemPrompt string `toml:"system_prompt"` - UseOllama bool `toml:"use_ollama"` - Models LLMModelsSection `toml:"models"` + SystemPrompt string `toml:"system_prompt"` + SystemPromptFile string `toml:"system_prompt_file"` + UseOllama bool `toml:"use_ollama"` + Models LLMModelsSection `toml:"models"` } // GitSection holds [git] config for git identity inside the container. @@ -242,12 +258,18 @@ func Merge(global, project CellConfig) CellConfig { if project.Cell.DockerPrivileged { out.Cell.DockerPrivileged = true } + if project.Cell.PerSessionImage != nil { + out.Cell.PerSessionImage = project.Cell.PerSessionImage + } // LLM: project wins for scalars, providers accumulate out.LLM = global.LLM if project.LLM.SystemPrompt != "" { out.LLM.SystemPrompt = project.LLM.SystemPrompt } + if project.LLM.SystemPromptFile != "" { + out.LLM.SystemPromptFile = project.LLM.SystemPromptFile + } if project.LLM.UseOllama { out.LLM.UseOllama = true } @@ -328,6 +350,10 @@ func ApplyEnv(c *CellConfig, getenv func(string) string) { if p := getenv("DEVCELL_NIXHOME_PATH"); p != "" { c.Cell.NixhomePath = p } + if v := getenv("DEVCELL_PER_SESSION_IMAGE"); v == "true" || v == "1" { + b := true + c.Cell.PerSessionImage = &b + } } // LoadLayered loads global + project files, merges them, then applies env overrides. diff --git a/internal/config/config.go b/internal/config/config.go index d79f2a0e..90c7d3a9 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -57,8 +57,8 @@ func Load(cwd string, getenv func(string) string) Config { ContainerName: "cell-" + appName + "-run", Hostname: "cell-" + appName, PortPrefix: portPrefix, - VNCPort: portPrefix + "50", - RDPPort: portPrefix + "89", + VNCPort: clampPort(portPrefix + "50"), + RDPPort: clampPort(portPrefix + "89"), BaseDir: cwd, HostUser: getenv("USER"), HostHome: home, @@ -150,6 +150,24 @@ func resolveAvailablePort(preferred string) string { return preferred } +// clampPort ensures a port string represents a valid TCP port (1024–65535). +// If the value exceeds 65535, it subtracts 65535 repeatedly until it fits, +// then floors at 1024 to stay out of the privileged range. +// Pure arithmetic — no I/O. Port availability is handled by ResolveAvailablePorts. +func clampPort(s string) string { + p, err := strconv.Atoi(s) + if err != nil || p <= 65535 { + return s + } + for p > 65535 { + p -= 65535 + } + if p < 1024 { + p += 1024 + } + return strconv.Itoa(p) +} + // isPortAvailable reports whether a TCP port can be bound on all interfaces. func isPortAvailable(port int) bool { ln, err := net.Listen("tcp", ":"+strconv.Itoa(port)) diff --git a/internal/config/config_test.go b/internal/config/config_test.go index 8421189b..4f8a1c42 100644 --- a/internal/config/config_test.go +++ b/internal/config/config_test.go @@ -228,6 +228,28 @@ func TestVNCPort_ParseableAsUint16(t *testing.T) { } } +func TestVNCPort_HighCellID_Clamped(t *testing.T) { + // CELL_ID=682 → portPrefix="682", VNCPort would be "68250" > 65535 + c := config.Load("/cwd", env("CELL_ID", "682")) + n, err := strconv.ParseUint(c.VNCPort, 10, 16) + if err != nil || n == 0 || n > 65535 { + t.Errorf("CellID=682 VNCPort=%q should be clamped to valid range, got parsed=%d err=%v", c.VNCPort, n, err) + } +} + +func TestVNCPort_PrefixPlusCellID_Clamped(t *testing.T) { + // SESSION_PORT_PREFIX="681" + CELL_ID="50" → portPrefix="68150", VNCPort would be "6815050" + c := config.Load("/cwd", env("SESSION_PORT_PREFIX", "681", "CELL_ID", "50")) + n, err := strconv.ParseUint(c.VNCPort, 10, 16) + if err != nil || n == 0 || n > 65535 { + t.Errorf("VNCPort=%q should be clamped to valid range", c.VNCPort) + } + n2, err := strconv.ParseUint(c.RDPPort, 10, 16) + if err != nil || n2 == 0 || n2 > 65535 { + t.Errorf("RDPPort=%q should be clamped to valid range", c.RDPPort) + } +} + // --- ContainerName --- func TestContainerName(t *testing.T) { diff --git a/internal/logger/logger.go b/internal/logger/logger.go index d7164859..e41b14ee 100644 --- a/internal/logger/logger.go +++ b/internal/logger/logger.go @@ -7,6 +7,7 @@ import ( "strings" charmlog "github.com/charmbracelet/log" + "github.com/muesli/termenv" ) var ( @@ -29,10 +30,17 @@ func Initialize(logLevel string, plain bool) { level = charmlog.InfoLevel } - logger := charmlog.NewWithOptions(os.Stderr, charmlog.Options{ + opts := charmlog.Options{ Level: level, - ReportTimestamp: false, - }) + ReportTimestamp: plain, + } + logger := charmlog.NewWithOptions(os.Stderr, opts) + if plain { + logger.SetFormatter(charmlog.TextFormatter) + logger.SetStyles(charmlog.DefaultStyles()) // reset to avoid nil + // Force no-color output for plain/server mode + logger.SetColorProfile(termenv.Ascii) + } defaultLogger = slog.New(logger) } diff --git a/internal/op/resolve.go b/internal/op/resolve.go index acb5eec2..775efa56 100644 --- a/internal/op/resolve.go +++ b/internal/op/resolve.go @@ -20,16 +20,23 @@ type item struct { // ResolveItems calls `op item get` for each item name and returns a merged // map of label→value for all fields that have both a label and a value. // Items later in the slice win on key conflict. -func ResolveItems(items []string) (map[string]string, error) { +// +// Resolution is optimistic: a failure on one item is recorded in the returned +// []error slice and the loop continues with the next item. The caller is +// expected to surface the per-item errors and apply whatever did resolve. +func ResolveItems(items []string) (map[string]string, []error) { env := make(map[string]string) + var errs []error for _, name := range items { out, err := exec.Command("op", "item", "get", name, "--format", "json", "--reveal", "--cache").Output() if err != nil { - return nil, fmt.Errorf("op item get %s: %w", name, err) + errs = append(errs, fmt.Errorf("op item get %s: %w", name, err)) + continue } var it item if err := json.Unmarshal(out, &it); err != nil { - return nil, fmt.Errorf("op item get %s: parse JSON: %w", name, err) + errs = append(errs, fmt.Errorf("op item get %s: parse JSON: %w", name, err)) + continue } for _, f := range it.Fields { if f.Label != "" && f.Value != "" { @@ -37,5 +44,5 @@ func ResolveItems(items []string) (map[string]string, error) { } } } - return env, nil + return env, errs } diff --git a/internal/runner/runner.go b/internal/runner/runner.go index f0687fdf..5155d056 100644 --- a/internal/runner/runner.go +++ b/internal/runner/runner.go @@ -3,12 +3,15 @@ package runner import ( "bytes" "context" + "crypto/sha256" + "encoding/hex" "encoding/json" "fmt" "io" "os" "os/exec" "path/filepath" + "sort" "strings" "syscall" "time" @@ -27,6 +30,18 @@ const ( // at startup; defaults to DefaultRegistry. var Registry = DefaultRegistry +// Stack is the resolved nix stack name (e.g. "ultimate", "go"). +// Set from CellConfig at startup; defaults to "base". +var Stack = "base" + +// Modules is the list of extra nix modules composed on top of the stack. +// Set from CellConfig at startup. +var Modules []string + +// PerSessionImage tags user images per tmux session instead of per stack. +// Set from CellConfig at startup; defaults to false (stack-based). +var PerSessionImage bool + // BaseImageTag returns the base image tag used in scaffold FROM, // allowing override via DEVCELL_BASE_IMAGE env var (local dev, CI, tests). func BaseImageTag() string { @@ -36,20 +51,41 @@ func BaseImageTag() string { return fmt.Sprintf("%s:%s-core", Registry, version.Version) } -// UserImageTag returns the per-session user image tag. -// Format: devcell-user: (e.g. devcell-user:main). +// UserImageTag returns the user image tag. +// Default (stack-based): devcell-user: or devcell-user:--- +// Legacy (per_session_image=true): devcell-user: (one image per tmux session) // Override with DEVCELL_USER_IMAGE env var (used by tests). func UserImageTag() string { if tag := os.Getenv("DEVCELL_USER_IMAGE"); tag != "" { return tag } - session := "main" + if PerSessionImage { + return "devcell-user:" + resolveSession() + } + tag := Stack + if tag == "" { + tag = "base" + } + if len(Modules) > 0 { + sorted := make([]string, len(Modules)) + copy(sorted, Modules) + sort.Strings(sorted) + tag += "-" + strings.Join(sorted, "-") + h := sha256.Sum256([]byte(strings.Join(sorted, ","))) + tag += "-" + hex.EncodeToString(h[:])[:8] + } + return "devcell-user:" + tag +} + +// resolveSession returns the session name from env vars (legacy per-session mode). +func resolveSession() string { if s := os.Getenv("DEVCELL_SESSION_NAME"); s != "" { - session = s - } else if s := os.Getenv("TMUX_SESSION_NAME"); s != "" { - session = s + return s + } + if s := os.Getenv("TMUX_SESSION_NAME"); s != "" { + return s } - return "devcell-user:" + session + return "main" } // FS abstracts filesystem stat for testability. @@ -337,6 +373,12 @@ func BuildImage(ctx context.Context, configDir string, noCache bool, verbose boo if noCache { args = append(args, "--no-cache", "--build-arg", "NIX_REFRESH=--refresh") } + // DEVCELL_DOCKER_BUILD_ARGS: space-separated extra --build-arg pairs (e.g. "FOO=bar BAZ=qux"). + if extra := os.Getenv("DEVCELL_DOCKER_BUILD_ARGS"); extra != "" { + for _, kv := range strings.Fields(extra) { + args = append(args, "--build-arg", kv) + } + } args = append(args, configDir) cmd := exec.CommandContext(ctx, "docker", args...) // Detach from the controlling TTY so Docker Desktop's BuildKit progress diff --git a/internal/runner/runner_test.go b/internal/runner/runner_test.go index 09c126ea..79388872 100644 --- a/internal/runner/runner_test.go +++ b/internal/runner/runner_test.go @@ -532,57 +532,114 @@ func min(a, b int) int { return b } -// --- UserImageTag per-session --- +// --- UserImageTag stack-based (default) --- -func TestUserImageTag_DefaultSession(t *testing.T) { +func withCleanImageState(t *testing.T) { + t.Helper() t.Setenv("DEVCELL_USER_IMAGE", "") t.Setenv("DEVCELL_SESSION_NAME", "") t.Setenv("TMUX_SESSION_NAME", "") + origStack := runner.Stack + origModules := runner.Modules + origPerSession := runner.PerSessionImage + t.Cleanup(func() { + runner.Stack = origStack + runner.Modules = origModules + runner.PerSessionImage = origPerSession + }) + runner.Stack = "base" + runner.Modules = nil + runner.PerSessionImage = false +} + +func TestUserImageTag_DefaultStack(t *testing.T) { + withCleanImageState(t) got := runner.UserImageTag() - if got != "devcell-user:main" { - t.Errorf("default session: want devcell-user:main, got %q", got) + if got != "devcell-user:base" { + t.Errorf("default stack: want devcell-user:base, got %q", got) } } -func TestUserImageTag_SessionName(t *testing.T) { - t.Setenv("DEVCELL_USER_IMAGE", "") - t.Setenv("DEVCELL_SESSION_NAME", "webdev") - t.Setenv("TMUX_SESSION_NAME", "") +func TestUserImageTag_UltimateStack(t *testing.T) { + withCleanImageState(t) + runner.Stack = "ultimate" got := runner.UserImageTag() - if got != "devcell-user:webdev" { - t.Errorf("session name: want devcell-user:webdev, got %q", got) + if got != "devcell-user:ultimate" { + t.Errorf("ultimate stack: want devcell-user:ultimate, got %q", got) } } -func TestUserImageTag_TmuxSessionFallback(t *testing.T) { - t.Setenv("DEVCELL_USER_IMAGE", "") - t.Setenv("DEVCELL_SESSION_NAME", "") - t.Setenv("TMUX_SESSION_NAME", "tmux-dev") +func TestUserImageTag_StackWithModules(t *testing.T) { + withCleanImageState(t) + runner.Stack = "ultimate" + runner.Modules = []string{"nixos", "electronics"} got := runner.UserImageTag() - if got != "devcell-user:tmux-dev" { - t.Errorf("tmux fallback: want devcell-user:tmux-dev, got %q", got) + // Modules sorted: electronics, nixos + if !strings.HasPrefix(got, "devcell-user:ultimate-electronics-nixos-") { + t.Errorf("stack+modules: want prefix devcell-user:ultimate-electronics-nixos-, got %q", got) + } + // sha8 suffix + parts := strings.Split(got, "-") + sha := parts[len(parts)-1] + if len(sha) != 8 { + t.Errorf("sha suffix: want 8 chars, got %d in %q", len(sha), got) } } -func TestUserImageTag_SessionNameBeatssTmux(t *testing.T) { - t.Setenv("DEVCELL_USER_IMAGE", "") - t.Setenv("DEVCELL_SESSION_NAME", "explicit") - t.Setenv("TMUX_SESSION_NAME", "tmux-session") - got := runner.UserImageTag() - if got != "devcell-user:explicit" { - t.Errorf("precedence: want devcell-user:explicit, got %q", got) +func TestUserImageTag_ModuleOrderDoesNotMatter(t *testing.T) { + withCleanImageState(t) + runner.Stack = "go" + runner.Modules = []string{"b", "a", "c"} + tag1 := runner.UserImageTag() + runner.Modules = []string{"c", "a", "b"} + tag2 := runner.UserImageTag() + if tag1 != tag2 { + t.Errorf("module order should not matter: %q != %q", tag1, tag2) } } func TestUserImageTag_EnvOverrideWins(t *testing.T) { + withCleanImageState(t) t.Setenv("DEVCELL_USER_IMAGE", "custom:override") - t.Setenv("DEVCELL_SESSION_NAME", "ignored") + runner.Stack = "ultimate" got := runner.UserImageTag() if got != "custom:override" { t.Errorf("override: want custom:override, got %q", got) } } +// --- UserImageTag per-session (legacy) --- + +func TestUserImageTag_PerSession_Default(t *testing.T) { + withCleanImageState(t) + runner.PerSessionImage = true + got := runner.UserImageTag() + if got != "devcell-user:main" { + t.Errorf("per-session default: want devcell-user:main, got %q", got) + } +} + +func TestUserImageTag_PerSession_TmuxFallback(t *testing.T) { + withCleanImageState(t) + runner.PerSessionImage = true + t.Setenv("TMUX_SESSION_NAME", "DIMM") + got := runner.UserImageTag() + if got != "devcell-user:DIMM" { + t.Errorf("per-session tmux: want devcell-user:DIMM, got %q", got) + } +} + +func TestUserImageTag_PerSession_ExplicitBeatssTmux(t *testing.T) { + withCleanImageState(t) + runner.PerSessionImage = true + t.Setenv("DEVCELL_SESSION_NAME", "explicit") + t.Setenv("TMUX_SESSION_NAME", "tmux-session") + got := runner.UserImageTag() + if got != "devcell-user:explicit" { + t.Errorf("per-session precedence: want devcell-user:explicit, got %q", got) + } +} + // --- ParseImageMetadata --- func TestParseImageMetadata_ValidJSON(t *testing.T) { diff --git a/internal/runner/systemprompt.go b/internal/runner/systemprompt.go index cad2d012..a3a855b6 100644 --- a/internal/runner/systemprompt.go +++ b/internal/runner/systemprompt.go @@ -1,31 +1,45 @@ +// Package runner builds the system prompt that devcell injects into agent +// CLIs (claude, opencode, codex) and the cell serve HTTP server. +// +// The prompt has two distinct conceptual layers, always concatenated in +// order — see ContainerContext and ResolveSystemPrompt — and a third +// per-request layer that lives outside this package (cell serve merges +// per-request `instructions` / `system` role from the API body into the +// user prompt directly). package runner import ( "fmt" + "os" + "path/filepath" "strings" "github.com/DimmKirr/devcell/internal/cfg" "github.com/DimmKirr/devcell/internal/config" ) -// BuildSystemPrompt generates the --append-system-prompt content for Claude Code. -// It describes the container environment, bind mounts, and host path mappings -// so Claude understands its runtime context. -func BuildSystemPrompt(c config.Config, cellCfg cfg.CellConfig) string { +// ContainerContext returns the auto-generated filesystem/runtime preamble +// — bind mounts, host path mappings, hard constraints — describing the +// devcell container the agent is running inside. Pure container facts; +// no user-controllable content. +// +// This is what makes the agent file-aware: when the user mentions a host +// path, the agent can translate it to the matching container path. Every +// surface that ships a system prompt (cell claude, cell serve) prepends +// this so the agent reasons correctly about its filesystem. +func ContainerContext(c config.Config, cellCfg cfg.CellConfig) string { var b strings.Builder appDir := "/" + c.AppName // e.g. /devcell-85 hostDir := c.BaseDir // e.g. /Users/dmitry/dev/dimmkirr/devcell homeDir := "/home/" + c.HostUser - // Container and project identity fmt.Fprintf(&b, "Environment: Docker container (cell-%s)\n", c.AppName) fmt.Fprintf(&b, "Project: %s (alias for %s on host)\n", appDir, hostDir) fmt.Fprintf(&b, "Both paths are bind-mounted from the same host directory and resolve to the same filesystem.\n") fmt.Fprintf(&b, "Working directory is %s. If the user mentions host paths like %s/..., they map to %s/...\n", appDir, hostDir, appDir) b.WriteString("\n") - // Bind mounts — standard b.WriteString("Bind mounts:\n") fmt.Fprintf(&b, " %s = %s (project source, read-write)\n", appDir, hostDir) fmt.Fprintf(&b, " %s (persistent home, survives container restarts)\n", homeDir) @@ -34,7 +48,6 @@ func BuildSystemPrompt(c config.Config, cellCfg cfg.CellConfig) string { fmt.Fprintf(&b, " %s/.claude/agents (read-only, from host)\n", homeDir) fmt.Fprintf(&b, " /etc/devcell/config = %s (user build config)\n", c.ConfigDir) - // User-defined volumes from devcell.toml [[volumes]] for _, vol := range cellCfg.Volumes { parts := strings.SplitN(vol.Mount, ":", 3) if len(parts) >= 2 { @@ -47,7 +60,6 @@ func BuildSystemPrompt(c config.Config, cellCfg cfg.CellConfig) string { } b.WriteString("\n") - // Host path mapping b.WriteString("Host path mapping (use these to translate paths the user mentions):\n") fmt.Fprintf(&b, " host: %s → container: %s\n", hostDir, hostDir) fmt.Fprintf(&b, " host: %s → container: %s\n", c.HostHome, homeDir) @@ -59,20 +71,112 @@ func BuildSystemPrompt(c config.Config, cellCfg cfg.CellConfig) string { } b.WriteString("\n") - // Key constraints b.WriteString("Constraints:\n") b.WriteString(" - /opt/devcell is the nix environment — do not modify at runtime\n") - fmt.Fprintf(&b, " - Nix profile: /opt/devcell/.local/state/nix/profiles/profile\n") - - // Custom system prompt from [llm] system_prompt - if cellCfg.LLM.SystemPrompt != "" { - b.WriteString("\n") - b.WriteString("Project context:\n") - b.WriteString(cellCfg.LLM.SystemPrompt) - if !strings.HasSuffix(cellCfg.LLM.SystemPrompt, "\n") { - b.WriteString("\n") + b.WriteString(" - Nix profile: /opt/devcell/.local/state/nix/profiles/profile\n") + + return b.String() +} + +// ResolveOpts bundles every input source the system-prompt resolver looks +// at. Surfaces wire only the inputs they have — `cell claude` leaves the +// flag fields empty; `cell serve` populates everything. +type ResolveOpts struct { + // FlagFile / FlagInline are the --system-prompt-file / --system-prompt + // CLI flags. Currently exposed only on `cell serve`. + FlagFile, FlagInline string + // EnvFile / EnvInline are the DEVCELL_SYSTEM_PROMPT_FILE / + // DEVCELL_SYSTEM_PROMPT env vars. Read by every surface. + EnvFile, EnvInline string + // CellCfg supplies [llm].system_prompt and [llm].system_prompt_file + // from the merged devcell.toml. + CellCfg cfg.CellConfig + // CfgBaseDir is the project base dir, used to resolve a relative + // `[llm].system_prompt_file` path. Empty disables relative resolution + // (absolute paths still work). + CfgBaseDir string +} + +// ResolveSystemPrompt walks the seven-tier source chain in order — flags, +// env, TOML — returning the first match. Within a tier, setting both the +// file and inline form is rejected as ambiguous so the caller never has +// to guess which one won. Across tiers, higher silently shadows lower: +// the layering is the whole point of having multiple sources. +// +// Returns ("", nil) when no source is set — callers concatenate this +// with ContainerContext via AssembleSystemPrompt. +// +// Resolution order (first match wins): +// +// 1. opts.FlagFile (--system-prompt-file) +// 2. opts.FlagInline (--system-prompt) +// 3. opts.EnvFile (DEVCELL_SYSTEM_PROMPT_FILE) +// 4. opts.EnvInline (DEVCELL_SYSTEM_PROMPT) +// 5. CellCfg.LLM.SystemPromptFile ([llm].system_prompt_file) +// 6. CellCfg.LLM.SystemPrompt ([llm].system_prompt) +// 7. "" +func ResolveSystemPrompt(opts ResolveOpts) (string, error) { + if opts.FlagFile != "" && opts.FlagInline != "" { + return "", fmt.Errorf("--system-prompt and --system-prompt-file are mutually exclusive") + } + if opts.FlagFile != "" { + return readPromptFile(opts.FlagFile, "--system-prompt-file") + } + if opts.FlagInline != "" { + return opts.FlagInline, nil + } + + if opts.EnvFile != "" && opts.EnvInline != "" { + return "", fmt.Errorf("DEVCELL_SYSTEM_PROMPT and DEVCELL_SYSTEM_PROMPT_FILE are mutually exclusive") + } + if opts.EnvFile != "" { + return readPromptFile(opts.EnvFile, "DEVCELL_SYSTEM_PROMPT_FILE") + } + if opts.EnvInline != "" { + return opts.EnvInline, nil + } + + tomlFile := opts.CellCfg.LLM.SystemPromptFile + tomlInline := opts.CellCfg.LLM.SystemPrompt + if tomlFile != "" && tomlInline != "" { + return "", fmt.Errorf("[llm].system_prompt and [llm].system_prompt_file are mutually exclusive") + } + if tomlFile != "" { + // Resolve relative paths against the project base dir, matching + // the convention `[[volumes]]` already uses. + path := tomlFile + if !filepath.IsAbs(path) && opts.CfgBaseDir != "" { + path = filepath.Join(opts.CfgBaseDir, path) } + return readPromptFile(path, "[llm].system_prompt_file") } + return tomlInline, nil +} - return b.String() +// AssembleSystemPrompt is the single entry point callers should use to +// build the string passed to claude's --append-system-prompt (or any +// future agent's equivalent). It prepends ContainerContext to the +// resolved prompt with a blank-line separator. When the resolved prompt +// is empty, returns just ContainerContext. +func AssembleSystemPrompt(c config.Config, cellCfg cfg.CellConfig, opts ResolveOpts) (string, error) { + resolved, err := ResolveSystemPrompt(opts) + if err != nil { + return "", err + } + ctx := ContainerContext(c, cellCfg) + if resolved == "" { + return ctx, nil + } + if !strings.HasSuffix(resolved, "\n") { + resolved += "\n" + } + return ctx + "\n" + resolved, nil +} + +func readPromptFile(path, source string) (string, error) { + b, err := os.ReadFile(path) + if err != nil { + return "", fmt.Errorf("read %s: %w", source, err) + } + return string(b), nil } diff --git a/internal/runner/systemprompt_test.go b/internal/runner/systemprompt_test.go index 492e00e0..a69db073 100644 --- a/internal/runner/systemprompt_test.go +++ b/internal/runner/systemprompt_test.go @@ -1,6 +1,8 @@ package runner import ( + "os" + "path/filepath" "strings" "testing" @@ -8,89 +10,249 @@ import ( "github.com/DimmKirr/devcell/internal/config" ) -func TestBuildSystemPrompt(t *testing.T) { - c := config.Config{ +func sampleConfig() config.Config { + return config.Config{ AppName: "devcell-85", BaseDir: "/Users/dmitry/dev/dimmkirr/devcell", HostUser: "dmitry", HostHome: "/Users/dmitry", } +} + +func TestContainerContext_DescribesMountsAndConstraints(t *testing.T) { cellCfg := cfg.CellConfig{ Volumes: []cfg.VolumeMount{ {Mount: "~/work/secrets:/run/secrets:ro"}, }, } - prompt := BuildSystemPrompt(c, cellCfg) + ctx := ContainerContext(sampleConfig(), cellCfg) - checks := []struct { - name string - want string - }{ - {"container identity", "Docker container"}, - {"project alias", "/devcell-85"}, - {"host base dir", "/Users/dmitry/dev/dimmkirr/devcell"}, - {"same filesystem", "same filesystem"}, - {"host path mapping", "host paths"}, - {"persistent home", "/home/dmitry"}, - {"skills mount", ".claude/skills"}, - {"user volume", "/run/secrets"}, - {"user volume ro", "read-only"}, - {"host mapping", "host: /Users/dmitry/dev/dimmkirr/devcell"}, - {"nix constraint", "/opt/devcell"}, - } - - for _, tc := range checks { - if !strings.Contains(prompt, tc.want) { - t.Errorf("%s: prompt missing %q\n\nFull prompt:\n%s", tc.name, tc.want, prompt) + checks := map[string]string{ + "container identity": "Docker container", + "project alias": "/devcell-85", + "host base dir": "/Users/dmitry/dev/dimmkirr/devcell", + "same filesystem": "same filesystem", + "persistent home": "/home/dmitry", + "skills mount": ".claude/skills", + "user volume": "/run/secrets", + "user volume ro": "read-only", + "host mapping prefix": "host: /Users/dmitry/dev/dimmkirr/devcell", + "nix constraint": "/opt/devcell", + } + + for name, want := range checks { + if !strings.Contains(ctx, want) { + t.Errorf("%s: container context missing %q\n\nFull:\n%s", name, want, ctx) } } } -func TestBuildSystemPrompt_WithCustomPrompt(t *testing.T) { - c := config.Config{ - AppName: "myproject-1", - BaseDir: "/Users/dev/myproject", - HostUser: "dev", - HostHome: "/Users/dev", - } +func TestContainerContext_NoProjectContextWrapper(t *testing.T) { + // `Project context:` was the legacy wrapper that glued user prompt + // content into the container preamble. ContainerContext now contains + // only container facts — the wrapper is gone (resolver outputs the + // user prompt as a separate concatenated layer). cellCfg := cfg.CellConfig{ - LLM: cfg.LLMSection{ - SystemPrompt: "This project uses PostgreSQL 16 with pgx/v5.\nAPI at /api/v2/.", - }, + LLM: cfg.LLMSection{SystemPrompt: "use postgres 16"}, + } + ctx := ContainerContext(sampleConfig(), cellCfg) + if strings.Contains(ctx, "Project context:") { + t.Errorf("ContainerContext leaked the legacy `Project context:` wrapper:\n%s", ctx) + } + if strings.Contains(ctx, "use postgres 16") { + t.Errorf("ContainerContext leaked user prompt content:\n%s", ctx) } +} - prompt := BuildSystemPrompt(c, cellCfg) +func TestResolveSystemPrompt_PrecedenceAndAmbiguity(t *testing.T) { + dir := t.TempDir() + flagFile := filepath.Join(dir, "flag.md") + envFile := filepath.Join(dir, "env.md") + tomlFile := filepath.Join(dir, "toml.md") + if err := os.WriteFile(flagFile, []byte("from-flag-file\n"), 0o600); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(envFile, []byte("from-env-file\n"), 0o600); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(tomlFile, []byte("from-toml-file\n"), 0o600); err != nil { + t.Fatal(err) + } - if !strings.Contains(prompt, "Project context:") { - t.Error("prompt should contain 'Project context:' header") + tests := []struct { + name string + opts ResolveOpts + want string + wantErr string + }{ + {name: "all empty", want: ""}, + + { + name: "tier 1: flag file wins over everything", + opts: ResolveOpts{ + FlagFile: flagFile, + FlagInline: "", + EnvFile: envFile, + EnvInline: "from-env-inline", + CellCfg: cfg.CellConfig{LLM: cfg.LLMSection{SystemPrompt: "from-toml-inline"}}, + }, + want: "from-flag-file\n", + }, + { + name: "tier 2: flag inline wins when no flag file", + opts: ResolveOpts{ + FlagInline: "from-flag-inline", + EnvFile: envFile, + EnvInline: "from-env-inline", + CellCfg: cfg.CellConfig{LLM: cfg.LLMSection{SystemPrompt: "from-toml-inline"}}, + }, + want: "from-flag-inline", + }, + { + name: "tier 3: env file when no flag (within tier inline must be empty)", + opts: ResolveOpts{ + EnvFile: envFile, + CellCfg: cfg.CellConfig{LLM: cfg.LLMSection{SystemPrompt: "from-toml-inline"}}, + }, + want: "from-env-file\n", + }, + { + name: "tier 4: env inline wins when no env file", + opts: ResolveOpts{ + EnvInline: "from-env-inline", + CellCfg: cfg.CellConfig{LLM: cfg.LLMSection{SystemPrompt: "from-toml-inline"}}, + }, + want: "from-env-inline", + }, + { + name: "tier 5: toml file when no env", + opts: ResolveOpts{ + CellCfg: cfg.CellConfig{LLM: cfg.LLMSection{SystemPromptFile: tomlFile}}, + }, + want: "from-toml-file\n", + }, + { + name: "tier 6: toml inline when no toml file", + opts: ResolveOpts{ + CellCfg: cfg.CellConfig{LLM: cfg.LLMSection{SystemPrompt: "from-toml-inline"}}, + }, + want: "from-toml-inline", + }, + + { + name: "ambiguity: flag file + flag inline", + opts: ResolveOpts{FlagFile: flagFile, FlagInline: "x"}, + wantErr: "--system-prompt and --system-prompt-file are mutually exclusive", + }, + { + name: "ambiguity: env file + env inline", + opts: ResolveOpts{EnvFile: envFile, EnvInline: "x"}, + wantErr: "DEVCELL_SYSTEM_PROMPT and DEVCELL_SYSTEM_PROMPT_FILE", + }, + { + name: "ambiguity: TOML file + TOML inline", + opts: ResolveOpts{ + CellCfg: cfg.CellConfig{LLM: cfg.LLMSection{ + SystemPrompt: "x", + SystemPromptFile: tomlFile, + }}, + }, + wantErr: "[llm].system_prompt and [llm].system_prompt_file", + }, + + { + name: "higher tier silences lower-tier ambiguity", + opts: ResolveOpts{ + FlagInline: "winning", + EnvFile: envFile, + EnvInline: "ambiguous", + }, + want: "winning", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got, err := ResolveSystemPrompt(tt.opts) + if tt.wantErr != "" { + if err == nil { + t.Fatalf("want error containing %q, got nil (value=%q)", tt.wantErr, got) + } + if !strings.Contains(err.Error(), tt.wantErr) { + t.Fatalf("want error containing %q, got %q", tt.wantErr, err.Error()) + } + return + } + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if got != tt.want { + t.Fatalf("got %q, want %q", got, tt.want) + } + }) } - if !strings.Contains(prompt, "PostgreSQL 16") { - t.Error("prompt should contain custom system prompt content") +} + +func TestResolveSystemPrompt_TomlFileRelativePath(t *testing.T) { + // `[llm].system_prompt_file = "./SYSTEM.md"` should resolve relative + // to the project's base dir (where devcell.toml lives). + dir := t.TempDir() + if err := os.WriteFile(filepath.Join(dir, "SYSTEM.md"), []byte("from-relative\n"), 0o600); err != nil { + t.Fatal(err) } - if !strings.Contains(prompt, "/api/v2/") { - t.Error("prompt should contain custom system prompt content") + + got, err := ResolveSystemPrompt(ResolveOpts{ + CellCfg: cfg.CellConfig{LLM: cfg.LLMSection{SystemPromptFile: "SYSTEM.md"}}, + CfgBaseDir: dir, + }) + if err != nil { + t.Fatalf("unexpected error: %v", err) } - // Custom prompt should come after container environment - envIdx := strings.Index(prompt, "Docker container") - customIdx := strings.Index(prompt, "Project context:") - if customIdx <= envIdx { - t.Error("custom prompt should appear after container environment section") + if got != "from-relative\n" { + t.Fatalf("got %q, want %q", got, "from-relative\n") } } -func TestBuildSystemPrompt_EmptyCustomPrompt(t *testing.T) { - c := config.Config{ - AppName: "myproject-1", - BaseDir: "/Users/dev/myproject", - HostUser: "dev", - HostHome: "/Users/dev", +func TestResolveSystemPrompt_FileNotFound(t *testing.T) { + _, err := ResolveSystemPrompt(ResolveOpts{FlagFile: "/no/such/file"}) + if err == nil { + t.Fatal("expected error for missing flag file, got nil") + } + if !strings.Contains(err.Error(), "--system-prompt-file") { + t.Fatalf("error lost source context: %v", err) } - cellCfg := cfg.CellConfig{} +} - prompt := BuildSystemPrompt(c, cellCfg) +func TestAssembleSystemPrompt_PrependsContainerContext(t *testing.T) { + out, err := AssembleSystemPrompt(sampleConfig(), cfg.CellConfig{}, ResolveOpts{ + FlagInline: "be terse", + }) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if !strings.Contains(out, "Docker container") { + t.Error("assembled prompt missing container context") + } + if !strings.Contains(out, "be terse") { + t.Error("assembled prompt missing resolved user prompt") + } + envIdx := strings.Index(out, "Docker container") + userIdx := strings.Index(out, "be terse") + if userIdx <= envIdx { + t.Error("user prompt should appear after container context") + } +} - if strings.Contains(prompt, "Project context:") { - t.Error("prompt should NOT contain 'Project context:' when system_prompt is empty") +func TestAssembleSystemPrompt_EmptyResolverReturnsContextOnly(t *testing.T) { + out, err := AssembleSystemPrompt(sampleConfig(), cfg.CellConfig{}, ResolveOpts{}) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if !strings.Contains(out, "Docker container") { + t.Error("assembled prompt missing container context") + } + if strings.HasSuffix(out, "\n\n") { + t.Errorf("expected no trailing blank when resolver empty, got %q", out) } } diff --git a/internal/serve/claude_json.go b/internal/serve/claude_json.go new file mode 100644 index 00000000..7b1501a1 --- /dev/null +++ b/internal/serve/claude_json.go @@ -0,0 +1,83 @@ +package serve + +import ( + "encoding/json" + "fmt" +) + +// claudeJSONResult mirrors `claude --output-format=json` output. +// +// Captured live (2026-04-28, claude-opus-4-7) — only the fields devcell +// actually consumes are typed; iterations, modelUsage, permission_denials, +// server_tool_use, cache_creation breakdowns, etc. are deliberately +// dropped to keep the surface tight. Unknown fields are ignored by +// json.Unmarshal — adding a new typed field later is non-breaking. +// +// Real example (truncated for clarity): +// +// { +// "type": "result", +// "subtype": "success", +// "is_error": false, +// "api_error_status": null, +// "duration_ms": 2879, +// "duration_api_ms": 2817, +// "num_turns": 1, +// "result": "Hi.", +// "stop_reason": "end_turn", +// "session_id": "68876525-...", +// "total_cost_usd": 0.36413124999999996, +// "usage": { +// "input_tokens": 5, +// "cache_creation_input_tokens": 58225, +// "cache_read_input_tokens": 0, +// "output_tokens": 8, +// "service_tier": "standard" +// } +// } +type claudeJSONResult struct { + Type string `json:"type"` // expected: "result" + Subtype string `json:"subtype"` // "success" / "error_during_execution" / ... + IsError bool `json:"is_error"` // true on API or tool error + APIErrorStatus *string `json:"api_error_status"` // populated when claude got an HTTP error from the upstream API + Result string `json:"result"` // the assistant's text reply + StopReason string `json:"stop_reason"` // "end_turn", "max_tokens", "tool_use", ... + SessionID string `json:"session_id"` + NumTurns int `json:"num_turns"` + DurationMs int `json:"duration_ms"` + DurationAPIMs int `json:"duration_api_ms"` + TotalCostUSD float64 `json:"total_cost_usd"` + Usage claudeJSONUsageObj `json:"usage"` +} + +// claudeJSONUsageObj is claude's per-turn token accounting (the four fields +// matter for billing). The remaining sub-objects (server_tool_use, +// cache_creation, iterations, ...) are ignored. +type claudeJSONUsageObj struct { + InputTokens int `json:"input_tokens"` + OutputTokens int `json:"output_tokens"` + CacheCreationInputTokens int `json:"cache_creation_input_tokens"` + CacheReadInputTokens int `json:"cache_read_input_tokens"` +} + +// parseClaudeJSON decodes claude --output-format=json output into the +// internal Usage shape and returns the assistant's text. Returns an error +// only when the bytes don't decode as JSON or aren't a "result" envelope — +// the caller falls back to raw stdout in that case. +func parseClaudeJSON(stdout []byte) (text string, usage Usage, err error) { + var r claudeJSONResult + if err := json.Unmarshal(stdout, &r); err != nil { + return "", Usage{}, fmt.Errorf("decode claude json: %w", err) + } + if r.Type != "result" { + return "", Usage{}, fmt.Errorf("unexpected claude json envelope type %q", r.Type) + } + usage = Usage{ + InputTokens: r.Usage.InputTokens, + OutputTokens: r.Usage.OutputTokens, + CacheCreationInputTokens: r.Usage.CacheCreationInputTokens, + CacheReadInputTokens: r.Usage.CacheReadInputTokens, + TotalCostUSD: r.TotalCostUSD, + } + return r.Result, usage, nil +} diff --git a/internal/serve/claude_json_test.go b/internal/serve/claude_json_test.go new file mode 100644 index 00000000..6e56b159 --- /dev/null +++ b/internal/serve/claude_json_test.go @@ -0,0 +1,71 @@ +package serve + +import ( + "strings" + "testing" +) + +// realClaudeJSON is a real `claude --output-format=json` capture from +// claude-opus-4-7 (2026-04-28). Used as the canonical fixture so the +// parser is exercised against actual wire bytes, not a hand-written +// approximation. +const realClaudeJSON = `{"type":"result","subtype":"success","is_error":false,"api_error_status":null,"duration_ms":2879,"duration_api_ms":2817,"num_turns":1,"result":"Hi.","stop_reason":"end_turn","session_id":"68876525-f79d-4369-945c-1acd2e7ba665","total_cost_usd":0.36413124999999996,"usage":{"input_tokens":5,"cache_creation_input_tokens":58225,"cache_read_input_tokens":0,"output_tokens":8,"server_tool_use":{"web_search_requests":0,"web_fetch_requests":0},"service_tier":"standard","cache_creation":{"ephemeral_1h_input_tokens":58225,"ephemeral_5m_input_tokens":0},"inference_geo":"","iterations":[{"input_tokens":5,"output_tokens":8,"cache_read_input_tokens":0,"cache_creation_input_tokens":58225,"cache_creation":{"ephemeral_5m_input_tokens":0,"ephemeral_1h_input_tokens":58225},"type":"message"}],"speed":"standard"},"modelUsage":{"claude-opus-4-7[1m]":{"inputTokens":5,"outputTokens":8,"cacheReadInputTokens":0,"cacheCreationInputTokens":58225,"webSearchRequests":0,"costUSD":0.36413124999999996,"contextWindow":1000000,"maxOutputTokens":64000}},"permission_denials":[],"terminal_reason":"completed","fast_mode_state":"off","uuid":"9dc17684-8c6f-4212-8c32-074c337bb1ec"}` + +func TestParseClaudeJSON_RealCapture(t *testing.T) { + text, usage, err := parseClaudeJSON([]byte(realClaudeJSON)) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if text != "Hi." { + t.Errorf("text = %q, want %q", text, "Hi.") + } + if usage.InputTokens != 5 { + t.Errorf("InputTokens = %d, want 5", usage.InputTokens) + } + if usage.OutputTokens != 8 { + t.Errorf("OutputTokens = %d, want 8", usage.OutputTokens) + } + if usage.CacheCreationInputTokens != 58225 { + t.Errorf("CacheCreationInputTokens = %d, want 58225", usage.CacheCreationInputTokens) + } + if usage.CacheReadInputTokens != 0 { + t.Errorf("CacheReadInputTokens = %d, want 0", usage.CacheReadInputTokens) + } + if usage.TotalCostUSD <= 0 { + t.Errorf("TotalCostUSD should be > 0, got %v", usage.TotalCostUSD) + } +} + +func TestParseClaudeJSON_NotJSON(t *testing.T) { + _, _, err := parseClaudeJSON([]byte("plain text from a non-json claude run")) + if err == nil { + t.Fatal("expected error decoding non-JSON") + } + if !strings.Contains(err.Error(), "decode claude json") { + t.Errorf("error context lost: %v", err) + } +} + +func TestParseClaudeJSON_WrongEnvelope(t *testing.T) { + _, _, err := parseClaudeJSON([]byte(`{"type":"system","subtype":"init"}`)) + if err == nil { + t.Fatal("expected error for non-result envelope") + } + if !strings.Contains(err.Error(), `envelope type "system"`) { + t.Errorf("error didn't name the unexpected type: %v", err) + } +} + +func TestParseClaudeJSON_MinimalShape(t *testing.T) { + // Smallest valid envelope — proves we don't require optional fields. + got, usage, err := parseClaudeJSON([]byte(`{"type":"result","result":"ok","usage":{"input_tokens":3,"output_tokens":2}}`)) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if got != "ok" { + t.Errorf("text = %q, want ok", got) + } + if usage.InputTokens != 3 || usage.OutputTokens != 2 { + t.Errorf("usage = %+v", usage) + } +} diff --git a/internal/serve/claude_stream.go b/internal/serve/claude_stream.go new file mode 100644 index 00000000..bba1b9b3 --- /dev/null +++ b/internal/serve/claude_stream.go @@ -0,0 +1,217 @@ +package serve + +import ( + "bufio" + "encoding/json" + "fmt" + "io" +) + +// claude `--output-format=stream-json --include-partial-messages` emits +// JSONL on stdout: one JSON object per line. The outer envelope has a +// `type` discriminator. Most lines are `stream_event` wrappers around a +// raw Anthropic Messages-API event (message_start / content_block_delta +// / message_stop / etc.). The terminal line is `type=result` with the +// same shape claude_json.go decodes for non-streamed runs. +// +// We model only the events devcell consumes: +// +// {"type":"stream_event","event":{"type":"message_start", "message":{…}}} +// {"type":"stream_event","event":{"type":"content_block_start","index":N,"content_block":{"type":"text"|"tool_use"|…}}} +// {"type":"stream_event","event":{"type":"content_block_delta","index":N,"delta":{"type":"text_delta","text":"…"}}} +// {"type":"stream_event","event":{"type":"content_block_stop","index":N}} +// {"type":"stream_event","event":{"type":"message_delta","delta":{"stop_reason":"…"},"usage":{…}}} +// {"type":"stream_event","event":{"type":"message_stop"}} +// {"type":"result", …} +// +// Other top-level types (system/init, system/hook_*, assistant full-turn, +// user/tool_result) are intentionally skipped for v1. + +// streamEnvelope is the outer wrapper for every JSONL line. +type streamEnvelope struct { + Type string `json:"type"` + Event json.RawMessage `json:"event,omitempty"` // populated for type="stream_event" + Subtype string `json:"subtype,omitempty"` // populated for type="result" (and others we ignore) +} + +// streamInnerEvent is the Anthropic Messages-API event nested inside a +// stream_event wrapper. +type streamInnerEvent struct { + Type string `json:"type"` + Index int `json:"index,omitempty"` + Message *streamMessage `json:"message,omitempty"` // message_start + ContentBlock *streamContentBlock `json:"content_block,omitempty"` // content_block_start + Delta *streamDelta `json:"delta,omitempty"` // content_block_delta + message_delta share this field name with different shapes +} + +type streamMessage struct { + ID string `json:"id"` + Model string `json:"model"` + Role string `json:"role"` +} + +type streamContentBlock struct { + Type string `json:"type"` // "text" | "tool_use" | … + Text string `json:"text"` // populated for type="text" (typically empty at start) +} + +// streamDelta carries either a content-block delta (text_delta) or a +// message-stop delta (stop_reason) — distinguished by the `type` field +// when present (text_delta) or by the surrounding event type. +type streamDelta struct { + Type string `json:"type,omitempty"` // "text_delta" for content_block_delta + Text string `json:"text,omitempty"` // text_delta payload + StopReason string `json:"stop_reason,omitempty"` // populated on message_delta +} + +// StreamEventKind discriminates the canonical event sent on the channel. +type StreamEventKind int + +const ( + streamEventInvalid StreamEventKind = iota + StreamEventMessageStart + StreamEventTextDelta + StreamEventMessageStop + StreamEventResult + StreamEventError +) + +// StreamEvent is the canonical, OpenAI-agnostic event the SSE formatters +// consume. One source (claude scanner), two sinks (Chat Completions and +// Responses). +type StreamEvent struct { + Kind StreamEventKind + // MessageID populated on MessageStart (the upstream message id). + MessageID string + // Model populated on MessageStart. + Model string + // Delta populated on TextDelta — incremental text since the previous + // delta, not cumulative. + Delta string + // StopReason populated on MessageStop — "end_turn" / "max_tokens" / … + StopReason string + // Final populated on Result — reuses claude_json.go's typed envelope + // for the terminal usage + cost payload. + Final *claudeJSONResult + // Err populated on Error — terminal; the caller should stop reading. + Err error +} + +// scanClaudeStream reads JSONL lines from r and emits canonical events on +// the returned channel. The channel closes when the scanner reaches EOF +// or a fatal decode error. Non-fatal lines (unknown envelopes, skipped +// event types) are silently dropped. +// +// The caller is responsible for cancelling its source (typically via +// killing the claude subprocess) — this function only reads. +func scanClaudeStream(r io.Reader) <-chan StreamEvent { + out := make(chan StreamEvent, 16) + go func() { + defer close(out) + sc := bufio.NewScanner(r) + // claude can emit single lines >64KB (e.g. when an assistant + // turn carries a large final content block). Bump the buffer. + buf := make([]byte, 0, 64*1024) + sc.Buffer(buf, 4*1024*1024) + for sc.Scan() { + line := sc.Bytes() + if len(line) == 0 { + continue + } + ev, ok := decodeStreamLine(line) + if !ok { + continue + } + out <- ev + } + if err := sc.Err(); err != nil { + out <- StreamEvent{Kind: StreamEventError, Err: fmt.Errorf("claude stream scan: %w", err)} + } + }() + return out +} + +// decodeStreamLine returns (event, true) when the line carries a +// canonical event we want to forward; (zero, false) otherwise. Decode +// failures are logged via the returned Error event only when they look +// like real claude output (i.e. valid JSON with a type field) — random +// non-JSON lines are silently dropped to be tolerant of stderr noise +// occasionally landing on stdout. +func decodeStreamLine(line []byte) (StreamEvent, bool) { + var env streamEnvelope + if err := json.Unmarshal(line, &env); err != nil { + return StreamEvent{}, false + } + switch env.Type { + case "stream_event": + return decodeInnerEvent(env.Event) + case "result": + // Reuse the same parser as the buffered (non-stream) path so + // the terminal usage shape is identical. + text, usage, err := parseClaudeJSON(line) + if err != nil { + return StreamEvent{Kind: StreamEventError, Err: err}, true + } + return StreamEvent{ + Kind: StreamEventResult, + Final: &claudeJSONResult{ + Type: "result", + Subtype: env.Subtype, + Result: text, + TotalCostUSD: usage.TotalCostUSD, + Usage: claudeJSONUsageObj{ + InputTokens: usage.InputTokens, + OutputTokens: usage.OutputTokens, + CacheCreationInputTokens: usage.CacheCreationInputTokens, + CacheReadInputTokens: usage.CacheReadInputTokens, + }, + }, + }, true + default: + // system/init, system/hook_*, assistant (full turn), user (tool + // result). Out of scope for v1. + return StreamEvent{}, false + } +} + +func decodeInnerEvent(raw json.RawMessage) (StreamEvent, bool) { + var ev streamInnerEvent + if err := json.Unmarshal(raw, &ev); err != nil { + return StreamEvent{}, false + } + switch ev.Type { + case "message_start": + if ev.Message == nil { + return StreamEvent{}, false + } + return StreamEvent{ + Kind: StreamEventMessageStart, + MessageID: ev.Message.ID, + Model: ev.Message.Model, + }, true + + case "content_block_delta": + // Only forward text_delta — tool_use deltas are out of scope. + if ev.Delta == nil || ev.Delta.Type != "text_delta" || ev.Delta.Text == "" { + return StreamEvent{}, false + } + return StreamEvent{Kind: StreamEventTextDelta, Delta: ev.Delta.Text}, true + + case "message_delta": + stop := "" + if ev.Delta != nil { + stop = ev.Delta.StopReason + } + return StreamEvent{Kind: StreamEventMessageStop, StopReason: stop}, true + + case "content_block_start", "content_block_stop", "message_stop": + // Boundary markers — not needed for the OpenAI mappings we + // emit. message_stop arrives after message_delta with no extra + // payload; the formatter has already handled the terminator + // once it sees Result. + return StreamEvent{}, false + + default: + return StreamEvent{}, false + } +} diff --git a/internal/serve/claude_stream_test.go b/internal/serve/claude_stream_test.go new file mode 100644 index 00000000..b23448e7 --- /dev/null +++ b/internal/serve/claude_stream_test.go @@ -0,0 +1,112 @@ +package serve + +import ( + "strings" + "testing" +) + +// realStreamCapture is a verbatim claude --output-format=stream-json +// --include-partial-messages capture (2026-04-28, claude-opus-4-7) for +// the prompt "Say 'one two three four five' on five separate lines". +// Used as the canonical fixture so the scanner is exercised against +// real wire bytes. +const realStreamCapture = `{"type":"system","subtype":"init","cwd":"/devcell-186","session_id":"ac12982f","tools":[]} +{"type":"stream_event","event":{"type":"message_start","message":{"model":"claude-opus-4-7","id":"msg_017bhZn5","type":"message","role":"assistant","content":[],"stop_reason":null,"usage":{"input_tokens":5,"cache_creation_input_tokens":30957,"cache_read_input_tokens":27019,"output_tokens":1}}},"session_id":"ac12982f"} +{"type":"stream_event","event":{"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}},"session_id":"ac12982f"} +{"type":"stream_event","event":{"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"one"}},"session_id":"ac12982f"} +{"type":"stream_event","event":{"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"\ntwo\nthree\nfour\nfive"}},"session_id":"ac12982f"} +{"type":"assistant","message":{"model":"claude-opus-4-7","id":"msg_017bhZn5","content":[{"type":"text","text":"one\ntwo\nthree\nfour\nfive"}]}} +{"type":"stream_event","event":{"type":"content_block_stop","index":0},"session_id":"ac12982f"} +{"type":"stream_event","event":{"type":"message_delta","delta":{"stop_reason":"end_turn"},"usage":{"output_tokens":13}},"session_id":"ac12982f"} +{"type":"stream_event","event":{"type":"message_stop"},"session_id":"ac12982f"} +{"type":"result","subtype":"success","is_error":false,"result":"one\ntwo\nthree\nfour\nfive","stop_reason":"end_turn","session_id":"ac12982f","total_cost_usd":0.001,"usage":{"input_tokens":5,"cache_creation_input_tokens":30957,"cache_read_input_tokens":27019,"output_tokens":13}} +` + +func TestScanClaudeStream_RealCapture(t *testing.T) { + ch := scanClaudeStream(strings.NewReader(realStreamCapture)) + + var ( + gotStart *StreamEvent + gotDeltas []string + gotStop *StreamEvent + gotResult *StreamEvent + gotErr error + eventCount int + ) + for ev := range ch { + eventCount++ + switch ev.Kind { + case StreamEventMessageStart: + e := ev + gotStart = &e + case StreamEventTextDelta: + gotDeltas = append(gotDeltas, ev.Delta) + case StreamEventMessageStop: + e := ev + gotStop = &e + case StreamEventResult: + e := ev + gotResult = &e + case StreamEventError: + gotErr = ev.Err + } + } + + if gotErr != nil { + t.Fatalf("scanner emitted error: %v", gotErr) + } + if gotStart == nil || gotStart.MessageID != "msg_017bhZn5" || gotStart.Model != "claude-opus-4-7" { + t.Errorf("MessageStart event missing or wrong: %+v", gotStart) + } + if got, want := strings.Join(gotDeltas, ""), "one\ntwo\nthree\nfour\nfive"; got != want { + t.Errorf("text deltas concatenated = %q, want %q", got, want) + } + if gotStop == nil || gotStop.StopReason != "end_turn" { + t.Errorf("MessageStop event missing or wrong: %+v", gotStop) + } + if gotResult == nil || gotResult.Final == nil { + t.Fatal("Result event missing") + } + if gotResult.Final.Result != "one\ntwo\nthree\nfour\nfive" { + t.Errorf("Result.Final.Result = %q", gotResult.Final.Result) + } + if gotResult.Final.Usage.OutputTokens != 13 { + t.Errorf("Result.Final.Usage.OutputTokens = %d, want 13", gotResult.Final.Usage.OutputTokens) + } + + // Skipped types (system, assistant full-turn, content_block_*, + // message_stop) must not emit canonical events. + wantEvents := 1 /*start*/ + len(gotDeltas) + 1 /*stop*/ + 1 /*result*/ + if eventCount != wantEvents { + t.Errorf("emitted %d canonical events, want %d (skipped types leaked through?)", eventCount, wantEvents) + } +} + +func TestScanClaudeStream_IgnoresGarbageLines(t *testing.T) { + // Random non-JSON noise (could happen if stderr leaks into stdout) + // must be silently dropped, not turned into Error events. + input := "not json\n" + + `{"type":"stream_event","event":{"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"hi"}}}` + "\n" + var deltas []string + for ev := range scanClaudeStream(strings.NewReader(input)) { + if ev.Kind == StreamEventTextDelta { + deltas = append(deltas, ev.Delta) + } + if ev.Kind == StreamEventError { + t.Fatalf("garbage line produced an Error event: %v", ev.Err) + } + } + if len(deltas) != 1 || deltas[0] != "hi" { + t.Errorf("deltas = %v, want [hi]", deltas) + } +} + +func TestScanClaudeStream_OnlyTextDeltasForwarded(t *testing.T) { + // tool_use deltas (out of scope for v1) must not leak as text. + input := `{"type":"stream_event","event":{"type":"content_block_delta","index":0,"delta":{"type":"input_json_delta","partial_json":"{\"foo\":"}}}` + "\n" + for ev := range scanClaudeStream(strings.NewReader(input)) { + if ev.Kind == StreamEventTextDelta { + t.Fatalf("non-text_delta leaked as TextDelta: %+v", ev) + } + } +} diff --git a/internal/serve/exec.go b/internal/serve/exec.go index 7841b33f..68d4baba 100644 --- a/internal/serve/exec.go +++ b/internal/serve/exec.go @@ -3,36 +3,79 @@ package serve import ( "bytes" "os/exec" + "strings" "syscall" + "time" + + "github.com/DimmKirr/devcell/internal/logger" ) // ShellExecutor runs agent binaries as subprocesses. type ShellExecutor struct{} -// Run executes the agent binary with the given prompt and optional model. -func (e *ShellExecutor) Run(agent, prompt, model string) ExecResult { +// claudeArgs builds the claude argv shared by the buffered (Run, format=json) +// and streaming (streamClaude, format=stream-json) execution paths. +// +// --dangerously-skip-permissions matches cell claude's default +// (cmd/claude.go:40). Without it, any tool use (Read/Bash/Write/...) +// hits claude's permission gate; since stdin is a bytes.Buffer and +// not a TTY, the gate either fails or hangs the HTTP request. The +// operator's auth boundary on cell serve is DEVCELL_API_KEY, so +// the permission gate is not adding a meaningful second layer here. +// +// --output-format=json wraps the assistant text in a result envelope with +// per-turn usage (input/output/cache tokens) and total cost; --output-format +// =stream-json (with --include-partial-messages) emits the same `result` +// envelope as the terminal line, prefixed by a stream of Anthropic +// Messages-API `stream_event` wrappers carrying token-level deltas. +func claudeArgs(opts ExecOpts, format string) []string { + args := []string{ + "--dangerously-skip-permissions", + "--output-format", format, + } + if format == "stream-json" { + args = append(args, "--include-partial-messages") + } + args = append(args, "-p", opts.Prompt) + if opts.Model != "" { + args = append(args, "--model", opts.Model) + } + if opts.Effort != "" { + args = append(args, "--effort", opts.Effort) + } + if opts.SystemPrompt != "" { + args = append(args, "--append-system-prompt", opts.SystemPrompt) + } + return args +} + +// Run executes the agent binary with the given options. +func (e *ShellExecutor) Run(opts ExecOpts) ExecResult { var args []string - switch agent { + switch opts.Agent { case "claude": - args = append(args, "-p", prompt) - if model != "" { - args = append(args, "--model", model) - } + args = claudeArgs(opts, "json") case "opencode": // opencode doesn't have a one-shot prompt mode yet; // pass prompt as positional arg for now. - args = append(args, prompt) - if model != "" { - args = append(args, "--model", model) + args = append(args, opts.Prompt) + if opts.Model != "" { + args = append(args, "--model", opts.Model) } + // opencode has no --effort or --append-system-prompt equivalent; ignore. } - cmd := exec.Command(agent, args...) + logger.Debug("exec agent", "agent", opts.Agent, "model", opts.Model, "effort", opts.Effort) + + cmd := exec.Command(opts.Agent, args...) var stdout, stderr bytes.Buffer cmd.Stdout = &stdout cmd.Stderr = &stderr + start := time.Now() err := cmd.Run() + duration := time.Since(start) + exitCode := 0 if err != nil { if exitErr, ok := err.(*exec.ExitError); ok { @@ -47,9 +90,42 @@ func (e *ShellExecutor) Run(agent, prompt, model string) ExecResult { } } - return ExecResult{ - Stdout: stdout.String(), + if exitCode != 0 { + logger.Warn("agent failed", "agent", opts.Agent, "exit_code", exitCode, "duration", duration.String()) + } else { + logger.Info("agent completed", "agent", opts.Agent, "duration", duration.String()) + } + + // Agent CLIs (claude, opencode) terminate stdout with a trailing newline, + // which would leak into output_text on /v1/responses and message.content + // on /v1/chat/completions. Strip only trailing newlines — preserves any + // intentional leading whitespace and indentation inside the answer. + res := ExecResult{ + Stdout: strings.TrimRight(stdout.String(), "\n"), Stderr: stderr.String(), ExitCode: exitCode, } + + // Decode claude's --output-format=json envelope into Stdout (the text) + // + Usage (token + cost telemetry). On parse failure (claude crashed + // mid-output, version skew, etc.) we keep the raw bytes as Stdout — + // degraded but functional, with a Warn line so the operator notices. + if opts.Agent == "claude" && exitCode == 0 && len(res.Stdout) > 0 { + text, usage, perr := parseClaudeJSON([]byte(res.Stdout)) + if perr != nil { + logger.Warn("claude json parse failed; falling back to raw stdout", + "err", perr.Error(), "raw_first_200", truncate(res.Stdout, 200)) + } else { + res.Stdout = text + res.Usage = usage + } + } + return res +} + +func truncate(s string, n int) string { + if len(s) <= n { + return s + } + return s[:n] + "..." } diff --git a/internal/serve/exec_stream.go b/internal/serve/exec_stream.go new file mode 100644 index 00000000..a1eea31a --- /dev/null +++ b/internal/serve/exec_stream.go @@ -0,0 +1,58 @@ +package serve + +import ( + "context" + "fmt" + "os/exec" + + "github.com/DimmKirr/devcell/internal/logger" +) + +// streamClaude spawns claude in --output-format=stream-json mode and +// returns a channel of canonical StreamEvents (see claude_stream.go). +// The channel closes when claude exits or stdout reaches EOF. +// +// Cancellation: the caller controls lifetime via ctx. exec.CommandContext +// kills the process when ctx is cancelled — used by SSE handlers to stop +// claude when the HTTP client closes the connection. +// +// Opencode is rejected with an error: it has no stream-json equivalent. +// SSE handlers fall back to the buffered path for opencode. +func streamClaude(ctx context.Context, opts ExecOpts) (<-chan StreamEvent, error) { + if opts.Agent != "claude" { + return nil, fmt.Errorf("streaming is only supported for claude (got %q)", opts.Agent) + } + + cmd := exec.CommandContext(ctx, "claude", claudeArgs(opts, "stream-json")...) + stdout, err := cmd.StdoutPipe() + if err != nil { + return nil, fmt.Errorf("claude stdout pipe: %w", err) + } + // Capture stderr — useful in WARN logs if claude exits non-zero + // before emitting a `result` envelope. + cmd.Stderr = nil + + if err := cmd.Start(); err != nil { + return nil, fmt.Errorf("start claude: %w", err) + } + + scan := scanClaudeStream(stdout) + + // Wrap so we can join `cmd.Wait` after the channel closes — without + // this, killed-by-ctx subprocesses would leave zombie entries until + // the Go runtime reaps them on its own schedule. + out := make(chan StreamEvent, 16) + go func() { + defer close(out) + for ev := range scan { + out <- ev + } + if err := cmd.Wait(); err != nil && ctx.Err() == nil { + // Don't surface cancellation as an error — that's the + // expected path when the HTTP client disconnects. + logger.Warn("claude stream exited non-zero", "err", err.Error()) + out <- StreamEvent{Kind: StreamEventError, Err: fmt.Errorf("claude exited: %w", err)} + } + }() + return out, nil +} diff --git a/internal/serve/exec_stream_test.go b/internal/serve/exec_stream_test.go new file mode 100644 index 00000000..bbb2b493 --- /dev/null +++ b/internal/serve/exec_stream_test.go @@ -0,0 +1,123 @@ +package serve + +import ( + "context" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +// makeStreamingStubAgent writes a shell script at /claude that emits +// the given JSONL lines (one per element) on stdout. Used to exercise +// streamClaude end-to-end without spawning the real claude binary. +func makeStreamingStubAgent(t *testing.T, lines []string) string { + t.Helper() + dir := t.TempDir() + var b strings.Builder + b.WriteString("#!/bin/sh\n") + for _, ln := range lines { + // Use single-quoted printf with explicit %s\n; JSON has no + // embedded single quotes so this is safe. + b.WriteString("printf '%s\\n' '") + b.WriteString(ln) + b.WriteString("'\n") + } + path := filepath.Join(dir, "claude") + if err := os.WriteFile(path, []byte(b.String()), 0o755); err != nil { + t.Fatalf("write stub: %v", err) + } + return dir +} + +func TestStreamClaude_EndToEnd(t *testing.T) { + lines := []string{ + `{"type":"system","subtype":"init","session_id":"s1"}`, + `{"type":"stream_event","event":{"type":"message_start","message":{"id":"msg_1","model":"claude-opus-4-7","role":"assistant"}}}`, + `{"type":"stream_event","event":{"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"Hi"}}}`, + `{"type":"stream_event","event":{"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" there"}}}`, + `{"type":"stream_event","event":{"type":"message_delta","delta":{"stop_reason":"end_turn"}}}`, + `{"type":"result","subtype":"success","is_error":false,"result":"Hi there","usage":{"input_tokens":3,"output_tokens":2}}`, + } + dir := makeStreamingStubAgent(t, lines) + withPath(t, dir) + + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + + ch, err := streamClaude(ctx, ExecOpts{Agent: "claude", Prompt: "hi"}) + if err != nil { + t.Fatalf("streamClaude: %v", err) + } + + var ( + gotStart bool + gotDeltas []string + gotResult bool + ) + for ev := range ch { + switch ev.Kind { + case StreamEventMessageStart: + gotStart = true + case StreamEventTextDelta: + gotDeltas = append(gotDeltas, ev.Delta) + case StreamEventResult: + gotResult = true + case StreamEventError: + t.Fatalf("unexpected error: %v", ev.Err) + } + } + if !gotStart { + t.Error("MessageStart not observed") + } + if got := strings.Join(gotDeltas, ""); got != "Hi there" { + t.Errorf("deltas concatenated = %q, want %q", got, "Hi there") + } + if !gotResult { + t.Error("Result not observed") + } +} + +func TestStreamClaude_OpencodeRejected(t *testing.T) { + _, err := streamClaude(context.Background(), ExecOpts{Agent: "opencode", Prompt: "hi"}) + if err == nil { + t.Fatal("opencode should be rejected by streamClaude") + } + if !strings.Contains(err.Error(), "claude") { + t.Errorf("error should mention claude-only support: %v", err) + } +} + +func TestStreamClaude_ContextCancelStopsClaude(t *testing.T) { + // Stub that sleeps forever — cancelling ctx must kill it. Using + // `sleep 60` keeps it simple; the test fails on the 5s timeout if + // the process isn't killed promptly. + dir := t.TempDir() + script := "#!/bin/sh\nsleep 60\n" + if err := os.WriteFile(filepath.Join(dir, "claude"), []byte(script), 0o755); err != nil { + t.Fatal(err) + } + withPath(t, dir) + + ctx, cancel := context.WithCancel(context.Background()) + ch, err := streamClaude(ctx, ExecOpts{Agent: "claude", Prompt: "hi"}) + if err != nil { + t.Fatalf("streamClaude: %v", err) + } + // Cancel almost immediately and verify the channel closes promptly + // (stub would otherwise sleep 60s). + cancel() + deadline := time.NewTimer(5 * time.Second) + defer deadline.Stop() + for { + select { + case _, open := <-ch: + if !open { + return // channel closed → process was killed + } + case <-deadline.C: + t.Fatal("channel still open 5s after cancel — claude not killed") + } + } +} diff --git a/internal/serve/exec_test.go b/internal/serve/exec_test.go new file mode 100644 index 00000000..339fc575 --- /dev/null +++ b/internal/serve/exec_test.go @@ -0,0 +1,282 @@ +package serve + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +// makeStubAgent writes a shell script at / that records its argv to +// /.args and prints a fixed string. Returns the dir to put on PATH. +func makeStubAgent(t *testing.T, name, stdout string) string { + t.Helper() + dir := t.TempDir() + script := "#!/bin/sh\n" + + "echo \"$@\" > \"" + filepath.Join(dir, name+".args") + "\"\n" + + "printf '%s' '" + stdout + "'\n" + path := filepath.Join(dir, name) + if err := os.WriteFile(path, []byte(script), 0o755); err != nil { + t.Fatalf("write stub: %v", err) + } + return dir +} + +func readArgs(t *testing.T, dir, name string) string { + t.Helper() + b, err := os.ReadFile(filepath.Join(dir, name+".args")) + if err != nil { + t.Fatalf("read args: %v", err) + } + return strings.TrimSpace(string(b)) +} + +func withPath(t *testing.T, dir string) { + t.Helper() + old := os.Getenv("PATH") + t.Cleanup(func() { os.Setenv("PATH", old) }) + os.Setenv("PATH", dir) +} + +func TestShellExecutor_ClaudePermissionsBypassDefault(t *testing.T) { + // cell serve must mirror cell claude's --dangerously-skip-permissions + // default — otherwise tool use hangs the HTTP request on the + // permission gate (no TTY for stdin). + dir := makeStubAgent(t, "claude", "ok") + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{Agent: "claude", Prompt: "hi"}) + if res.ExitCode != 0 { + t.Fatalf("exit = %d, stderr=%q", res.ExitCode, res.Stderr) + } + args := readArgs(t, dir, "claude") + if !strings.Contains(args, "--dangerously-skip-permissions") { + t.Errorf("expected --dangerously-skip-permissions in argv, got %q", args) + } +} + +func TestShellExecutor_OpenCodeNoPermissionsFlag(t *testing.T) { + // opencode has no equivalent flag; ensure we don't accidentally pass + // --dangerously-skip-permissions to it. + dir := makeStubAgent(t, "opencode", "ok") + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{Agent: "opencode", Prompt: "hi"}) + if res.ExitCode != 0 { + t.Fatalf("exit = %d", res.ExitCode) + } + args := readArgs(t, dir, "opencode") + if strings.Contains(args, "--dangerously-skip-permissions") { + t.Errorf("opencode should not receive --dangerously-skip-permissions, got argv %q", args) + } +} + +func TestShellExecutor_ClaudeAppendsEffortFlag(t *testing.T) { + dir := makeStubAgent(t, "claude", "ok") + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{ + Agent: "claude", + Prompt: "hi", + Model: "sonnet", + Effort: "high", + }) + if res.ExitCode != 0 { + t.Fatalf("exit = %d, stderr=%q", res.ExitCode, res.Stderr) + } + if res.Stdout != "ok" { + t.Errorf("stdout = %q, want ok", res.Stdout) + } + + args := readArgs(t, dir, "claude") + if !strings.Contains(args, "--effort high") { + t.Errorf("expected --effort high in argv, got %q", args) + } + if !strings.Contains(args, "--model sonnet") { + t.Errorf("expected --model sonnet in argv, got %q", args) + } + // -p hi must come before flags, but we don't pin exact order — just presence. + if !strings.Contains(args, "-p hi") { + t.Errorf("expected -p hi in argv, got %q", args) + } +} + +func TestShellExecutor_ClaudeNoEffortNoFlag(t *testing.T) { + dir := makeStubAgent(t, "claude", "ok") + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{Agent: "claude", Prompt: "hi", Model: "sonnet"}) + if res.ExitCode != 0 { + t.Fatalf("exit = %d", res.ExitCode) + } + args := readArgs(t, dir, "claude") + if strings.Contains(args, "--effort") { + t.Errorf("expected no --effort flag when Effort empty, got %q", args) + } +} + +func TestShellExecutor_OpenCodeIgnoresEffort(t *testing.T) { + // opencode has no --effort flag; ExecOpts.Effort should not produce one. + dir := makeStubAgent(t, "opencode", "ok") + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{Agent: "opencode", Prompt: "hi", Effort: "high"}) + if res.ExitCode != 0 { + t.Fatalf("exit = %d", res.ExitCode) + } + args := readArgs(t, dir, "opencode") + if strings.Contains(args, "--effort") { + t.Errorf("opencode should not receive --effort, got argv %q", args) + } +} + +func TestShellExecutor_ClaudeAppendsSystemPromptFlag(t *testing.T) { + dir := makeStubAgent(t, "claude", "ok") + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{ + Agent: "claude", + Prompt: "hi", + SystemPrompt: "you are concise", + }) + if res.ExitCode != 0 { + t.Fatalf("exit = %d, stderr=%q", res.ExitCode, res.Stderr) + } + args := readArgs(t, dir, "claude") + if !strings.Contains(args, "--append-system-prompt you are concise") { + t.Errorf("expected --append-system-prompt flag in argv, got %q", args) + } +} + +func TestShellExecutor_ClaudeNoSystemPromptNoFlag(t *testing.T) { + dir := makeStubAgent(t, "claude", "ok") + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{Agent: "claude", Prompt: "hi"}) + if res.ExitCode != 0 { + t.Fatalf("exit = %d", res.ExitCode) + } + args := readArgs(t, dir, "claude") + if strings.Contains(args, "--append-system-prompt") { + t.Errorf("expected no --append-system-prompt flag when SystemPrompt empty, got %q", args) + } +} + +func TestShellExecutor_OpenCodeIgnoresSystemPrompt(t *testing.T) { + dir := makeStubAgent(t, "opencode", "ok") + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{Agent: "opencode", Prompt: "hi", SystemPrompt: "you are concise"}) + if res.ExitCode != 0 { + t.Fatalf("exit = %d", res.ExitCode) + } + args := readArgs(t, dir, "opencode") + if strings.Contains(args, "--append-system-prompt") { + t.Errorf("opencode should not receive --append-system-prompt, got argv %q", args) + } +} + +func TestShellExecutor_ClaudeUsesJSONOutputFormat(t *testing.T) { + dir := makeStubAgent(t, "claude", "ok") + withPath(t, dir) + + e := &ShellExecutor{} + _ = e.Run(ExecOpts{Agent: "claude", Prompt: "hi"}) + + args := readArgs(t, dir, "claude") + if !strings.Contains(args, "--output-format json") { + t.Errorf("expected --output-format json in claude argv, got %q", args) + } +} + +func TestShellExecutor_ClaudeJSONResultExtractedToStdout(t *testing.T) { + // Stub claude that emits the real JSON envelope. Stdout must end up + // as the unwrapped "result" string, and Usage must reflect the token + // counts from the JSON. + dir := makeStubAgent(t, "claude", realClaudeJSON) + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{Agent: "claude", Prompt: "hi"}) + + if res.ExitCode != 0 { + t.Fatalf("exit = %d, stderr=%q", res.ExitCode, res.Stderr) + } + if res.Stdout != "Hi." { + t.Errorf("Stdout = %q, want %q (should be unwrapped from JSON envelope)", res.Stdout, "Hi.") + } + if res.Usage.InputTokens != 5 || res.Usage.OutputTokens != 8 { + t.Errorf("usage not parsed: %+v", res.Usage) + } + if res.Usage.CacheCreationInputTokens != 58225 { + t.Errorf("CacheCreationInputTokens = %d, want 58225", res.Usage.CacheCreationInputTokens) + } +} + +func TestShellExecutor_ClaudeJSONParseFailureFallsBackToRaw(t *testing.T) { + // Garbled output (claude crashed mid-write, version skew, etc.). We + // must not lose the bytes — keep them as Stdout, leave Usage zero. + dir := makeStubAgent(t, "claude", "not actually json output") + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{Agent: "claude", Prompt: "hi"}) + + if res.ExitCode != 0 { + t.Fatalf("exit = %d", res.ExitCode) + } + if res.Stdout != "not actually json output" { + t.Errorf("Stdout = %q, want raw fallback", res.Stdout) + } + if (res.Usage != Usage{}) { + t.Errorf("Usage should be zero on parse failure, got %+v", res.Usage) + } +} + +func TestShellExecutor_OpenCodeNoJSONFlag(t *testing.T) { + // opencode doesn't emit JSON — it must NOT receive --output-format, + // and Usage stays zero. + dir := makeStubAgent(t, "opencode", "plain text reply") + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{Agent: "opencode", Prompt: "hi"}) + + args := readArgs(t, dir, "opencode") + if strings.Contains(args, "--output-format") { + t.Errorf("opencode should not receive --output-format, got argv %q", args) + } + if res.Stdout != "plain text reply" { + t.Errorf("Stdout = %q", res.Stdout) + } + if (res.Usage != Usage{}) { + t.Errorf("Usage should be zero for opencode, got %+v", res.Usage) + } +} + +func TestShellExecutor_NonZeroExitPropagated(t *testing.T) { + dir := t.TempDir() + script := "#!/bin/sh\necho oops 1>&2\nexit 7\n" + path := filepath.Join(dir, "claude") + if err := os.WriteFile(path, []byte(script), 0o755); err != nil { + t.Fatalf("write stub: %v", err) + } + withPath(t, dir) + + e := &ShellExecutor{} + res := e.Run(ExecOpts{Agent: "claude", Prompt: "hi"}) + if res.ExitCode != 7 { + t.Errorf("exit = %d, want 7", res.ExitCode) + } + if !strings.Contains(res.Stderr, "oops") { + t.Errorf("stderr = %q, want contains oops", res.Stderr) + } +} diff --git a/internal/serve/handler.go b/internal/serve/handler.go index fd035eca..6b4bf3c3 100644 --- a/internal/serve/handler.go +++ b/internal/serve/handler.go @@ -1,13 +1,17 @@ package serve import ( + "bytes" "crypto/rand" "encoding/hex" "encoding/json" "fmt" + "io" "net/http" "strings" "time" + + "github.com/DimmKirr/devcell/internal/logger" ) // agentForPrefix maps model prefix to the binary name. @@ -19,7 +23,28 @@ var agentForPrefix = map[string]string{ // Executor runs an agent command and returns the result. type Executor interface { - Run(agent, prompt, model string) ExecResult + Run(opts ExecOpts) ExecResult +} + +// ExecOpts is the bundle of arguments passed to Executor.Run. +// +// Adding a new CLI-flag mapping (e.g. --max-budget-usd) means adding a field +// here rather than widening Run's signature. +type ExecOpts struct { + // Agent is the binary name ("claude" or "opencode"). + Agent string + // Prompt is the assembled prompt string passed via -p / positional arg. + Prompt string + // Model is the optional sub-model (e.g. "sonnet" or "claude-sonnet-4-5"). Empty = agent default. + Model string + // Effort, when set, is passed as --effort to the claude CLI. + // Valid values: "low", "medium", "high". Empty = CLI default. + Effort string + // SystemPrompt, when set, is passed as --append-system-prompt to claude. + // Operator-level baseline (set on `cell serve` startup), composes with + // any per-request `instructions` / `system` role from the OpenAI body — + // it does NOT override them. Ignored for opencode (no equivalent flag). + SystemPrompt string } // ExecResult holds the output of an agent execution. @@ -27,42 +52,123 @@ type ExecResult struct { Stdout string Stderr string ExitCode int + // Usage carries token-and-cost telemetry parsed from the agent's + // machine-readable output (claude --output-format=json today). + // Zero-valued when the agent doesn't emit usage data (opencode) or + // when JSON parsing falls back to raw stdout. + Usage Usage +} + +// Usage is the agent-side token/cost view for one request. Mapped into +// OpenAI's per-API shape (ChatUsage / ResponsesUsage) at handler time. +// +// Field naming follows Anthropic's wire format so the parser is a 1:1 +// JSON decode of claude's `usage` object — the OpenAI mapping +// (prompt_tokens = input + cache_creation + cache_read; completion = output) +// is done in the handlers, not here, so other agents can plug in with +// their own native shape. +type Usage struct { + InputTokens int + OutputTokens int + CacheCreationInputTokens int + CacheReadInputTokens int + // TotalCostUSD is what claude reports; opencode doesn't surface cost. + TotalCostUSD float64 +} + +// validEffort reports whether v is one of the OpenAI-spec effort levels. +// +// We deliberately accept only "low", "medium", "high" — the values OpenAI +// documents — and silently drop unknown values (including Claude's "xhigh" +// and "max" extensions) to match permissive parsing of other fields. +func validEffort(v string) bool { + switch v { + case "low", "medium", "high": + return true + } + return false } // ChatMessage is an OpenAI-compatible message. type ChatMessage struct { - Role string `json:"role"` - Content string `json:"content"` + // Role of the message author: "system", "user", or "assistant". + Role string `json:"role" example:"user"` + // The message content (prompt text for user, response text for assistant). + Content string `json:"content" example:"Explain the main function in this repo"` } // ChatRequest is the OpenAI-compatible chat completions request. type ChatRequest struct { - Model string `json:"model"` + // Model selects the agent. Use "claude", "anthropic", or "opencode" as a prefix. + // Append a sub-model with a slash: "claude/claude-sonnet-4-5". + Model string `json:"model" example:"claude"` + // Messages is the conversation history. The last user message is used as the prompt. Messages []ChatMessage `json:"messages"` + // ReasoningEffort, when set, controls Claude's thinking budget for this request. + // Valid values: "low", "medium", "high". Other values are silently dropped. + // Maps to the `claude --effort` CLI flag. Has no effect on the opencode agent. + ReasoningEffort string `json:"reasoning_effort,omitempty" example:"high"` + // Stream, when true, emits Server-Sent Events with token-level deltas. + // Supported only for the claude agent (opencode falls back to buffered). + Stream bool `json:"stream,omitempty" example:"false"` + // StreamOptions configures streaming behavior. Honored only when Stream is true. + StreamOptions *ChatStreamOptions `json:"stream_options,omitempty"` +} + +// ChatStreamOptions mirrors OpenAI's stream_options object. +type ChatStreamOptions struct { + // IncludeUsage, when true, embeds the token-usage object in the + // final SSE chunk. Off by default to match OpenAI's contract. + IncludeUsage bool `json:"include_usage,omitempty" example:"true"` } // ChatChoice is a single choice in the response. type ChatChoice struct { - Index int `json:"index"` - Message ChatMessage `json:"message"` - FinishReason string `json:"finish_reason"` + // Index of this choice (always 0 — single-choice responses). + Index int `json:"index" example:"0"` + // The assistant's response message. + Message ChatMessage `json:"message"` + // Finish reason: "stop" on success, "error" if the agent exited non-zero. + FinishReason string `json:"finish_reason" example:"stop"` } -// ChatUsage tracks token usage (stubbed for now). +// ChatUsage tracks token usage. Populated from claude --output-format=json +// (input + cache_creation + cache_read merged into prompt_tokens to match +// OpenAI semantics). Zero-valued for opencode and for claude paths where +// JSON parsing fell back to raw stdout. type ChatUsage struct { - PromptTokens int `json:"prompt_tokens"` - CompletionTokens int `json:"completion_tokens"` - TotalTokens int `json:"total_tokens"` + PromptTokens int `json:"prompt_tokens" example:"42"` + CompletionTokens int `json:"completion_tokens" example:"7"` + TotalTokens int `json:"total_tokens" example:"49"` +} + +// chatUsageFromExec maps the agent-native Usage into the OpenAI Chat +// Completions shape. OpenAI's prompt_tokens is "input + cached input +// (creation + read)" — Anthropic separates these on the wire, so we +// flatten them here. +func chatUsageFromExec(u Usage) ChatUsage { + prompt := u.InputTokens + u.CacheCreationInputTokens + u.CacheReadInputTokens + return ChatUsage{ + PromptTokens: prompt, + CompletionTokens: u.OutputTokens, + TotalTokens: prompt + u.OutputTokens, + } } // ChatResponse is the OpenAI-compatible chat completions response. type ChatResponse struct { - ID string `json:"id"` - Object string `json:"object"` - Created int64 `json:"created"` - Model string `json:"model"` + // Unique completion ID (format: chatcmpl-). + ID string `json:"id" example:"chatcmpl-a1b2c3d4e5f6"` + // Object type (always "chat.completion"). + Object string `json:"object" example:"chat.completion"` + // Unix timestamp of when the response was created. + Created int64 `json:"created" example:"1714000000"` + // The model that was requested. + Model string `json:"model" example:"claude"` + // Response choices (always a single element). Choices []ChatChoice `json:"choices"` - Usage ChatUsage `json:"usage"` + // Token usage (stubbed, reserved for future use). + Usage ChatUsage `json:"usage"` } // parseModel extracts agent and submodel from the model string. @@ -81,15 +187,52 @@ func chatcmplID() string { } // NewChatHandler returns an http.Handler for POST /v1/chat/completions. -func NewChatHandler(exec Executor) http.Handler { +// +// @Summary Send a chat completion request +// @Description Accepts an OpenAI-compatible chat completion request and routes it to the appropriate +// @Description LLM agent binary (Claude Code or OpenCode) running inside the DevCell container. +// @Description +// @Description The `model` field determines which agent handles the request: +// @Description - `"claude"` or `"anthropic"` — routes to Claude Code CLI +// @Description - `"opencode"` — routes to OpenCode CLI +// @Description - `"claude/claude-sonnet-4-5"` — routes to Claude Code with a specific sub-model +// @Description +// @Description Only the **last user message** in the `messages` array is sent as the prompt to the agent. +// @Description The response is a single-choice completion with finish_reason "stop" on success or "error" on failure. +// @Description +// @Description **Honored fields:** +// @Description - `reasoning_effort` (`low` / `medium` / `high`) → maps to the `claude --effort` CLI flag +// @Description to control thinking budget. Other values (including Claude's `xhigh`/`max`) are silently dropped. +// @Description +// @Description **Example request:** +// @Description ```json +// @Description {"model": "claude", "messages": [{"role": "user", "content": "explain this repo"}]} +// @Description ``` +// @Tags chat +// @Accept json +// @Produce json +// @Param request body ChatRequest true "Chat completion request" +// @Success 200 {object} ChatResponse "Successful completion" +// @Failure 400 {string} string "Invalid JSON, missing model, empty messages, or unknown model prefix" +// @Failure 401 {string} string "Missing or invalid Bearer token" +// @Failure 405 {string} string "Only POST is allowed" +// @Security BearerAuth +// @Router /v1/chat/completions [post] +func NewChatHandler(exec Executor, logPrompts bool, systemPrompt string) http.Handler { return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { if r.Method != http.MethodPost { http.Error(w, "method not allowed", http.StatusMethodNotAllowed) return } + body, err := io.ReadAll(r.Body) + if err != nil { + http.Error(w, fmt.Sprintf("read body: %v", err), http.StatusBadRequest) + return + } + var req ChatRequest - if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + if err := json.NewDecoder(bytes.NewReader(body)).Decode(&req); err != nil { http.Error(w, fmt.Sprintf("invalid JSON: %v", err), http.StatusBadRequest) return } @@ -114,7 +257,42 @@ func NewChatHandler(exec Executor) http.Handler { // Use the last user message as the prompt. prompt := req.Messages[len(req.Messages)-1].Content - result := exec.Run(agent, prompt, submodel) + // reasoning_effort is OpenAI's per-request thinking-budget hint + // (low|medium|high). Maps to claude's --effort flag. Silently drop + // unknown values rather than 400 — matches our permissive parsing + // of other not-fully-supported fields. + effort := req.ReasoningEffort + if effort != "" && !validEffort(effort) { + effort = "" + } + + id := chatcmplID() + + if logPrompts { + logger.Info("chat prompt", + "id", id, "model", req.Model, "agent", agent, + "effort", effort, "prompt", prompt, + "body", json.RawMessage(body)) + } + + opts := ExecOpts{ + Agent: agent, + Prompt: prompt, + Model: submodel, + Effort: effort, + SystemPrompt: systemPrompt, + } + + // Streaming path: only claude has a token-level streaming surface + // (claude --output-format=stream-json). Opencode falls through to + // the buffered path. + if req.Stream && agent == "claude" { + includeUsage := req.StreamOptions != nil && req.StreamOptions.IncludeUsage + streamChatCompletions(w, r, opts, id, req.Model, includeUsage) + return + } + + result := exec.Run(opts) finishReason := "stop" content := result.Stdout @@ -125,8 +303,13 @@ func NewChatHandler(exec Executor) http.Handler { } } + if logPrompts { + logger.Info("chat response", + "id", id, "finish_reason", finishReason, "content", content) + } + resp := ChatResponse{ - ID: chatcmplID(), + ID: id, Object: "chat.completion", Created: time.Now().Unix(), Model: req.Model, @@ -137,6 +320,7 @@ func NewChatHandler(exec Executor) http.Handler { FinishReason: finishReason, }, }, + Usage: chatUsageFromExec(result.Usage), } w.Header().Set("Content-Type", "application/json") diff --git a/internal/serve/handler_test.go b/internal/serve/handler_test.go index 7b7a47ea..8e9f0225 100644 --- a/internal/serve/handler_test.go +++ b/internal/serve/handler_test.go @@ -11,21 +11,25 @@ import ( // fakeExec records what was called and returns canned output. type fakeExec struct { - called bool - agent string - prompt string - model string + called bool + agent string + prompt string + model string + effort string + systemPrompt string stdout string stderr string exitCode int } -func (f *fakeExec) Run(agent, prompt, model string) ExecResult { +func (f *fakeExec) Run(opts ExecOpts) ExecResult { f.called = true - f.agent = agent - f.prompt = prompt - f.model = model + f.agent = opts.Agent + f.prompt = opts.Prompt + f.model = opts.Model + f.effort = opts.Effort + f.systemPrompt = opts.SystemPrompt return ExecResult{ Stdout: f.stdout, Stderr: f.stderr, @@ -44,7 +48,7 @@ func postChat(t *testing.T, handler http.Handler, body string) *httptest.Respons func TestHandler_ValidClaude(t *testing.T) { fe := &fakeExec{stdout: "hello back", exitCode: 0} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{"model":"anthropic/sonnet","messages":[{"role":"user","content":"hello"}]}`) @@ -83,7 +87,7 @@ func TestHandler_ValidClaude(t *testing.T) { func TestHandler_ValidOpencode(t *testing.T) { fe := &fakeExec{stdout: "opencode result"} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{"model":"opencode","messages":[{"role":"user","content":"hello"}]}`) @@ -97,7 +101,7 @@ func TestHandler_ValidOpencode(t *testing.T) { func TestHandler_ModelWithSubmodel(t *testing.T) { fe := &fakeExec{stdout: "ok"} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{"model":"anthropic/opus","messages":[{"role":"user","content":"hello"}]}`) @@ -114,7 +118,7 @@ func TestHandler_ModelWithSubmodel(t *testing.T) { func TestHandler_MissingModel(t *testing.T) { fe := &fakeExec{} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{"messages":[{"role":"user","content":"hello"}]}`) @@ -131,7 +135,7 @@ func TestHandler_MissingModel(t *testing.T) { func TestHandler_MissingMessages(t *testing.T) { fe := &fakeExec{} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{"model":"anthropic/sonnet"}`) @@ -145,7 +149,7 @@ func TestHandler_MissingMessages(t *testing.T) { func TestHandler_EmptyMessages(t *testing.T) { fe := &fakeExec{} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{"model":"anthropic/sonnet","messages":[]}`) @@ -159,7 +163,7 @@ func TestHandler_EmptyMessages(t *testing.T) { func TestHandler_UnknownAgent(t *testing.T) { fe := &fakeExec{} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{"model":"foo","messages":[{"role":"user","content":"hello"}]}`) @@ -174,7 +178,7 @@ func TestHandler_UnknownAgent(t *testing.T) { func TestHandler_EmptyBody(t *testing.T) { fe := &fakeExec{} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") req := httptest.NewRequest(http.MethodPost, "/v1/chat/completions", &bytes.Buffer{}) req.Header.Set("Content-Type", "application/json") @@ -188,7 +192,7 @@ func TestHandler_EmptyBody(t *testing.T) { func TestHandler_InvalidJSON(t *testing.T) { fe := &fakeExec{} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{broken`) @@ -199,7 +203,7 @@ func TestHandler_InvalidJSON(t *testing.T) { func TestHandler_MethodNotAllowed(t *testing.T) { fe := &fakeExec{} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") req := httptest.NewRequest(http.MethodGet, "/v1/chat/completions", nil) rec := httptest.NewRecorder() @@ -212,7 +216,7 @@ func TestHandler_MethodNotAllowed(t *testing.T) { func TestHandler_MultipleMessages_UsesLast(t *testing.T) { fe := &fakeExec{stdout: "ok"} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{"model":"anthropic/sonnet","messages":[{"role":"user","content":"first"},{"role":"user","content":"second"}]}`) @@ -226,7 +230,7 @@ func TestHandler_MultipleMessages_UsesLast(t *testing.T) { func TestHandler_ExecFailure(t *testing.T) { fe := &fakeExec{stderr: "something broke", exitCode: 1} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{"model":"anthropic/sonnet","messages":[{"role":"user","content":"hello"}]}`) @@ -246,7 +250,7 @@ func TestHandler_ExecFailure(t *testing.T) { func TestHandler_ResponseHasID(t *testing.T) { fe := &fakeExec{stdout: "ok"} - h := NewChatHandler(fe) + h := NewChatHandler(fe, false, "") rec := postChat(t, h, `{"model":"anthropic/sonnet","messages":[{"role":"user","content":"hello"}]}`) @@ -259,3 +263,76 @@ func TestHandler_ResponseHasID(t *testing.T) { t.Errorf("id = %q, want prefix 'chatcmpl-'", resp.ID) } } + +// --- reasoning_effort → claude --effort mapping (Chat Completions) --- +// +// Mirrors the Responses-API tests. Chat Completions sends `reasoning_effort` +// at the request root (flat), not nested under `reasoning`. Both endpoints +// must produce identical executor behavior for the same effort value. + +func TestHandler_Effort_OpenAISpecValuesPassThrough(t *testing.T) { + for _, v := range []string{"low", "medium", "high"} { + t.Run(v, func(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewChatHandler(fe, false, "") + body := `{"model":"anthropic/sonnet","reasoning_effort":"` + v + + `","messages":[{"role":"user","content":"hi"}]}` + rec := postChat(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + if fe.effort != v { + t.Errorf("effort reaching executor = %q, want %q", fe.effort, v) + } + }) + } +} + +func TestHandler_Effort_ClaudeOnlyValuesDropped(t *testing.T) { + // Same rule as Responses: Claude's "xhigh"/"max" are not OpenAI spec, drop them. + for _, v := range []string{"xhigh", "max"} { + t.Run(v, func(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewChatHandler(fe, false, "") + body := `{"model":"anthropic/sonnet","reasoning_effort":"` + v + + `","messages":[{"role":"user","content":"hi"}]}` + rec := postChat(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d", rec.Code) + } + if fe.effort != "" { + t.Errorf("non-OpenAI-spec effort %q leaked to executor (got %q)", v, fe.effort) + } + }) + } +} + +func TestHandler_Effort_UnknownValuesDropped(t *testing.T) { + for _, v := range []string{"extreme", "minimal", "LOW", "High", "auto"} { + t.Run(v, func(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewChatHandler(fe, false, "") + body := `{"model":"anthropic/sonnet","reasoning_effort":"` + v + + `","messages":[{"role":"user","content":"hi"}]}` + rec := postChat(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d", rec.Code) + } + if fe.effort != "" { + t.Errorf("unknown effort %q leaked to executor (got %q)", v, fe.effort) + } + }) + } +} + +func TestHandler_Effort_AbsentNoFlag(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewChatHandler(fe, false, "") + rec := postChat(t, h, `{"model":"anthropic/sonnet","messages":[{"role":"user","content":"hi"}]}`) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d", rec.Code) + } + if fe.effort != "" { + t.Errorf("absent reasoning_effort produced executor effort=%q", fe.effort) + } +} diff --git a/internal/serve/logging.go b/internal/serve/logging.go new file mode 100644 index 00000000..382e7196 --- /dev/null +++ b/internal/serve/logging.go @@ -0,0 +1,52 @@ +package serve + +import ( + "net/http" + "time" + + "github.com/DimmKirr/devcell/internal/logger" +) + +// statusWriter wraps http.ResponseWriter to capture the status code. +type statusWriter struct { + http.ResponseWriter + code int +} + +func (w *statusWriter) WriteHeader(code int) { + w.code = code + w.ResponseWriter.WriteHeader(code) +} + +// LoggingMiddleware logs every HTTP request with method, path, status, and duration. +func LoggingMiddleware(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + start := time.Now() + sw := &statusWriter{ResponseWriter: w, code: http.StatusOK} + next.ServeHTTP(sw, r) + duration := time.Since(start) + + if sw.code >= 500 { + logger.Error("request", + "method", r.Method, + "path", r.URL.Path, + "status", sw.code, + "duration", duration.String(), + ) + } else if sw.code >= 400 { + logger.Warn("request", + "method", r.Method, + "path", r.URL.Path, + "status", sw.code, + "duration", duration.String(), + ) + } else { + logger.Info("request", + "method", r.Method, + "path", r.URL.Path, + "status", sw.code, + "duration", duration.String(), + ) + } + }) +} diff --git a/internal/serve/models.go b/internal/serve/models.go index e1d9544f..910c4338 100644 --- a/internal/serve/models.go +++ b/internal/serve/models.go @@ -10,16 +10,22 @@ import ( // ModelInfo represents a single model in the OpenAI /v1/models response. type ModelInfo struct { - ID string `json:"id"` - Object string `json:"object"` - Created int64 `json:"created"` - OwnedBy string `json:"owned_by"` + // Model identifier — use this value in the chat completions "model" field. + ID string `json:"id" example:"anthropic/claude-sonnet-4-5-20250514"` + // Object type (always "model"). + Object string `json:"object" example:"model"` + // Unix timestamp when the model was discovered. + Created int64 `json:"created" example:"1714000000"` + // Owner of the model: "anthropic" for API-discovered models, "devcell" for local agents. + OwnedBy string `json:"owned_by" example:"anthropic"` } // ModelsResponse is the OpenAI-compatible /v1/models response. type ModelsResponse struct { - Object string `json:"object"` - Data []ModelInfo `json:"data"` + // Object type (always "list"). + Object string `json:"object" example:"list"` + // Available models discovered from installed agents and the Anthropic API. + Data []ModelInfo `json:"data"` } // LookPathFunc matches exec.LookPath signature. @@ -108,6 +114,24 @@ func (c *RealAnthropicClient) FetchModels() ([]ModelInfo, error) { } // NewModelsHandler returns an http.Handler for GET /v1/models. +// +// @Summary List available models +// @Description Returns all models that can be used in chat completion requests. +// @Description +// @Description Models are discovered dynamically at request time: +// @Description 1. If the `claude` binary is found, the server tries the Anthropic API to list real model IDs +// @Description (e.g. `anthropic/claude-sonnet-4-5-20250514`). If the API is unreachable, it falls back to +// @Description aliases: `anthropic/opus`, `anthropic/sonnet`, `anthropic/haiku`. +// @Description 2. If the `opencode` binary is found, `opencode` is added as an available model. +// @Description +// @Description Use any returned `id` value as the `model` field in `/v1/chat/completions`. +// @Tags models +// @Produce json +// @Success 200 {object} ModelsResponse "List of available models" +// @Failure 401 {string} string "Missing or invalid Bearer token" +// @Failure 405 {string} string "Only GET is allowed" +// @Security BearerAuth +// @Router /v1/models [get] func NewModelsHandler(lookPath LookPathFunc, ac AnthropicClient) http.Handler { return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { if r.Method != http.MethodGet { diff --git a/internal/serve/openai_compat_test.go b/internal/serve/openai_compat_test.go index 35e3840a..b42bafd5 100644 --- a/internal/serve/openai_compat_test.go +++ b/internal/serve/openai_compat_test.go @@ -15,7 +15,7 @@ type fakeExecCompat struct { stdout string } -func (f *fakeExecCompat) Run(agent, prompt, model string) serve.ExecResult { +func (f *fakeExecCompat) Run(opts serve.ExecOpts) serve.ExecResult { return serve.ExecResult{Stdout: f.stdout} } @@ -142,9 +142,9 @@ type routingExec struct { onRun func(agent, prompt, model string) } -func (r *routingExec) Run(agent, prompt, model string) serve.ExecResult { +func (r *routingExec) Run(opts serve.ExecOpts) serve.ExecResult { if r.onRun != nil { - r.onRun(agent, prompt, model) + r.onRun(opts.Agent, opts.Prompt, opts.Model) } return serve.ExecResult{Stdout: "ok"} } diff --git a/internal/serve/responses.go b/internal/serve/responses.go new file mode 100644 index 00000000..b3c443b0 --- /dev/null +++ b/internal/serve/responses.go @@ -0,0 +1,574 @@ +package serve + +import ( + "bytes" + "crypto/rand" + "encoding/hex" + "encoding/json" + "fmt" + "io" + "net/http" + "strings" + "time" + + "github.com/DimmKirr/devcell/internal/logger" +) + +// ResponsesRequest is the OpenAI Responses-API request body. +// +// All fields are accepted, but only Model, Input, Instructions, and Stream +// affect behavior. Tools, ResponseFormat, Reasoning, sampling params and +// other fields are decoded for compatibility and silently ignored — devcell +// shells out to a CLI agent and cannot honor them. +type ResponsesRequest struct { + // Model selects the agent. Use "claude", "anthropic", or "opencode" as a prefix. + // Append a sub-model with a slash: "anthropic/sonnet" or "anthropic/claude-sonnet-4-5". + Model string `json:"model" example:"anthropic/sonnet"` + + // Input is either a string OR an array of input items. + // String form: a single user message. + // Array form: a multi-turn conversation; each item has role + content. + Input json.RawMessage `json:"input" swaggertype:"string" example:"hello"` + + // Instructions is an optional system prompt prepended to the conversation. + Instructions string `json:"instructions,omitempty" example:"be brief"` + + // Stream, if true, returns 400 — streaming is not supported. + Stream bool `json:"stream,omitempty"` + + // Reasoning carries reasoning-model controls. Only `reasoning.effort` + // (low|medium|high) is honored — it maps to `claude --effort`. Other + // fields (summary, generate_summary) are accepted and ignored. + Reasoning *ResponsesReasoningConfig `json:"reasoning,omitempty"` + + // Accepted, ignored — kept for client compatibility. + PreviousResponseID string `json:"previous_response_id,omitempty"` + Tools json.RawMessage `json:"tools,omitempty" swaggerignore:"true"` + ToolChoice json.RawMessage `json:"tool_choice,omitempty" swaggerignore:"true"` + ResponseFormat json.RawMessage `json:"response_format,omitempty" swaggerignore:"true"` + Text json.RawMessage `json:"text,omitempty" swaggerignore:"true"` + Temperature *float64 `json:"temperature,omitempty"` + TopP *float64 `json:"top_p,omitempty"` + MaxOutputTokens *int `json:"max_output_tokens,omitempty"` + Metadata json.RawMessage `json:"metadata,omitempty" swaggerignore:"true"` + Store *bool `json:"store,omitempty"` + ParallelToolCalls *bool `json:"parallel_tool_calls,omitempty"` + Truncation string `json:"truncation,omitempty"` + User string `json:"user,omitempty"` + Background *bool `json:"background,omitempty"` + Include []string `json:"include,omitempty"` +} + +// ResponsesReasoningConfig is the OpenAI-spec reasoning object. +// +// The Responses API allows clients to control thinking budget on reasoning +// models. devcell honors `effort` (mapping it to `claude --effort`) and +// ignores other fields. +type ResponsesReasoningConfig struct { + // Effort: "low", "medium", or "high". Other values are silently dropped. + Effort string `json:"effort,omitempty" example:"high"` + // Summary controls reasoning summary verbosity. Accepted, ignored. + Summary string `json:"summary,omitempty"` + // GenerateSummary is a deprecated alias of Summary. Accepted, ignored. + GenerateSummary string `json:"generate_summary,omitempty"` +} + +// ResponsesOutputContentPart is one part of an output message's content array. +type ResponsesOutputContentPart struct { + // Always "output_text" for text responses. + Type string `json:"type" example:"output_text"` + // The actual text. + Text string `json:"text" example:"Hello, world!"` + // Annotations on the text (always empty — reserved for future use). + Annotations []any `json:"annotations"` +} + +// ResponsesOutputItem is a single item in the output array. +// +// devcell only emits "message" items (assistant messages). Reasoning, +// tool_call, and other variants are not produced. +type ResponsesOutputItem struct { + // Item type — always "message". + Type string `json:"type" example:"message"` + // Unique item ID (format: msg_). + ID string `json:"id" example:"msg_a1b2c3d4e5f6"` + // Status of this item — "completed" on success. + Status string `json:"status" example:"completed"` + // Role of the message author — always "assistant". + Role string `json:"role" example:"assistant"` + // Content parts of the message. + Content []ResponsesOutputContentPart `json:"content"` +} + +// ResponsesUsage tracks token usage. Populated from claude +// --output-format=json. The Responses API has its own naming +// (input_tokens / output_tokens) and exposes cached-token detail under +// input_tokens_details.cached_tokens — that's claude's +// cache_read_input_tokens and is real money saved, so we surface it. +type ResponsesUsage struct { + InputTokens int `json:"input_tokens" example:"42"` + InputTokensDetails *ResponsesInputTokensDetails `json:"input_tokens_details,omitempty"` + OutputTokens int `json:"output_tokens" example:"7"` + OutputTokensDetails *ResponsesOutputTokensDetails `json:"output_tokens_details,omitempty"` + TotalTokens int `json:"total_tokens" example:"49"` +} + +// ResponsesInputTokensDetails carries the cached-input breakdown. +type ResponsesInputTokensDetails struct { + CachedTokens int `json:"cached_tokens" example:"0"` +} + +// ResponsesOutputTokensDetails is reserved for reasoning-model splits +// (reasoning_tokens vs visible output). Always zero today; included for +// schema completeness. +type ResponsesOutputTokensDetails struct { + ReasoningTokens int `json:"reasoning_tokens" example:"0"` +} + +// responsesUsageFromExec maps the agent-native Usage into the Responses +// shape. input_tokens here = OpenAI's prompt_tokens definition (input + +// cached); cached_tokens surfaces only the read-side cache hit. +func responsesUsageFromExec(u Usage) ResponsesUsage { + input := u.InputTokens + u.CacheCreationInputTokens + u.CacheReadInputTokens + return ResponsesUsage{ + InputTokens: input, + InputTokensDetails: &ResponsesInputTokensDetails{CachedTokens: u.CacheReadInputTokens}, + OutputTokens: u.OutputTokens, + TotalTokens: input + u.OutputTokens, + } +} + +// ResponsesError describes a model-side failure (exit != 0 from the agent CLI). +// +// Note: HTTP-level errors (400, 401, 405) use a different envelope at the +// top of the response — see APIError. +type ResponsesError struct { + // Short error code, e.g. "server_error". + Code string `json:"code,omitempty" example:"server_error"` + // Human-readable message — typically the agent's stderr. + Message string `json:"message" example:"agent failed"` +} + +// ResponsesIncompleteDetails is reserved — always null in devcell. +type ResponsesIncompleteDetails struct { + Reason string `json:"reason,omitempty"` +} + +// ResponsesObject is the OpenAI Responses-API response body. +type ResponsesObject struct { + // Unique response ID (format: resp_). + ID string `json:"id" example:"resp_a1b2c3d4e5f6"` + // Object type — always "response". + Object string `json:"object" example:"response"` + // Unix timestamp (seconds) when the response was created. + CreatedAt int64 `json:"created_at" example:"1714000000"` + // Status: "completed" on success, "failed" if the agent exited non-zero. + Status string `json:"status" example:"completed"` + // Model echo of the requested model string. + Model string `json:"model" example:"anthropic/sonnet"` + // Output items generated by the model. + Output []ResponsesOutputItem `json:"output"` + // Convenience field: concatenation of all output_text parts in Output. + // Most clients (n8n, simple scripts) read this directly. + OutputText string `json:"output_text" example:"Hello, world!"` + // Token usage (stubbed at zero — reserved for future use). + Usage ResponsesUsage `json:"usage"` + // Error populated when Status == "failed", null otherwise. + Error *ResponsesError `json:"error"` + // IncompleteDetails is reserved — always null. + IncompleteDetails *ResponsesIncompleteDetails `json:"incomplete_details"` + // Echo of the input instructions, or null if not set. + Instructions *string `json:"instructions"` + // Echo of input metadata, or null. + Metadata map[string]string `json:"metadata"` + // ParallelToolCalls — echo of input or default true. + ParallelToolCalls bool `json:"parallel_tool_calls" example:"true"` + // PreviousResponseID — always null (stateless). + PreviousResponseID *string `json:"previous_response_id"` + // Reasoning config — echoes the input reasoning object (with normalized + // effort if it was applied), or null if no reasoning was sent. + Reasoning *ResponsesReasoningConfig `json:"reasoning"` + // Store flag — echo of input or default true. + Store bool `json:"store" example:"true"` + // Sampling temperature — echo of input or default 1.0 (devcell ignores it). + Temperature float64 `json:"temperature" example:"1.0"` + // ToolChoice — always "auto". + ToolChoice string `json:"tool_choice" example:"auto"` + // Tools — always empty array (no tools bridged). + Tools []any `json:"tools"` + // Top-p — echo of input or default 1.0 (devcell ignores it). + TopP float64 `json:"top_p" example:"1.0"` + // Truncation — "disabled" by default. + Truncation string `json:"truncation" example:"disabled"` + // User — echo of input or empty. + User string `json:"user,omitempty"` +} + +// APIError is the OpenAI-shaped error envelope returned by /v1/responses +// for HTTP-level errors (4xx / 5xx). +type APIError struct { + Error APIErrorBody `json:"error"` +} + +// APIErrorBody is the inner error object. +type APIErrorBody struct { + Message string `json:"message" example:"streaming is not supported"` + Type string `json:"type" example:"invalid_request_error"` + Code string `json:"code,omitempty" example:"streaming_unsupported"` +} + +func writeAPIError(w http.ResponseWriter, status int, errType, code, message string) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(status) + json.NewEncoder(w).Encode(APIError{Error: APIErrorBody{ + Message: message, + Type: errType, + Code: code, + }}) +} + +func responseID() string { + b := make([]byte, 12) + rand.Read(b) + return "resp_" + hex.EncodeToString(b) +} + +func messageID() string { + b := make([]byte, 12) + rand.Read(b) + return "msg_" + hex.EncodeToString(b) +} + +// inputItem is a single message in the array form of `input`. +// +// Matches the SDK's EasyInputMessageParam: role + content (string or part array). +type inputItem struct { + Type string `json:"type"` // optional — "message" if present + Role string `json:"role"` + Content json.RawMessage `json:"content"` +} + +// inputContentPart matches both `input_text` and `output_text` parts. +type inputContentPart struct { + Type string `json:"type"` + Text string `json:"text"` +} + +// renderContent extracts text from a content field that may be a string +// or an array of typed parts. +func renderContent(raw json.RawMessage) (string, error) { + if len(raw) == 0 { + return "", nil + } + // Try string first. + var s string + if err := json.Unmarshal(raw, &s); err == nil { + return s, nil + } + // Else try array of parts. + var parts []inputContentPart + if err := json.Unmarshal(raw, &parts); err != nil { + return "", fmt.Errorf("content must be a string or an array of content parts") + } + var out []string + for _, p := range parts { + if p.Text != "" { + out = append(out, p.Text) + } + } + return strings.Join(out, " "), nil +} + +// buildPrompt serializes instructions + input into a single prompt string. +// +// Returns (prompt, error). An empty input returns ("", error). +func buildPrompt(instructions string, input json.RawMessage) (string, error) { + var b strings.Builder + + if instructions != "" { + b.WriteString("[system]: ") + b.WriteString(instructions) + b.WriteString("\n") + } + + if len(input) == 0 || string(input) == "null" { + if b.Len() == 0 { + return "", fmt.Errorf("input is required") + } + return "", fmt.Errorf("input is required") + } + + // Try string form. + var s string + if err := json.Unmarshal(input, &s); err == nil { + if s == "" { + return "", fmt.Errorf("input is required") + } + b.WriteString("[user]: ") + b.WriteString(s) + b.WriteString("\n") + return b.String(), nil + } + + // Try array form. + var items []inputItem + if err := json.Unmarshal(input, &items); err != nil { + return "", fmt.Errorf("input must be a string or an array of input items") + } + if len(items) == 0 { + return "", fmt.Errorf("input is required") + } + + any := false + for _, it := range items { + text, err := renderContent(it.Content) + if err != nil { + return "", err + } + if text == "" { + continue + } + role := it.Role + if role == "" { + role = "user" + } + // "developer" is OpenAI's newer alias for system. + if role == "developer" { + role = "system" + } + fmt.Fprintf(&b, "[%s]: %s\n", role, text) + any = true + } + if !any { + return "", fmt.Errorf("input is required") + } + return b.String(), nil +} + +// NewResponsesHandler returns an http.Handler for POST /v1/responses. +// +// @Summary Create a model response (Responses API) +// @Description OpenAI-compatible Responses API endpoint. Accepts a request shaped like +// @Description `client.responses.create` from the official SDKs and returns a Response object. +// @Description +// @Description The `model` field selects the agent (same routing as `/v1/chat/completions`): +// @Description - `"anthropic/sonnet"`, `"claude/"` — routes to the Claude Code CLI +// @Description - `"opencode"` — routes to the OpenCode CLI +// @Description +// @Description The `input` field is either a string or an array of input items +// @Description (`{"role": "user|assistant|system", "content": "..."}` or with typed content parts). +// @Description The `instructions` field is prepended as a system message. +// @Description +// @Description **Statelessness:** devcell does not persist responses. `previous_response_id` +// @Description is accepted for compatibility but ignored — clients must send full conversation +// @Description history every request. +// @Description +// @Description **Streaming is not supported.** Requests with `"stream": true` return 400. +// @Description +// @Description **Honored fields beyond core:** +// @Description - `reasoning.effort` (`low` / `medium` / `high`) → maps to the `claude --effort` CLI flag +// @Description to control thinking budget. Other values (including Claude's `xhigh`/`max`) are silently dropped. +// @Description +// @Description **Unsupported fields** (`tools`, `response_format`, `temperature`, `top_p`, +// @Description `max_output_tokens`, `metadata`, `store`, `service_tier`, etc.) are accepted to keep +// @Description SDK clients happy but have no effect — devcell shells out to a CLI agent and +// @Description cannot honor them. +// @Description +// @Description **Example request:** +// @Description ```json +// @Description {"model": "anthropic/sonnet", "input": "What is 2+2?"} +// @Description ``` +// @Description +// @Description **Reading the response:** most clients use the top-level `output_text` field. +// @Description SDK clients use the `output[].content[].text` structure. +// @Tags responses +// @Accept json +// @Produce json +// @Param request body ResponsesRequest true "Responses API request" +// @Success 200 {object} ResponsesObject "Successful response" +// @Failure 400 {object} APIError "Invalid JSON, missing model/input, unknown model, or streaming requested" +// @Failure 401 {string} string "Missing or invalid Bearer token" +// @Failure 405 {object} APIError "Only POST is allowed" +// @Security BearerAuth +// @Router /v1/responses [post] +func NewResponsesHandler(exec Executor, logPrompts bool, systemPrompt string) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodPost { + writeAPIError(w, http.StatusMethodNotAllowed, + "invalid_request_error", "method_not_allowed", + "only POST is allowed") + return + } + + body, err := io.ReadAll(r.Body) + if err != nil { + writeAPIError(w, http.StatusBadRequest, + "invalid_request_error", "invalid_body", + fmt.Sprintf("read body: %v", err)) + return + } + + var req ResponsesRequest + if err := json.NewDecoder(bytes.NewReader(body)).Decode(&req); err != nil { + writeAPIError(w, http.StatusBadRequest, + "invalid_request_error", "invalid_json", + fmt.Sprintf("invalid JSON: %v", err)) + return + } + + if req.Model == "" { + writeAPIError(w, http.StatusBadRequest, + "invalid_request_error", "model_required", + `"model" is required`) + return + } + + prefix, submodel := parseModel(req.Model) + agent, ok := agentForPrefix[prefix] + if !ok { + writeAPIError(w, http.StatusBadRequest, + "invalid_request_error", "unknown_model", + fmt.Sprintf("unknown model %q; valid prefixes: anthropic, claude, opencode", prefix)) + return + } + + prompt, err := buildPrompt(req.Instructions, req.Input) + if err != nil { + code := "input_required" + if !strings.Contains(err.Error(), "required") { + code = "invalid_input" + } + writeAPIError(w, http.StatusBadRequest, + "invalid_request_error", code, + err.Error()) + return + } + + // reasoning.effort is OpenAI's per-request thinking-budget hint + // (low|medium|high). Maps to claude's --effort flag. Silently drop + // unknown values rather than 400 — matches our permissive parsing + // of other not-fully-supported fields. + var effort string + if req.Reasoning != nil && validEffort(req.Reasoning.Effort) { + effort = req.Reasoning.Effort + } + + id := responseID() + + if logPrompts { + logger.Info("responses prompt", + "id", id, "model", req.Model, "agent", agent, + "effort", effort, "prompt", prompt, + "body", json.RawMessage(body)) + } + + opts := ExecOpts{ + Agent: agent, + Prompt: prompt, + Model: submodel, + Effort: effort, + SystemPrompt: systemPrompt, + } + + // Streaming path: claude has a token-level surface; opencode + // falls through to the buffered path even when stream=true. + if req.Stream && agent == "claude" { + var instructionsEcho *string + if req.Instructions != "" { + s := req.Instructions + instructionsEcho = &s + } + streamResponses(w, r, opts, id, req.Model, instructionsEcho) + return + } + + result := exec.Run(opts) + + status := "completed" + text := result.Stdout + var apiErr *ResponsesError + var output []ResponsesOutputItem + if result.ExitCode != 0 { + status = "failed" + msg := result.Stderr + if msg == "" { + msg = "agent failed" + } + apiErr = &ResponsesError{Code: "server_error", Message: msg} + text = "" + output = []ResponsesOutputItem{} + } else { + output = []ResponsesOutputItem{ + { + Type: "message", + ID: messageID(), + Status: "completed", + Role: "assistant", + Content: []ResponsesOutputContentPart{ + {Type: "output_text", Text: text, Annotations: []any{}}, + }, + }, + } + } + + var instructionsEcho *string + if req.Instructions != "" { + s := req.Instructions + instructionsEcho = &s + } + + temperature := 1.0 + if req.Temperature != nil { + temperature = *req.Temperature + } + topP := 1.0 + if req.TopP != nil { + topP = *req.TopP + } + store := true + if req.Store != nil { + store = *req.Store + } + parallel := true + if req.ParallelToolCalls != nil { + parallel = *req.ParallelToolCalls + } + truncation := "disabled" + if req.Truncation != "" { + truncation = req.Truncation + } + + if logPrompts { + logger.Info("responses response", + "id", id, "status", status, "output_text", text) + } + + resp := ResponsesObject{ + ID: id, + Object: "response", + CreatedAt: time.Now().Unix(), + Status: status, + Model: req.Model, + Output: output, + OutputText: text, + Usage: responsesUsageFromExec(result.Usage), + Error: apiErr, + IncompleteDetails: nil, + Instructions: instructionsEcho, + Metadata: map[string]string{}, + ParallelToolCalls: parallel, + PreviousResponseID: nil, + Reasoning: req.Reasoning, + Store: store, + Temperature: temperature, + ToolChoice: "auto", + Tools: []any{}, + TopP: topP, + Truncation: truncation, + User: req.User, + } + + w.Header().Set("Content-Type", "application/json") + json.NewEncoder(w).Encode(resp) + }) +} diff --git a/internal/serve/responses_compat_test.go b/internal/serve/responses_compat_test.go new file mode 100644 index 00000000..0193ad06 --- /dev/null +++ b/internal/serve/responses_compat_test.go @@ -0,0 +1,169 @@ +package serve_test + +import ( + "context" + "testing" + + "github.com/DimmKirr/devcell/internal/serve" + "github.com/openai/openai-go" + "github.com/openai/openai-go/option" + "github.com/openai/openai-go/responses" + "github.com/openai/openai-go/shared" +) + +// fakeExecResponses captures input for assertions and returns canned output. +type fakeExecResponses struct { + gotAgent string + gotPrompt string + gotModel string + gotEffort string + stdout string +} + +func (f *fakeExecResponses) Run(opts serve.ExecOpts) serve.ExecResult { + f.gotAgent = opts.Agent + f.gotPrompt = opts.Prompt + f.gotModel = opts.Model + f.gotEffort = opts.Effort + return serve.ExecResult{Stdout: f.stdout} +} + +// TestOpenAISDK_Responses_StringInput verifies the official openai-go SDK +// can call client.Responses.New against our /v1/responses endpoint. +func TestOpenAISDK_Responses_StringInput(t *testing.T) { + fe := &fakeExecResponses{stdout: "Hello, world!"} + srv := serve.NewServer(fe, 0) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + addr, _ := srv.Start(ctx) + if addr == "" { + t.Fatal("server failed to start") + } + + client := openai.NewClient( + option.WithBaseURL("http://"+addr+"/v1"), + option.WithAPIKey("test-key"), + ) + + resp, err := client.Responses.New(ctx, responses.ResponseNewParams{ + Model: shared.ResponsesModel("anthropic/sonnet"), + Input: responses.ResponseNewParamsInputUnion{ + OfString: openai.String("Say hello."), + }, + }) + if err != nil { + t.Fatalf("Responses.New failed: %v", err) + } + + if resp.Object != "response" { + t.Errorf("object = %q, want response", resp.Object) + } + if resp.Status != "completed" { + t.Errorf("status = %q, want completed", resp.Status) + } + if resp.OutputText() != "Hello, world!" { + t.Errorf("OutputText() = %q, want %q", resp.OutputText(), "Hello, world!") + } + if len(resp.Output) != 1 { + t.Fatalf("output len = %d, want 1", len(resp.Output)) + } + if resp.Output[0].Type != "message" { + t.Errorf("output[0].type = %q, want message", resp.Output[0].Type) + } +} + +// TestOpenAISDK_Responses_ModelRouting verifies SDK -> agent routing. +func TestOpenAISDK_Responses_ModelRouting(t *testing.T) { + fe := &fakeExecResponses{stdout: "ok"} + srv := serve.NewServer(fe, 0) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + addr, _ := srv.Start(ctx) + client := openai.NewClient( + option.WithBaseURL("http://"+addr+"/v1"), + option.WithAPIKey("test-key"), + ) + + _, err := client.Responses.New(ctx, responses.ResponseNewParams{ + Model: shared.ResponsesModel("anthropic/sonnet"), + Input: responses.ResponseNewParamsInputUnion{ + OfString: openai.String("hi"), + }, + }) + if err != nil { + t.Fatalf("Responses.New failed: %v", err) + } + if fe.gotAgent != "claude" { + t.Errorf("agent = %q, want claude", fe.gotAgent) + } + if fe.gotModel != "sonnet" { + t.Errorf("submodel = %q, want sonnet", fe.gotModel) + } +} + +// TestOpenAISDK_Responses_PreviousResponseIDIgnored verifies that sending +// previous_response_id (which devcell can't honor) doesn't error — the +// stateless server just ignores it. +func TestOpenAISDK_Responses_PreviousResponseIDIgnored(t *testing.T) { + fe := &fakeExecResponses{stdout: "fresh response"} + srv := serve.NewServer(fe, 0) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + addr, _ := srv.Start(ctx) + client := openai.NewClient( + option.WithBaseURL("http://"+addr+"/v1"), + option.WithAPIKey("test-key"), + ) + + resp, err := client.Responses.New(ctx, responses.ResponseNewParams{ + Model: shared.ResponsesModel("anthropic/sonnet"), + Input: responses.ResponseNewParamsInputUnion{ + OfString: openai.String("continue"), + }, + PreviousResponseID: openai.String("resp_does_not_exist"), + Temperature: openai.Float(0.7), + MaxOutputTokens: openai.Int(100), + }) + if err != nil { + t.Fatalf("Responses.New failed: %v", err) + } + if resp.OutputText() != "fresh response" { + t.Errorf("OutputText() = %q, want fresh response", resp.OutputText()) + } +} + +// TestOpenAISDK_Responses_Instructions verifies system prompt routing. +func TestOpenAISDK_Responses_Instructions(t *testing.T) { + fe := &fakeExecResponses{stdout: "brief"} + srv := serve.NewServer(fe, 0) + + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + addr, _ := srv.Start(ctx) + client := openai.NewClient( + option.WithBaseURL("http://"+addr+"/v1"), + option.WithAPIKey("test-key"), + ) + + _, err := client.Responses.New(ctx, responses.ResponseNewParams{ + Model: shared.ResponsesModel("anthropic/sonnet"), + Instructions: openai.String("be brief"), + Input: responses.ResponseNewParamsInputUnion{ + OfString: openai.String("hello"), + }, + }) + if err != nil { + t.Fatalf("Responses.New failed: %v", err) + } + wantPrefix := "[system]: be brief\n[user]: hello\n" + if fe.gotPrompt != wantPrefix { + t.Errorf("prompt = %q, want %q", fe.gotPrompt, wantPrefix) + } +} diff --git a/internal/serve/responses_test.go b/internal/serve/responses_test.go new file mode 100644 index 00000000..28f40008 --- /dev/null +++ b/internal/serve/responses_test.go @@ -0,0 +1,503 @@ +package serve + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" +) + +func postResponses(t *testing.T, handler http.Handler, body string) *httptest.ResponseRecorder { + t.Helper() + req := httptest.NewRequest(http.MethodPost, "/v1/responses", strings.NewReader(body)) + req.Header.Set("Content-Type", "application/json") + rec := httptest.NewRecorder() + handler.ServeHTTP(rec, req) + return rec +} + +func decodeResponse(t *testing.T, rec *httptest.ResponseRecorder) ResponsesObject { + t.Helper() + var r ResponsesObject + if err := json.NewDecoder(rec.Body).Decode(&r); err != nil { + t.Fatalf("decode response: %v\nbody: %s", err, rec.Body.String()) + } + return r +} + +func decodeAPIError(t *testing.T, rec *httptest.ResponseRecorder) APIError { + t.Helper() + var e APIError + if err := json.NewDecoder(rec.Body).Decode(&e); err != nil { + t.Fatalf("decode error: %v\nbody: %s", err, rec.Body.String()) + } + return e +} + +func TestResponses_StringInput(t *testing.T) { + fe := &fakeExec{stdout: "world"} + h := NewResponsesHandler(fe, false, "") + + rec := postResponses(t, h, `{"model":"anthropic/sonnet","input":"hello"}`) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + + r := decodeResponse(t, rec) + if r.Object != "response" { + t.Errorf("object = %q, want response", r.Object) + } + if r.Status != "completed" { + t.Errorf("status = %q, want completed", r.Status) + } + if r.OutputText != "world" { + t.Errorf("output_text = %q, want world", r.OutputText) + } + if len(r.Output) != 1 { + t.Fatalf("output len = %d, want 1", len(r.Output)) + } + out := r.Output[0] + if out.Type != "message" || out.Role != "assistant" || out.Status != "completed" { + t.Errorf("output[0] = %+v", out) + } + if len(out.Content) != 1 || out.Content[0].Type != "output_text" || out.Content[0].Text != "world" { + t.Errorf("output[0].content = %+v", out.Content) + } + if r.Error != nil { + t.Errorf("error = %+v, want nil", r.Error) + } + if !strings.HasPrefix(r.ID, "resp_") { + t.Errorf("id = %q, want resp_ prefix", r.ID) + } + if !strings.HasPrefix(out.ID, "msg_") { + t.Errorf("output[0].id = %q, want msg_ prefix", out.ID) + } + + // Routing: "anthropic/sonnet" → agent "claude", submodel "sonnet" + if fe.agent != "claude" || fe.model != "sonnet" { + t.Errorf("routing: agent=%q model=%q, want claude/sonnet", fe.agent, fe.model) + } + if !strings.Contains(fe.prompt, "[user]: hello") { + t.Errorf("prompt = %q, want [user]: hello", fe.prompt) + } +} + +func TestResponses_ArrayInput(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + + body := `{ + "model": "anthropic/sonnet", + "input": [ + {"role":"user","content":"hi"}, + {"role":"assistant","content":"hello"}, + {"role":"user","content":"what did I say?"} + ] + }` + rec := postResponses(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + + want := "[user]: hi\n[assistant]: hello\n[user]: what did I say?\n" + if fe.prompt != want { + t.Errorf("prompt =\n%q\nwant\n%q", fe.prompt, want) + } +} + +func TestResponses_ContentParts(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + + body := `{ + "model": "anthropic/sonnet", + "input": [ + {"role":"user","content":[{"type":"input_text","text":"part one"},{"type":"input_text","text":"part two"}]} + ] + }` + rec := postResponses(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + if !strings.Contains(fe.prompt, "[user]: part one part two") { + t.Errorf("prompt = %q", fe.prompt) + } +} + +func TestResponses_Instructions(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + + body := `{"model":"anthropic/sonnet","instructions":"be brief","input":"hi"}` + rec := postResponses(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + + want := "[system]: be brief\n[user]: hi\n" + if fe.prompt != want { + t.Errorf("prompt = %q, want %q", fe.prompt, want) + } + + // Echoed in response. + r := decodeResponse(t, rec) + if r.Instructions == nil || *r.Instructions != "be brief" { + t.Errorf("instructions echo = %v, want \"be brief\"", r.Instructions) + } +} + +func TestResponses_SystemRoleInArray(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + + // system role inside input[] flattens to [system]: line. + body := `{ + "model": "anthropic/sonnet", + "input": [ + {"role":"system","content":"you are helpful"}, + {"role":"user","content":"hi"} + ] + }` + rec := postResponses(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + want := "[system]: you are helpful\n[user]: hi\n" + if fe.prompt != want { + t.Errorf("prompt = %q, want %q", fe.prompt, want) + } +} + +func TestResponses_DeveloperRoleAliasesToSystem(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + + body := `{ + "model": "anthropic/sonnet", + "input": [ + {"role":"developer","content":"dev instructions"}, + {"role":"user","content":"hi"} + ] + }` + rec := postResponses(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + if !strings.Contains(fe.prompt, "[system]: dev instructions") { + t.Errorf("prompt = %q", fe.prompt) + } +} + +// TestResponses_StreamFallsBackForOpencode covers the routing rule: +// stream:true is honored only for the claude agent (which has a token- +// level CLI surface). Opencode has no equivalent, so a stream:true +// opencode request falls back to the buffered Run path. The dedicated +// claude streaming path is exercised by the streamClaude tests, which +// avoid spawning the real binary in this unit test file. +func TestResponses_StreamFallsBackForOpencode(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + + rec := postResponses(t, h, `{"model":"opencode","input":"hi","stream":true}`) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200 (buffered fallback), got %d body=%s", rec.Code, rec.Body.String()) + } + if ct := rec.Header().Get("Content-Type"); !strings.HasPrefix(ct, "application/json") { + t.Errorf("content-type = %q, want application/json (buffered, not SSE)", ct) + } + if !fe.called { + t.Error("executor should have been called via buffered path") + } +} + +func TestResponses_BadJSON(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + + rec := postResponses(t, h, `{not json}`) + if rec.Code != http.StatusBadRequest { + t.Fatalf("expected 400, got %d", rec.Code) + } + e := decodeAPIError(t, rec) + if e.Error.Code != "invalid_json" { + t.Errorf("code = %q, want invalid_json", e.Error.Code) + } +} + +func TestResponses_EmptyModel(t *testing.T) { + h := NewResponsesHandler(&fakeExec{}, false, "") + rec := postResponses(t, h, `{"input":"hi"}`) + if rec.Code != http.StatusBadRequest { + t.Fatalf("expected 400, got %d", rec.Code) + } + e := decodeAPIError(t, rec) + if e.Error.Code != "model_required" { + t.Errorf("code = %q, want model_required", e.Error.Code) + } +} + +func TestResponses_UnknownModel(t *testing.T) { + h := NewResponsesHandler(&fakeExec{}, false, "") + rec := postResponses(t, h, `{"model":"gpt-4","input":"hi"}`) + if rec.Code != http.StatusBadRequest { + t.Fatalf("expected 400, got %d", rec.Code) + } + e := decodeAPIError(t, rec) + if e.Error.Code != "unknown_model" { + t.Errorf("code = %q, want unknown_model", e.Error.Code) + } +} + +func TestResponses_EmptyInputString(t *testing.T) { + h := NewResponsesHandler(&fakeExec{}, false, "") + rec := postResponses(t, h, `{"model":"anthropic/sonnet","input":""}`) + if rec.Code != http.StatusBadRequest { + t.Fatalf("expected 400, got %d", rec.Code) + } + e := decodeAPIError(t, rec) + if e.Error.Code != "input_required" { + t.Errorf("code = %q, want input_required", e.Error.Code) + } +} + +func TestResponses_EmptyInputArray(t *testing.T) { + h := NewResponsesHandler(&fakeExec{}, false, "") + rec := postResponses(t, h, `{"model":"anthropic/sonnet","input":[]}`) + if rec.Code != http.StatusBadRequest { + t.Fatalf("expected 400, got %d", rec.Code) + } + e := decodeAPIError(t, rec) + if e.Error.Code != "input_required" { + t.Errorf("code = %q, want input_required", e.Error.Code) + } +} + +func TestResponses_MissingInput(t *testing.T) { + h := NewResponsesHandler(&fakeExec{}, false, "") + rec := postResponses(t, h, `{"model":"anthropic/sonnet"}`) + if rec.Code != http.StatusBadRequest { + t.Fatalf("expected 400, got %d", rec.Code) + } + e := decodeAPIError(t, rec) + if e.Error.Code != "input_required" { + t.Errorf("code = %q, want input_required", e.Error.Code) + } +} + +func TestResponses_NonPOST(t *testing.T) { + h := NewResponsesHandler(&fakeExec{}, false, "") + req := httptest.NewRequest(http.MethodGet, "/v1/responses", nil) + rec := httptest.NewRecorder() + h.ServeHTTP(rec, req) + if rec.Code != http.StatusMethodNotAllowed { + t.Errorf("expected 405, got %d", rec.Code) + } + e := decodeAPIError(t, rec) + if e.Error.Code != "method_not_allowed" { + t.Errorf("code = %q, want method_not_allowed", e.Error.Code) + } +} + +func TestResponses_ExitCodeFailure(t *testing.T) { + fe := &fakeExec{stderr: "boom", exitCode: 1} + h := NewResponsesHandler(fe, false, "") + + rec := postResponses(t, h, `{"model":"anthropic/sonnet","input":"hi"}`) + // Failure is a 200 with status: "failed" and error populated — matches OpenAI. + if rec.Code != http.StatusOK { + t.Fatalf("expected 200 (failure mode), got %d", rec.Code) + } + r := decodeResponse(t, rec) + if r.Status != "failed" { + t.Errorf("status = %q, want failed", r.Status) + } + if r.Error == nil || r.Error.Message != "boom" { + t.Errorf("error = %+v, want {message: boom}", r.Error) + } + if r.OutputText != "" { + t.Errorf("output_text = %q, want empty on failure", r.OutputText) + } + if len(r.Output) != 0 { + t.Errorf("output len = %d, want 0 on failure", len(r.Output)) + } +} + +func TestResponses_IgnoredFieldsTolerated(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + + // All these fields should decode cleanly and have no effect. + body := `{ + "model": "anthropic/sonnet", + "input": "hi", + "previous_response_id": "resp_old123", + "tools": [{"type":"function","function":{"name":"foo"}}], + "tool_choice": "auto", + "response_format": {"type":"json_object"}, + "reasoning": {"effort":"high"}, + "temperature": 0.5, + "top_p": 0.9, + "max_output_tokens": 100, + "metadata": {"a":"b"}, + "store": false, + "parallel_tool_calls": false, + "truncation": "auto", + "user": "u1", + "include": ["message.output_text.logprobs"] + }` + rec := postResponses(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + r := decodeResponse(t, rec) + if r.Status != "completed" { + t.Errorf("status = %q, want completed", r.Status) + } + // Echoes + if r.Temperature != 0.5 { + t.Errorf("temperature echo = %v, want 0.5", r.Temperature) + } + if r.TopP != 0.9 { + t.Errorf("top_p echo = %v, want 0.9", r.TopP) + } + if r.Store { + t.Errorf("store echo = true, want false") + } + if r.ParallelToolCalls { + t.Errorf("parallel_tool_calls echo = true, want false") + } + if r.Truncation != "auto" { + t.Errorf("truncation echo = %q, want auto", r.Truncation) + } + if r.User != "u1" { + t.Errorf("user echo = %q, want u1", r.User) + } +} + +// --- reasoning.effort → claude --effort mapping --- +// +// OpenAI documents `low`, `medium`, `high` as the valid values for +// `reasoning.effort`. Claude CLI accepts those plus `xhigh` and `max`. +// We deliberately accept only the OpenAI-spec values and silently drop +// non-spec values (including Claude's extensions) — same permissive +// pattern we use for other partially-supported fields. + +func TestResponses_Effort_OpenAISpecValuesPassThrough(t *testing.T) { + for _, v := range []string{"low", "medium", "high"} { + t.Run(v, func(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + body := `{"model":"anthropic/sonnet","input":"hi","reasoning":{"effort":"` + v + `"}}` + rec := postResponses(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + if fe.effort != v { + t.Errorf("effort reaching executor = %q, want %q", fe.effort, v) + } + }) + } +} + +func TestResponses_Effort_ClaudeOnlyValuesDropped(t *testing.T) { + // "xhigh" and "max" are Claude CLI extensions, NOT in the OpenAI spec. + // Clients sending them are misusing the OpenAI surface — we drop the + // value rather than silently passing a non-spec string to the CLI. + for _, v := range []string{"xhigh", "max"} { + t.Run(v, func(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + body := `{"model":"anthropic/sonnet","input":"hi","reasoning":{"effort":"` + v + `"}}` + rec := postResponses(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + if fe.effort != "" { + t.Errorf("non-OpenAI-spec effort %q leaked to executor (got %q), should be dropped", + v, fe.effort) + } + }) + } +} + +func TestResponses_Effort_UnknownValuesDropped(t *testing.T) { + for _, v := range []string{"extreme", "minimal", "LOW", "High", "auto", "none"} { + t.Run(v, func(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + body := `{"model":"anthropic/sonnet","input":"hi","reasoning":{"effort":"` + v + `"}}` + rec := postResponses(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d", rec.Code) + } + if fe.effort != "" { + t.Errorf("unknown effort %q leaked to executor (got %q)", v, fe.effort) + } + }) + } +} + +func TestResponses_Effort_AbsentNoFlag(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + rec := postResponses(t, h, `{"model":"anthropic/sonnet","input":"hi"}`) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d", rec.Code) + } + if fe.effort != "" { + t.Errorf("absent reasoning.effort still produced executor effort=%q", fe.effort) + } +} + +func TestResponses_Effort_OtherReasoningFieldsIgnored(t *testing.T) { + // `summary` and `generate_summary` decode but have no effect. + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + body := `{ + "model":"anthropic/sonnet","input":"hi", + "reasoning":{"effort":"medium","summary":"detailed","generate_summary":"auto"} + }` + rec := postResponses(t, h, body) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d: %s", rec.Code, rec.Body.String()) + } + if fe.effort != "medium" { + t.Errorf("effort = %q, want medium", fe.effort) + } +} + +func TestResponses_Effort_EchoedInResponse(t *testing.T) { + // The reasoning object should be echoed back verbatim (with whatever + // fields the client sent) so SDK round-trips don't drop information. + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + rec := postResponses(t, h, + `{"model":"anthropic/sonnet","input":"hi","reasoning":{"effort":"high","summary":"auto"}}`) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d", rec.Code) + } + r := decodeResponse(t, rec) + if r.Reasoning == nil { + t.Fatalf("expected reasoning to be echoed, got nil") + } + if r.Reasoning.Effort != "high" { + t.Errorf("echo effort = %q, want high", r.Reasoning.Effort) + } + if r.Reasoning.Summary != "auto" { + t.Errorf("echo summary = %q, want auto", r.Reasoning.Summary) + } +} + +func TestResponses_OpenCodeRouting(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + h := NewResponsesHandler(fe, false, "") + + rec := postResponses(t, h, `{"model":"opencode","input":"hi"}`) + if rec.Code != http.StatusOK { + t.Fatalf("expected 200, got %d", rec.Code) + } + if fe.agent != "opencode" || fe.model != "" { + t.Errorf("routing: agent=%q model=%q, want opencode/(empty)", fe.agent, fe.model) + } +} diff --git a/internal/serve/server.go b/internal/serve/server.go index e6e34fa7..fb618640 100644 --- a/internal/serve/server.go +++ b/internal/serve/server.go @@ -8,6 +8,10 @@ import ( "net/http" "os/exec" "time" + + "github.com/DimmKirr/devcell/internal/version" + "github.com/swaggo/swag" + httpSwagger "github.com/swaggo/http-swagger/v2" ) // DefaultPort is the default listen port for devcell serve. @@ -15,11 +19,13 @@ const DefaultPort = 8484 // Server is the devcell HTTP API server. type Server struct { - exec Executor - port int - lookPath LookPathFunc - anthropic AnthropicClient - apiKey string // empty = no auth + exec Executor + port int + lookPath LookPathFunc + anthropic AnthropicClient + apiKey string // empty = no auth + logPrompts bool // when true, handlers log full prompt + response bodies + systemPrompt string // when non-empty, passed as --append-system-prompt to claude } // NewServer creates a Server. Use port=0 to let the OS pick a free port. @@ -53,6 +59,24 @@ func (s *Server) APIKey() string { return s.apiKey } +// SetLogPrompts enables or disables full prompt + response body logging. +// +// When true, /v1/chat/completions and /v1/responses handlers log the +// assembled prompt and the model's reply at INFO level. Off by default — +// prompts often contain secrets, PII, or large pasted content from +// upstream tools (n8n flows, agents, etc.), so this is opt-in. +func (s *Server) SetLogPrompts(v bool) { + s.logPrompts = v +} + +// SetSystemPrompt sets the operator-level system prompt passed to claude +// as --append-system-prompt on every /v1/chat/completions and /v1/responses +// request. Empty disables the flag (default). Composes with — does not +// override — any per-request `instructions` / `system` role from the body. +func (s *Server) SetSystemPrompt(p string) { + s.systemPrompt = p +} + func execLookPath(name string) (string, error) { return exec.LookPath(name) } @@ -61,16 +85,19 @@ func execLookPath(name string) (string, error) { // The server shuts down when ctx is cancelled. func (s *Server) Start(ctx context.Context) (addr string, errCh chan error) { mux := http.NewServeMux() - mux.Handle("/v1/chat/completions", AuthMiddleware(s.apiKey, NewChatHandler(s.exec))) + mux.Handle("/v1/chat/completions", AuthMiddleware(s.apiKey, NewChatHandler(s.exec, s.logPrompts, s.systemPrompt))) + mux.Handle("/v1/responses", AuthMiddleware(s.apiKey, NewResponsesHandler(s.exec, s.logPrompts, s.systemPrompt))) mux.Handle("/v1/models", AuthMiddleware(s.apiKey, NewModelsHandler(s.lookPath, s.anthropic))) - mux.HandleFunc("/api/v1/health", func(w http.ResponseWriter, r *http.Request) { - if r.Method != http.MethodGet { - http.Error(w, "method not allowed", http.StatusMethodNotAllowed) - return - } + mux.HandleFunc("/healthz", healthHandler) + mux.HandleFunc("/api/v1/health", healthHandler) + mux.HandleFunc("/api/openapi.json", func(w http.ResponseWriter, r *http.Request) { + doc, _ := swag.ReadDoc() w.Header().Set("Content-Type", "application/json") - json.NewEncoder(w).Encode(map[string]string{"status": "ok"}) + w.Write([]byte(doc)) }) + mux.Handle("/swagger/", httpSwagger.Handler( + httpSwagger.URL("/api/openapi.json"), + )) ln, err := net.Listen("tcp", fmt.Sprintf(":%d", s.port)) if err != nil { @@ -79,7 +106,7 @@ func (s *Server) Start(ctx context.Context) (addr string, errCh chan error) { return "", errCh } - srv := &http.Server{Handler: mux} + srv := &http.Server{Handler: LoggingMiddleware(mux)} errCh = make(chan error, 1) go func() { @@ -95,3 +122,59 @@ func (s *Server) Start(ctx context.Context) (addr string, errCh chan error) { return ln.Addr().String(), errCh } + +// HealthResponse is the health check response body. +// +// Version fields are injected at build time via -ldflags by the +// `task cell:build` / `task swag:generate` flow. An unbuilt-via-task binary +// will report defaults: version=v0.0.0, commit=none, build_date=unknown. +type HealthResponse struct { + // Server status — "ok" when the server is running and ready to accept requests. + Status string `json:"status" example:"ok"` + // Composite version string matching `cell --version` output. + // Format: `--`. + Version string `json:"version" example:"v0.1.0-2026-04-26-abc1234"` + // Semantic version tag from the build (e.g. "v0.1.0"). + VersionTag string `json:"version_tag" example:"v0.1.0"` + // Git commit hash this binary was built from. + Commit string `json:"commit" example:"abc1234"` + // Build date (UTC). + BuildDate string `json:"build_date" example:"2026-04-26"` +} + +// healthHandler handles GET /healthz and GET /api/v1/health. +// +// @Summary Health check +// @Description Returns server health status and the build version of the devcell binary +// @Description currently serving the request. No authentication required. +// @Description +// @Description The `version` field matches the output of `cell --version` exactly +// @Description (`--`). The individual `version_tag`, `commit`, +// @Description and `build_date` fields are also exposed for tooling that wants to parse them +// @Description without splitting the composite string. All values are injected at build time +// @Description via Go ldflags; a binary built outside the task pipeline reports defaults +// @Description (`v0.0.0`, `none`, `unknown`). +// @Description +// @Description Available at two paths: +// @Description - `/healthz` — Kubernetes convention for liveness/readiness probes and load balancers +// @Description - `/api/v1/health` — REST API convention for application-level client health checks +// @Tags health +// @Produce json +// @Success 200 {object} HealthResponse "Server is healthy" +// @Failure 405 {string} string "Only GET is allowed" +// @Router /healthz [get] +// @Router /api/v1/health [get] +func healthHandler(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodGet { + http.Error(w, "method not allowed", http.StatusMethodNotAllowed) + return + } + w.Header().Set("Content-Type", "application/json") + json.NewEncoder(w).Encode(HealthResponse{ + Status: "ok", + Version: version.Full(), + VersionTag: version.Version, + Commit: version.GitCommit, + BuildDate: version.BuildDate, + }) +} diff --git a/internal/serve/server_test.go b/internal/serve/server_test.go index 7912460c..7df75b12 100644 --- a/internal/serve/server_test.go +++ b/internal/serve/server_test.go @@ -7,6 +7,8 @@ import ( "net/http" "testing" "time" + + "github.com/DimmKirr/devcell/internal/version" ) func TestServer_ListensOnConfiguredPort(t *testing.T) { @@ -92,10 +94,111 @@ func TestHealth_Returns200(t *testing.T) { if resp.StatusCode != http.StatusOK { t.Errorf("status = %d, want 200", resp.StatusCode) } - var body map[string]string - json.NewDecoder(resp.Body).Decode(&body) - if body["status"] != "ok" { - t.Errorf("status = %q, want %q", body["status"], "ok") + var body HealthResponse + if err := json.NewDecoder(resp.Body).Decode(&body); err != nil { + t.Fatalf("decode: %v", err) + } + if body.Status != "ok" { + t.Errorf("status = %q, want %q", body.Status, "ok") + } + // Version fields must be populated even without ldflags injection + // (defaults: v0.0.0, none, unknown). They're informational, never empty. + if body.VersionTag == "" { + t.Error("version_tag should never be empty (default v0.0.0)") + } + if body.Commit == "" { + t.Error("commit should never be empty (default 'none')") + } + if body.BuildDate == "" { + t.Error("build_date should never be empty (default 'unknown')") + } + if body.Version == "" { + t.Error("version composite should never be empty") + } +} + +// TestHealth_VersionMatchesLdflagsInjection proves the response reflects +// values from internal/version (which the build pipeline overrides via +// -ldflags). We mutate the package vars directly here to simulate that +// injection without rebuilding. +// TestServer_LogPromptsToggle verifies the server stores the LogPrompts +// flag and threads it into the handlers. We don't assert on log output +// directly (the logger captures stderr at init time and isn't trivially +// redirectable mid-test) — instead we verify the wiring contract. +func TestServer_LogPromptsToggle(t *testing.T) { + srv := NewServer(&fakeExec{}, 0) + if srv.logPrompts { + t.Error("logPrompts should default to false") + } + srv.SetLogPrompts(true) + if !srv.logPrompts { + t.Error("SetLogPrompts(true) did not flip the field") + } + srv.SetLogPrompts(false) + if srv.logPrompts { + t.Error("SetLogPrompts(false) did not flip the field") + } +} + +// TestServer_SystemPromptThreadedToExec proves SetSystemPrompt on the Server +// reaches ExecOpts.SystemPrompt on every chat-completions request — the +// contract that lets `cell serve --system-prompt` actually take effect. +func TestServer_SystemPromptThreadedToExec(t *testing.T) { + fe := &fakeExec{stdout: "ok"} + srv := NewServer(fe, 0) + srv.SetSystemPrompt("you are a backend assistant") + + h := NewChatHandler(fe, false, srv.systemPrompt) + rec := postChat(t, h, `{"model":"claude","messages":[{"role":"user","content":"hi"}]}`) + if rec.Code != http.StatusOK { + t.Fatalf("status = %d, body=%s", rec.Code, rec.Body.String()) + } + if fe.systemPrompt != "you are a backend assistant" { + t.Errorf("ExecOpts.SystemPrompt = %q, want operator-level value", fe.systemPrompt) + } +} + +func TestHealth_VersionMatchesLdflagsInjection(t *testing.T) { + // Patch package-level version vars (same vars that `-X .../version.GitCommit=...` + // writes at link time) and restore on cleanup. + origV, origC, origD := version.Version, version.GitCommit, version.BuildDate + t.Cleanup(func() { + version.Version = origV + version.GitCommit = origC + version.BuildDate = origD + }) + version.Version = "v9.9.9" + version.GitCommit = "deadbeef" + version.BuildDate = "2026-04-26" + + fe := &fakeExec{} + srv := NewServer(fe, 0) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + addr, _ := srv.Start(ctx) + + resp, err := http.Get("http://" + addr + "/healthz") + if err != nil { + t.Fatalf("GET /healthz: %v", err) + } + defer resp.Body.Close() + + var body HealthResponse + if err := json.NewDecoder(resp.Body).Decode(&body); err != nil { + t.Fatalf("decode: %v", err) + } + if body.VersionTag != "v9.9.9" { + t.Errorf("version_tag = %q, want v9.9.9", body.VersionTag) + } + if body.Commit != "deadbeef" { + t.Errorf("commit = %q, want deadbeef", body.Commit) + } + if body.BuildDate != "2026-04-26" { + t.Errorf("build_date = %q, want 2026-04-26", body.BuildDate) + } + want := "v9.9.9-2026-04-26-deadbeef" + if body.Version != want { + t.Errorf("version = %q, want %q", body.Version, want) } } diff --git a/internal/serve/sse_chat.go b/internal/serve/sse_chat.go new file mode 100644 index 00000000..f2bd8844 --- /dev/null +++ b/internal/serve/sse_chat.go @@ -0,0 +1,238 @@ +package serve + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "time" +) + +// ChatStreamChoice is one element of a chat.completion.chunk's choices +// array. The `delta` carries the incremental content; `finish_reason` is +// null on every chunk except the last. +type ChatStreamChoice struct { + Index int `json:"index"` + Delta ChatStreamDelta `json:"delta"` + FinishReason *string `json:"finish_reason"` +} + +// ChatStreamDelta is the OpenAI per-chunk delta. Role is sent on the +// first chunk; content on each text-delta chunk; both empty on the +// terminal chunk. +type ChatStreamDelta struct { + Role string `json:"role,omitempty"` + Content string `json:"content,omitempty"` +} + +// ChatStreamChunk is the JSON payload of one SSE `data:` frame. +type ChatStreamChunk struct { + ID string `json:"id"` + Object string `json:"object"` // always "chat.completion.chunk" + Created int64 `json:"created"` + Model string `json:"model"` + Choices []ChatStreamChoice `json:"choices"` + Usage *ChatUsage `json:"usage,omitempty"` +} + +// streamChatCompletions runs claude in stream mode and writes Chat +// Completions SSE to w. Closes when claude finishes or the HTTP client +// disconnects (r.Context() cancellation propagates to claude via +// streamClaude/exec.CommandContext). +// +// includeUsage corresponds to OpenAI's stream_options.include_usage — +// when true the final chunk carries a usage object; when false (default) +// it doesn't. +func streamChatCompletions( + w http.ResponseWriter, + r *http.Request, + opts ExecOpts, + chatID, model string, + includeUsage bool, +) { + flusher, ok := w.(http.Flusher) + if !ok { + http.Error(w, "streaming requires an http.Flusher", http.StatusInternalServerError) + return + } + + w.Header().Set("Content-Type", "text/event-stream") + w.Header().Set("Cache-Control", "no-cache") + w.Header().Set("Connection", "keep-alive") + w.Header().Set("X-Accel-Buffering", "no") // disable nginx response buffering + + ctx, cancel := context.WithCancel(r.Context()) + defer cancel() + + events, err := streamClaude(ctx, opts) + if err != nil { + writeChatErrorChunk(w, flusher, chatID, model, err) + return + } + pumpChatSSE(ctx, w, flusher, events, chatID, model, includeUsage, 15*time.Second) +} + +// pumpChatSSE consumes a stream of canonical events and writes Chat +// Completions SSE frames. Pure formatter — used by streamChatCompletions +// in production and exercised directly by tests with a synthetic channel. +// +// heartbeatInterval = 0 disables heartbeats (tests use this so they +// don't have to manage timers). +func pumpChatSSE( + ctx context.Context, + w io.Writer, + flusher http.Flusher, + events <-chan StreamEvent, + chatID, model string, + includeUsage bool, + heartbeatInterval time.Duration, +) { + created := time.Now().Unix() + roleSent := false + + var heartbeatC <-chan time.Time + if heartbeatInterval > 0 { + t := time.NewTicker(heartbeatInterval) + defer t.Stop() + heartbeatC = t.C + } + + for { + select { + case <-ctx.Done(): + return + + case <-heartbeatC: + fmt.Fprint(w, ":keepalive\n\n") + flusher.Flush() + + case ev, open := <-events: + if !open { + // Channel closed without a Result — emit a terminal + // chunk so the client doesn't hang. + writeChatTerminalChunk(w, flusher, chatID, model, created, "stop", nil, includeUsage) + return + } + + switch ev.Kind { + case StreamEventMessageStart: + if !roleSent { + writeChatChunk(w, flusher, ChatStreamChunk{ + ID: chatID, + Object: "chat.completion.chunk", + Created: created, + Model: model, + Choices: []ChatStreamChoice{{ + Index: 0, + Delta: ChatStreamDelta{Role: "assistant"}, + }}, + }) + roleSent = true + } + + case StreamEventTextDelta: + if !roleSent { + // Defensive: some streams skip message_start. Emit + // the role chunk on the first delta we see. + writeChatChunk(w, flusher, ChatStreamChunk{ + ID: chatID, + Object: "chat.completion.chunk", + Created: created, + Model: model, + Choices: []ChatStreamChoice{{ + Index: 0, + Delta: ChatStreamDelta{Role: "assistant"}, + }}, + }) + roleSent = true + } + writeChatChunk(w, flusher, ChatStreamChunk{ + ID: chatID, + Object: "chat.completion.chunk", + Created: created, + Model: model, + Choices: []ChatStreamChoice{{ + Index: 0, + Delta: ChatStreamDelta{Content: ev.Delta}, + }}, + }) + + case StreamEventResult: + finish := "stop" + if ev.Final != nil && ev.Final.IsError { + finish = "error" + } + var usage *ChatUsage + if includeUsage && ev.Final != nil { + u := chatUsageFromExec(Usage{ + InputTokens: ev.Final.Usage.InputTokens, + OutputTokens: ev.Final.Usage.OutputTokens, + CacheCreationInputTokens: ev.Final.Usage.CacheCreationInputTokens, + CacheReadInputTokens: ev.Final.Usage.CacheReadInputTokens, + }) + usage = &u + } + writeChatTerminalChunk(w, flusher, chatID, model, created, finish, usage, includeUsage) + return + + case StreamEventError: + writeChatErrorChunk(w, flusher, chatID, model, ev.Err) + return + + default: + // MessageStop and friends — nothing to emit; Result + // is the OpenAI terminator. + } + } + } +} + +func writeChatChunk(w io.Writer, flusher http.Flusher, c ChatStreamChunk) { + b, _ := json.Marshal(c) + fmt.Fprintf(w, "data: %s\n\n", b) + flusher.Flush() +} + +func writeChatTerminalChunk( + w io.Writer, flusher http.Flusher, + chatID, model string, created int64, + finish string, usage *ChatUsage, includeUsage bool, +) { + finishPtr := finish + chunk := ChatStreamChunk{ + ID: chatID, + Object: "chat.completion.chunk", + Created: created, + Model: model, + Choices: []ChatStreamChoice{{ + Index: 0, + Delta: ChatStreamDelta{}, + FinishReason: &finishPtr, + }}, + } + if includeUsage { + chunk.Usage = usage + } + writeChatChunk(w, flusher, chunk) + fmt.Fprint(w, "data: [DONE]\n\n") + flusher.Flush() +} + +func writeChatErrorChunk(w io.Writer, flusher http.Flusher, chatID, model string, err error) { + finish := "error" + c := ChatStreamChunk{ + ID: chatID, + Object: "chat.completion.chunk", + Created: time.Now().Unix(), + Model: model, + Choices: []ChatStreamChoice{{ + Index: 0, + Delta: ChatStreamDelta{Content: err.Error()}, + FinishReason: &finish, + }}, + } + writeChatChunk(w, flusher, c) + fmt.Fprint(w, "data: [DONE]\n\n") + flusher.Flush() +} diff --git a/internal/serve/sse_chat_test.go b/internal/serve/sse_chat_test.go new file mode 100644 index 00000000..081b9ab9 --- /dev/null +++ b/internal/serve/sse_chat_test.go @@ -0,0 +1,129 @@ +package serve + +import ( + "context" + "net/http/httptest" + "strings" + "testing" +) + +// pumpChatSSEHelper feeds canonical events through pumpChatSSE and returns +// the rendered SSE body as a string. Heartbeats disabled so tests are +// deterministic. +func pumpChatSSEHelper(t *testing.T, events []StreamEvent, includeUsage bool) string { + t.Helper() + rec := httptest.NewRecorder() + ch := make(chan StreamEvent, len(events)) + for _, ev := range events { + ch <- ev + } + close(ch) + pumpChatSSE(context.Background(), rec, rec, ch, "chatcmpl-test", "anthropic/sonnet", includeUsage, 0) + return rec.Body.String() +} + +func TestPumpChatSSE_HappyPath(t *testing.T) { + events := []StreamEvent{ + {Kind: StreamEventMessageStart, MessageID: "msg_1", Model: "claude-opus-4-7"}, + {Kind: StreamEventTextDelta, Delta: "Hi"}, + {Kind: StreamEventTextDelta, Delta: ", "}, + {Kind: StreamEventTextDelta, Delta: "world"}, + {Kind: StreamEventMessageStop, StopReason: "end_turn"}, + {Kind: StreamEventResult, Final: &claudeJSONResult{ + Type: "result", Subtype: "success", Result: "Hi, world", + Usage: claudeJSONUsageObj{InputTokens: 5, OutputTokens: 3, CacheCreationInputTokens: 100}, + }}, + } + body := pumpChatSSEHelper(t, events, false) + + // Frame ordering check. + checks := []string{ + `"delta":{"role":"assistant"}`, + `"delta":{"content":"Hi"}`, + `"delta":{"content":", "}`, + `"delta":{"content":"world"}`, + `"finish_reason":"stop"`, + "data: [DONE]", + } + prev := -1 + for _, c := range checks { + idx := strings.Index(body, c) + if idx < 0 { + t.Errorf("missing frame fragment %q\n\nfull body:\n%s", c, body) + continue + } + if idx <= prev { + t.Errorf("fragment %q appeared out of order (idx=%d, prev=%d)\n\nfull body:\n%s", c, idx, prev, body) + } + prev = idx + } + // Without include_usage, the final chunk must NOT carry usage. + if strings.Contains(body, `"usage":`) { + t.Errorf("usage leaked when stream_options.include_usage=false:\n%s", body) + } +} + +func TestPumpChatSSE_IncludeUsage(t *testing.T) { + events := []StreamEvent{ + {Kind: StreamEventMessageStart, MessageID: "msg_1", Model: "x"}, + {Kind: StreamEventTextDelta, Delta: "ok"}, + {Kind: StreamEventResult, Final: &claudeJSONResult{ + Type: "result", Result: "ok", + Usage: claudeJSONUsageObj{InputTokens: 5, OutputTokens: 2, CacheReadInputTokens: 100}, + }}, + } + body := pumpChatSSEHelper(t, events, true) + + if !strings.Contains(body, `"prompt_tokens":105`) { + t.Errorf("prompt_tokens should be 5+0+100=105 (input+cache_creation+cache_read), got body:\n%s", body) + } + if !strings.Contains(body, `"completion_tokens":2`) { + t.Errorf("completion_tokens should be 2:\n%s", body) + } + if !strings.Contains(body, `"total_tokens":107`) { + t.Errorf("total_tokens should be 107:\n%s", body) + } +} + +func TestPumpChatSSE_DeltaSentBeforeRoleFallsBackGracefully(t *testing.T) { + // Defensive path: streams that skip message_start should still get + // a synthetic role chunk before the first content delta. + events := []StreamEvent{ + {Kind: StreamEventTextDelta, Delta: "x"}, + {Kind: StreamEventResult, Final: &claudeJSONResult{Type: "result"}}, + } + body := pumpChatSSEHelper(t, events, false) + + rolePos := strings.Index(body, `"role":"assistant"`) + contentPos := strings.Index(body, `"content":"x"`) + if rolePos < 0 || contentPos < 0 || rolePos > contentPos { + t.Errorf("role chunk must precede first content delta\n%s", body) + } +} + +func TestPumpChatSSE_ChannelClosedWithoutResultStillTerminates(t *testing.T) { + // Claude crash mid-stream — channel closes with no Result. Client + // must still see a terminal chunk + [DONE] so it doesn't hang. + events := []StreamEvent{ + {Kind: StreamEventMessageStart, MessageID: "x"}, + {Kind: StreamEventTextDelta, Delta: "partial"}, + } + body := pumpChatSSEHelper(t, events, false) + if !strings.Contains(body, "data: [DONE]") { + t.Errorf("missing terminator on premature channel close:\n%s", body) + } + if !strings.Contains(body, `"finish_reason":"stop"`) { + t.Errorf("missing finish_reason on premature close:\n%s", body) + } +} + +func TestPumpChatSSE_ResultWithIsErrorSendsFinishReasonError(t *testing.T) { + events := []StreamEvent{ + {Kind: StreamEventMessageStart}, + {Kind: StreamEventResult, Final: &claudeJSONResult{Type: "result", IsError: true, Subtype: "error_during_execution"}}, + } + body := pumpChatSSEHelper(t, events, false) + if !strings.Contains(body, `"finish_reason":"error"`) { + t.Errorf("missing finish_reason=error:\n%s", body) + } +} diff --git a/internal/serve/sse_responses.go b/internal/serve/sse_responses.go new file mode 100644 index 00000000..6da1a94f --- /dev/null +++ b/internal/serve/sse_responses.go @@ -0,0 +1,314 @@ +package serve + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "time" +) + +// Responses SSE shape (per OpenAI Responses API): +// +// event: +// data: +// +// Events emitted by devcell, in order, on a successful turn: +// +// response.created — initial response object +// response.in_progress — status update +// response.output_item.added — assistant message item starts +// response.content_part.added — output_text part starts +// response.output_text.delta — incremental text (one per claude delta) +// response.output_text.done — text part assembled +// response.content_part.done — content part finalized +// response.output_item.done — message item finalized +// response.completed — terminal, with usage +// +// On error: response.failed with response.error: {code, message}. +// No `[DONE]` sentinel — response.completed is the terminator. + +type responseSSEPayload struct { + Type string `json:"type"` + Response *ResponsesObject `json:"response,omitempty"` + // item / content_part / delta envelopes + OutputIndex *int `json:"output_index,omitempty"` + ContentIndex *int `json:"content_index,omitempty"` + ItemID string `json:"item_id,omitempty"` + Item json.RawMessage `json:"item,omitempty"` + Part json.RawMessage `json:"part,omitempty"` + Delta string `json:"delta,omitempty"` + Text string `json:"text,omitempty"` +} + +// streamResponses runs claude in stream mode and writes Responses-API +// SSE to w. +func streamResponses( + w http.ResponseWriter, + r *http.Request, + opts ExecOpts, + respID, model string, + instructions *string, +) { + flusher, ok := w.(http.Flusher) + if !ok { + http.Error(w, "streaming requires an http.Flusher", http.StatusInternalServerError) + return + } + + w.Header().Set("Content-Type", "text/event-stream") + w.Header().Set("Cache-Control", "no-cache") + w.Header().Set("Connection", "keep-alive") + w.Header().Set("X-Accel-Buffering", "no") + + ctx, cancel := context.WithCancel(r.Context()) + defer cancel() + + events, err := streamClaude(ctx, opts) + if err != nil { + writeResponsesError(w, flusher, respID, model, err) + return + } + pumpResponsesSSE(ctx, w, flusher, events, respID, model, instructions, 15*time.Second) +} + +// pumpResponsesSSE consumes canonical events and writes Responses SSE +// frames. Pure formatter for tests. +func pumpResponsesSSE( + ctx context.Context, + w io.Writer, + flusher http.Flusher, + events <-chan StreamEvent, + respID, model string, + instructions *string, + heartbeatInterval time.Duration, +) { + created := time.Now().Unix() + itemID := messageID() + zero := 0 + + // Track whether we've started the message item — emitted lazily on + // the first delta so an instant-error path can skip it. + itemStarted := false + + // Accumulate text so output_text.done / output_item.done carry the + // final assembled text (the OpenAI spec requires it). + var accumulated string + + emitCreated := func() { + obj := &ResponsesObject{ + ID: respID, + Object: "response", + CreatedAt: created, + Status: "in_progress", + Model: model, + Output: []ResponsesOutputItem{}, + OutputText: "", + Usage: ResponsesUsage{}, + Error: nil, + IncompleteDetails: nil, + Instructions: instructions, + Metadata: map[string]string{}, + ParallelToolCalls: true, + Reasoning: nil, + Store: true, + Temperature: 1.0, + ToolChoice: "auto", + Tools: []any{}, + TopP: 1.0, + Truncation: "disabled", + } + writeResponseEvent(w, flusher, "response.created", responseSSEPayload{ + Type: "response.created", Response: obj, + }) + writeResponseEvent(w, flusher, "response.in_progress", responseSSEPayload{ + Type: "response.in_progress", Response: obj, + }) + } + + startItem := func() { + if itemStarted { + return + } + itemStarted = true + item, _ := json.Marshal(map[string]any{ + "id": itemID, + "type": "message", + "status": "in_progress", + "role": "assistant", + "content": []any{}, + }) + part, _ := json.Marshal(map[string]any{ + "type": "output_text", + "text": "", + "annotations": []any{}, + }) + writeResponseEvent(w, flusher, "response.output_item.added", responseSSEPayload{ + Type: "response.output_item.added", + OutputIndex: &zero, + Item: item, + }) + writeResponseEvent(w, flusher, "response.content_part.added", responseSSEPayload{ + Type: "response.content_part.added", + ItemID: itemID, + OutputIndex: &zero, + ContentIndex: &zero, + Part: part, + }) + } + + finalize := func(final *claudeJSONResult) { + if !itemStarted { + startItem() + } + writeResponseEvent(w, flusher, "response.output_text.done", responseSSEPayload{ + Type: "response.output_text.done", ItemID: itemID, + OutputIndex: &zero, ContentIndex: &zero, Text: accumulated, + }) + part, _ := json.Marshal(map[string]any{ + "type": "output_text", "text": accumulated, "annotations": []any{}, + }) + writeResponseEvent(w, flusher, "response.content_part.done", responseSSEPayload{ + Type: "response.content_part.done", ItemID: itemID, + OutputIndex: &zero, ContentIndex: &zero, Part: part, + }) + item, _ := json.Marshal(map[string]any{ + "id": itemID, "type": "message", "status": "completed", "role": "assistant", + "content": []map[string]any{{"type": "output_text", "text": accumulated, "annotations": []any{}}}, + }) + writeResponseEvent(w, flusher, "response.output_item.done", responseSSEPayload{ + Type: "response.output_item.done", OutputIndex: &zero, Item: item, + }) + + var usage ResponsesUsage + if final != nil { + usage = responsesUsageFromExec(Usage{ + InputTokens: final.Usage.InputTokens, + OutputTokens: final.Usage.OutputTokens, + CacheCreationInputTokens: final.Usage.CacheCreationInputTokens, + CacheReadInputTokens: final.Usage.CacheReadInputTokens, + }) + } + obj := &ResponsesObject{ + ID: respID, + Object: "response", + CreatedAt: created, + Status: "completed", + Model: model, + Output: []ResponsesOutputItem{{ + Type: "message", + ID: itemID, + Status: "completed", + Role: "assistant", + Content: []ResponsesOutputContentPart{ + {Type: "output_text", Text: accumulated, Annotations: []any{}}, + }, + }}, + OutputText: accumulated, + Usage: usage, + Instructions: instructions, + Metadata: map[string]string{}, + Tools: []any{}, + ToolChoice: "auto", + Truncation: "disabled", + Temperature: 1.0, + TopP: 1.0, + Store: true, + } + writeResponseEvent(w, flusher, "response.completed", responseSSEPayload{ + Type: "response.completed", Response: obj, + }) + } + + failed := func(message string) { + errObj := &ResponsesError{Code: "server_error", Message: message} + obj := &ResponsesObject{ + ID: respID, + Object: "response", + CreatedAt: created, + Status: "failed", + Model: model, + Error: errObj, + } + writeResponseEvent(w, flusher, "response.failed", responseSSEPayload{ + Type: "response.failed", Response: obj, + }) + } + + emitCreated() + + var heartbeatC <-chan time.Time + if heartbeatInterval > 0 { + t := time.NewTicker(heartbeatInterval) + defer t.Stop() + heartbeatC = t.C + } + + for { + select { + case <-ctx.Done(): + return + + case <-heartbeatC: + fmt.Fprint(w, ":keepalive\n\n") + flusher.Flush() + + case ev, open := <-events: + if !open { + finalize(nil) + return + } + switch ev.Kind { + case StreamEventMessageStart: + startItem() + + case StreamEventTextDelta: + startItem() + accumulated += ev.Delta + writeResponseEvent(w, flusher, "response.output_text.delta", responseSSEPayload{ + Type: "response.output_text.delta", + ItemID: itemID, + OutputIndex: &zero, + ContentIndex: &zero, + Delta: ev.Delta, + }) + + case StreamEventResult: + if ev.Final != nil && ev.Final.IsError { + failed(ev.Final.Subtype) + return + } + finalize(ev.Final) + return + + case StreamEventError: + failed(ev.Err.Error()) + return + + default: + // MessageStop is implicit in Result for our mapping. + } + } + } +} + +func writeResponseEvent(w io.Writer, flusher http.Flusher, name string, payload responseSSEPayload) { + b, _ := json.Marshal(payload) + fmt.Fprintf(w, "event: %s\ndata: %s\n\n", name, b) + flusher.Flush() +} + +func writeResponsesError(w io.Writer, flusher http.Flusher, respID, model string, err error) { + obj := &ResponsesObject{ + ID: respID, + Object: "response", + CreatedAt: time.Now().Unix(), + Status: "failed", + Model: model, + Error: &ResponsesError{Code: "server_error", Message: err.Error()}, + } + writeResponseEvent(w, flusher, "response.failed", responseSSEPayload{ + Type: "response.failed", Response: obj, + }) +} diff --git a/internal/serve/sse_responses_test.go b/internal/serve/sse_responses_test.go new file mode 100644 index 00000000..1eeb093e --- /dev/null +++ b/internal/serve/sse_responses_test.go @@ -0,0 +1,118 @@ +package serve + +import ( + "context" + "net/http/httptest" + "strings" + "testing" +) + +func pumpResponsesSSEHelper(t *testing.T, events []StreamEvent) string { + t.Helper() + rec := httptest.NewRecorder() + ch := make(chan StreamEvent, len(events)) + for _, ev := range events { + ch <- ev + } + close(ch) + pumpResponsesSSE(context.Background(), rec, rec, ch, "resp_test", "anthropic/sonnet", nil, 0) + return rec.Body.String() +} + +func TestPumpResponsesSSE_HappyPathFrameOrder(t *testing.T) { + events := []StreamEvent{ + {Kind: StreamEventMessageStart, MessageID: "msg_1"}, + {Kind: StreamEventTextDelta, Delta: "Hi"}, + {Kind: StreamEventTextDelta, Delta: ", world"}, + {Kind: StreamEventResult, Final: &claudeJSONResult{ + Type: "result", Subtype: "success", Result: "Hi, world", + Usage: claudeJSONUsageObj{InputTokens: 5, OutputTokens: 4, CacheReadInputTokens: 100}, + }}, + } + body := pumpResponsesSSEHelper(t, events) + + expected := []string{ + "event: response.created", + "event: response.in_progress", + "event: response.output_item.added", + "event: response.content_part.added", + `event: response.output_text.delta`, + `"delta":"Hi"`, + `"delta":", world"`, + "event: response.output_text.done", + `"text":"Hi, world"`, + "event: response.content_part.done", + "event: response.output_item.done", + "event: response.completed", + } + prev := -1 + for _, want := range expected { + idx := strings.Index(body, want) + if idx < 0 { + t.Errorf("missing event %q\n\n%s", want, body) + continue + } + if idx <= prev { + t.Errorf("event %q out of order (idx=%d, prev=%d)\n\n%s", want, idx, prev, body) + } + prev = idx + } + + // No `[DONE]` sentinel — Responses uses response.completed as terminator. + if strings.Contains(body, "[DONE]") { + t.Errorf("Responses SSE must NOT include [DONE] sentinel:\n%s", body) + } +} + +func TestPumpResponsesSSE_UsageOnCompletedFrame(t *testing.T) { + events := []StreamEvent{ + {Kind: StreamEventMessageStart}, + {Kind: StreamEventTextDelta, Delta: "ok"}, + {Kind: StreamEventResult, Final: &claudeJSONResult{ + Type: "result", Result: "ok", + Usage: claudeJSONUsageObj{InputTokens: 5, OutputTokens: 2, CacheCreationInputTokens: 50, CacheReadInputTokens: 100}, + }}, + } + body := pumpResponsesSSEHelper(t, events) + + // input_tokens flattens (input + cache_creation + cache_read) = 155 + if !strings.Contains(body, `"input_tokens":155`) { + t.Errorf("input_tokens should be 5+50+100=155 (flattened):\n%s", body) + } + if !strings.Contains(body, `"output_tokens":2`) { + t.Errorf("output_tokens=2 missing:\n%s", body) + } + if !strings.Contains(body, `"cached_tokens":100`) { + t.Errorf("cached_tokens detail (cache_read=100) missing:\n%s", body) + } + if !strings.Contains(body, `"total_tokens":157`) { + t.Errorf("total_tokens should be 157:\n%s", body) + } +} + +func TestPumpResponsesSSE_ResultIsErrorEmitsFailed(t *testing.T) { + events := []StreamEvent{ + {Kind: StreamEventMessageStart}, + {Kind: StreamEventResult, Final: &claudeJSONResult{ + Type: "result", IsError: true, Subtype: "error_during_execution", + }}, + } + body := pumpResponsesSSEHelper(t, events) + if !strings.Contains(body, "event: response.failed") { + t.Errorf("expected response.failed event, got:\n%s", body) + } + if strings.Contains(body, "event: response.completed") { + t.Errorf("response.completed must not be emitted on error:\n%s", body) + } +} + +func TestPumpResponsesSSE_ChannelClosedWithoutResultStillCompletes(t *testing.T) { + events := []StreamEvent{ + {Kind: StreamEventMessageStart}, + {Kind: StreamEventTextDelta, Delta: "partial"}, + } + body := pumpResponsesSSEHelper(t, events) + if !strings.Contains(body, "event: response.completed") { + t.Errorf("missing response.completed on premature channel close:\n%s", body) + } +} diff --git a/nixhome/flake.lock b/nixhome/flake.lock index 82bb65b4..0b94087a 100644 --- a/nixhome/flake.lock +++ b/nixhome/flake.lock @@ -97,11 +97,11 @@ }, "nixpkgs-edge": { "locked": { - "lastModified": 1774559443, - "narHash": "sha256-DaTXhYbeUlJMMiCYb786lgHaHC3lyAL93ifALTXNA2U=", + "lastModified": 1777030977, + "narHash": "sha256-NszQhLt5+Fll97Kg++amaRzJzZEwnqpOS99hZ00jlFQ=", "owner": "NixOS", "repo": "nixpkgs", - "rev": "79016ca515681ea95825a30e746c9155c5a6f9fb", + "rev": "591688422bef21534907a0d326ba32888e48d315", "type": "github" }, "original": { diff --git a/nixhome/modules/go.nix b/nixhome/modules/go.nix index b3c3f9b6..bb1bab04 100644 --- a/nixhome/modules/go.nix +++ b/nixhome/modules/go.nix @@ -14,6 +14,7 @@ golangci-lint gopls gotools # goimports, godoc, etc. + go-swag # swagger doc generator (use: swag init) ]; home.sessionVariables = { diff --git a/nixhome/modules/infra.nix b/nixhome/modules/infra.nix index 4fe1b4a9..bfe668d8 100644 --- a/nixhome/modules/infra.nix +++ b/nixhome/modules/infra.nix @@ -15,6 +15,15 @@ exec ${pkgs.uv}/bin/uvx awslabs.cloudwatch-mcp-server "$@" ''; + # Notion MCP — official local stdio server (https://github.com/makenotion/notion-mcp-server). + # Wraps `npx -y @notionhq/notion-mcp-server@` so cold starts pull from the + # persistent npm cache (~/.npm) instead of redownloading. Same pattern as the AWS + # MCPs above (uvx wrapper). + notionMcpVersion = "2.2.1"; + notionMcpServer = pkgs.writeShellScriptBin "notion-mcp-server" '' + exec ${pkgs.nodejs_22}/bin/npx -y @notionhq/notion-mcp-server@${notionMcpVersion} "$@" + ''; + # AWS read-only session policy — used by credential_process to scope down creds. # Based on AWS managed ReadOnlyAccess: allows all read/list/describe/get actions. awsReadOnlyPolicy = pkgs.writeText "aws-readonly-policy.json" (builtins.toJSON { @@ -195,6 +204,7 @@ in { opentofuMcp # OpenTofu Registry MCP server (use: opentofu-mcp-server) awsApiMcpServer # AWS API MCP server via uvx (use: aws-api-mcp-server) cloudwatchMcpServer # CloudWatch MCP server via uvx (use: cloudwatch-mcp-server) + notionMcpServer # Notion API MCP server via npx (use: notion-mcp-server) ]; # AWS API MCP — wraps all 200+ AWS services. Uses standard AWS credential chain. @@ -214,9 +224,27 @@ in { command = "${bin}/opentofu-mcp-server"; args = []; }; - # Notion — remote HTTP MCP server. - # Auth: OAuth 2.1 flow on first use (run /mcp in Claude session to authenticate). - devcell.managedMcp.servers.notion = { + # Notion — two variants registered side-by-side. + # + # `notion-api` (default, enabled): official local stdio server + # (@notionhq/notion-mcp-server). Auth via NOTION_TOKEN env var, sourced from + # NOTION_API_KEY (DEVCELL_SECRET_KEYS). Token must be a Notion *internal + # integration* token (ntn_…) with relevant pages/databases shared to it. + # Non-interactive — works for headless agents and skills. + devcell.managedMcp.servers."notion-api" = { + command = "${bin}/notion-mcp-server"; + args = []; + env = { + NOTION_TOKEN = "\${NOTION_API_KEY}"; + }; + }; + + # `notion-oauth` (disabled by default): hosted remote MCP at mcp.notion.com, + # OAuth 2.1 flow on first use (run /mcp in a Claude session to authenticate). + # Useful when per-user OAuth scopes are preferred over a workspace-wide + # integration token. Flip `enabled = true` to stage it. + devcell.managedMcp.servers."notion-oauth" = { + enabled = false; type = "http"; url = "https://mcp.notion.com/mcp"; }; diff --git a/nixhome/modules/llm/claude.nix b/nixhome/modules/llm/claude.nix index b242b37d..46df098b 100644 --- a/nixhome/modules/llm/claude.nix +++ b/nixhome/modules/llm/claude.nix @@ -36,12 +36,15 @@ env = s.env or {}; }; + # Skip servers explicitly disabled (enabled = false). Default: enabled. + enabledServers = lib.filterAttrs (_: s: (s.enabled or true)) mcpCfg.servers; + claudeConfig = json.generate "claude-nix-mcp-servers.json" { backupBeforeMerge = mcpCfg.backupBeforeMerge; - mcpServers = lib.mapAttrs toClaudeServer mcpCfg.servers; + mcpServers = lib.mapAttrs toClaudeServer enabledServers; }; - hasServers = mcpCfg.servers != {}; + hasServers = enabledServers != {}; in { options.devcell.managedClaude = { settings = lib.mkOption { diff --git a/nixhome/modules/llm/codex.nix b/nixhome/modules/llm/codex.nix index 18488861..8c6f7a5c 100644 --- a/nixhome/modules/llm/codex.nix +++ b/nixhome/modules/llm/codex.nix @@ -12,7 +12,10 @@ toml = pkgs.formats.toml {}; # Only stdio servers — Codex doesn't support HTTP transport. - stdioServers = lib.filterAttrs (_: s: (s.type or "stdio") == "stdio") mcpCfg.servers; + # Also skip servers explicitly disabled (enabled = false). Default: enabled. + stdioServers = lib.filterAttrs ( + _: s: (s.type or "stdio") == "stdio" && (s.enabled or true) + ) mcpCfg.servers; toCodexServer = _: s: { diff --git a/nixhome/modules/llm/mcp.nix b/nixhome/modules/llm/mcp.nix index 070c7ed9..fded83d2 100644 --- a/nixhome/modules/llm/mcp.nix +++ b/nixhome/modules/llm/mcp.nix @@ -12,7 +12,16 @@ servers = lib.mkOption { type = lib.types.attrsOf lib.types.anything; default = {}; - description = "Canonical MCP server definitions. Each entry: { command, args?, env? }."; + description = '' + Canonical MCP server definitions. Each entry: + { command, args?, env? } # stdio (default) + { type = "http"; url; } # http (Claude only) + { ...; enabled = false; } # registered but skipped during merge + + `enabled` defaults to true. Set `enabled = false` to keep an entry + documented in nix without staging it into Claude/OpenCode/Codex configs + (useful for opt-in / experimental / alternative-auth variants). + ''; }; backupBeforeMerge = lib.mkOption { type = lib.types.bool; diff --git a/nixhome/modules/llm/opencode.nix b/nixhome/modules/llm/opencode.nix index 864c9340..655c7a76 100644 --- a/nixhome/modules/llm/opencode.nix +++ b/nixhome/modules/llm/opencode.nix @@ -20,7 +20,10 @@ # OpenCode MCP config derivation (from mcp.nix servers) # Only stdio servers — OpenCode doesn't support HTTP transport. - stdioServers = lib.filterAttrs (_: s: (s.type or "stdio") == "stdio") mcpCfg.servers; + # Also skip servers explicitly disabled (enabled = false). Default: enabled. + stdioServers = lib.filterAttrs ( + _: s: (s.type or "stdio") == "stdio" && (s.enabled or true) + ) mcpCfg.servers; toOpenCodeServer = _: s: { diff --git a/nixhome/modules/project-management.nix b/nixhome/modules/project-management.nix index d02ff637..63b8487d 100644 --- a/nixhome/modules/project-management.nix +++ b/nixhome/modules/project-management.nix @@ -1,4 +1,4 @@ -# project-management.nix — Project management and time-tracking MCP servers +# project-management.nix — Project management, time-tracking, and workflow-automation MCP servers {pkgs, config, ...}: let bin = config.devcell.managedMcp.nixBinPrefix; # hubstaff-mcp: Python MCP server for Hubstaff time tracking and project management. @@ -23,9 +23,34 @@ ]; doCheck = false; }; + + # n8n-mcp: Node MCP server bridging Claude/agents to an n8n workflow-automation instance. + # https://github.com/czlonkowski/n8n-mcp + n8nMcp = pkgs.buildNpmPackage { + pname = "n8n-mcp"; + version = "2.47.14"; + src = pkgs.fetchFromGitHub { + owner = "czlonkowski"; + repo = "n8n-mcp"; + rev = "v2.47.14"; + hash = "sha256-nHuWh3hMkvXnUZQcex5pmxF627UlZwVP01ekTI7QCdI="; + }; + npmDepsHash = "sha256-x/gzRVq7rhnNGNGzG3UU/V4SSwCD0FXspvtx5gLf5iE="; + # --legacy-peer-deps: upstream's lockfile has unresolvable peer-dep conflicts + # (langchain/langgraph vs langchain/core, huggingface/inference vs langchain/community). + # Without it, npm's FOD prefetch silently skips conflicting transitive deps + # (e.g. @azure/search-documents) and the offline build phase fails ENOTCACHED. + # --ignore-scripts: esbuild's postinstall does a strict version-match against + # its native binary; nixpkgs' esbuild version drifts from upstream's pin and + # the script throws. The package's own `npm run build` still runs (driven + # separately by buildNpmPackage), so TS→JS compilation is unaffected. + npmFlags = ["--legacy-peer-deps" "--ignore-scripts"]; + nodejs = pkgs.nodejs_22; + }; in { home.packages = [ hubstaffMcp # Hubstaff MCP server for time tracking (use: hubstaff-mcp) + n8nMcp # n8n MCP server for workflow automation (use: n8n-mcp) ]; devcell.managedMcp.servers."hubstaff-mcp" = { @@ -40,4 +65,16 @@ in { type = "http"; url = "https://mcp.linear.app/mcp"; }; + + # n8n — workflow automation. Talks to a self-hosted or cloud n8n instance via its REST API. + # Required env vars: N8N_API_URL (e.g. https://n8n.example.com), N8N_API_KEY (instance API key). + # The \${VAR} escape produces literal ${VAR} in the generated JSON, which Claude expands at spawn time. + devcell.managedMcp.servers."n8n" = { + command = "${bin}/n8n-mcp"; + args = []; + env = { + N8N_API_URL = "\${N8N_API_URL}"; + N8N_API_KEY = "\${N8N_API_KEY}"; + }; + }; } diff --git a/nixhome/modules/security.nix b/nixhome/modules/security.nix index 2a92e720..bce635f6 100644 --- a/nixhome/modules/security.nix +++ b/nixhome/modules/security.nix @@ -89,6 +89,18 @@ in { apkeep # APK downloader from Google Play / APKPure (use: apkeep -a com.example.app .) jadx # APK/DEX decompiler → readable Java source (use: jadx app.apk -d out/) + # binary analysis & reverse engineering (PE / ELF / Mach-O) + ghidra # NSA RE suite — best-in-class PE decompiler (use: ghidra) + radare2 # swiss-army RE framework, strong PE support (use: r2 file.exe) + rizin # radare2 fork, cleaner codebase (use: rizin file.exe) + binwalk # firmware/binary analyzer + carving (use: binwalk -e file.exe) + yara # pattern matching for malware/PE ID (use: yara rules.yar file.exe) + upx # executable packer/unpacker (use: upx -d file.exe) + pev # PE-specific toolkit: pestr/pesec/pedis/pescan (use: readpe file.exe) + detect-it-easy # PE compiler/packer/protector ID (use: diec file.exe) + capstone # multi-arch disassembly engine (lib + cstool: cstool x86 …) + python312Packages.ropper # ROP gadget finder for exploit dev (use: ropper -f file.exe) + # parameter discovery arjun # HTTP parameter discovery (use: arjun -u https://target.com/endpoint)