Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -23,3 +23,4 @@ test/results
.devcell.toml
.gocache
.gomodcache
docs/
4 changes: 4 additions & 0 deletions .goreleaser.dev.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,10 @@ version: 2

project_name: cell

before:
hooks:
- go run github.com/swaggo/swag/cmd/swag@latest init -g cmd/serve.go -o docs --parseDependency --parseInternal

builds:
- id: cell
main: ./cmd
Expand Down
4 changes: 4 additions & 0 deletions .goreleaser.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,10 @@ version: 2

project_name: cell

before:
hooks:
- go run github.com/swaggo/swag/cmd/swag@latest init -g cmd/serve.go -o docs --parseDependency --parseInternal

builds:
- id: cell
main: ./cmd
Expand Down
9 changes: 8 additions & 1 deletion Taskfile.yml
Original file line number Diff line number Diff line change
Expand Up @@ -13,9 +13,16 @@ env:

tasks:

swag:generate:
desc: Regenerate Swagger docs from annotations
dir: "{{.TASKFILE_DIR}}"
cmds:
- go run github.com/swaggo/swag/cmd/swag@latest init -g cmd/serve.go -o docs --parseDependency --parseInternal

cell:build:
desc: Build cell CLI binary
dir: "{{.TASKFILE_DIR}}"
deps: [swag:generate]
cmds:
- CGO_ENABLED=0 go build -ldflags "{{.CELL_LDFLAGS}}" -o ~/.local/bin/cell ./cmd/

Expand Down Expand Up @@ -66,7 +73,7 @@ tasks:
desc: Build ci group and push to registry (multi-arch)
silent: true
cmds:
- GIT_COMMIT={{.GIT_COMMIT_HASH}} docker buildx bake --file {{.TASKFILE_DIR}}/docker-bake.hcl --set '*.output=type=image,push=true,compression=zstd,compression-level=3,force-compression=true' ci {{.CLI_ARGS}}
- GIT_COMMIT={{.GIT_COMMIT_HASH}} docker buildx bake --file {{.TASKFILE_DIR}}/docker-bake.hcl --push ci {{.CLI_ARGS}}

nix:validate:
desc: Validate all nixhome stacks — syntax check then attr check (no build, no activation)
Expand Down
43 changes: 29 additions & 14 deletions cmd/root.go
Original file line number Diff line number Diff line change
Expand Up @@ -243,6 +243,11 @@ func runAgent(binary string, defaultFlags, userArgs []string, extraEnv map[strin

cellCfg := cfg.LoadFromOS(c.ConfigDir, c.BaseDir)

// Set stack/modules so UserImageTag() produces stack-based tags.
runner.Stack = cellCfg.Cell.ResolvedStack()
runner.Modules = cellCfg.Cell.Modules
runner.PerSessionImage = cellCfg.Cell.ResolvedPerSessionImage()

// Resolve available GUI ports — probe and bump if already bound
if cellCfg.Cell.ResolvedGUI() {
c.ResolveAvailablePorts()
Expand Down Expand Up @@ -356,10 +361,21 @@ func runAgent(binary string, defaultFlags, userArgs []string, extraEnv map[strin
imageID = ""
}

// Inject system prompt for Claude Code — describes container environment,
// bind mounts, and host path mappings so Claude understands its runtime context.
// Inject system prompt for Claude Code — container context (mounts,
// host paths, constraints) plus the operator/project prompt resolved
// from env vars and devcell.toml. See runner.AssembleSystemPrompt for
// the full source-precedence chain (cell claude doesn't expose flags
// today; cell serve does).
if binary == "claude" {
prompt := runner.BuildSystemPrompt(c, cellCfg)
prompt, err := runner.AssembleSystemPrompt(c, cellCfg, runner.ResolveOpts{
EnvFile: os.Getenv("DEVCELL_SYSTEM_PROMPT_FILE"),
EnvInline: os.Getenv("DEVCELL_SYSTEM_PROMPT"),
CellCfg: cellCfg,
CfgBaseDir: c.BaseDir,
})
if err != nil {
return fmt.Errorf("system prompt: %w", err)
}
defaultFlags = append(defaultFlags, "--append-system-prompt", prompt)
}

Expand Down Expand Up @@ -392,18 +408,17 @@ func runAgent(binary string, defaultFlags, userArgs []string, extraEnv map[strin
}
ux.Debugf("1Password: resolving %d document(s): %v", len(opDocs), opDocs)
if _, err := exec.LookPath("op"); err == nil {
resolved, err := op.ResolveItems(opDocs)
if err != nil {
fmt.Fprintf(os.Stderr, "warning: 1Password: %v\n", err)
} else {
keys := make([]string, 0, len(resolved))
for k, v := range resolved {
os.Setenv(k, v)
inheritEnv = append(inheritEnv, k)
keys = append(keys, k)
}
ux.Debugf("1Password: resolved %d secret(s): %v", len(keys), keys)
resolved, errs := op.ResolveItems(opDocs)
for _, e := range errs {
fmt.Fprintf(os.Stderr, "warning: 1Password: %v\n", e)
}
keys := make([]string, 0, len(resolved))
for k, v := range resolved {
os.Setenv(k, v)
inheritEnv = append(inheritEnv, k)
keys = append(keys, k)
}
ux.Debugf("1Password: resolved %d secret(s) from %d document(s) (%d failed): %v", len(keys), len(opDocs)-len(errs), len(errs), keys)
} else {
ux.Debugf("1Password: op CLI not found, skipping secret resolution")
}
Expand Down
203 changes: 191 additions & 12 deletions cmd/serve.go
Original file line number Diff line number Diff line change
Expand Up @@ -5,46 +5,197 @@ import (
"fmt"
"os"
"os/signal"
"strconv"
"syscall"

_ "github.com/DimmKirr/devcell/docs" // swagger docs (generated by swag init)
"github.com/DimmKirr/devcell/internal/cfg"
"github.com/DimmKirr/devcell/internal/config"
"github.com/DimmKirr/devcell/internal/logger"
"github.com/DimmKirr/devcell/internal/runner"
"github.com/DimmKirr/devcell/internal/serve"
"github.com/spf13/cobra"
)

// @title DevCell Serve API
// @version 1.0
// @description DevCell Serve exposes an OpenAI-compatible HTTP API that proxies LLM requests to agent binaries (Claude Code, OpenCode) running inside a DevCell container.
// @description
// @description ## Why use this?
// @description
// @description Any tool that speaks the OpenAI protocol — Cursor, Continue, n8n, custom scripts, CI pipelines, the OpenAI Agents SDK — can target DevCell Serve as its backend. The server routes requests to the appropriate agent binary based on the `model` field — no SDK or CLI wrapper needed.
// @description
// @description ## Endpoints
// @description
// @description - **`POST /v1/chat/completions`** — Chat Completions API (the most widely supported OpenAI surface). Use this for traditional chat clients.
// @description - **`POST /v1/responses`** — Responses API (newer, used by OpenAI Agents SDK and n8n's "Message a Model" node). Same model routing as chat completions; stateless (no `previous_response_id` chain).
// @description - **`GET /v1/models`** — list available models discovered from installed agents.
// @description
// @description ## Quick start
// @description
// @description 1. Start the server: `cell serve` (or `cell serve --port 9090`)
// @description 2. The server prints an API key on startup. Set `DEVCELL_API_KEY` env var to use a fixed key.
// @description 3. Send requests with `Authorization: Bearer <key>` header.
// @description
// @description ## Model routing
// @description
// @description The `model` field selects which agent handles the prompt (same for both `/v1/chat/completions` and `/v1/responses`):
// @description - `"claude"` or `"anthropic"` → routes to the Claude Code CLI
// @description - `"opencode"` → routes to the OpenCode CLI
// @description - `"anthropic/sonnet"`, `"claude/claude-sonnet-4-5"` → Claude Code with a specific sub-model
// @description
// @description ## Limitations
// @description
// @description - **Streaming is not supported.** Requests with `"stream": true` to `/v1/responses` return 400; `/v1/chat/completions` returns the full response synchronously regardless.
// @description - **No tool calling.** The `tools` field is accepted for compatibility but never invokes a tool — the underlying CLI agents have their own internal tool loop.
// @description - **Stateless.** `previous_response_id` is accepted and ignored; clients must re-send full conversation history each request.
// @description - **Token usage is stubbed at zero** in responses.
// @description
// @description ## Reasoning effort
// @description
// @description Both endpoints honor the OpenAI `reasoning_effort` / `reasoning.effort` field (values: `low`, `medium`, `high`). It maps to the `claude --effort` CLI flag, controlling thinking budget on a per-request basis. Non-spec values (e.g. Claude's `xhigh`/`max`) are silently dropped.
// @description
// @description ## Debug logging
// @description
// @description By default the server logs only request metadata (method, path, status, duration) plus agent metadata at DEBUG level. Prompt and response bodies are **never** logged — they often contain secrets, PII, or large pasted content from upstream tools.
// @description
// @description Set `DEVCELL_LOG_PROMPTS=1` (combined with `LOG_LEVEL=info` or lower) to log the full assembled prompt and the model's reply for every `/v1/chat/completions` and `/v1/responses` request. Use only for debugging client integrations; do not leave on in production.
// @description
// @description ## Example curl
// @description
// @description Chat Completions:
// @description ```bash
// @description curl http://localhost:8484/v1/chat/completions \
// @description -H "Authorization: Bearer $DEVCELL_API_KEY" \
// @description -H "Content-Type: application/json" \
// @description -d '{"model":"anthropic/sonnet","messages":[{"role":"user","content":"explain this repo"}]}'
// @description ```
// @description
// @description Responses API:
// @description ```bash
// @description curl http://localhost:8484/v1/responses \
// @description -H "Authorization: Bearer $DEVCELL_API_KEY" \
// @description -H "Content-Type: application/json" \
// @description -d '{"model":"anthropic/sonnet","input":"explain this repo"}'
// @description ```
// @host localhost:8484
// @BasePath /
// @securityDefinitions.apikey BearerAuth
// @in header
// @name Authorization
// @description Bearer token — set via DEVCELL_API_KEY env var or use the auto-generated key printed on startup. Format: `Bearer dcl-abc123...`

var serveCmd = &cobra.Command{
Use: "serve",
Short: "Start HTTP API server for LLM commands",
Long: `Starts an OpenAI-compatible HTTP server that proxies chat completions
Long: `Starts an OpenAI-compatible HTTP server that proxies requests
to LLM agent binaries (claude, opencode).

Endpoints:

POST /v1/chat/completions — OpenAI chat completions API
GET /api/v1/health — health check
POST /v1/chat/completions — OpenAI Chat Completions API
POST /v1/responses — OpenAI Responses API (newer; n8n, Agents SDK)
GET /v1/models — list available models
GET /healthz — health check (k8s convention)
GET /api/v1/health — health check (REST convention)
GET /api/openapi.json — OpenAPI spec
GET /swagger/ — Swagger UI

The model field selects the agent: "claude", "opencode", or
"claude/claude-sonnet-4-5" (agent/submodel).

Request:

{"model": "claude", "messages": [{"role": "user", "content": "explain this"}]}
"anthropic/sonnet", "claude/claude-sonnet-4-5" (agent/submodel).

Chat Completions request:

{"model": "anthropic/sonnet",
"messages": [{"role": "user", "content": "explain this"}]}

Responses API request:

{"model": "anthropic/sonnet", "input": "explain this"}

Streaming is supported on /v1/chat/completions and /v1/responses for
the claude agent (set "stream": true). Server-Sent Events are emitted
in the standard OpenAI shape: chat.completion.chunk frames terminated
by data: [DONE] for chat, response.<event-name> frames terminated by
response.completed for responses. Heartbeat ":keepalive" comments are
sent every 15s while idle so long-running agentic turns survive proxy
idle timeouts. Opencode has no streaming surface and falls back to
buffered. previous_response_id and tools are accepted but ignored.

The claude binary is invoked with --dangerously-skip-permissions (same
default cell claude uses) — without it any tool call would block on the
permission gate, since the served claude has no TTY for stdin. The
operator's auth boundary is the bearer API key (DEVCELL_API_KEY).

Environment:

DEVCELL_API_KEY Bearer token (auto-generated if empty)
PORT Listen port (overridden by --port)
LOG_LEVEL debug|info|warn|error (default: warn)
DEVCELL_LOG_PROMPTS=1 Log full prompt + response bodies at INFO level
(off by default; prompts may contain secrets)
DEVCELL_SYSTEM_PROMPT Inline system prompt (overridden by --system-prompt)
DEVCELL_SYSTEM_PROMPT_FILE Path to a file used as the system prompt
(overridden by --system-prompt-file)

System-prompt resolution order (first match wins):
1. --system-prompt-file
2. --system-prompt
3. DEVCELL_SYSTEM_PROMPT_FILE
4. DEVCELL_SYSTEM_PROMPT
5. [llm].system_prompt_file in devcell.toml (path relative to project)
6. [llm].system_prompt in devcell.toml (inline)

A container-context preamble (bind mounts, host paths, runtime
constraints) is auto-prepended to whichever prompt resolves above.
Per-request 'instructions' (Responses) / 'system' role (Chat) from
the API body still merge into the user prompt independently.

Examples:

cell serve
cell serve --port 9090`,
cell serve --port 9090
cell serve --system-prompt-file ./SYSTEM.md
cell serve --system-prompt "You are a backend code assistant for project X."
DEVCELL_LOG_PROMPTS=1 LOG_LEVEL=info cell serve # debug a client integration`,
RunE: runServe,
}

var servePort int
var (
servePort int
serveSystemPrompt string
serveSystemPromptFile string
)

func init() {
serveCmd.Flags().IntVar(&servePort, "port", serve.DefaultPort, "port to listen on")
serveCmd.Flags().StringVar(&serveSystemPrompt, "system-prompt", "",
"system prompt passed to claude as --append-system-prompt on every request "+
"(env: DEVCELL_SYSTEM_PROMPT). Composes with per-request `instructions`/`system` from the API body.")
serveCmd.Flags().StringVar(&serveSystemPromptFile, "system-prompt-file", "",
"path to a file whose contents are used as the system prompt "+
"(env: DEVCELL_SYSTEM_PROMPT_FILE). Mutually exclusive with --system-prompt.")
}

func runServe(cmd *cobra.Command, args []string) error {
logLevel := os.Getenv("LOG_LEVEL")
if logLevel == "" {
logLevel = "warn"
}
logger.Initialize(logLevel, true) // plain text, no colors for server logs

// Port priority: --port flag > PORT env > default
if !cmd.Flags().Changed("port") {
if envPort := os.Getenv("PORT"); envPort != "" {
if p, err := strconv.Atoi(envPort); err == nil {
servePort = p
}
}
}

apiKey := os.Getenv("DEVCELL_API_KEY")
if apiKey == "" {
generated := apiKey == ""
if generated {
apiKey = serve.GenerateAPIKey()
}

Expand All @@ -56,17 +207,45 @@ func runServe(cmd *cobra.Command, args []string) error {
ctx, cancel := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM)
defer cancel()

c, err := config.LoadFromOS()
if err != nil {
return fmt.Errorf("load config: %w", err)
}
cellCfg := cfg.LoadFromOS(c.ConfigDir, c.BaseDir)

systemPrompt, err := runner.AssembleSystemPrompt(c, cellCfg, runner.ResolveOpts{
FlagFile: serveSystemPromptFile,
FlagInline: serveSystemPrompt,
EnvFile: os.Getenv("DEVCELL_SYSTEM_PROMPT_FILE"),
EnvInline: os.Getenv("DEVCELL_SYSTEM_PROMPT"),
CellCfg: cellCfg,
CfgBaseDir: c.BaseDir,
})
if err != nil {
return fmt.Errorf("system prompt: %w", err)
}

exec := &serve.ShellExecutor{}
srv := serve.NewServer(exec, servePort)
srv.SetAPIKey(apiKey)
srv.SetSystemPrompt(systemPrompt)
// Off by default. Setting DEVCELL_LOG_PROMPTS=1 makes /v1/chat/completions
// and /v1/responses log full prompt + response text at INFO level. Useful
// for debugging client integrations; risky for prod logs because prompts
// often carry secrets / PII / large pasted content.
if os.Getenv("DEVCELL_LOG_PROMPTS") == "1" {
srv.SetLogPrompts(true)
}

addr, errCh := srv.Start(ctx)
if addr == "" {
return <-errCh
}

fmt.Fprintf(os.Stderr, "devcell serve listening on %s\n", addr)
fmt.Fprintf(os.Stderr, "API key: %s\n", apiKey)
if generated {
fmt.Fprintf(os.Stderr, "API key: %s\n", apiKey)
}

return <-errCh
}
Loading
Loading