diff --git a/.github/workflows/gateway-check.yml b/.github/workflows/gateway-check.yml new file mode 100644 index 000000000..827d5a6cb --- /dev/null +++ b/.github/workflows/gateway-check.yml @@ -0,0 +1,98 @@ +name: Gateway Checks + +# Guards the gateway plugin on every change under gateway/: +# - go: build + vet + unit tests for the plugin packages (auth, +# adminapi, hooks, …). The root package is a Go plugin (no main +# func; -buildmode=plugin, CGO, pinned Bifrost checkout), so the +# plugin binary itself is only built by the Dockerfile — a plain +# `go build .` fails by design. ./internal/... is the compile +# surface that matters for review. +# - tygo-check: regenerates ui/src/api/types.ts from the Go structs +# in gateway/internal/adminapi and fails on any diff, so a Go +# response-shape change can't land without its committed TS +# counterpart (phase-8 ship gate; drift would silently break the +# dashboard's typed fetch layer). +# - ui: type-checks (tsc -b) and bundles the SPA, catching imports +# the regenerated types.ts no longer satisfies. + +on: + pull_request: + paths: + - "gateway/**" + - ".github/workflows/gateway-check.yml" + push: + branches: [main] + paths: + - "gateway/**" + - ".github/workflows/gateway-check.yml" + +jobs: + go: + runs-on: ubuntu-latest + timeout-minutes: 15 + defaults: + run: + working-directory: gateway + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-go@v5 + with: + go-version-file: gateway/go.mod + cache-dependency-path: gateway/go.sum + + - name: Build + run: go build ./internal/... + + - name: Vet + run: go vet ./internal/... + + - name: Test + run: go test ./internal/... + + tygo-check: + runs-on: ubuntu-latest + timeout-minutes: 15 + defaults: + run: + working-directory: gateway + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-go@v5 + with: + go-version-file: gateway/go.mod + cache-dependency-path: gateway/go.sum + + # Pinned so the generated output is byte-stable — a tygo release + # changing its formatting would otherwise fail every PR. Bump in + # lockstep with the version noted in gateway/Makefile and + # gateway/tygo.yaml. + - name: Install tygo + run: go install github.com/gzuidhof/tygo@v0.2.21 + + - name: Check generated TS types are up to date + run: make tygo-check + + ui: + runs-on: ubuntu-latest + timeout-minutes: 15 + defaults: + run: + working-directory: gateway + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-node@v4 + with: + node-version: 24 + cache: npm + # package-lock.json is git-ignored for this SPA, so key the + # npm cache on package.json instead. + cache-dependency-path: gateway/internal/adminapi/ui/package.json + + - name: Install SPA dependencies + run: make ui-install + + - name: Type-check and build SPA + run: make ui-build diff --git a/gateway/Makefile b/gateway/Makefile index 9d01de629..fc3b66bab 100644 --- a/gateway/Makefile +++ b/gateway/Makefile @@ -90,10 +90,13 @@ ui: ui-install ui-build # ─── tygo (Go -> TS struct codegen) ─────────────────────────────────── # -# Install once per dev machine: -# go install github.com/gzuidhof/tygo@latest +# Install once per dev machine (CI pins the same version in +# .github/workflows/gateway-check.yml — bump both together): +# go install github.com/gzuidhof/tygo@v0.2.21 tygo: + @command -v tygo >/dev/null 2>&1 || { \ + echo "tygo not found: go install github.com/gzuidhof/tygo@v0.2.21"; exit 1; } tygo generate # CI hook: regenerate, then fail the build if the diff is non-empty. diff --git a/gateway/internal/adminapi/evals.go b/gateway/internal/adminapi/evals.go index cf5df79ab..30d6d04f8 100644 --- a/gateway/internal/adminapi/evals.go +++ b/gateway/internal/adminapi/evals.go @@ -104,8 +104,8 @@ type evalRunRequest struct { Agent string `json:"agent,omitempty"` } -// evalRefResponse is the create/link acknowledgement. -type evalRefResponse struct { +// EvalRefResponse is the create/link acknowledgement. +type EvalRefResponse struct { RefID string `json:"ref_id"` Linked bool `json:"linked,omitempty"` } @@ -234,7 +234,7 @@ func (h *evalHandlers) createOrLinkForAgent(w http.ResponseWriter, r *http.Reque writeError(w, http.StatusBadGateway, "catalog_write_failed", "neo4j write failed") return } - writeJSON(w, http.StatusOK, evalRefResponse{RefID: req.SetID, Linked: true}) + writeJSON(w, http.StatusOK, EvalRefResponse{RefID: req.SetID, Linked: true}) return } @@ -243,7 +243,7 @@ func (h *evalHandlers) createOrLinkForAgent(w http.ResponseWriter, r *http.Reque writeError(w, http.StatusBadRequest, "missing_field", "name (or set_id) is required") return } - var created evalRefResponse + var created EvalRefResponse if err := h.hive.call(ctx, http.MethodPost, "/api/gateway/evals", map[string]any{"name": req.Name, "description": req.Description}, &created); err != nil { relayHiveError(w, err) @@ -262,7 +262,7 @@ func (h *evalHandlers) createOrLinkForAgent(w http.ResponseWriter, r *http.Reque "set created but linking to agent failed") return } - writeJSON(w, http.StatusOK, evalRefResponse{RefID: created.RefID}) + writeJSON(w, http.StatusOK, EvalRefResponse{RefID: created.RefID}) } // linkEdge MERGEs HiveAgent-[:HAS_EVAL_SET]->EvalSet and clears any @@ -457,7 +457,7 @@ func (h *evalHandlers) createSet(w http.ResponseWriter, r *http.Request) { writeError(w, http.StatusBadRequest, "missing_field", "name is required") return } - var created evalRefResponse + var created EvalRefResponse if err := h.hive.call(r.Context(), http.MethodPost, "/api/gateway/evals", map[string]any{"name": req.Name, "description": req.Description}, &created); err != nil { relayHiveError(w, err) @@ -513,7 +513,7 @@ func (h *evalHandlers) createRequirement(w http.ResponseWriter, r *http.Request, writeError(w, http.StatusBadRequest, "missing_field", "name is required") return } - var created evalRefResponse + var created EvalRefResponse if err := h.hive.call(r.Context(), http.MethodPost, "/api/gateway/evals/"+urlSeg(setID)+"/requirements", req, &created); err != nil { relayHiveError(w, err) diff --git a/gateway/internal/adminapi/histogram.go b/gateway/internal/adminapi/histogram.go new file mode 100644 index 000000000..1f75f560b --- /dev/null +++ b/gateway/internal/adminapi/histogram.go @@ -0,0 +1,259 @@ +package adminapi + +import ( + "math" + "net/http" + "sort" + "time" +) + +// Phase-7 token and latency histograms. Same shape family as +// /_plugin/histogram/cost (observability.go): ?window=, ?bucket=, +// ?dimension=, optional ?user_id= / ?agent_name= scoping, epoch- +// aligned buckets, empty buckets omitted, series sorted heaviest +// first. Bifrost's own /api/logs/histogram/*/by-dimension can't group +// by metadata columns, so these bucket in Go like cost does. + +// TokenHistogramPoint is one bucket of a per-dimension token series. +type TokenHistogramPoint struct { + Timestamp string `json:"ts"` + PromptTokens int64 `json:"prompt_tokens"` + CompletionTokens int64 `json:"completion_tokens"` + TotalTokens int64 `json:"total_tokens"` +} + +// TokenHistogramSeries is one dimension value's line. +type TokenHistogramSeries struct { + DimensionValue string `json:"dimension_value"` + Points []TokenHistogramPoint `json:"points"` +} + +// HistogramTokensResponse is the envelope for /_plugin/histogram/tokens. +type HistogramTokensResponse struct { + BucketSizeSeconds int64 `json:"bucket_size_seconds"` + Dimension string `json:"dimension"` + Series []TokenHistogramSeries `json:"series"` +} + +// LatencyHistogramPoint is one bucket of a per-dimension latency +// series: nearest-rank percentiles (ms) over the bucket's calls plus +// the call count the percentiles were taken from, so a p99 over +// three calls can be read with the right skepticism. +type LatencyHistogramPoint struct { + Timestamp string `json:"ts"` + P50 float64 `json:"p50"` + P95 float64 `json:"p95"` + P99 float64 `json:"p99"` + Count int64 `json:"count"` +} + +// LatencyHistogramSeries is one dimension value's line. +type LatencyHistogramSeries struct { + DimensionValue string `json:"dimension_value"` + Points []LatencyHistogramPoint `json:"points"` +} + +// HistogramLatencyResponse is the envelope for /_plugin/histogram/latency. +type HistogramLatencyResponse struct { + BucketSizeSeconds int64 `json:"bucket_size_seconds"` + Dimension string `json:"dimension"` + Series []LatencyHistogramSeries `json:"series"` +} + +// histogramArgs parses the three histogram params in the order the +// cost handler does (window → bucket → dimension), so every histogram +// rejects bad input with the same messages. When ok is false a 400 +// has been written. +func histogramArgs(w http.ResponseWriter, r *http.Request) (start, end time.Time, bucket time.Duration, dimension string, ok bool) { + _, start, end, ok = parseWindow(w, r) + if !ok { + return + } + bucket, ok = parseBucket(w, r, end.Sub(start)) + if !ok { + return + } + dimension, ok = parseDimensionParam(w, r) + return +} + +// bucketStart floors a row's timestamp to its epoch-aligned bucket. +// Rows with an unparseable timestamp are skipped (ok=false) rather +// than piled into bucket zero. +func bucketStart(ts string, bucketSec int64) (int64, bool) { + t, err := time.Parse(time.RFC3339Nano, ts) + if err != nil { + return 0, false + } + return (t.Unix() / bucketSec) * bucketSec, true +} + +func bucketLabel(start int64) string { + return time.Unix(start, 0).UTC().Format(time.RFC3339) +} + +// ─── /_plugin/histogram/tokens ─────────────────────────────────────── + +func (h *observabilityHandlers) histogramTokens(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodGet { + methodNotAllowed(w, http.MethodGet) + return + } + start, end, bucket, dimension, ok := histogramArgs(w, r) + if !ok { + return + } + logs, err := h.logs.searchAll(r.Context(), searchOpts{ + StartTime: &start, + EndTime: &end, + Metadata: metadataFilterFromQuery(r), + }, 1000, 200_000) + if err != nil { + writeUpstreamError(w, err, "histogram.tokens") + return + } + + bucketSec := int64(bucket.Seconds()) + type agg struct{ prompt, completion, total int64 } + series := map[string]map[int64]*agg{} + for _, l := range logs { + dim := dimensionValue(l, dimension) + if dim == "" || l.TokenUsage == nil { + continue + } + b, ok := bucketStart(l.Timestamp, bucketSec) + if !ok { + continue + } + if _, ok := series[dim]; !ok { + series[dim] = map[int64]*agg{} + } + a, ok := series[dim][b] + if !ok { + a = &agg{} + series[dim][b] = a + } + a.prompt += l.TokenUsage.PromptTokens + a.completion += l.TokenUsage.CompletionTokens + a.total += l.tokens() + } + + out := HistogramTokensResponse{ + BucketSizeSeconds: bucketSec, + Dimension: dimension, + Series: make([]TokenHistogramSeries, 0, len(series)), + } + totals := map[string]int64{} + for dim, buckets := range series { + pts := make([]TokenHistogramPoint, 0, len(buckets)) + for b, a := range buckets { + pts = append(pts, TokenHistogramPoint{ + Timestamp: bucketLabel(b), + PromptTokens: a.prompt, + CompletionTokens: a.completion, + TotalTokens: a.total, + }) + totals[dim] += a.total + } + sort.Slice(pts, func(i, j int) bool { return pts[i].Timestamp < pts[j].Timestamp }) + out.Series = append(out.Series, TokenHistogramSeries{DimensionValue: dim, Points: pts}) + } + sort.SliceStable(out.Series, func(i, j int) bool { + ti, tj := totals[out.Series[i].DimensionValue], totals[out.Series[j].DimensionValue] + if ti != tj { + return ti > tj + } + return out.Series[i].DimensionValue < out.Series[j].DimensionValue + }) + writeJSON(w, http.StatusOK, out) +} + +// ─── /_plugin/histogram/latency ────────────────────────────────────── + +func (h *observabilityHandlers) histogramLatency(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodGet { + methodNotAllowed(w, http.MethodGet) + return + } + start, end, bucket, dimension, ok := histogramArgs(w, r) + if !ok { + return + } + logs, err := h.logs.searchAll(r.Context(), searchOpts{ + StartTime: &start, + EndTime: &end, + Metadata: metadataFilterFromQuery(r), + }, 1000, 200_000) + if err != nil { + writeUpstreamError(w, err, "histogram.latency") + return + } + + bucketSec := int64(bucket.Seconds()) + series := map[string]map[int64][]float64{} + for _, l := range logs { + dim := dimensionValue(l, dimension) + if dim == "" || l.Latency <= 0 { + continue // no latency on errored / in-flight rows + } + b, ok := bucketStart(l.Timestamp, bucketSec) + if !ok { + continue + } + if _, ok := series[dim]; !ok { + series[dim] = map[int64][]float64{} + } + series[dim][b] = append(series[dim][b], l.Latency) + } + + out := HistogramLatencyResponse{ + BucketSizeSeconds: bucketSec, + Dimension: dimension, + Series: make([]LatencyHistogramSeries, 0, len(series)), + } + counts := map[string]int64{} + for dim, buckets := range series { + pts := make([]LatencyHistogramPoint, 0, len(buckets)) + for b, lat := range buckets { + sort.Float64s(lat) + pts = append(pts, LatencyHistogramPoint{ + Timestamp: bucketLabel(b), + P50: percentile(lat, 0.50), + P95: percentile(lat, 0.95), + P99: percentile(lat, 0.99), + Count: int64(len(lat)), + }) + counts[dim] += int64(len(lat)) + } + sort.Slice(pts, func(i, j int) bool { return pts[i].Timestamp < pts[j].Timestamp }) + out.Series = append(out.Series, LatencyHistogramSeries{DimensionValue: dim, Points: pts}) + } + sort.SliceStable(out.Series, func(i, j int) bool { + ci, cj := counts[out.Series[i].DimensionValue], counts[out.Series[j].DimensionValue] + if ci != cj { + return ci > cj + } + return out.Series[i].DimensionValue < out.Series[j].DimensionValue + }) + writeJSON(w, http.StatusOK, out) +} + +// percentile is the nearest-rank percentile of an ascending-sorted +// sample: the smallest value with at least p of the sample at or +// below it. No interpolation — with the small per-bucket counts a +// dashboard sees, a real observed latency is more honest than a +// synthetic one between two. +func percentile(sorted []float64, p float64) float64 { + n := len(sorted) + if n == 0 { + return 0 + } + rank := int(math.Ceil(p*float64(n))) - 1 + if rank < 0 { + rank = 0 + } + if rank >= n { + rank = n - 1 + } + return sorted[rank] +} diff --git a/gateway/internal/adminapi/hotstate.go b/gateway/internal/adminapi/hotstate.go index 8fa646bb9..f0beb74a4 100644 --- a/gateway/internal/adminapi/hotstate.go +++ b/gateway/internal/adminapi/hotstate.go @@ -41,6 +41,13 @@ type KillAgentResponse struct { // RunStateResponse is the wire shape for GET /_plugin/runs/:id/state — // the run's live phase-6 accumulators. A run that has never made a // call reads as all-zero with ttl_seconds = -2 (no key), not 404. +// +// The cap fields come from meta:run:, which the accumulator +// stamps from the verified macaroon chain. `null` means the run has +// no state yet or the layer declared no cap — either way there is +// nothing to draw a meter against. `ancestors` walks the parent +// links outward (nearest parent first), one entry per budgeted run +// above this one, so the UI can render a meter per layer. type RunStateResponse struct { RunID string `json:"run_id"` CostUSD float64 `json:"cost_usd"` @@ -50,8 +57,37 @@ type RunStateResponse struct { // TTLSeconds is the remaining lifetime of the cost accumulator: // -2 when the run has no state yet, -1 when it has no expiry. TTLSeconds int64 `json:"ttl_seconds"` + + MaxCostUSD *float64 `json:"max_cost_usd"` + MaxSteps *int64 `json:"max_steps"` + // Exp is the macaroon layer's expiry (RFC3339); empty when the + // run has no meta yet. + Exp string `json:"exp,omitempty"` + // AgentName / UserID are recorded by the run's own calls. An + // ancestor that has only been seen through a child's chain has + // neither. + AgentName string `json:"agent_name,omitempty"` + UserID string `json:"user_id,omitempty"` + Ancestors []RunAncestorState `json:"ancestors"` } +// RunAncestorState is one budgeted run above the requested one in +// its macaroon chain: the same accumulators and caps, minus the tool +// history. Nearest parent first. +type RunAncestorState struct { + RunID string `json:"run_id"` + CostUSD float64 `json:"cost_usd"` + Steps int64 `json:"steps"` + Killed bool `json:"killed"` + MaxCostUSD *float64 `json:"max_cost_usd"` + MaxSteps *int64 `json:"max_steps"` +} + +// maxAncestorHops bounds the parent walk. Macaroon chains are a +// handful of layers deep in practice; the bound guards against a +// corrupted parent link forming a cycle. +const maxAncestorHops = 8 + // AgentStateResponse is the wire shape for GET /_plugin/agents/:name/state. type AgentStateResponse struct { AgentName string `json:"agent_name"` @@ -102,14 +138,64 @@ func (h *hotStateHandlers) runState(w http.ResponseWriter, r *http.Request, runI writeHotStateErr(w, err, "runs.state") return } - writeJSON(w, http.StatusOK, RunStateResponse{ + out := RunStateResponse{ RunID: st.RunID, CostUSD: st.CostUSD, Steps: st.Steps, Tools: st.Tools, Killed: st.Killed, TTLSeconds: st.TTLSeconds, - }) + MaxCostUSD: capUSD(st), + MaxSteps: capSteps(st), + Exp: st.Exp, + AgentName: st.AgentName, + UserID: st.UserID, + Ancestors: []RunAncestorState{}, + } + + // Walk the parent links. A missing ancestor (its keys expired + // before the child's) ends the walk rather than erroring: the + // meters we can draw are still worth returning. + seen := map[string]bool{runID: true} + for parent := st.Parent; parent != "" && !seen[parent] && len(out.Ancestors) < maxAncestorHops; { + seen[parent] = true + ps, err := auth.GetRunState(r.Context(), parent) + if err != nil { + writeHotStateErr(w, err, "runs.state.ancestor") + return + } + if !ps.HasMeta { + break + } + out.Ancestors = append(out.Ancestors, RunAncestorState{ + RunID: ps.RunID, + CostUSD: ps.CostUSD, + Steps: ps.Steps, + Killed: ps.Killed, + MaxCostUSD: capUSD(ps), + MaxSteps: capSteps(ps), + }) + parent = ps.Parent + } + writeJSON(w, http.StatusOK, out) +} + +// capUSD / capSteps turn the accumulator's "0 = no cap, and also 0 = +// unknown" into the wire's explicit null. +func capUSD(st auth.RunState) *float64 { + if !st.HasMeta || st.MaxCostUSD <= 0 { + return nil + } + v := st.MaxCostUSD + return &v +} + +func capSteps(st auth.RunState) *int64 { + if !st.HasMeta || st.MaxSteps <= 0 { + return nil + } + v := st.MaxSteps + return &v } func (h *hotStateHandlers) agentKill(w http.ResponseWriter, r *http.Request, name string) { diff --git a/gateway/internal/adminapi/hotstate_test.go b/gateway/internal/adminapi/hotstate_test.go index 254e69a6c..0cec2862f 100644 --- a/gateway/internal/adminapi/hotstate_test.go +++ b/gateway/internal/adminapi/hotstate_test.go @@ -364,3 +364,64 @@ func TestRevokeUser_DefaultsToNow(t *testing.T) { t.Fatalf("cutoff %v not ≈ now", got) } } + +func TestRunState_CapsAndAncestors(t *testing.T) { + srv, mr := newBudgetTestServer(t, nil) + // Chain: r_root (invocation) → r_mid → r_leaf. Only the leaf has + // been seen as a leaf, so only it carries agent/user. + mr.HSet("bifrost:meta:run:r_root", "max_cost_usd", "20", "max_steps", "500", "exp", "2026-05-14T18:00:00Z", "parent", "") + mr.HSet("bifrost:cost:run:r_root", "total", "3.5") + mr.HSet("bifrost:steps:run:r_root", "total", "40") + mr.HSet("bifrost:meta:run:r_mid", "max_cost_usd", "5", "max_steps", "0", "exp", "2026-05-14T12:00:00Z", "parent", "r_root") + mr.HSet("bifrost:cost:run:r_mid", "total", "1.5") + mr.HSet("bifrost:steps:run:r_mid", "total", "12") + mr.Set("bifrost:kill:r_mid", "1") + mr.HSet("bifrost:meta:run:r_leaf", "max_cost_usd", "2", "max_steps", "50", "exp", "2026-05-14T10:30:00Z", + "parent", "r_mid", "agent", "coder", "user", "u_alice") + mr.HSet("bifrost:cost:run:r_leaf", "total", "0.75") + mr.HSet("bifrost:steps:run:r_leaf", "total", "3") + + var st RunStateResponse + decodeBody(t, bearerDo(t, srv, http.MethodGet, "/_plugin/runs/r_leaf/state", ""), &st) + if st.MaxCostUSD == nil || *st.MaxCostUSD != 2 || st.MaxSteps == nil || *st.MaxSteps != 50 { + t.Fatalf("leaf caps: %+v", st) + } + if st.Exp != "2026-05-14T10:30:00Z" || st.AgentName != "coder" || st.UserID != "u_alice" { + t.Errorf("leaf identity: %+v", st) + } + if len(st.Ancestors) != 2 { + t.Fatalf("ancestors: %+v", st.Ancestors) + } + mid, root := st.Ancestors[0], st.Ancestors[1] + if mid.RunID != "r_mid" || mid.CostUSD != 1.5 || mid.Steps != 12 || !mid.Killed || + mid.MaxCostUSD == nil || *mid.MaxCostUSD != 5 || mid.MaxSteps != nil { + t.Errorf("mid (max_steps 0 must read as null): %+v", mid) + } + if root.RunID != "r_root" || root.CostUSD != 3.5 || root.MaxSteps == nil || *root.MaxSteps != 500 || root.Killed { + t.Errorf("root: %+v", root) + } + + // A run with no meta: caps null, ancestors empty (not null). + var none RunStateResponse + decodeBody(t, bearerDo(t, srv, http.MethodGet, "/_plugin/runs/r_never/state", ""), &none) + if none.MaxCostUSD != nil || none.MaxSteps != nil || none.Ancestors == nil || len(none.Ancestors) != 0 { + t.Errorf("no-meta run: %+v", none) + } + + // A dangling parent (its keys expired) ends the walk quietly. + mr.HSet("bifrost:meta:run:r_orphan", "max_cost_usd", "1", "max_steps", "1", "exp", "", "parent", "r_gone") + var orphan RunStateResponse + decodeBody(t, bearerDo(t, srv, http.MethodGet, "/_plugin/runs/r_orphan/state", ""), &orphan) + if len(orphan.Ancestors) != 0 { + t.Errorf("dangling parent: %+v", orphan.Ancestors) + } + + // A parent cycle terminates. + mr.HSet("bifrost:meta:run:r_a", "max_cost_usd", "1", "max_steps", "1", "exp", "", "parent", "r_b") + mr.HSet("bifrost:meta:run:r_b", "max_cost_usd", "1", "max_steps", "1", "exp", "", "parent", "r_a") + var cyc RunStateResponse + decodeBody(t, bearerDo(t, srv, http.MethodGet, "/_plugin/runs/r_a/state", ""), &cyc) + if len(cyc.Ancestors) != 1 || cyc.Ancestors[0].RunID != "r_b" { + t.Errorf("cycle: %+v", cyc.Ancestors) + } +} diff --git a/gateway/internal/adminapi/logstore_client.go b/gateway/internal/adminapi/logstore_client.go index 62abb0175..88deb645a 100644 --- a/gateway/internal/adminapi/logstore_client.go +++ b/gateway/internal/adminapi/logstore_client.go @@ -95,6 +95,34 @@ type logstoreLog struct { // "metadata"). Decoding as map[string]string covers every dim // header the plugin canonicalises. Metadata map[string]string `json:"metadata"` + + // TokenUsage is Bifrost's provider-reported usage. The list + // endpoint selects the denormalised prompt/completion/total + // columns and its AfterFind hook rebuilds `token_usage` from them + // (framework/logstore/tables.go), so every row that had usage + // carries it here; rows without (errors, embeddings) decode nil. + TokenUsage *logstoreTokenUsage `json:"token_usage,omitempty"` +} + +// logstoreTokenUsage is the slice of schemas.BifrostLLMUsage the +// aggregations read. Cached-token splits stay on the detail row. +type logstoreTokenUsage struct { + PromptTokens int64 `json:"prompt_tokens"` + CompletionTokens int64 `json:"completion_tokens"` + TotalTokens int64 `json:"total_tokens"` +} + +// tokens returns the row's total token count, 0 when Bifrost +// recorded no usage. Prompt + completion is used when the provider +// left total_tokens at zero (some streaming responses do). +func (l logstoreLog) tokens() int64 { + if l.TokenUsage == nil { + return 0 + } + if l.TokenUsage.TotalTokens > 0 { + return l.TokenUsage.TotalTokens + } + return l.TokenUsage.PromptTokens + l.TokenUsage.CompletionTokens } // logstoreSearchResult is the trimmed shape of /api/logs. @@ -298,6 +326,49 @@ type logstoreUserRanking struct { // matching JSON shape, so phase 9 can swap in if it wants the // trend deltas. +// ─── governance (bifrost customer budgets) ─────────────────────────── + +// logstoreCustomer mirrors the slice of Bifrost's TableCustomer that +// /_plugin/users/:id/quota reads: the customer's budgets (Hive's +// reconciler provisions one per (workspace × user), see +// llm-governance-v2.md §"Hive as credential broker"). Same loopback +// admin API and Basic auth as /api/logs, so the client is shared even +// though the name says "logstore". +type logstoreCustomer struct { + ID string `json:"id"` + Name string `json:"name"` + Budgets []logstoreBudget `json:"budgets"` +} + +// logstoreBudget mirrors Bifrost's TableBudget: a cap, the window it +// resets on (Bifrost duration vocabulary), and the running usage +// Bifrost's governance plugin maintains. +type logstoreBudget struct { + ID string `json:"id"` + MaxLimit float64 `json:"max_limit"` + ResetDuration string `json:"reset_duration"` + LastReset string `json:"last_reset"` + CurrentUsage float64 `json:"current_usage"` +} + +// customer fetches GET /api/governance/customers/{id}. Returns +// (nil, nil) on 404 — a user without a provisioned Customer is a +// normal state (pre-reconciler traffic), not an upstream failure. +func (c *logstoreClient) customer(ctx context.Context, id string) (*logstoreCustomer, error) { + var out struct { + Customer *logstoreCustomer `json:"customer"` + } + err := c.getJSON(ctx, "/api/governance/customers/"+url.PathEscape(id), &out) + if err != nil { + var ue *upstreamError + if errors.As(err, &ue) && ue.status == http.StatusNotFound { + return nil, nil + } + return nil, err + } + return out.Customer, nil +} + // ─── helpers ───────────────────────────────────────────────────────── // getJSON GETs `path` (must include leading slash) against Bifrost's diff --git a/gateway/internal/adminapi/logstore_client_test.go b/gateway/internal/adminapi/logstore_client_test.go new file mode 100644 index 000000000..ecf268f8d --- /dev/null +++ b/gateway/internal/adminapi/logstore_client_test.go @@ -0,0 +1,269 @@ +package adminapi + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "net/http" + "net/http/httptest" + "net/url" + "strconv" + "testing" + "time" +) + +// Unit tests for the loopback client itself (phase-7 wire-up +// checklist: "logstore_client_test.go … hits a fake Bifrost"). The +// handler tests cover the shapes end-to-end; these pin the query +// composition, paging, auth header and error mapping the handlers +// rely on without going through HTTP twice. + +// recordingBifrost captures every request the client makes and +// answers with a canned handler. +type recordingBifrost struct { + srv *httptest.Server + reqs []*http.Request + h http.HandlerFunc +} + +func newRecordingBifrost(t *testing.T, h http.HandlerFunc) *recordingBifrost { + t.Helper() + rb := &recordingBifrost{h: h} + rb.srv = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + rb.reqs = append(rb.reqs, r.Clone(context.Background())) + rb.h(w, r) + })) + t.Cleanup(rb.srv.Close) + return rb +} + +func (rb *recordingBifrost) client() *logstoreClient { + return &logstoreClient{ + base: rb.srv.URL, + httpClient: &http.Client{Timeout: 2 * time.Second}, + authHeader: basicAuth("admin", "hunter2"), + } +} + +func emptyPage(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + fmt.Fprint(w, `{"logs":[],"pagination":{"limit":50,"offset":0,"total_count":0},"stats":{"total_requests":0,"total_cost":0,"total_tokens":0},"has_logs":false}`) +} + +func TestLogstoreClient_SearchComposesQuery(t *testing.T) { + rb := newRecordingBifrost(t, emptyPage) + start := time.Date(2026, 5, 14, 9, 0, 0, 0, time.UTC) + end := start.Add(time.Hour) + + _, err := rb.client().search(context.Background(), searchOpts{ + StartTime: &start, + EndTime: &end, + Metadata: map[string]string{"run-id": "r1", "agent-name": "coder"}, + Limit: 25, + Offset: 50, + SortBy: "cost", + Order: "asc", + }) + if err != nil { + t.Fatal(err) + } + if len(rb.reqs) != 1 { + t.Fatalf("requests: %d", len(rb.reqs)) + } + r := rb.reqs[0] + if r.URL.Path != "/api/logs" { + t.Errorf("path: %s", r.URL.Path) + } + if got := r.Header.Get("Authorization"); got != basicAuth("admin", "hunter2") { + t.Errorf("auth header: %q", got) + } + q := r.URL.Query() + want := url.Values{ + "start_time": {start.Format(time.RFC3339Nano)}, + "end_time": {end.Format(time.RFC3339Nano)}, + "metadata_run-id": {"r1"}, + "metadata_agent-name": {"coder"}, + "limit": {"25"}, + "offset": {"50"}, + "sort_by": {"cost"}, + "order": {"asc"}, + } + for k, v := range want { + if q.Get(k) != v[0] { + t.Errorf("query %s = %q, want %q", k, q.Get(k), v[0]) + } + } + if len(q) != len(want) { + t.Errorf("unexpected extra params: %v", q) + } +} + +func TestLogstoreClient_SearchAll_PagesAndCaps(t *testing.T) { + // 2500 rows served in pages of whatever `limit` asks for. + rb := newRecordingBifrost(t, func(w http.ResponseWriter, r *http.Request) { + limit, _ := strconv.Atoi(r.URL.Query().Get("limit")) + offset, _ := strconv.Atoi(r.URL.Query().Get("offset")) + const total = 2500 + var rows []map[string]any + for i := offset; i < offset+limit && i < total; i++ { + rows = append(rows, map[string]any{"id": strconv.Itoa(i), "cost": 0.01}) + } + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(map[string]any{ + "logs": rows, + "pagination": map[string]any{"limit": limit, "offset": offset, "total_count": total}, + "stats": map[string]any{}, + }) + }) + c := rb.client() + + all, err := c.searchAll(context.Background(), searchOpts{}, 1000, 0) + if err != nil { + t.Fatal(err) + } + if len(all) != 2500 { + t.Fatalf("rows = %d, want 2500", len(all)) + } + // 1000 + 1000 + 500: the short page ends the walk. + if len(rb.reqs) != 3 { + t.Fatalf("requests = %d, want 3", len(rb.reqs)) + } + if all[2499].ID != "2499" { + t.Errorf("last row: %+v", all[2499]) + } + + // maxRows stops the walk even though more pages exist. + rb.reqs = nil + capped, err := c.searchAll(context.Background(), searchOpts{}, 1000, 1500) + if err != nil { + t.Fatal(err) + } + if len(capped) != 2000 || len(rb.reqs) != 2 { + t.Errorf("capped walk: rows=%d requests=%d (cap applies after the page that crosses it)", len(capped), len(rb.reqs)) + } + // Oversized page size clamps to Bifrost's 1000 ceiling. + rb.reqs = nil + _, _ = c.searchAll(context.Background(), searchOpts{}, 5000, 100) + if got := rb.reqs[0].URL.Query().Get("limit"); got != "1000" { + t.Errorf("page size not clamped: limit=%s", got) + } +} + +func TestLogstoreClient_FindByID_404IsNil(t *testing.T) { + rb := newRecordingBifrost(t, func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/api/logs/known" { + w.Header().Set("Content-Type", "application/json") + fmt.Fprint(w, `{"id":"known","provider":"anthropic","raw_response":"{}","stream":true}`) + return + } + w.WriteHeader(http.StatusNotFound) + }) + c := rb.client() + + got, err := c.findByID(context.Background(), "known") + if err != nil || got == nil || got.ID != "known" || !got.Stream { + t.Fatalf("known: %+v (%v)", got, err) + } + missing, err := c.findByID(context.Background(), "nope") + if err != nil || missing != nil { + t.Fatalf("404 must be (nil, nil): %+v (%v)", missing, err) + } + // Path-escaping: an id with a slash can't walk the URL. + rb.reqs = nil + _, _ = c.findByID(context.Background(), "a/b") + if rb.reqs[0].URL.EscapedPath() != "/api/logs/a%2Fb" { + t.Errorf("id not escaped: %s", rb.reqs[0].URL.EscapedPath()) + } +} + +func TestLogstoreClient_UpstreamErrors(t *testing.T) { + // Non-2xx maps to upstreamError carrying the status + a body excerpt. + rb := newRecordingBifrost(t, func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusBadGateway) + fmt.Fprint(w, `{"error":"db locked"}`) + }) + _, err := rb.client().search(context.Background(), searchOpts{}) + var ue *upstreamError + if !errors.As(err, &ue) || ue.status != http.StatusBadGateway || ue.body != `{"error":"db locked"}` { + t.Fatalf("non-2xx: %v", err) + } + + // Unreachable maps to upstreamError with a cause. + dead := &logstoreClient{ + base: "http://127.0.0.1:1", // nothing listens on port 1 + httpClient: &http.Client{Timeout: time.Second}, + authHeader: basicAuth("a", "b"), + } + _, err = dead.search(context.Background(), searchOpts{}) + if !errors.As(err, &ue) || ue.cause == nil { + t.Fatalf("unreachable: %v", err) + } + + // A 2xx with a non-JSON body is a decode error, not upstreamError — + // the handler maps that to 500 rather than 502. + rb2 := newRecordingBifrost(t, func(w http.ResponseWriter, _ *http.Request) { + fmt.Fprint(w, "oops") + }) + _, err = rb2.client().search(context.Background(), searchOpts{}) + if err == nil || errors.As(err, &ue) { + t.Fatalf("decode failure must not be upstreamError: %v", err) + } +} + +func TestLogstoreClient_Customer(t *testing.T) { + rb := newRecordingBifrost(t, func(w http.ResponseWriter, r *http.Request) { + switch r.URL.Path { + case "/api/governance/customers/u_alice": + w.Header().Set("Content-Type", "application/json") + fmt.Fprint(w, `{"customer":{"id":"u_alice","name":"alice","budgets":[{"id":"b1","max_limit":1000,"reset_duration":"1d","last_reset":"2026-05-14T00:00:00Z","current_usage":42.5}]}}`) + default: + w.WriteHeader(http.StatusNotFound) + fmt.Fprint(w, `{"error":"Customer not found"}`) + } + }) + c := rb.client() + + got, err := c.customer(context.Background(), "u_alice") + if err != nil || got == nil { + t.Fatalf("customer: %+v (%v)", got, err) + } + if got.ID != "u_alice" || len(got.Budgets) != 1 || got.Budgets[0].MaxLimit != 1000 || + got.Budgets[0].ResetDuration != "1d" || got.Budgets[0].CurrentUsage != 42.5 { + t.Errorf("decoded: %+v", got) + } + missing, err := c.customer(context.Background(), "u_ghost") + if err != nil || missing != nil { + t.Fatalf("404 must be (nil, nil): %+v (%v)", missing, err) + } + if got := rb.reqs[0].Header.Get("Authorization"); got != basicAuth("admin", "hunter2") { + t.Errorf("governance call must carry the admin basic auth: %q", got) + } +} + +func TestLogstoreLog_Tokens(t *testing.T) { + var l logstoreLog + if l.tokens() != 0 { + t.Error("nil usage must be 0") + } + if err := json.Unmarshal([]byte(`{"id":"1","token_usage":{"prompt_tokens":10,"completion_tokens":5,"total_tokens":15}}`), &l); err != nil { + t.Fatal(err) + } + if l.tokens() != 15 { + t.Errorf("tokens = %d", l.tokens()) + } + // total_tokens left at 0 by the provider: fall back to the sum. + l.TokenUsage.TotalTokens = 0 + if l.tokens() != 15 { + t.Errorf("fallback tokens = %d", l.tokens()) + } +} + +func TestNewLogstoreClient_RequiresCreds(t *testing.T) { + if newLogstoreClient("", "x") != nil || newLogstoreClient("x", "") != nil { + t.Error("missing creds must yield nil (routes skipped), not a client that 401s") + } + if c := newLogstoreClient("u", "p"); c == nil || c.base != logstoreBaseURL { + t.Errorf("client: %+v", c) + } +} diff --git a/gateway/internal/adminapi/observability.go b/gateway/internal/adminapi/observability.go index 6f8ebb894..dd0feb44f 100644 --- a/gateway/internal/adminapi/observability.go +++ b/gateway/internal/adminapi/observability.go @@ -9,6 +9,7 @@ import ( "strings" "time" + "github.com/stakwork/stakgraph/gateway/internal/duration" "github.com/stakwork/stakgraph/gateway/internal/pluginlog" ) @@ -260,7 +261,7 @@ func (h *observabilityHandlers) spendByAgent(w http.ResponseWriter, r *http.Requ type agg struct { cost float64 - tokens int64 // not in our trimmed Log; left 0 in phase 8 + tokens int64 count int64 } by := map[string]*agg{} @@ -273,6 +274,7 @@ func (h *observabilityHandlers) spendByAgent(w http.ResponseWriter, r *http.Requ by[name] = &agg{} } by[name].cost += l.Cost + by[name].tokens += l.tokens() by[name].count++ } @@ -353,6 +355,7 @@ func (h *observabilityHandlers) spendByUser(w http.ResponseWriter, r *http.Reque by[uid] = &agg{} } by[uid].cost += l.Cost + by[uid].tokens += l.tokens() by[uid].count++ } @@ -443,6 +446,7 @@ func (h *observabilityHandlers) spendByAgentUser(w http.ResponseWriter, r *http. by[k] = a } a.cost += l.Cost + a.tokens += l.tokens() a.count++ // Provider is recorded on every Bifrost log row; unattributed // rows would skip the loop above before reaching here. @@ -625,8 +629,8 @@ func dimensionValue(l logstoreLog, dim string) string { // Routed under `/_plugin/runs/` (subtree); the trailing segments // dispatch by shape: // -// /_plugin/runs/{run_id} → runDetail (list) -// /_plugin/runs/{run_id}/calls/{call_id} → runCallDetail (body) +// /_plugin/runs/{run_id} → runDetail (list) +// /_plugin/runs/{run_id}/calls/{call_id} → runCallDetail (body) // // /:id/state and /:id/kill live under the same prefix but are // routed to the phase-6 hot-state handlers before this one runs @@ -713,12 +717,12 @@ func (h *observabilityHandlers) runDetailList(w http.ResponseWriter, r *http.Req // (Bifrost's primary key is the call id), but we verify it matches // `metadata.run-id` on the row before returning. Two reasons: // -// 1. Defence in depth — keeps the call-detail URL self-describing -// and prevents the SPA from being tricked into enumerating -// call IDs across runs. -// 2. Symmetric with /_plugin/runs/{id} which is already run-scoped. -// Operators reasonably expect /runs/A/calls/X and /runs/B/calls/X -// to give 404 for whichever doesn't actually contain X. +// 1. Defence in depth — keeps the call-detail URL self-describing +// and prevents the SPA from being tricked into enumerating +// call IDs across runs. +// 2. Symmetric with /_plugin/runs/{id} which is already run-scoped. +// Operators reasonably expect /runs/A/calls/X and /runs/B/calls/X +// to give 404 for whichever doesn't actually contain X. func (h *observabilityHandlers) runCallDetail(w http.ResponseWriter, r *http.Request, runID, callID string) { log, err := h.logs.findByID(r.Context(), callID) if err != nil { @@ -740,38 +744,45 @@ func (h *observabilityHandlers) runCallDetail(w http.ResponseWriter, r *http.Req } writeJSON(w, http.StatusOK, CallDetailResponse{ - ID: log.ID, - RunID: runID, - Timestamp: log.Timestamp, - Provider: log.Provider, - Model: log.Model, - Status: log.Status, - Cost: log.Cost, - Latency: log.Latency, - CustomerID: log.CustomerID, - Metadata: log.Metadata, + ID: log.ID, + RunID: runID, + Timestamp: log.Timestamp, + Provider: log.Provider, + Model: log.Model, + Status: log.Status, + Cost: log.Cost, + Latency: log.Latency, + CustomerID: log.CustomerID, + Metadata: log.Metadata, StopReason: log.StopReason, Stream: log.Stream, NumberOfRetries: log.NumberOfRetries, FallbackIndex: log.FallbackIndex, TokenUsage: log.TokenUsage, CacheDebug: log.CacheDebug, - InputHistory: log.InputHistory, - OutputMessage: log.OutputMessage, - Params: log.Params, - Tools: log.Tools, - ErrorDetails: log.ErrorDetails, - RawRequest: log.RawRequest, - RawResponse: log.RawResponse, - ContentSummary: log.ContentSummary, + InputHistory: log.InputHistory, + OutputMessage: log.OutputMessage, + Params: log.Params, + Tools: log.Tools, + ErrorDetails: log.ErrorDetails, + RawRequest: log.RawRequest, + RawResponse: log.RawResponse, + ContentSummary: log.ContentSummary, }) } // ─── parameter parsing ─────────────────────────────────────────────── -// parseWindow reads ?window=1h|24h|7d|30d and returns the canonical -// label plus the resolved start/end pair (UTC). 24h is the default -// when unset. +// parseWindow reads ?window= and returns the label plus the resolved +// start/end pair (UTC). Any Bifrost duration is accepted (phase 7 +// "Query parameters": 1h, 6h, 24h, 1d, 7d, 1w, 30d, 1M, 1Y — see +// internal/duration for the vocabulary); the SPA's picker uses a +// four-option subset. Analytics windows are rolling — "1d" is the +// last 24 hours ending now, not the calendar day — which is what +// "what happened recently" dashboards expect. Calendar alignment is +// the agent-budget bucket's concern (budgets.go), not this one's. +// 24h is the default when unset; 1Y is the ceiling (the 200k-row +// scan cap bounds the work regardless). // // Returns (window, start, end, ok). When ok is false the handler // has already written a 400 and must return. @@ -780,31 +791,22 @@ func parseWindow(w http.ResponseWriter, r *http.Request) (string, time.Time, tim if q == "" { q = "24h" } - now := time.Now().UTC() - var from time.Time - switch q { - case "1h": - from = now.Add(-time.Hour) - case "6h": - from = now.Add(-6 * time.Hour) - case "24h": - from = now.Add(-24 * time.Hour) - case "7d": - from = now.AddDate(0, 0, -7) - case "30d": - from = now.AddDate(0, 0, -30) - default: + d, err := duration.Parse(q) + if err != nil || d.Length() > 366*24*time.Hour { writeError(w, http.StatusBadRequest, "bad_request", - "window must be one of: 1h, 6h, 24h, 7d, 30d") + "window must be a Bifrost duration (e.g. 1h, 6h, 24h, 1d, 7d, 1w, 30d, 1M), at most 1Y") return "", time.Time{}, time.Time{}, false } - return q, from, now, true + now := time.Now().UTC() + return q, now.Add(-d.Length()), now, true } -// parseBucket reads ?bucket=… and enforces a small whitelist. Must -// not exceed the window (otherwise a 7d window with a 30d bucket -// would produce one point — useless, and the kind of silent -// degradation that's confusing to debug). +// parseBucket reads ?bucket=… (any Bifrost duration; 1h default) and +// rejects buckets shorter than a minute or longer than the window +// (otherwise a 7d window with a 30d bucket would produce one point — +// useless, and the kind of silent degradation that's confusing to +// debug). Buckets are aligned to the unix epoch, never the calendar, +// so "1d" here is 86400-second slots. func parseBucket(w http.ResponseWriter, r *http.Request, window time.Duration) (time.Duration, bool) { q := r.URL.Query().Get("bucket") if q == "" { @@ -812,25 +814,18 @@ func parseBucket(w http.ResponseWriter, r *http.Request, window time.Duration) ( // today" use case the dashboard does on first paint. q = "1h" } - d, ok := map[string]time.Duration{ - "1m": time.Minute, - "5m": 5 * time.Minute, - "10m": 10 * time.Minute, - "1h": time.Hour, - "6h": 6 * time.Hour, - "1d": 24 * time.Hour, - }[q] - if !ok { + d, err := duration.Parse(q) + if err != nil || d.Length() < time.Minute { writeError(w, http.StatusBadRequest, "bad_request", - "bucket must be one of: 1m, 5m, 10m, 1h, 6h, 1d") + "bucket must be a Bifrost duration of at least 1m (e.g. 1m, 5m, 10m, 1h, 6h, 1d)") return 0, false } - if d > window { + if d.Length() > window { writeError(w, http.StatusBadRequest, "bad_request", "bucket must be ≤ window") return 0, false } - return d, true + return d.Length(), true } // parseDimensionParam reads ?dimension=… and enforces the set of diff --git a/gateway/internal/adminapi/observability_phase7_test.go b/gateway/internal/adminapi/observability_phase7_test.go new file mode 100644 index 000000000..8facfc679 --- /dev/null +++ b/gateway/internal/adminapi/observability_phase7_test.go @@ -0,0 +1,468 @@ +package adminapi + +import ( + "encoding/json" + "net/http" + "testing" + "time" + + "github.com/alicebob/miniredis/v2" + "github.com/redis/go-redis/v9" + + "github.com/stakwork/stakgraph/gateway/internal/redisclient" +) + +// Phase-7 remainder: by-session / by-model rollups, token and latency +// histograms, session drill-down, user spend + quota, agent spend, +// and the widened window / bucket vocabulary. Same fakeBifrost +// harness as observability_test.go. + +// phase7Logs is a fixture with every dim stamped and token usage on +// each row, spanning two sessions, three runs, two users, two models. +// Timestamps are offsets from `now` truncated to a 10-minute epoch +// boundary, so every row sits inside the last hour (window=1h sees +// everything) and the first two rows always share one 10m histogram +// bucket regardless of the wall clock. +func phase7Logs(now time.Time) []fakeLog { + base := now.Truncate(10 * time.Minute) + ts := func(minBefore int) string { + return base.Add(-time.Duration(minBefore) * time.Minute).Format(time.RFC3339Nano) + } + tu := func(p, c int64) json.RawMessage { + return json.RawMessage(`{"prompt_tokens":` + itoa(p) + `,"completion_tokens":` + itoa(c) + `,"total_tokens":` + itoa(p+c) + `}`) + } + md := func(agent, run, user, session string) map[string]string { + return map[string]string{"agent-name": agent, "run-id": run, "user-id": user, "session-id": session} + } + return []fakeLog{ + {ID: "1", Timestamp: ts(29), Provider: "anthropic", Model: "claude-3-5-haiku", Status: "success", + Cost: 0.05, Latency: 800, CustomerID: "u_alice", Metadata: md("coder", "r1", "u_alice", "s1"), TokenUsage: tu(100, 20)}, + {ID: "2", Timestamp: ts(28), Provider: "anthropic", Model: "claude-3-5-haiku", Status: "success", + Cost: 0.10, Latency: 1200, CustomerID: "u_alice", Metadata: md("coder", "r1", "u_alice", "s1"), TokenUsage: tu(200, 40)}, + {ID: "3", Timestamp: ts(19), Provider: "openai", Model: "gpt-4o-mini", Status: "success", + Cost: 0.02, Latency: 600, CustomerID: "u_alice", Metadata: md("web-search", "r2", "u_alice", "s1"), TokenUsage: tu(50, 10)}, + {ID: "4", Timestamp: ts(9), Provider: "openai", Model: "gpt-4o-mini", Status: "success", + Cost: 0.04, Latency: 400, CustomerID: "u_bob", Metadata: md("coder", "r3", "u_bob", "s2"), TokenUsage: tu(80, 20)}, + // Errored call: no latency, no usage. Excluded from the latency + // histogram and contributes zero tokens. + {ID: "5", Timestamp: ts(4), Provider: "openai", Model: "gpt-4o-mini", Status: "error", + Cost: 0, Latency: 0, CustomerID: "u_bob", Metadata: md("coder", "r3", "u_bob", "s2")}, + } +} + +func itoa(v int64) string { return json.Number(formatInt(v)).String() } + +func formatInt(v int64) string { + b, _ := json.Marshal(v) + return string(b) +} + +func decodeOK[T any](t *testing.T, resp *http.Response) T { + t.Helper() + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK { + t.Fatalf("status %d", resp.StatusCode) + } + var out T + if err := json.NewDecoder(resp.Body).Decode(&out); err != nil { + t.Fatal(err) + } + return out +} + +// withMiniRedis points redisclient at a fresh miniredis for the test +// and returns the handle for seeding. Observability tests otherwise +// run with no Redis, which is exactly the "redis_available=false" +// path the quota endpoint has to survive. +func withMiniRedis(t *testing.T) *miniredis.Miniredis { + t.Helper() + mr := miniredis.RunT(t) + rc := redis.NewClient(&redis.Options{Addr: mr.Addr()}) + redisclient.SetClientForTest(rc) + t.Cleanup(func() { + _ = rc.Close() + redisclient.SetClientForTest(nil) + }) + return mr +} + +// ─── spend.by-session / by-model ───────────────────────────────────── + +func TestSpendBySession(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + out := decodeOK[SpendBySessionResponse](t, bearerGet(t, srv, "/_plugin/spend/by-session?window=1h")) + if out.Window != "1h" || len(out.Results) != 2 { + t.Fatalf("unexpected: %+v", out) + } + s1 := out.Results[0] // 0.17 > 0.04 + if s1.SessionID != "s1" || s1.UserID != "u_alice" || s1.RequestCount != 3 || s1.RunCount != 2 { + t.Errorf("s1: %+v", s1) + } + if s1.TotalTokens != 120+240+60 || s1.TotalCost < 0.169 || s1.TotalCost > 0.171 { + t.Errorf("s1 totals: %+v", s1) + } + if s1.FirstSeen >= s1.LastSeen { + t.Errorf("s1 span: first=%s last=%s", s1.FirstSeen, s1.LastSeen) + } + if out.Results[1].SessionID != "s2" || out.Results[1].RequestCount != 2 || out.Results[1].RunCount != 1 { + t.Errorf("s2: %+v", out.Results[1]) + } +} + +func TestSpendByModel(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + out := decodeOK[SpendByModelResponse](t, bearerGet(t, srv, "/_plugin/spend/by-model?window=1h")) + if len(out.Results) != 2 { + t.Fatalf("unexpected: %+v", out) + } + if out.Results[0].Model != "claude-3-5-haiku" || out.Results[0].Provider != "anthropic" || + out.Results[0].RequestCount != 2 || out.Results[0].TotalTokens != 360 { + t.Errorf("haiku: %+v", out.Results[0]) + } + // gpt-4o-mini: three rows incl. the errored one (by-model excludes nothing). + if out.Results[1].Model != "gpt-4o-mini" || out.Results[1].RequestCount != 3 || out.Results[1].TotalTokens != 160 { + t.Errorf("gpt: %+v", out.Results[1]) + } +} + +// ─── histogram.tokens / histogram.latency ──────────────────────────── + +func TestHistogramTokens_ByAgent(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + out := decodeOK[HistogramTokensResponse](t, bearerGet(t, srv, + "/_plugin/histogram/tokens?window=1h&bucket=10m&dimension=agent-name")) + if out.BucketSizeSeconds != 600 || out.Dimension != "agent-name" || len(out.Series) != 2 { + t.Fatalf("unexpected: %+v", out) + } + coder := out.Series[0] // 460 tokens > web-search's 60 + if coder.DimensionValue != "coder" { + t.Fatalf("series order: %+v", out.Series) + } + var total int64 + for _, p := range coder.Points { + total += p.TotalTokens + if p.PromptTokens+p.CompletionTokens != p.TotalTokens { + t.Errorf("point split: %+v", p) + } + } + if total != 120+240+100 { + t.Errorf("coder tokens = %d", total) + } + // Rows 1 and 2 (29 and 28 minutes before the 10m-aligned base) + // land in the same bucket, so coder has exactly 2 points (that + // bucket + row 4's). + if len(coder.Points) != 2 { + t.Errorf("expected folded buckets, got %d points", len(coder.Points)) + } + for i := 1; i < len(coder.Points); i++ { + if coder.Points[i-1].Timestamp >= coder.Points[i].Timestamp { + t.Errorf("points not ascending: %+v", coder.Points) + } + } +} + +func TestHistogramLatency_Percentiles(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + // One bucket spanning the whole window so every call folds together. + out := decodeOK[HistogramLatencyResponse](t, bearerGet(t, srv, + "/_plugin/histogram/latency?window=1h&bucket=1h&dimension=user-id")) + if len(out.Series) != 2 || out.Series[0].DimensionValue != "u_alice" { + t.Fatalf("unexpected: %+v", out) + } + // Both of alice's calls may straddle an epoch-aligned hour + // boundary; sum counts across points and check the percentiles + // of whichever point holds the most calls. + var count int64 + var top LatencyHistogramPoint + for _, p := range out.Series[0].Points { + count += p.Count + if p.Count > top.Count { + top = p + } + } + if count != 3 { + t.Fatalf("alice latency samples = %d, want 3 (errored rows excluded)", count) + } + if top.P50 <= 0 || top.P95 < top.P50 || top.P99 < top.P95 { + t.Errorf("percentiles not monotone: %+v", top) + } + // bob: one successful call at 400ms; the errored one is dropped. + bob := out.Series[1] + if bob.DimensionValue != "u_bob" || len(bob.Points) != 1 || bob.Points[0].Count != 1 || + bob.Points[0].P50 != 400 || bob.Points[0].P99 != 400 { + t.Errorf("bob: %+v", bob) + } +} + +func TestPercentile_NearestRank(t *testing.T) { + s := []float64{100, 200, 300, 400, 500} + cases := []struct { + p float64 + want float64 + }{{0.5, 300}, {0.95, 500}, {0.99, 500}, {0.2, 100}, {0.0, 100}} + for _, c := range cases { + if got := percentile(s, c.p); got != c.want { + t.Errorf("p%v = %v, want %v", c.p, got, c.want) + } + } + if percentile(nil, 0.5) != 0 { + t.Error("empty sample must be 0") + } +} + +// ─── sessions ──────────────────────────────────────────────────────── + +func TestSessionDetail_PagesByMetadata(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + out := decodeOK[SessionDetailResponse](t, bearerGet(t, srv, "/_plugin/sessions/s1?limit=2")) + if out.SessionID != "s1" || len(out.Logs) != 2 || out.TotalCount != 3 { + t.Fatalf("unexpected: %+v", out) + } + if out.Stats.TotalRequests != 3 || out.Stats.TotalTokens != 420 { + t.Errorf("stats over whole session: %+v", out.Stats) + } + for _, l := range out.Logs { + if l.Metadata["session-id"] != "s1" { + t.Errorf("leaked row: %+v", l) + } + } +} + +func TestSessionSummary(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + out := decodeOK[SessionSummaryResponse](t, bearerGet(t, srv, "/_plugin/sessions/s1/summary")) + if out.SessionID != "s1" || out.UserID != "u_alice" || out.RequestCount != 3 || out.TotalTokens != 420 { + t.Fatalf("unexpected: %+v", out) + } + if len(out.Agents) != 2 || out.Agents[0] != "coder" || out.Agents[1] != "web-search" { + t.Errorf("agents: %v", out.Agents) + } + if len(out.Runs) != 2 || out.Runs[0] != "r1" || out.Runs[1] != "r2" { + t.Errorf("runs: %v", out.Runs) + } + // base−29m → base−19m: exactly ten minutes. + if out.DurationMS != 10*60_000 { + t.Errorf("duration_ms = %d", out.DurationMS) + } + if out.StartedAt >= out.LatestAt { + t.Errorf("span: %s .. %s", out.StartedAt, out.LatestAt) + } + + // Unknown session: empty, not 404 (same as an empty logs.db). + empty := decodeOK[SessionSummaryResponse](t, bearerGet(t, srv, "/_plugin/sessions/nope/summary")) + if empty.RequestCount != 0 || empty.Agents == nil || empty.Runs == nil || empty.DurationMS != 0 { + t.Errorf("empty summary: %+v", empty) + } +} + +func TestSessions_404OnBadShape(t *testing.T) { + srv := newObservabilityTestServer(t, newFakeBifrost(t, nil)) + for _, p := range []string{"/_plugin/sessions/", "/_plugin/sessions/s1/bogus", "/_plugin/sessions/s1/summary/x"} { + resp := bearerGet(t, srv, p) + resp.Body.Close() + if resp.StatusCode != http.StatusNotFound { + t.Errorf("%s: want 404, got %d", p, resp.StatusCode) + } + } +} + +// ─── users/:id/spend · users/:id/quota ─────────────────────────────── + +func TestUserSpend(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + out := decodeOK[UserSpendResponse](t, bearerGet(t, srv, "/_plugin/users/u_alice/spend?window=1h")) + if out.UserID != "u_alice" || out.Window != "1h" || out.RequestCount != 3 || out.TotalTokens != 420 { + t.Fatalf("unexpected: %+v", out) + } + if out.TotalCost < 0.169 || out.TotalCost > 0.171 { + t.Errorf("cost: %v", out.TotalCost) + } + // The rollup at /users/:id must still answer. + roll := decodeOK[UserDetailResponse](t, bearerGet(t, srv, "/_plugin/users/u_alice?window=1h")) + if roll.RequestCount != 3 { + t.Errorf("rollup: %+v", roll) + } + resp := bearerGet(t, srv, "/_plugin/users/u_alice/bogus") + resp.Body.Close() + if resp.StatusCode != http.StatusNotFound { + t.Errorf("bad subpath: want 404, got %d", resp.StatusCode) + } +} + +func TestUserQuota_BudgetAndInflight(t *testing.T) { + now := time.Now().UTC() + bf := newFakeBifrost(t, phase7Logs(now)) + bf.customers = map[string]fakeCustomer{ + "u_alice": {ID: "u_alice", Name: "alice", Budgets: []fakeBudget{{ + ID: "b1", MaxLimit: 1000, ResetDuration: "1d", CurrentUsage: 12.5, + LastReset: now.Add(-3 * time.Hour).Format(time.RFC3339), + }}}, + "u_nobudget": {ID: "u_nobudget", Name: "nobody"}, + } + srv := newObservabilityTestServer(t, bf) + mr := withMiniRedis(t) + + // Two indexed runs: one live, one whose exp has passed (pruned). + future := float64(now.Add(2 * time.Hour).Unix()) + past := float64(now.Add(-2 * time.Hour).Unix()) + mr.ZAdd("bifrost:runs:user:u_alice", future, "r_live") + mr.ZAdd("bifrost:runs:user:u_alice", past, "r_dead") + mr.HSet("bifrost:cost:run:r_live", "total", "1.25") + mr.HSet("bifrost:steps:run:r_live", "total", "7") + mr.HSet("bifrost:meta:run:r_live", "max_cost_usd", "5", "max_steps", "100", + "exp", now.Add(2*time.Hour).Format(time.RFC3339), "parent", "", "agent", "coder", "user", "u_alice") + mr.Set("bifrost:kill:r_live", "1") + + out := decodeOK[UserQuotaResponse](t, bearerGet(t, srv, "/_plugin/users/u_alice/quota")) + if !out.CustomerFound || out.BudgetUSD == nil || *out.BudgetUSD != 1000 || out.BudgetWindow != "1d" || + out.SpentUSD != 12.5 || out.RemainingUSD == nil || *out.RemainingUSD != 987.5 || out.BudgetLastReset == "" { + t.Fatalf("budget half: %+v", out) + } + if !out.RedisAvailable || len(out.InflightRuns) != 1 { + t.Fatalf("live half: %+v", out) + } + run := out.InflightRuns[0] + if run.RunID != "r_live" || run.AgentName != "coder" || run.CostUSD != 1.25 || run.Steps != 7 || + run.MaxCostUSD == nil || *run.MaxCostUSD != 5 || run.MaxSteps == nil || *run.MaxSteps != 100 || !run.Killed { + t.Errorf("inflight run: %+v", run) + } + if mr.Exists("bifrost:runs:user:u_alice") { + if members, _ := mr.ZMembers("bifrost:runs:user:u_alice"); len(members) != 1 { + t.Errorf("expired run not pruned: %v", members) + } + } + + // Customer without a budget: found, but nothing to draw. + nb := decodeOK[UserQuotaResponse](t, bearerGet(t, srv, "/_plugin/users/u_nobudget/quota")) + if !nb.CustomerFound || nb.BudgetUSD != nil || nb.RemainingUSD != nil || !nb.RedisAvailable || len(nb.InflightRuns) != 0 { + t.Errorf("no-budget customer: %+v", nb) + } + + // Unknown customer: 404 upstream folds into customer_found=false. + unk := decodeOK[UserQuotaResponse](t, bearerGet(t, srv, "/_plugin/users/u_ghost/quota")) + if unk.CustomerFound || unk.BudgetUSD != nil || !unk.RedisAvailable { + t.Errorf("unknown customer: %+v", unk) + } +} + +func TestUserQuota_RedisOffDegrades(t *testing.T) { + now := time.Now().UTC() + bf := newFakeBifrost(t, phase7Logs(now)) + bf.customers = map[string]fakeCustomer{ + "u_alice": {ID: "u_alice", Budgets: []fakeBudget{{MaxLimit: 10, ResetDuration: "1d", CurrentUsage: 2}}}, + } + srv := newObservabilityTestServer(t, bf) + redisclient.SetClientForTest(nil) + + resp := bearerGet(t, srv, "/_plugin/users/u_alice/quota") + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK { + t.Fatalf("status %d", resp.StatusCode) + } + raw := map[string]any{} + _ = json.NewDecoder(resp.Body).Decode(&raw) + if raw["redis_available"] != false || raw["inflight_runs"] != nil || raw["budget_usd"] != 10.0 { + t.Errorf("degraded quota: %+v", raw) + } +} + +// ─── agents/:name/spend ────────────────────────────────────────────── + +func TestAgentSpend(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + out := decodeOK[AgentSpendResponse](t, bearerGet(t, srv, "/_plugin/agents/coder/spend?window=1h")) + if out.AgentName != "coder" || out.Window != "1h" || out.RequestCount != 4 || out.TotalTokens != 460 { + t.Fatalf("unexpected: %+v", out) + } + if out.TotalCost < 0.189 || out.TotalCost > 0.191 { + t.Errorf("cost: %v", out.TotalCost) + } + none := decodeOK[AgentSpendResponse](t, bearerGet(t, srv, "/_plugin/agents/ghost/spend")) + if none.RequestCount != 0 || none.Window != "24h" { + t.Errorf("unknown agent: %+v", none) + } +} + +func TestAgentSpend_404WithoutLogstore(t *testing.T) { + srv, _ := newBudgetTestServer(t, nil) // no logstore in routeDeps + resp := bearerDo(t, srv, http.MethodGet, "/_plugin/agents/coder/spend", "") + resp.Body.Close() + if resp.StatusCode != http.StatusNotFound { + t.Fatalf("want 404, got %d", resp.StatusCode) + } +} + +// ─── window / bucket vocabulary ────────────────────────────────────── + +func TestWindowVocabulary(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + for _, w := range []string{"1h", "6h", "24h", "1d", "7d", "1w", "30d", "1M", "1Y"} { + resp := bearerGet(t, srv, "/_plugin/spend/by-agent?window="+w) + var out SpendByAgentResponse + _ = json.NewDecoder(resp.Body).Decode(&out) + resp.Body.Close() + if resp.StatusCode != http.StatusOK || out.Window != w { + t.Errorf("window=%s: status %d window %q", w, resp.StatusCode, out.Window) + } + } + for _, w := range []string{"99y", "2Y", "0d", "1", "d", "01h", "1m5"} { + resp := bearerGet(t, srv, "/_plugin/spend/by-agent?window="+w) + resp.Body.Close() + if resp.StatusCode != http.StatusBadRequest { + t.Errorf("window=%s: want 400, got %d", w, resp.StatusCode) + } + } +} + +func TestBucketVocabulary(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + for _, c := range []struct { + q string + want int + }{ + {"window=1d&bucket=1h", 200}, + {"window=1d&bucket=15m", 200}, + {"window=1w&bucket=1d", 200}, + {"window=1h&bucket=1d", 400}, // bucket > window + {"window=1h&bucket=30s", 400}, // sub-minute + {"window=1h&bucket=x", 400}, + } { + resp := bearerGet(t, srv, "/_plugin/histogram/cost?"+c.q+"&dimension=agent-name") + resp.Body.Close() + if resp.StatusCode != c.want { + t.Errorf("%s: want %d, got %d", c.q, c.want, resp.StatusCode) + } + } +} + +// ─── tokens now flow through the phase-8 rollups too ───────────────── + +func TestSpendByAgent_TokensFilled(t *testing.T) { + now := time.Now().UTC() + srv := newObservabilityTestServer(t, newFakeBifrost(t, phase7Logs(now))) + + out := decodeOK[SpendByAgentResponse](t, bearerGet(t, srv, "/_plugin/spend/by-agent?window=1h")) + if len(out.Results) != 2 || out.Results[0].AgentName != "coder" || out.Results[0].TotalTokens != 460 { + t.Fatalf("tokens on by-agent: %+v", out.Results) + } +} diff --git a/gateway/internal/adminapi/observability_test.go b/gateway/internal/adminapi/observability_test.go index 155324d8a..34da4c4fd 100644 --- a/gateway/internal/adminapi/observability_test.go +++ b/gateway/internal/adminapi/observability_test.go @@ -60,6 +60,27 @@ type fakeBifrost struct { authPass string requireAuth bool failNextWith int // if non-zero, the next call returns this status + + // customers backs GET /api/governance/customers/{id} (phase-7 + // quota). Keyed by customer id; a miss is a 404 like Bifrost's. + customers map[string]fakeCustomer +} + +// fakeCustomer is the slice of Bifrost's TableCustomer the quota +// endpoint reads, wrapped in the `{"customer": …}` envelope the +// governance handler emits. +type fakeCustomer struct { + ID string `json:"id"` + Name string `json:"name"` + Budgets []fakeBudget `json:"budgets"` +} + +type fakeBudget struct { + ID string `json:"id"` + MaxLimit float64 `json:"max_limit"` + ResetDuration string `json:"reset_duration"` + LastReset string `json:"last_reset"` + CurrentUsage float64 `json:"current_usage"` } func newFakeBifrost(t *testing.T, logs []fakeLog) *fakeBifrost { @@ -94,11 +115,25 @@ func (f *fakeBifrost) handle(w http.ResponseWriter, r *http.Request) { f.serveLogs(w, r) case strings.HasPrefix(r.URL.Path, "/api/logs/"): f.serveLogByID(w, r, strings.TrimPrefix(r.URL.Path, "/api/logs/")) + case strings.HasPrefix(r.URL.Path, "/api/governance/customers/"): + f.serveCustomer(w, strings.TrimPrefix(r.URL.Path, "/api/governance/customers/")) default: w.WriteHeader(http.StatusNotFound) } } +// serveCustomer mimics Bifrost's GET /api/governance/customers/{id}. +func (f *fakeBifrost) serveCustomer(w http.ResponseWriter, id string) { + c, ok := f.customers[id] + if !ok { + w.WriteHeader(http.StatusNotFound) + fmt.Fprintf(w, `{"error":"Customer not found"}`) + return + } + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(map[string]any{"customer": c}) +} + // serveLogByID mimics Bifrost's GET /api/logs/{id} — returns the // full log row (body fields included) or 404. func (f *fakeBifrost) serveLogByID(w http.ResponseWriter, _ *http.Request, id string) { @@ -153,10 +188,20 @@ func (f *fakeBifrost) serveLogs(w http.ResponseWriter, r *http.Request) { } page := rows[start:end] - // Compute stats over the *filtered* set (matches Bifrost). + // Compute stats over the *filtered* set (matches Bifrost). Tokens + // come from token_usage the way Bifrost's SearchStats sums the + // denormalised total_tokens column. var totalCost float64 + var totalTokens int64 for _, l := range rows { totalCost += l.Cost + if len(l.TokenUsage) > 0 { + var tu struct { + TotalTokens int64 `json:"total_tokens"` + } + _ = json.Unmarshal(l.TokenUsage, &tu) + totalTokens += tu.TotalTokens + } } resp := map[string]any{ @@ -169,7 +214,7 @@ func (f *fakeBifrost) serveLogs(w http.ResponseWriter, r *http.Request) { "stats": map[string]any{ "total_requests": total, "total_cost": totalCost, - "total_tokens": int64(0), + "total_tokens": totalTokens, }, "has_logs": total > 0, } diff --git a/gateway/internal/adminapi/server.go b/gateway/internal/adminapi/server.go index 89d022399..0466ef900 100644 --- a/gateway/internal/adminapi/server.go +++ b/gateway/internal/adminapi/server.go @@ -263,9 +263,16 @@ func registerRoutes(mux *http.ServeMux, deps routeDeps) { mux.HandleFunc("/_plugin/spend/by-user", cookieOrBearer(obs.spendByUser)) mux.HandleFunc("/_plugin/spend/by-agent-user", cookieOrBearer(obs.spendByAgentUser)) mux.HandleFunc("/_plugin/histogram/cost", cookieOrBearer(obs.histogramCost)) - // /_plugin/users/ takes a trailing path segment as user-id. - // Phase-8 only exposes the rollup (KPIs + agents-used + - // runs); phase-9 adds /:id/quota for spend-vs-cap. + // Phase-7 remainder: the session / model rollups, token and + // latency histograms, and the session drill-down. + mux.HandleFunc("/_plugin/spend/by-session", cookieOrBearer(obs.spendBySession)) + mux.HandleFunc("/_plugin/spend/by-model", cookieOrBearer(obs.spendByModel)) + mux.HandleFunc("/_plugin/histogram/tokens", cookieOrBearer(obs.histogramTokens)) + mux.HandleFunc("/_plugin/histogram/latency", cookieOrBearer(obs.histogramLatency)) + mux.HandleFunc("/_plugin/sessions/", cookieOrBearer(obs.sessions)) + // /_plugin/users/ subtree: `` is the phase-8 rollup, + // `/spend` the windowed totals, `/quota` the Bifrost + // customer budget blended with the user's in-flight runs. mux.HandleFunc("/_plugin/users/", cookieOrBearer(obs.userDetail)) } @@ -328,6 +335,17 @@ func registerRoutes(mux *http.ServeMux, deps routeDeps) { hot.agentState(w, r, parts[0]) return } + // `/spend` (GET): the agent's windowed totals from + // logs.db — phase 7. 404 when the logstore isn't configured, + // like the /runs/ drill-down. + if len(parts) == 2 && parts[0] != "" && parts[1] == "spend" { + if obs == nil { + http.NotFound(w, r) + return + } + obs.agentSpend(w, r, parts[0]) + return + } // `/_plugin/agents/catalog` (single segment) is the catalog // list — every registry agent, traffic or not. Distinct from // `/catalog` (two segments) which is one agent's detail. diff --git a/gateway/internal/adminapi/sessions.go b/gateway/internal/adminapi/sessions.go new file mode 100644 index 000000000..985853e83 --- /dev/null +++ b/gateway/internal/adminapi/sessions.go @@ -0,0 +1,163 @@ +package adminapi + +import ( + "net/http" + "sort" + "strings" + "time" +) + +// Phase-7 session drill-down: `/_plugin/sessions/:id` (the paginated +// call log, run-detail's sibling) and `/_plugin/sessions/:id/summary` +// (one-shot totals + span). +// +// A "session" here is the x-bf-dim-session-id dim — Hive's chat or +// workflow session, which spans runs. It is NOT Bifrost's own +// session_id filter: that one aliases parent_request_id (the +// multi-turn linkage Bifrost's UI draws), so the native +// /api/logs/sessions/{id} routes would answer a different question. +// Both endpoints filter on metadata.session-id, exactly like +// /runs/:id filters on metadata.run-id. + +// SessionDetailResponse is the envelope for /_plugin/sessions/:id. +// `logs` is one page (?limit=, ?offset=, newest first); `stats` and +// `total_count` cover the whole session. +type SessionDetailResponse struct { + SessionID string `json:"session_id"` + Logs []RunLogEntry `json:"logs"` + Stats RunStats `json:"stats"` + TotalCount int64 `json:"total_count"` +} + +// SessionSummaryResponse is the envelope for +// /_plugin/sessions/:id/summary. Timestamps are RFC3339; duration is +// latest − started, 0 for a single-call session. +type SessionSummaryResponse struct { + SessionID string `json:"session_id"` + UserID string `json:"user_id,omitempty"` + TotalCost float64 `json:"total_cost"` + TotalTokens int64 `json:"total_tokens"` + RequestCount int64 `json:"request_count"` + StartedAt string `json:"started_at,omitempty"` + LatestAt string `json:"latest_at,omitempty"` + DurationMS int64 `json:"duration_ms"` + Agents []string `json:"agents"` + Runs []string `json:"runs"` +} + +// sessions dispatches the /_plugin/sessions/ subtree: +// +// /_plugin/sessions/{session_id} → sessionDetail +// /_plugin/sessions/{session_id}/summary → sessionSummary +// +// Anything else 404s. +func (h *observabilityHandlers) sessions(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodGet { + methodNotAllowed(w, http.MethodGet) + return + } + rest := strings.TrimPrefix(r.URL.Path, "/_plugin/sessions/") + parts := strings.Split(rest, "/") + switch { + case len(parts) == 1 && parts[0] != "": + h.sessionDetail(w, r, parts[0]) + case len(parts) == 2 && parts[0] != "" && parts[1] == "summary": + h.sessionSummary(w, r, parts[0]) + default: + http.NotFound(w, r) + } +} + +func (h *observabilityHandlers) sessionDetail(w http.ResponseWriter, r *http.Request, sessionID string) { + limit, offset, ok := parsePagination(w, r) + if !ok { + return + } + res, err := h.logs.search(r.Context(), searchOpts{ + Metadata: map[string]string{"session-id": sessionID}, + Limit: limit, + Offset: offset, + SortBy: "timestamp", + Order: "desc", + }) + if err != nil { + writeUpstreamError(w, err, "sessions.detail") + return + } + out := SessionDetailResponse{ + SessionID: sessionID, + Logs: make([]RunLogEntry, 0, len(res.Logs)), + Stats: RunStats{ + TotalRequests: res.Stats.TotalRequests, + TotalCost: res.Stats.TotalCost, + TotalTokens: res.Stats.TotalTokens, + }, + TotalCount: res.Pagination.TotalCount, + } + for _, l := range res.Logs { + out.Logs = append(out.Logs, RunLogEntry{ + ID: l.ID, + Timestamp: l.Timestamp, + Provider: l.Provider, + Model: l.Model, + Status: l.Status, + Cost: l.Cost, + Latency: l.Latency, + Metadata: l.Metadata, + }) + } + writeJSON(w, http.StatusOK, out) +} + +// sessionSummary scans every row of the session (no window — a +// session is finite; the 200k-row ceiling still applies) and folds +// it into totals, the time span, and the distinct agents / runs. +func (h *observabilityHandlers) sessionSummary(w http.ResponseWriter, r *http.Request, sessionID string) { + logs, err := h.logs.searchAll(r.Context(), searchOpts{ + Metadata: map[string]string{"session-id": sessionID}, + }, 1000, 200_000) + if err != nil { + writeUpstreamError(w, err, "sessions.summary") + return + } + + out := SessionSummaryResponse{ + SessionID: sessionID, + Agents: []string{}, + Runs: []string{}, + } + agents := map[string]bool{} + runs := map[string]bool{} + for _, l := range logs { + out.TotalCost += l.Cost + out.TotalTokens += l.tokens() + out.RequestCount++ + if out.StartedAt == "" || l.Timestamp < out.StartedAt { + out.StartedAt = l.Timestamp + } + if l.Timestamp > out.LatestAt { + out.LatestAt = l.Timestamp + } + if out.UserID == "" { + out.UserID = l.Metadata["user-id"] + } + if a := l.Metadata["agent-name"]; a != "" && !agents[a] { + agents[a] = true + out.Agents = append(out.Agents, a) + } + if rid := l.Metadata["run-id"]; rid != "" && !runs[rid] { + runs[rid] = true + out.Runs = append(out.Runs, rid) + } + } + sort.Strings(out.Agents) + sort.Strings(out.Runs) + if out.StartedAt != "" && out.LatestAt != "" { + s, err1 := time.Parse(time.RFC3339Nano, out.StartedAt) + e, err2 := time.Parse(time.RFC3339Nano, out.LatestAt) + if err1 == nil && err2 == nil { + out.DurationMS = e.Sub(s).Milliseconds() + } + } + writeJSON(w, http.StatusOK, out) +} diff --git a/gateway/internal/adminapi/spend.go b/gateway/internal/adminapi/spend.go new file mode 100644 index 000000000..f0761141b --- /dev/null +++ b/gateway/internal/adminapi/spend.go @@ -0,0 +1,258 @@ +package adminapi + +import ( + "net/http" + "sort" +) + +// Phase-7 spend rollups that phase 8 didn't ship: by-session, +// by-model, and the single-agent total. Same strategy as the +// by-agent / by-user handlers in observability.go — page the window +// out of Bifrost's /api/logs and bucket in Go, because the dims live +// in `metadata` and Bifrost's native group-bys are column-bound. +// Same 200k-row ceiling, same "rows missing the dim are excluded" +// policy. + +// SessionSpend is one row of /_plugin/spend/by-session. A session is +// whatever the caller stamped on x-bf-dim-session-id (Hive's chat / +// workflow session); it spans runs, so `run_count` and the first / +// last timestamps are included for a Sessions list page. +type SessionSpend struct { + SessionID string `json:"session_id"` + UserID string `json:"user_id"` + TotalCost float64 `json:"total_cost"` + TotalTokens int64 `json:"total_tokens"` + RequestCount int64 `json:"request_count"` + RunCount int64 `json:"run_count"` + FirstSeen string `json:"first_seen,omitempty"` + LastSeen string `json:"last_seen,omitempty"` +} + +// SpendBySessionResponse is the envelope for /_plugin/spend/by-session. +type SpendBySessionResponse struct { + Window string `json:"window"` + Results []SessionSpend `json:"results"` +} + +// ModelSpend is one row of /_plugin/spend/by-model, keyed on +// (provider, model) — the same model name can be served by two +// providers at different prices. +type ModelSpend struct { + Model string `json:"model"` + Provider string `json:"provider"` + TotalCost float64 `json:"total_cost"` + TotalTokens int64 `json:"total_tokens"` + RequestCount int64 `json:"request_count"` +} + +// SpendByModelResponse is the envelope for /_plugin/spend/by-model. +type SpendByModelResponse struct { + Window string `json:"window"` + Results []ModelSpend `json:"results"` +} + +// AgentSpendResponse is the envelope for /_plugin/agents/:name/spend — +// one agent's totals over the window. +type AgentSpendResponse struct { + AgentName string `json:"agent_name"` + Window string `json:"window"` + TotalCost float64 `json:"total_cost"` + TotalTokens int64 `json:"total_tokens"` + RequestCount int64 `json:"request_count"` +} + +// windowedLogs is the shared front half of every rollup: parse +// ?window=, page the matching rows (plus any ?user_id= / ?agent_name= +// scoping) out of Bifrost, and map upstream failure to 502. When ok +// is false the response has been written. +func (h *observabilityHandlers) windowedLogs(w http.ResponseWriter, r *http.Request, where string) (string, []logstoreLog, bool) { + window, start, end, ok := parseWindow(w, r) + if !ok { + return "", nil, false + } + logs, err := h.logs.searchAll(r.Context(), searchOpts{ + StartTime: &start, + EndTime: &end, + Metadata: metadataFilterFromQuery(r), + }, 1000, 200_000) + if err != nil { + writeUpstreamError(w, err, where) + return "", nil, false + } + return window, logs, true +} + +// ─── /_plugin/spend/by-session ─────────────────────────────────────── + +func (h *observabilityHandlers) spendBySession(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodGet { + methodNotAllowed(w, http.MethodGet) + return + } + window, logs, ok := h.windowedLogs(w, r, "spend.by_session") + if !ok { + return + } + + type agg struct { + user string + cost float64 + tokens int64 + count int64 + runs map[string]bool + firstSeen string + lastSeen string + } + by := map[string]*agg{} + for _, l := range logs { + sid := l.Metadata["session-id"] + if sid == "" { + continue + } + a, ok := by[sid] + if !ok { + a = &agg{user: l.Metadata["user-id"], runs: map[string]bool{}} + by[sid] = a + } + a.cost += l.Cost + a.tokens += l.tokens() + a.count++ + if rid := l.Metadata["run-id"]; rid != "" { + a.runs[rid] = true + } + if a.firstSeen == "" || l.Timestamp < a.firstSeen { + a.firstSeen = l.Timestamp + } + if l.Timestamp > a.lastSeen { + a.lastSeen = l.Timestamp + } + } + + out := SpendBySessionResponse{ + Window: window, + Results: make([]SessionSpend, 0, len(by)), + } + for sid, a := range by { + out.Results = append(out.Results, SessionSpend{ + SessionID: sid, + UserID: a.user, + TotalCost: a.cost, + TotalTokens: a.tokens, + RequestCount: a.count, + RunCount: int64(len(a.runs)), + FirstSeen: a.firstSeen, + LastSeen: a.lastSeen, + }) + } + sort.Slice(out.Results, func(i, j int) bool { + if out.Results[i].TotalCost != out.Results[j].TotalCost { + return out.Results[i].TotalCost > out.Results[j].TotalCost + } + return out.Results[i].SessionID < out.Results[j].SessionID + }) + writeJSON(w, http.StatusOK, out) +} + +// ─── /_plugin/spend/by-model ───────────────────────────────────────── +// +// Provider and model are first-class columns on every Bifrost row, +// so unlike the dim rollups nothing is excluded here — an unattributed +// curl still shows up under the model it hit. That makes by-model the +// one rollup whose totals reconcile with Bifrost's own dashboard. + +func (h *observabilityHandlers) spendByModel(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodGet { + methodNotAllowed(w, http.MethodGet) + return + } + window, logs, ok := h.windowedLogs(w, r, "spend.by_model") + if !ok { + return + } + + type agg struct { + model string + provider string + cost float64 + tokens int64 + count int64 + } + by := map[string]*agg{} + for _, l := range logs { + prov := l.Provider + if prov == "" { + prov = "unknown" + } + model := l.Model + if model == "" { + model = "unknown" + } + k := prov + "\x00" + model + a, ok := by[k] + if !ok { + a = &agg{model: model, provider: prov} + by[k] = a + } + a.cost += l.Cost + a.tokens += l.tokens() + a.count++ + } + + out := SpendByModelResponse{ + Window: window, + Results: make([]ModelSpend, 0, len(by)), + } + for _, a := range by { + out.Results = append(out.Results, ModelSpend{ + Model: a.model, + Provider: a.provider, + TotalCost: a.cost, + TotalTokens: a.tokens, + RequestCount: a.count, + }) + } + sort.Slice(out.Results, func(i, j int) bool { + if out.Results[i].TotalCost != out.Results[j].TotalCost { + return out.Results[i].TotalCost > out.Results[j].TotalCost + } + if out.Results[i].Provider != out.Results[j].Provider { + return out.Results[i].Provider < out.Results[j].Provider + } + return out.Results[i].Model < out.Results[j].Model + }) + writeJSON(w, http.StatusOK, out) +} + +// ─── /_plugin/agents/:name/spend ───────────────────────────────────── +// +// One agent's totals. Bifrost computes SearchStats over the whole +// filtered set regardless of the page, so a single limit=1 call +// returns the numbers without paging rows through the plugin. + +func (h *observabilityHandlers) agentSpend(w http.ResponseWriter, r *http.Request, name string) { + if r.Method != http.MethodGet { + methodNotAllowed(w, http.MethodGet) + return + } + window, start, end, ok := parseWindow(w, r) + if !ok { + return + } + res, err := h.logs.search(r.Context(), searchOpts{ + StartTime: &start, + EndTime: &end, + Metadata: map[string]string{"agent-name": name}, + Limit: 1, + }) + if err != nil { + writeUpstreamError(w, err, "agents.spend") + return + } + writeJSON(w, http.StatusOK, AgentSpendResponse{ + AgentName: name, + Window: window, + TotalCost: res.Stats.TotalCost, + TotalTokens: res.Stats.TotalTokens, + RequestCount: res.Stats.TotalRequests, + }) +} diff --git a/gateway/internal/adminapi/ui/AGENTS.md b/gateway/internal/adminapi/ui/AGENTS.md index 8252e07dc..050dd8aa5 100644 --- a/gateway/internal/adminapi/ui/AGENTS.md +++ b/gateway/internal/adminapi/ui/AGENTS.md @@ -17,7 +17,7 @@ toggle. The whole bundle is ~134 KB / 50 KB gzipped. | Data | **`@tanstack/react-query`** aliased through `preact/compat` | | Charts | **`uPlot`** + a thin wrapper in `components/charts/UplotChart.tsx` | | Styling | Vanilla CSS + custom properties; two files only (`styles/base.css`, `styles/components.css`) | -| Types | Hand-maintained in `src/api/types.ts` (will be tygo-generated once `make tygo` runs in CI) | +| Types | `src/api/types.ts` is tygo-generated from the Go structs (`make tygo`; CI fails on drift); `src/api/manual.ts` holds the few non-struct types | Explicit non-choices: no Tailwind, no styled-components, no SSR, no react-router, no global state library beyond Tanstack Query, no @@ -37,7 +37,9 @@ ui/ ├── api/ │ ├── client.ts # typed fetch wrapper, 401 → UnauthorizedError │ ├── queries.ts # one hook per endpoint (useMe, useSpendByAgent, …) - │ ├── types.ts # mirrors Go response structs + │ ├── types.ts # GENERATED by `make tygo` from internal/adminapi — do not edit + │ ├── manual.ts # hand-maintained: Window/Bucket/Dimension unions, ApiError, + │ │ # trust mirrors, Bifrost pass-through chat/usage shapes │ └── window.ts # Window label → seconds (shared by pages + chart) ├── components/ │ ├── layout/ # Shell / Sidebar / Topbar @@ -163,6 +165,13 @@ four mutation hooks (`useKillRun`, `useUnkillRun`, `useKillAgent`, sets it on every request, so per-call code does nothing extra. - Not exposed: `enforce_macaroons` / `enforce_budgets`. The modal carries a static "only enforced when enforce_macaroons=true" hint. +- **Cap meters** on RunDetail's live-state card read `max_cost_usd` / + `max_steps` / `ancestors` off `/runs/:id/state` (the accumulator's + `meta:run:` record) and draw one cost + one steps bar per + budgeted run in the chain, this run first. A null cap renders as + "no cap", not an empty bar. `deriveRunStatus` takes an `exceeded` + flag from the same numbers, so a run at/over any cap (its own or an + ancestor's) badges as "exceeded". ## Auth model the SPA expects @@ -181,9 +190,13 @@ bootstrap (`/_plugin/admin-credentials`) are bearer-only by design. ## When adding a new page 1. Define the Go response struct in `gateway/internal/adminapi/` - (or extend an existing one). -2. Mirror it in `src/api/types.ts` (eventually tygo will generate - this from the Go struct — until then keep them in lockstep). + (or extend an existing one). Exported, named, `json:` tags — tygo + only emits what it can see. +2. Run `make tygo` from `gateway/` and commit the regenerated + `src/api/types.ts`. Never hand-edit it: CI's `make tygo-check` + (`.github/workflows/gateway-check.yml`) fails on any drift. A type + with no Go struct behind it (a string union, an error envelope) + goes in `src/api/manual.ts` instead. 3. Add a `useFoo` hook in `src/api/queries.ts` with the right polling cadence. 4. Create `src/pages/Foo.tsx` and wire a `` into `app.tsx`. diff --git a/gateway/internal/adminapi/ui/src/api/client.ts b/gateway/internal/adminapi/ui/src/api/client.ts index a2a392b58..93d19a5d4 100644 --- a/gateway/internal/adminapi/ui/src/api/client.ts +++ b/gateway/internal/adminapi/ui/src/api/client.ts @@ -12,7 +12,7 @@ // or auth-refresh logic here lights the trap of having two HTTP // stacks; Tanstack Query gives us all of that one layer up. -import type { ApiError } from "./types"; +import type { ApiError } from "./manual"; const PLUGIN_PREFIX = "/_plugin"; diff --git a/gateway/internal/adminapi/ui/src/api/manual.ts b/gateway/internal/adminapi/ui/src/api/manual.ts new file mode 100644 index 000000000..8a78c3bad --- /dev/null +++ b/gateway/internal/adminapi/ui/src/api/manual.ts @@ -0,0 +1,169 @@ +// Hand-maintained types that tygo cannot generate. Everything the +// SPA decodes from an exported Go struct in internal/adminapi lives +// in the sibling types.ts, which is generated by `make tygo` and +// checked in CI by `make tygo-check` — never hand-edit that file. +// +// What belongs here instead, and why: +// - String-literal unions (Window / Bucket / Dimension): the Go +// side validates these as plain strings, so there is no struct +// for tygo to translate. +// - ApiError: the phase-7 error envelope is built inline by +// adminapi's writeError helper (a map, not a struct). +// - TrustOrg / TrustStatus: mirror gateway/internal/trust, a +// package outside tygo.yaml's adminapi scope, re-named for the +// dashboard (trust.Org -> TrustOrg, trust.StatusResponse -> +// TrustStatus). +// - Chat / usage shapes (ChatMessage, TokenUsage, CacheDebug, …): +// Bifrost pass-through fields that Go carries as +// json.RawMessage (tygo emits `any`); typed here to the depth +// the RunDetail drawer actually renders. +// +// Keep each block in lockstep with the Go source it names. + +// ─── enums validated Go-side as strings ───────────────────────────── + +// Window options the SPA's picker exposes. The backend accepts any +// Bifrost duration (observability.go > parseWindow: 1h … 1Y, rolling +// from now); this union is the operator-facing subset, not the +// server's validation set. +export type Window = "1h" | "6h" | "24h" | "7d" | "30d"; + +// Bucket options for the histogram endpoints. Same story: the server +// takes any duration ≥ 1m and ≤ the window; these are the ones the +// SPA offers. +export type Bucket = "1m" | "5m" | "10m" | "1h" | "6h" | "1d"; + +// Dimension values the histogram endpoints accept (observability.go +// > parseDimensionParam). Phase 11 removed `realm-id` — every row in +// a swarm's logs.db is implicitly for that swarm's realm, and the +// realm is surfaced on the trust-status card instead of as a +// per-row column. +export type Dimension = "agent-name" | "run-id" | "session-id" | "user-id"; + +// ─── error envelope ───────────────────────────────────────────────── + +// Phase-7 error envelope (returned on 4xx/5xx). Mirrors the map +// built by internal/adminapi's writeError helper. +export interface ApiError { + error: { + code: string; + message: string; + }; +} + +// ─── trust registry (gateway/internal/trust) ──────────────────────── + +// Trust-registry Org entry — mirrors gateway/internal/trust.Org. +// Surfaced on the dashboard's Provenance card so an operator can +// see which org's signature authorized a run, plus the pubkey / +// issuer URL the plugin would verify against. +export interface TrustOrg { + org_id: string; + pubkey: string; + issuer_url: string; + revocation_poll_seconds: number; + grace_pubkeys?: string[]; + grace_until?: string; +} + +// Trust-registry status — mirrors gateway/internal/trust.StatusResponse. +// The Provenance card on RunDetail uses `realm_id` to show the +// swarm's self-identity ("this run was processed by swarm w1"), +// since phase 11 dropped the per-row realm-id metadata column. +export interface TrustStatus { + claimed: boolean; + org_count: number; + orgs: string[]; + seed_source: "" | "env" | "api"; + last_modified: string; + /** Set on multi-swarm deployments; absent / empty on single-swarm. */ + realm_id?: string; +} + +// ─── Bifrost pass-through shapes (json.RawMessage on the Go side) ─── + +// One message in a chat-style input_history / output_message. +// Mirrors Bifrost's `schemas.ChatMessage` to the depth the drawer +// renders: role, content (string OR an array of content blocks), +// optional tool-call list (assistant) and tool_call_id (tool reply). +// Everything is optional because providers vary in which fields +// they populate, and the drawer falls back to JSON for anything it +// doesn't recognize. +export interface ChatMessage { + role?: string; + name?: string; + /** OpenAI/Anthropic-style: either a plain string or an array of + * typed content blocks. */ + content?: string | ChatContentBlock[] | null; + /** Tool messages: which prior tool_call this is the result for. */ + tool_call_id?: string; + /** Assistant tool calls. */ + tool_calls?: ChatToolCall[]; + /** Anthropic / OpenAI reasoning summaries. */ + reasoning?: string; + refusal?: string; +} + +export interface ChatContentBlock { + type: string; + text?: string; + refusal?: string; + /** Anthropic-style cache marker. When present on a block the + * provider charged this block as a cache write (or read on a + * subsequent call). */ + cache_control?: { type?: string } | null; + cachePoint?: { type?: string } | null; + image_url?: unknown; + input_audio?: unknown; + file?: unknown; +} + +export interface ChatToolCall { + id?: string; + type?: string; + function?: { name?: string; arguments?: string }; +} + +/** Provider-reported usage breakdown. Source: Bifrost + * `schemas.BifrostLLMUsage`. The `prompt_tokens_details` sub- + * object is where the cache split lives — Anthropic populates + * `cached_write_tokens` (prompt cache writes) and + * `cached_read_tokens`; OpenAI uses `cached_read_tokens` only. */ +export interface TokenUsage { + prompt_tokens?: number; + completion_tokens?: number; + total_tokens?: number; + prompt_tokens_details?: { + text_tokens?: number; + audio_tokens?: number; + image_tokens?: number; + cached_read_tokens?: number; + cached_write_tokens?: number; + cached_write_token_details?: { + cached_write_tokens_5m?: number; + cached_write_tokens_1h?: number; + }; + }; + completion_tokens_details?: { + reasoning_tokens?: number; + accepted_prediction_tokens?: number; + }; + cost?: unknown; +} + +/** Semantic cache verdict. Source: `schemas.BifrostCacheDebug`. + * Distinct from prompt-cache token splits (those live in + * TokenUsage above). Present only when a semantic cache plugin + * is configured. */ +export interface CacheDebug { + cache_hit: boolean; + cache_id?: string; + hit_type?: string; + requested_provider?: string; + requested_model?: string; + provider_used?: string; + model_used?: string; + input_tokens?: number; + threshold?: number; + similarity?: number; +} diff --git a/gateway/internal/adminapi/ui/src/api/queries.ts b/gateway/internal/adminapi/ui/src/api/queries.ts index 2cfa5b75d..18f6d4f9f 100644 --- a/gateway/internal/adminapi/ui/src/api/queries.ts +++ b/gateway/internal/adminapi/ui/src/api/queries.ts @@ -30,13 +30,15 @@ import type { SpendByAgentResponse, SpendByAgentUserResponse, SpendByUserResponse, + UserDetailResponse, +} from "./types"; +import type { TrustOrg, TrustStatus, - UserDetailResponse, Window, Bucket, Dimension, -} from "./types"; +} from "./manual"; // ─── /me ───────────────────────────────────────────────────────────── // Fires once at boot, plus on tab refocus (Tanstack default). Cheap diff --git a/gateway/internal/adminapi/ui/src/api/types.ts b/gateway/internal/adminapi/ui/src/api/types.ts index 4865afaf0..fb1def8cc 100644 --- a/gateway/internal/adminapi/ui/src/api/types.ts +++ b/gateway/internal/adminapi/ui/src/api/types.ts @@ -1,216 +1,533 @@ // Code generated by tygo. DO NOT EDIT. // -// To regenerate from the Go source of truth: -// make tygo (or: tygo generate from gateway/) -// // Source: github.com/stakwork/stakgraph/gateway/internal/adminapi -// -// Phase 8 hand-maintains this file as a stop-gap until the tygo -// codegen runs in CI. The shapes below MUST stay byte-equivalent to -// the Go structs they mirror (LoginResponse, MeResponse, -// SpendByAgentResponse, SpendByUserResponse, HistogramCostResponse, -// RunDetailResponse and friends). Any drift fails the matching -// adminapi Go tests on next run. - +// Regenerate with `make tygo`; CI enforces via `make tygo-check`. +// Hand-maintained companions (unions, trust mirrors, Bifrost +// pass-through shapes) live in manual.ts. /* eslint-disable */ +////////// +// source: budgets.go + +/** + * AgentBudgetResponse is the wire shape for /_plugin/agents/:name/budget. + * Phase-8.5 read-only view — phase 9 grows the matching PUT/DELETE + * mutations on `/_plugin/config/agent_budgets/:name`. + * `CapUSD` and `Window` come from the plugin config (YAML / Redis + * override merged); `SpentUSD` comes from the live phase-6 Redis + * accumulator `bifrost:cost:agent::`. Either source + * being absent is fine — the response surfaces nil so the UI can + * render "no budget" or "no spend yet" rather than misleading zeros. + */ +export interface AgentBudgetResponse { + agent_name: string; + /** + * Cap is the configured maximum spend for the current window. + * Null when the agent has no entry in `agent_budgets`. + */ + cap_usd?: number /* float64 */; + /** + * Window is the Bifrost duration string the cap applies to + * (e.g. "1d", "1h"). Empty when no cap is set. + */ + window: string; + /** + * PeriodStart / PeriodEnd bracket the active bucket. Both are + * RFC3339 strings, UTC. Phase 8 only renders the start; future + * "resets in: 23h" badges can read end. + */ + period_start?: string; + period_end?: string; + /** + * SpentUSD is the running total against the cap for the + * current bucket. 0 (not null) when the bucket key simply + * doesn't exist yet — that's a real-world "fresh window, no + * calls yet" state, not a missing-data state. + */ + spent_usd: number /* float64 */; + /** + * RemainingUSD = max(cap - spent, 0). Null when cap is null. + */ + remaining_usd?: number /* float64 */; + /** + * Ratio = spent / cap, clamped to [0, 1+] (we don't clamp the + * upper bound so a runaway "150%" reads honestly). Null when + * cap is null. + */ + ratio?: number /* float64 */; +} + +////////// +// source: catalog.go + +/** + * AgentCatalogResponse is the merged view across all sources for one + * agent — what the agent is _made of_ (prompts/tools/skills), as + * opposed to the budget view's what it's _allowed to spend_. + */ +export interface AgentCatalogResponse { + name: string; + display_name?: string; + description?: string; + default_model?: string; + sources: string[]; + prompts: CatalogPrompt[]; + tools: CatalogTool[]; + skills: CatalogSkill[]; +} +/** + * CatalogAgentSummary is one row of the catalog list — enough to merge + * the registry into the spend-derived /agents table without pulling + * every agent's full prompt/tool/skill bodies. Counts let the list + * render "📄 2 · 🔧 5 · ✦ 1" badges cheaply. + */ +export interface CatalogAgentSummary { + name: string; + display_name?: string; + description?: string; + default_model?: string; + sources: string[]; + prompts: number /* int */; + tools: number /* int */; + skills: number /* int */; + updated_at: string; +} +/** + * CatalogListResponse is the whole registry — every HiveAgent node, + * traffic or not. The UI unions this with spend-by-agent so seeded + * agents that have never been invoked still appear. + */ +export interface CatalogListResponse { + agents: CatalogAgentSummary[]; +} +/** + * CatalogPrompt is one prompt linked to an agent. name/body come from + * the shared `:Prompt` node the agent links to; source/updated_at come + * from the HAS_PROMPT relationship (which system wired it, and when). + */ +export interface CatalogPrompt { + name: string; + body: string; + /** + * Role is the prompt's slot for this agent — "SYSTEM" or "USER" + * (the main/task prompt). Stored on the HAS_PROMPT relationship; + * empty when the wiring source didn't classify it. + */ + role?: string; + source: string; + updated_at: string; +} +export interface CatalogTool { + name: string; + description: string; + schema?: any /* json.RawMessage */; + source: string; + version?: string; + /** + * Enabled is the per-swarm operator toggle, mirroring skills: + * seeded enabled, flipped in the dashboard, preserved across + * Hive re-seeds. Legacy nodes with no value read back as true. + */ + enabled: boolean; + updated_at: string; +} +export interface CatalogSkill { + name: string; + description: string; + source: string; + version?: string; + /** + * Enabled is the per-swarm operator toggle. Skills seed enabled by + * default; an operator flips this in the dashboard and the gateway + * preserves it across Hive re-seeds (a push only refreshes the + * palette + metadata, never the toggle). Legacy nodes with no + * stored value read back as true (coalesce on the read query). + */ + enabled: boolean; + updated_at: string; +} + +////////// +// source: evals.go + +/** + * EvalSetSummary is one row of an agent's eval-set list. + */ +export interface EvalSetSummary { + ref_id: string; + name?: string; + description?: string; + requirements: number /* int */; +} +/** + * AgentEvalsResponse is the agent-detail Evals tab payload — the sets + * linked to this agent via HAS_EVAL_SET. + */ +export interface AgentEvalsResponse { + agent: string; + sets: EvalSetSummary[]; +} +/** + * EvalTriggerSummary is one captured trigger under a requirement, plus + * the outcome of its most-recent run (nil-ish fields when never run). + */ +export interface EvalTriggerSummary { + ref_id: string; + agent?: string; + source?: string; + environment?: string; + change_type?: string; + last_result?: string; // "pass" / "fail" / "" + last_score?: number /* float64 */; + last_notes?: string; + last_attempt?: number /* int */; +} +/** + * EvalRequirementDetail is a requirement with its triggers. + */ +export interface EvalRequirementDetail { + ref_id: string; + name?: string; + description?: string; + prompt_snippet?: string; + order: number /* int */; + triggers: EvalTriggerSummary[]; +} +/** + * EvalSetDetailResponse is the expanded view of one set. + */ +export interface EvalSetDetailResponse { + ref_id: string; + name?: string; + description?: string; + requirements: EvalRequirementDetail[]; +} +/** + * EvalRefResponse is the create/link acknowledgement. + */ +export interface EvalRefResponse { + ref_id: string; + linked?: boolean; +} + +////////// +// source: histogram.go + +/** + * TokenHistogramPoint is one bucket of a per-dimension token series. + */ +export interface TokenHistogramPoint { + ts: string; + prompt_tokens: number /* int64 */; + completion_tokens: number /* int64 */; + total_tokens: number /* int64 */; +} +/** + * TokenHistogramSeries is one dimension value's line. + */ +export interface TokenHistogramSeries { + dimension_value: string; + points: TokenHistogramPoint[]; +} +/** + * HistogramTokensResponse is the envelope for /_plugin/histogram/tokens. + */ +export interface HistogramTokensResponse { + bucket_size_seconds: number /* int64 */; + dimension: string; + series: TokenHistogramSeries[]; +} +/** + * LatencyHistogramPoint is one bucket of a per-dimension latency + * series: nearest-rank percentiles (ms) over the bucket's calls plus + * the call count the percentiles were taken from, so a p99 over + * three calls can be read with the right skepticism. + */ +export interface LatencyHistogramPoint { + ts: string; + p50: number /* float64 */; + p95: number /* float64 */; + p99: number /* float64 */; + count: number /* int64 */; +} +/** + * LatencyHistogramSeries is one dimension value's line. + */ +export interface LatencyHistogramSeries { + dimension_value: string; + points: LatencyHistogramPoint[]; +} +/** + * HistogramLatencyResponse is the envelope for /_plugin/histogram/latency. + */ +export interface HistogramLatencyResponse { + bucket_size_seconds: number /* int64 */; + dimension: string; + series: LatencyHistogramSeries[]; +} + +////////// +// source: hivecallback.go + + +////////// +// source: hotstate.go + +/** + * KillRunResponse is the wire shape for POST /_plugin/runs/:id/kill. + */ +export interface KillRunResponse { + run_id: string; + killed_at: string; // RFC3339 UTC +} +/** + * KillAgentResponse is the wire shape for POST /_plugin/agents/:name/kill. + */ +export interface KillAgentResponse { + agent_name: string; + killed_at: string; // RFC3339 UTC +} +/** + * RunStateResponse is the wire shape for GET /_plugin/runs/:id/state — + * the run's live phase-6 accumulators. A run that has never made a + * call reads as all-zero with ttl_seconds = -2 (no key), not 404. + * The cap fields come from meta:run:, which the accumulator + * stamps from the verified macaroon chain. `null` means the run has + * no state yet or the layer declared no cap — either way there is + * nothing to draw a meter against. `ancestors` walks the parent + * links outward (nearest parent first), one entry per budgeted run + * above this one, so the UI can render a meter per layer. + */ +export interface RunStateResponse { + run_id: string; + cost_usd: number /* float64 */; + steps: number /* int64 */; + tools: string[]; // last 10 tool names, most recent first + killed: boolean; + /** + * TTLSeconds is the remaining lifetime of the cost accumulator: + * -2 when the run has no state yet, -1 when it has no expiry. + */ + ttl_seconds: number /* int64 */; + max_cost_usd?: number /* float64 */; + max_steps?: number /* int64 */; + /** + * Exp is the macaroon layer's expiry (RFC3339); empty when the + * run has no meta yet. + */ + exp?: string; + /** + * AgentName / UserID are recorded by the run's own calls. An + * ancestor that has only been seen through a child's chain has + * neither. + */ + agent_name?: string; + user_id?: string; + ancestors: RunAncestorState[]; +} +/** + * RunAncestorState is one budgeted run above the requested one in + * its macaroon chain: the same accumulators and caps, minus the tool + * history. Nearest parent first. + */ +export interface RunAncestorState { + run_id: string; + cost_usd: number /* float64 */; + steps: number /* int64 */; + killed: boolean; + max_cost_usd?: number /* float64 */; + max_steps?: number /* int64 */; +} +/** + * AgentStateResponse is the wire shape for GET /_plugin/agents/:name/state. + */ +export interface AgentStateResponse { + agent_name: string; + window: string; + bucket_key: string; + current_spend_usd: number /* float64 */; + /** + * ConfiguredCapUSD is null when the agent has no agent_budgets + * entry; then `window` is informational (?window= or "1d"). + */ + configured_cap_usd?: number /* float64 */; + killed: boolean; +} + +////////// +// source: login.go + +/** + * LoginResponse is the JSON body returned by `POST /_plugin/login` + * on success. The SPA stashes the username so the topbar can render + * it without an extra `/me` round-trip. + * Exported because tygo emits TS bindings from this declaration + * (see gateway/tygo.yaml); rename in lockstep with the frontend. + */ export interface LoginResponse { user: string; } - +/** + * MeResponse is `GET /_plugin/me` — the SPA's boot probe to decide + * "am I authenticated?". Unix timestamps (seconds) keep the wire + * format compact and language-agnostic; the frontend converts to + * Date once. + */ export interface MeResponse { user: string; - iat: number; - last_seen: number; + iat: number /* int64 */; + last_seen: number /* int64 */; } +////////// +// source: logstore_client.go + + +////////// +// source: observability.go + +/** + * AgentSpend is one row of /_plugin/spend/by-agent. + */ export interface AgentSpend { agent_name: string; - total_cost: number; - total_tokens: number; - request_count: number; + total_cost: number /* float64 */; + total_tokens: number /* int64 */; + request_count: number /* int64 */; } - +/** + * SpendByAgentResponse is the envelope for /_plugin/spend/by-agent. + */ export interface SpendByAgentResponse { window: string; results: AgentSpend[]; } - +/** + * UserSpend is one row of /_plugin/spend/by-user. + */ export interface UserSpend { user_id: string; user_name: string; - total_cost: number; - total_tokens: number; - request_count: number; + total_cost: number /* float64 */; + total_tokens: number /* int64 */; + request_count: number /* int64 */; } - +/** + * SpendByUserResponse is the envelope for /_plugin/spend/by-user. + */ export interface SpendByUserResponse { window: string; results: UserSpend[]; } - -// Per-provider slice inside an AgentUserSpend row. Lets the canvas -// drive gateway→provider edge widths from real spend; same Bifrost -// call that computes the (agent × user) rollup also fills these, -// so no extra round-trips. +/** + * ProviderSpend is the per-provider slice carried inside an + * AgentUserSpend row. Lets the canvas drive gateway→provider edge + * widths (and the provider drawer) from real spend without a second + * round-trip — the data is already on every Bifrost log row, so + * surfacing it costs only a second bucket in the same pass. + */ export interface ProviderSpend { provider: string; - total_cost: number; - request_count: number; -} - -// One row of /_plugin/spend/by-agent-user — the (agent × user) fan-out -// the Canvas page uses to render one box per pairing without N round -// trips. Rows missing either dim are excluded server-side. -// -// `providers` sums to `total_cost` / `request_count` and is sorted by -// cost desc, with provider name as a stable tiebreaker. + total_cost: number /* float64 */; + request_count: number /* int64 */; +} +/** + * AgentUserSpend is one row of /_plugin/spend/by-agent-user — the + * fan-out crossing of (agent-name × user-id) so the flowchart UI can + * render one box per pairing in a single round-trip. Rows missing + * either dim are excluded (same policy as by-agent / by-user). + * `Providers` is the breakdown across providers for this pairing. + * Sums of `Providers[*].TotalCost` and `Providers[*].RequestCount` + * equal `TotalCost` and `RequestCount` — the breakdown is additive + * to the top-line totals, not a replacement. + */ export interface AgentUserSpend { agent_name: string; user_id: string; user_name: string; - total_cost: number; - total_tokens: number; - request_count: number; + total_cost: number /* float64 */; + total_tokens: number /* int64 */; + request_count: number /* int64 */; providers: ProviderSpend[]; } - +/** + * SpendByAgentUserResponse is the envelope for + * /_plugin/spend/by-agent-user. + */ export interface SpendByAgentUserResponse { window: string; results: AgentUserSpend[]; } - +/** + * HistogramPoint is one (timestamp, cost) datum in a per-dimension + * series. Timestamp is the bucket's start time in RFC3339. + */ export interface HistogramPoint { ts: string; - cost: number; + cost: number /* float64 */; } - +/** + * HistogramSeries is one line on a stacked-area chart. + */ export interface HistogramSeries { dimension_value: string; points: HistogramPoint[]; } - +/** + * HistogramCostResponse is the envelope for + * /_plugin/histogram/cost. + */ export interface HistogramCostResponse { - bucket_size_seconds: number; + bucket_size_seconds: number /* int64 */; dimension: string; series: HistogramSeries[]; } - +/** + * RunLogEntry is one row inside a RunDetailResponse's `logs` field. + * A trimmed view of Bifrost's Log — phase 8 only renders the columns + * the call-log table actually shows. + */ export interface RunLogEntry { id: string; timestamp: string; provider: string; model: string; status: string; - cost: number; - latency: number; - metadata: Record; + cost: number /* float64 */; + latency: number /* float64 */; + metadata: { [key: string]: string}; } - +/** + * RunStats is the aggregate-card summary at the top of /runs/:id. + */ export interface RunStats { - total_requests: number; - total_cost: number; - total_tokens: number; + total_requests: number /* int64 */; + total_cost: number /* float64 */; + total_tokens: number /* int64 */; } - +/** + * RunDetailResponse is the envelope for /_plugin/runs/:run_id. + */ export interface RunDetailResponse { run_id: string; logs: RunLogEntry[]; stats: RunStats; } - -// One message in a chat-style input_history / output_message. -// Mirrors Bifrost's `schemas.ChatMessage` to the depth the drawer -// renders: role, content (string OR an array of content blocks), -// optional tool-call list (assistant) and tool_call_id (tool reply). -// Everything is optional because providers vary in which fields -// they populate, and the drawer falls back to JSON for anything it -// doesn't recognize. -export interface ChatMessage { - role?: string; - name?: string; - /** OpenAI/Anthropic-style: either a plain string or an array of - * typed content blocks. */ - content?: string | ChatContentBlock[] | null; - /** Tool messages: which prior tool_call this is the result for. */ - tool_call_id?: string; - /** Assistant tool calls. */ - tool_calls?: ChatToolCall[]; - /** Anthropic / OpenAI reasoning summaries. */ - reasoning?: string; - refusal?: string; -} - -export interface ChatContentBlock { - type: string; - text?: string; - refusal?: string; - /** Anthropic-style cache marker. When present on a block the - * provider charged this block as a cache write (or read on a - * subsequent call). */ - cache_control?: { type?: string } | null; - cachePoint?: { type?: string } | null; - image_url?: unknown; - input_audio?: unknown; - file?: unknown; -} - -export interface ChatToolCall { - id?: string; - type?: string; - function?: { name?: string; arguments?: string }; -} - -/** Provider-reported usage breakdown. Source: Bifrost - * `schemas.BifrostLLMUsage`. The `prompt_tokens_details` sub- - * object is where the cache split lives — Anthropic populates - * `cached_write_tokens` (prompt cache writes) and - * `cached_read_tokens`; OpenAI uses `cached_read_tokens` only. */ -export interface TokenUsage { - prompt_tokens?: number; - completion_tokens?: number; - total_tokens?: number; - prompt_tokens_details?: { - text_tokens?: number; - audio_tokens?: number; - image_tokens?: number; - cached_read_tokens?: number; - cached_write_tokens?: number; - cached_write_token_details?: { - cached_write_tokens_5m?: number; - cached_write_tokens_1h?: number; - }; - }; - completion_tokens_details?: { - reasoning_tokens?: number; - accepted_prediction_tokens?: number; - }; - cost?: unknown; -} - -/** Semantic cache verdict. Source: `schemas.BifrostCacheDebug`. - * Distinct from prompt-cache token splits (those live in - * TokenUsage above). Present only when a semantic cache plugin - * is configured. */ -export interface CacheDebug { - cache_hit: boolean; - cache_id?: string; - hit_type?: string; - requested_provider?: string; - requested_model?: string; - provider_used?: string; - model_used?: string; - input_tokens?: number; - threshold?: number; - similarity?: number; -} - -// CallDetailResponse — /_plugin/runs/:run_id/calls/:call_id. -// -// The full body of one LLM call: same fields as a list row, plus -// the heavy JSON columns Bifrost strips from /api/logs. Chat-shaped -// fields (input_history / output_message) are typed so the drawer -// can render bubble UIs; everything else stays opaque. +/** + * CallDetailResponse is the envelope for + * /_plugin/runs/:run_id/calls/:call_id — the full request/response + * content for a single LLM call, fetched on-demand when the + * operator clicks a row in the RunDetail call log. + * Run-scoping rationale: we verify that the fetched log's + * metadata.run-id matches the URL's run_id before returning, so a + * caller can't enumerate other workspaces' logs by guessing IDs. + * Bifrost's /api/logs/{id} doesn't do this check itself (it just + * looks up by primary key), so the plugin enforces it. + * Body fields are pass-through json.RawMessage from Bifrost — the + * SPA pretty-prints them; the plugin doesn't introspect them. This + * keeps the schema coupling minimal: new fields upstream surface in + * the UI without code changes here. + */ export interface CallDetailResponse { id: string; run_id: string; @@ -218,295 +535,331 @@ export interface CallDetailResponse { provider: string; model: string; status: string; - cost: number; - latency: number; + cost: number /* float64 */; + latency: number /* float64 */; customer_id: string; - metadata: Record; - + metadata: { [key: string]: string}; + /** + * Per-request descriptors stamped by Bifrost. `stop_reason` + * tells the operator why the model stopped (stop, length, + * content_filter, tool_calls, refusal). `stream` flags + * streaming responses (which lack a single output_message and + * surface their content via Bifrost's stream chunk replay). + * Retries / fallback_index are zero on the happy path; non-zero + * means Bifrost had to retry the call or fall back to a + * different provider, which is useful provenance. + */ stop_reason?: string; stream: boolean; - number_of_retries: number; - fallback_index: number; - - token_usage?: TokenUsage; - cache_debug?: CacheDebug; - - input_history?: ChatMessage[]; - output_message?: ChatMessage; - params?: unknown; - tools?: unknown; - error_details?: unknown; + number_of_retries: number /* int */; + fallback_index: number /* int */; + /** + * TokenUsage is the provider-reported usage breakdown + * (BifrostLLMUsage). Includes prompt/completion/total totals + * plus cached read/write splits (Anthropic prompt-cache, + * OpenAI cached_tokens), audio/image token counts, and a + * per-call cost record. Pass-through JSON — the SPA introspects + * it. Bifrost's row-level prompt_tokens / completion_tokens / + * total_tokens columns are denormalized helpers tagged + * `json:"-"`, so this is the only place token data is on the + * wire. + */ + token_usage?: any /* json.RawMessage */; + /** + * CacheDebug carries Bifrost's *semantic* cache verdict for + * this call (hit/miss + similarity score). Distinct from + * prompt-cache tokens, which live in TokenUsage above. Absent + * when no semantic cache is configured for this swarm. + */ + cache_debug?: any /* json.RawMessage */; + /** + * All optional; missing on failures, realtime turns, or rows + * recorded before a given column existed. + */ + input_history?: any /* json.RawMessage */; + output_message?: any /* json.RawMessage */; + params?: any /* json.RawMessage */; + tools?: any /* json.RawMessage */; + error_details?: any /* json.RawMessage */; raw_request?: string; raw_response?: string; content_summary?: string; } -// One row of UserDetailResponse.AgentsUsed — which agents the user -// invoked in the window, and how much each cost. -export interface UserAgentUsage { - agent_name: string; - total_cost: number; - request_count: number; - last_seen?: string; -} +////////// +// source: ratelimit.go -// One row of UserDetailResponse.RecentRuns — links into RunDetail. -export interface UserRunSummary { - run_id: string; - agent_name: string; - total_cost: number; - request_count: number; - first_seen?: string; - last_seen?: string; -} -// /_plugin/users/:id response. -export interface UserDetailResponse { - user_id: string; - window: string; - total_cost: number; - request_count: number; - agents_used: UserAgentUsage[]; - recent_runs: UserRunSummary[]; - first_seen?: string; - last_seen?: string; -} +////////// +// source: revoke.go -// Trust-registry Org entry — mirrors gateway/internal/trust.Org. -// Surfaced on the dashboard's Provenance card so an operator can -// see which org's signature authorized a run, plus the pubkey / -// issuer URL the plugin would verify against. -export interface TrustOrg { - org_id: string; - pubkey: string; - issuer_url: string; - revocation_poll_seconds: number; - grace_pubkeys?: string[]; - grace_until?: string; -} - -// Trust-registry status — mirrors gateway/internal/trust.StatusResponse. -// The Provenance card on RunDetail uses `realm_id` to show the -// swarm's self-identity ("this run was processed by swarm w1"), -// since phase 11 dropped the per-row realm-id metadata column. -export interface TrustStatus { - claimed: boolean; - org_count: number; - orgs: string[]; - seed_source: "" | "env" | "api"; - last_modified: string; - /** Set on multi-swarm deployments; absent / empty on single-swarm. */ - realm_id?: string; -} - -// Per-agent budget (phase-8.5). Cap and the derived fields can be -// null when no budget is configured for the agent — the UI renders -// "no budget" rather than "$0". -export interface AgentBudgetResponse { - agent_name: string; - cap_usd: number | null; - window: string; - period_start?: string; - period_end?: string; - spent_usd: number; - remaining_usd: number | null; - ratio: number | null; +/** + * RevokeNonceRequest is the body for POST /_plugin/revoke/nonce/:nonce. + */ +export interface RevokeNonceRequest { + exp?: string; // RFC3339; layer expiry } - -// ─── agent catalog ────────────────────────────────────────────────── -// Mirrors gateway/internal/adminapi/catalog.go. The catalog answers -// "what is this agent _made of_?" (prompts/tools/skills), as opposed to -// the budget view's "what is it _allowed to spend_?". Sourced from the -// neo4j Hive* catalog subgraph via GET /_plugin/agents/:name/catalog. - -// A prompt linked to an agent. name/body come from the shared `:Prompt` -// node the agent links to (authored by the Stakwork prompt workflow); -// source/updated_at come from the HAS_PROMPT relationship. -export interface CatalogPrompt { - name: string; - body: string; - /** Prompt slot for this agent: "SYSTEM" or "USER" (the main/task - * prompt). Absent when the wiring source didn't classify it. */ - role?: string; - source: string; - updated_at: string; +/** + * RevokeNonceResponse is the wire shape for POST /_plugin/revoke/nonce/:nonce. + */ +export interface RevokeNonceResponse { + nonce: string; + expires_at: string; // RFC3339 UTC; when the tombstone lapses } - -export interface CatalogTool { - name: string; - description: string; - /** JSON parameter schema, passed through opaque — the UI renders it - * as pretty-printed JSON. Absent when the source didn't supply one. */ - schema?: unknown; - source: string; - version?: string; - /** Per-swarm operator toggle. Seeded enabled; preserved across - * Hive re-seeds. Flip via PATCH /_plugin/agents/:name/tools. */ - enabled: boolean; - updated_at: string; +/** + * RevokeUserRequest is the body for PUT /_plugin/revoke/user/:user_id. + */ +export interface RevokeUserRequest { + before?: string; // RFC3339; defaults to now } - -export interface CatalogSkill { - name: string; - description: string; - source: string; - version?: string; - /** Per-swarm operator toggle. Seeded enabled; preserved across - * Hive re-seeds. Flip via PATCH /_plugin/agents/:name/skills. */ - enabled: boolean; - updated_at: string; +/** + * RevokeUserResponse is the wire shape for PUT/GET /_plugin/revoke/user/:user_id. + */ +export interface RevokeUserResponse { + user_id: string; + before: string; // RFC3339 UTC } -// One row of the catalog list (GET /_plugin/agents/catalog) — identity -// + child counts, enough to merge the registry into the spend-derived -// /agents table without pulling every prompt/tool/skill body. -export interface CatalogAgentSummary { - name: string; - display_name?: string; - description?: string; - default_model?: string; - sources: string[]; - prompts: number; - tools: number; - skills: number; - updated_at: string; -} +////////// +// source: server.go +/* +Package adminapi is the gateway plugin's in-process HTTP server, +hosting the `/_plugin/*` route namespace. -// The whole registry — every catalog agent, traffic or not. -export interface CatalogListResponse { - agents: CatalogAgentSummary[]; -} +The wrapper binary (gateway/wrapper) reverse-proxies traffic on +`/_plugin/*` to this server (loopback only). Why a separate server +instead of routes registered through Bifrost's router: Bifrost +plugins (.so) cannot register arbitrary HTTP routes through +Bifrost's own router, and we also want routes that aren't behind +Bifrost's AuthMiddleware so Hive can bootstrap on a fresh swarm. -// Merged catalog view across all contributing sources for one agent. -export interface AgentCatalogResponse { - name: string; - display_name?: string; - description?: string; - /** Default LLM used for this agent (model shortcut or full id). */ - default_model?: string; - sources: string[]; - prompts: CatalogPrompt[]; - tools: CatalogTool[]; - skills: CatalogSkill[]; -} +Lifecycle +--------- + - Start() is called from the plugin's Init(). + - Stop() is called from the plugin's Cleanup() with a small grace + period. -// ─── Evals (agent-detail tab) ─────────────────────────────────────── -// Mirror of the Go structs in internal/adminapi/evals.go. Eval sets are -// Jarvis-authored nodes surfaced under an agent via HAS_EVAL_SET; the -// gateway reads them from neo4j and delegates writes/runs to Hive. +One server per process. Re-calling Start() is a no-op. +*/ -export interface EvalSetSummary { - ref_id: string; - name?: string; - description?: string; - requirements: number; -} -export interface AgentEvalsResponse { - agent: string; - sets: EvalSetSummary[]; -} +////////// +// source: session.go -export interface EvalTriggerSummary { - ref_id: string; - agent?: string; - source?: string; - environment?: string; - change_type?: string; - last_result?: string; // "pass" | "fail" | "" - last_score?: number; - last_notes?: string; - last_attempt?: number; -} -export interface EvalRequirementDetail { - ref_id: string; - name?: string; - description?: string; - prompt_snippet?: string; - order: number; - triggers: EvalTriggerSummary[]; -} +////////// +// source: sessions.go -export interface EvalSetDetailResponse { - ref_id: string; - name?: string; - description?: string; - requirements: EvalRequirementDetail[]; +/** + * SessionDetailResponse is the envelope for /_plugin/sessions/:id. + * `logs` is one page (?limit=, ?offset=, newest first); `stats` and + * `total_count` cover the whole session. + */ +export interface SessionDetailResponse { + session_id: string; + logs: RunLogEntry[]; + stats: RunStats; + total_count: number /* int64 */; +} +/** + * SessionSummaryResponse is the envelope for + * /_plugin/sessions/:id/summary. Timestamps are RFC3339; duration is + * latest − started, 0 for a single-call session. + */ +export interface SessionSummaryResponse { + session_id: string; + user_id?: string; + total_cost: number /* float64 */; + total_tokens: number /* int64 */; + request_count: number /* int64 */; + started_at?: string; + latest_at?: string; + duration_ms: number /* int64 */; + agents: string[]; + runs: string[]; +} + +////////// +// source: spend.go + +/** + * SessionSpend is one row of /_plugin/spend/by-session. A session is + * whatever the caller stamped on x-bf-dim-session-id (Hive's chat / + * workflow session); it spans runs, so `run_count` and the first / + * last timestamps are included for a Sessions list page. + */ +export interface SessionSpend { + session_id: string; + user_id: string; + total_cost: number /* float64 */; + total_tokens: number /* int64 */; + request_count: number /* int64 */; + run_count: number /* int64 */; + first_seen?: string; + last_seen?: string; } - -// Acknowledgement for create/link (ref_id of the set/requirement). -export interface EvalRefResponse { - ref_id: string; - linked?: boolean; +/** + * SpendBySessionResponse is the envelope for /_plugin/spend/by-session. + */ +export interface SpendBySessionResponse { + window: string; + results: SessionSpend[]; +} +/** + * ModelSpend is one row of /_plugin/spend/by-model, keyed on + * (provider, model) — the same model name can be served by two + * providers at different prices. + */ +export interface ModelSpend { + model: string; + provider: string; + total_cost: number /* float64 */; + total_tokens: number /* int64 */; + request_count: number /* int64 */; +} +/** + * SpendByModelResponse is the envelope for /_plugin/spend/by-model. + */ +export interface SpendByModelResponse { + window: string; + results: ModelSpend[]; } - -// Phase-7 error envelope (returned on 4xx/5xx). -export interface ApiError { - error: { - code: string; - message: string; - }; +/** + * AgentSpendResponse is the envelope for /_plugin/agents/:name/spend — + * one agent's totals over the window. + */ +export interface AgentSpendResponse { + agent_name: string; + window: string; + total_cost: number /* float64 */; + total_tokens: number /* int64 */; + request_count: number /* int64 */; } -// Window options the SPA exposes to the operator. Kept in lockstep -// with the Go-side validation in observability.go > parseWindow. -export type Window = "1h" | "6h" | "24h" | "7d" | "30d"; +////////// +// source: tickets.go -// Bucket options for the histogram endpoints — same source-of-truth -// note as Window. -export type Bucket = "1m" | "5m" | "10m" | "1h" | "6h" | "1d"; +/** + * TicketResponse is the JSON body returned by POST /_plugin/auth/ticket. + * Hive embeds the ticket in the iframe src as `?ticket=`. + */ +export interface TicketResponse { + ticket: string; + expires_in: number /* int */; // seconds; mirrors ticketTTL +} -// Dimension values the histogram endpoint accepts. Phase 11 removed -// `realm-id` — every row in a swarm's logs.db is implicitly for -// that swarm's realm, and the realm is surfaced on the trust-status -// card instead of as a per-row column. -export type Dimension = - | "agent-name" - | "run-id" - | "session-id" - | "user-id"; +////////// +// source: trust.go -// ─── hot state (phase-6 kill switches, phase-9 UI) ────────────────── -// Mirrors gateway/internal/adminapi/hotstate.go. Redis-backed live -// state: a run's phase-6 accumulators + kill flag, and an agent's -// current-bucket spend + kill flag. Every route 503s when the swarm -// has no Redis; the hooks in queries.ts fold that into `null` data. -// POST /_plugin/runs/:id/kill -export interface KillRunResponse { - run_id: string; - killed_at: string; // RFC3339 UTC -} +////////// +// source: users.go -// POST /_plugin/agents/:name/kill -export interface KillAgentResponse { +/** + * UserAgentUsage is one row in `UserDetailResponse.AgentsUsed` — + * a per-agent rollup scoped to this user's traffic in the window. + */ +export interface UserAgentUsage { agent_name: string; - killed_at: string; // RFC3339 UTC + total_cost: number /* float64 */; + request_count: number /* int64 */; + last_seen?: string; } - -// GET /_plugin/runs/:id/state — the run's live phase-6 accumulators. -// A run that has never made a call reads as all-zero with -// ttl_seconds = -2 (no key), not 404. -export interface RunStateResponse { +/** + * UserRunSummary is one row in `UserDetailResponse.RecentRuns`. + * Phase 8 keeps this lightweight — the dashboard renders cost + + * agent + first/last-seen and links into RunDetail for the full + * call log. + */ +export interface UserRunSummary { run_id: string; - cost_usd: number; - steps: number; - tools: string[]; // last 10 tool names, most recent first - killed: boolean; - /** Remaining lifetime of the cost accumulator: -2 when the run has - * no state yet, -1 when it has no expiry. */ - ttl_seconds: number; -} - -// GET /_plugin/agents/:name/state?window=1d -export interface AgentStateResponse { agent_name: string; + total_cost: number /* float64 */; + request_count: number /* int64 */; + first_seen?: string; + last_seen?: string; +} +/** + * UserDetailResponse is the wire shape for /_plugin/users/:id. + * Composition + * ----------- + * Everything is derived from a single paged scan of `logs.db` + * filtered by `metadata.user-id = ` (with a fallback to the + * indexed `customer_id` column on Bifrost's logs table, which + * equals the user-id per the v2 invariant). One round-trip to + * Bifrost, one in-memory aggregation pass; the dashboard renders + * the result without further fan-out. + * Once phase 6's PostLLMHook fills Redis cost accumulators, the + * `total_cost` field could be sourced from the Redis hash instead + * of summing logs — same number, less work. Phase 8 doesn't take + * that shortcut yet because the Redis bucket is per-(agent, day) + * not per-user. + */ +export interface UserDetailResponse { + user_id: string; window: string; - bucket_key: string; - current_spend_usd: number; - /** null when the agent has no agent_budgets entry; then `window` - * is informational (?window= or "1d"). */ - configured_cap_usd: number | null; + total_cost: number /* float64 */; + request_count: number /* int64 */; + agents_used: UserAgentUsage[]; + recent_runs: UserRunSummary[]; + first_seen?: string; + last_seen?: string; +} +/** + * UserSpendResponse is the envelope for /_plugin/users/:id/spend — + * the user's totals over the window, straight from Bifrost's + * SearchStats (one limit=1 call, no row paging). + */ +export interface UserSpendResponse { + user_id: string; + window: string; + total_cost: number /* float64 */; + total_tokens: number /* int64 */; + request_count: number /* int64 */; +} +/** + * UserQuotaResponse is the envelope for /_plugin/users/:id/quota: + * the user's Bifrost Customer budget (Hive's reconciler provisions + * one per workspace × user; llm-governance-v2.md "Hive as credential + * broker") blended with the runs the accumulator has indexed for + * them in Redis. + * Two independent degradations, both non-fatal: + * - no Customer for this id ⇒ customer_found=false, budget fields + * null (pre-reconciler traffic, or an id that was only ever a + * dim value). + * - Redis unavailable ⇒ redis_available=false, inflight_runs null. + * The budget window is whatever Bifrost's reset_duration says + * (Bifrost duration vocabulary); phase 7's sketch called it + * "daily", but the reconciler decides that, not the plugin. + */ +export interface UserQuotaResponse { + user_id: string; + customer_found: boolean; + budget_usd?: number /* float64 */; + budget_window?: string; + spent_usd: number /* float64 */; + remaining_usd?: number /* float64 */; + /** + * BudgetLastReset is Bifrost's last_reset (RFC3339); the next + * reset is last_reset + budget_window. + */ + budget_last_reset?: string; + redis_available: boolean; + inflight_runs: InflightRun[]; +} +/** + * InflightRun is one run in a user's quota view: the live + * accumulators and caps from Redis, as /runs/:id/state would report + * them. "In flight" means the macaroon layer hasn't expired; a run + * that finished early still lists until its exp passes. + */ +export interface InflightRun { + run_id: string; + agent_name?: string; + cost_usd: number /* float64 */; + steps: number /* int64 */; + max_cost_usd?: number /* float64 */; + max_steps?: number /* int64 */; + exp?: string; killed: boolean; } diff --git a/gateway/internal/adminapi/ui/src/api/window.ts b/gateway/internal/adminapi/ui/src/api/window.ts index 57e45d414..cff5b3979 100644 --- a/gateway/internal/adminapi/ui/src/api/window.ts +++ b/gateway/internal/adminapi/ui/src/api/window.ts @@ -2,7 +2,7 @@ // because the backend's parseWindow is the authority on this set; if // it grows a new option, this mapping needs to grow with it. -import type { Window } from "./types"; +import type { Window } from "./manual"; export function windowToSeconds(w: Window): number { switch (w) { diff --git a/gateway/internal/adminapi/ui/src/components/StatusBadge.tsx b/gateway/internal/adminapi/ui/src/components/StatusBadge.tsx index d76e2828b..005d81780 100644 --- a/gateway/internal/adminapi/ui/src/components/StatusBadge.tsx +++ b/gateway/internal/adminapi/ui/src/components/StatusBadge.tsx @@ -9,10 +9,13 @@ // killed state.killed. The kill key is set; the run's (or the // agent's runs') next LLM call is rejected when the swarm // has enforce_macaroons=true, logged otherwise. -// exceeded agents only: current_spend_usd >= configured_cap_usd. -// Runs carry their caps inside the macaroon, which /state -// doesn't surface — a run that hit its cap simply stops -// making calls and reads as "done". +// exceeded agents: current_spend_usd >= configured_cap_usd. +// runs: cost or steps at/over the macaroon layer's cap +// (/state surfaces max_cost_usd / max_steps from the +// accumulator's meta:run record), or any ancestor over +// its own cap — the cap walk rejects the child for that +// too. Only enforced when enforce_budgets=true; in shadow +// it is the operator's cue, not a hard stop. // running a call landed within RUN_ACTIVE_WINDOW_MS (either the // newest call-log row or the /state step counter moving // between polls). This is a heuristic: a run idling in a @@ -30,12 +33,15 @@ export const RUN_ACTIVE_WINDOW_MS = 5 * 60_000; export function deriveRunStatus(args: { killed: boolean; + /** Cost or steps at/over a cap on this run or an ancestor. */ + exceeded?: boolean; /** Epoch ms of the most recent evidence of activity, if any. */ lastActivityMs?: number; now?: number; }): Status { - const { killed, lastActivityMs, now = Date.now() } = args; + const { killed, exceeded = false, lastActivityMs, now = Date.now() } = args; if (killed) return "killed"; + if (exceeded) return "exceeded"; if ( lastActivityMs !== undefined && now - lastActivityMs < RUN_ACTIVE_WINDOW_MS diff --git a/gateway/internal/adminapi/ui/src/components/controls/WindowPicker.tsx b/gateway/internal/adminapi/ui/src/components/controls/WindowPicker.tsx index 7f0d26435..1d5aca09f 100644 --- a/gateway/internal/adminapi/ui/src/components/controls/WindowPicker.tsx +++ b/gateway/internal/adminapi/ui/src/components/controls/WindowPicker.tsx @@ -3,7 +3,7 @@ // set stays consistent (and stays in lockstep with the Go-side // parseWindow whitelist). -import type { Window } from "../../api/types"; +import type { Window } from "../../api/manual"; const OPTIONS: Window[] = ["1h", "24h", "7d", "30d"]; diff --git a/gateway/internal/adminapi/ui/src/pages/AgentDetail.tsx b/gateway/internal/adminapi/ui/src/pages/AgentDetail.tsx index 7fc443dbb..5562e0c92 100644 --- a/gateway/internal/adminapi/ui/src/pages/AgentDetail.tsx +++ b/gateway/internal/adminapi/ui/src/pages/AgentDetail.tsx @@ -38,7 +38,8 @@ import { useToggleTool, useUnkillAgent, } from "../api/queries"; -import type { HistogramCostResponse, Window } from "../api/types"; +import type { HistogramCostResponse } from "../api/types"; +import type { Window } from "../api/manual"; import { windowToSeconds } from "../api/window"; import { EvalsView } from "./EvalsView"; diff --git a/gateway/internal/adminapi/ui/src/pages/Agents.tsx b/gateway/internal/adminapi/ui/src/pages/Agents.tsx index dceab1566..2486f1079 100644 --- a/gateway/internal/adminapi/ui/src/pages/Agents.tsx +++ b/gateway/internal/adminapi/ui/src/pages/Agents.tsx @@ -18,7 +18,7 @@ import { useAgentStates, useSpendByAgent, } from "../api/queries"; -import type { Window } from "../api/types"; +import type { Window } from "../api/manual"; // A merged agent row: spend metrics (zeroed when registry-only) plus // catalog identity/counts (absent when traffic-only). diff --git a/gateway/internal/adminapi/ui/src/pages/Canvas.tsx b/gateway/internal/adminapi/ui/src/pages/Canvas.tsx index b1b10dd18..35d00c5af 100644 --- a/gateway/internal/adminapi/ui/src/pages/Canvas.tsx +++ b/gateway/internal/adminapi/ui/src/pages/Canvas.tsx @@ -34,7 +34,8 @@ import { useTrustOrg, useTrustStatus, } from "../api/queries"; -import type { AgentUserSpend, TrustOrg, Window } from "../api/types"; +import type { AgentUserSpend } from "../api/types"; +import type { TrustOrg, Window } from "../api/manual"; import { WindowPicker } from "../components/controls/WindowPicker"; import { canvasTheme, PROVIDER_DISPLAY, providerIcon } from "./canvasTheme"; diff --git a/gateway/internal/adminapi/ui/src/pages/Dashboard.tsx b/gateway/internal/adminapi/ui/src/pages/Dashboard.tsx index 94be8c10c..6e970a5bb 100644 --- a/gateway/internal/adminapi/ui/src/pages/Dashboard.tsx +++ b/gateway/internal/adminapi/ui/src/pages/Dashboard.tsx @@ -17,7 +17,8 @@ import { useSpendByAgent, useSpendByUser, } from "../api/queries"; -import type { AgentSpend, UserSpend, Window } from "../api/types"; +import type { AgentSpend, UserSpend } from "../api/types"; +import type { Window } from "../api/manual"; import { windowToSeconds } from "../api/window"; // LLM call costs can be fractions of a cent; clamping to 2 decimals diff --git a/gateway/internal/adminapi/ui/src/pages/People.tsx b/gateway/internal/adminapi/ui/src/pages/People.tsx index 091cdb976..7173254c0 100644 --- a/gateway/internal/adminapi/ui/src/pages/People.tsx +++ b/gateway/internal/adminapi/ui/src/pages/People.tsx @@ -14,7 +14,8 @@ import { WindowPicker } from "../components/controls/WindowPicker"; import { UserIcon } from "../components/icons"; import { getErrorMessage } from "../api/client"; import { useSpendByUser } from "../api/queries"; -import type { UserSpend, Window } from "../api/types"; +import type { UserSpend } from "../api/types"; +import type { Window } from "../api/manual"; const fmtUSD = (v: number) => { if (v === 0) return "$0.00"; diff --git a/gateway/internal/adminapi/ui/src/pages/RunDetail.tsx b/gateway/internal/adminapi/ui/src/pages/RunDetail.tsx index 82bdcf711..8ebbc66fd 100644 --- a/gateway/internal/adminapi/ui/src/pages/RunDetail.tsx +++ b/gateway/internal/adminapi/ui/src/pages/RunDetail.tsx @@ -26,15 +26,18 @@ import { useUnkillRun, } from "../api/queries"; import type { - CacheDebug, CallDetailResponse, + RunLogEntry, + RunStateResponse, +} from "../api/types"; +import type { + CacheDebug, ChatContentBlock, ChatMessage, ChatToolCall, - RunLogEntry, TokenUsage, TrustOrg, -} from "../api/types"; +} from "../api/manual"; interface Props { runID: string; @@ -374,7 +377,16 @@ function LiveStateCard({ (v): v is number => v !== undefined && !Number.isNaN(v), ); const lastActivityMs = evidence.length ? Math.max(...evidence) : undefined; - const status = deriveRunStatus({ killed: !!st?.killed, lastActivityMs, now }); + // Over cap on this run or any ancestor: the cap walk rejects the + // child for a parent's exhausted budget too. + const exceeded = + !!st && (overCap(st) || (st.ancestors ?? []).some((a) => overCap(a))); + const status = deriveRunStatus({ + killed: !!st?.killed, + exceeded, + lastActivityMs, + now, + }); useEffect(() => setInFlight(status === "running"), [status]); const unavailable = st === null; @@ -477,6 +489,7 @@ function LiveStateCard({ tone={st.killed ? "danger" : undefined} /> + {noState ? null : }
Last tools: @@ -501,6 +514,14 @@ function LiveStateCard({ catches the run's next call.
) : null} + {exceeded && !st.killed ? ( +
+ Over cap. The run's next LLM call is rejected + with 402 when the swarm has{" "} + enforce_budgets=true; in shadow mode + the overrun is logged and the call goes through. +
+ ) : null} {st.killed ? (
Kill flag set @@ -530,6 +551,160 @@ function LiveStateCard({ ); } +// ─── CapMeters ─────────────────────────────────────────────────────── +// +// "Cost vs max_cost_usd, steps vs max_steps, one meter per ancestor" +// (phase 9 "Run detail"). The caps come from the macaroon chain: the +// accumulator records each layer's caveats in meta:run: and +// /state serves them back as max_cost_usd / max_steps plus the +// ancestors walk. A null cap means the layer declared none — the +// meter cell says so rather than drawing an empty bar. Ancestors are +// nearest-parent first; each links to its own run page. + +type Capped = Pick< + RunStateResponse, + "run_id" | "cost_usd" | "steps" | "max_cost_usd" | "max_steps" +>; + +function overCap(c: Capped): boolean { + return ( + (c.max_cost_usd != null && c.cost_usd >= c.max_cost_usd) || + (c.max_steps != null && c.steps >= c.max_steps) + ); +} + +function meterTone(ratio: number): "ok" | "warning" | "danger" { + return ratio >= 1 ? "danger" : ratio >= 0.8 ? "warning" : "ok"; +} + +function CapMeters({ st }: { st: RunStateResponse }) { + const ancestors = st.ancestors ?? []; + const leafHasCap = st.max_cost_usd != null || st.max_steps != null; + if (!leafHasCap && ancestors.length === 0) { + return ( +
+ No caps recorded for this run — its macaroon declared neither + max_cost_usd nor max_steps. +
+ ); + } + return ( +
+ + {ancestors.map((a, i) => ( + + ))} +
+ ); +} + +function CapRow({ + who, + run, + link, + killed, +}: { + who: string; + run: Capped; + link?: boolean; + killed?: boolean; +}) { + return ( +
+
+ + {who} + {killed ? ( + <> + {" "} + · killed + + ) : null} + + {link ? ( + + + {run.run_id} + + + ) : ( + + {run.run_id} + + )} +
+ + +
+ ); +} + +function CapMeter({ + label, + value, + cap, + fmt, +}: { + label: string; + value: number; + cap: number | null | undefined; + fmt: (v: number) => string; +}) { + if (cap == null) { + return ( +
+
+ {label} + + {fmt(value)} · no cap + +
+ + ); + } + const ratio = cap > 0 ? value / cap : 0; + const pct = Math.min(100, Math.max(0, ratio * 100)); + const tone = meterTone(ratio); + return ( +
+
+ {label} + + {fmt(value)} of{" "} + {fmt(cap)} ({(ratio * 100).toFixed(0)}%) + +
+
+
+
+
+ ); +} + function Figure({ label, value, diff --git a/gateway/internal/adminapi/ui/src/pages/UserDetail.tsx b/gateway/internal/adminapi/ui/src/pages/UserDetail.tsx index d271ebf68..f026177a8 100644 --- a/gateway/internal/adminapi/ui/src/pages/UserDetail.tsx +++ b/gateway/internal/adminapi/ui/src/pages/UserDetail.tsx @@ -22,7 +22,8 @@ import { WindowPicker } from "../components/controls/WindowPicker"; import { BotIcon, UserIcon } from "../components/icons"; import { getErrorMessage } from "../api/client"; import { useHistogramCost, useUserDetail } from "../api/queries"; -import type { UserAgentUsage, UserRunSummary, Window } from "../api/types"; +import type { UserAgentUsage, UserRunSummary } from "../api/types"; +import type { Window } from "../api/manual"; import { windowToSeconds } from "../api/window"; interface Props { diff --git a/gateway/internal/adminapi/ui/src/styles/components.css b/gateway/internal/adminapi/ui/src/styles/components.css index c9b1cbbb2..b4e58a75b 100644 --- a/gateway/internal/adminapi/ui/src/styles/components.css +++ b/gateway/internal/adminapi/ui/src/styles/components.css @@ -1502,3 +1502,61 @@ h1 .badge { .hotstate-note .mono { color: var(--text); } + +/* Cap meters (phase-9 "cost vs max_cost_usd, steps vs max_steps"): + * one row per budgeted run in the chain — the run itself first, then + * each ancestor outward. Reuses the budget-meter bar. */ +.capmeters { + margin-top: var(--sp-4); + display: grid; + gap: var(--sp-3); +} +.capmeter-run { + display: grid; + grid-template-columns: minmax(120px, 1fr) 3fr 3fr; + gap: var(--sp-4); + align-items: center; +} +.capmeter-run.is-ancestor { + opacity: 0.85; +} +.capmeter-who { + font-size: 12px; + color: var(--text-muted); + display: flex; + flex-direction: column; + gap: 2px; + min-width: 0; +} +.capmeter-who .mono { + color: var(--text); + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} +.capmeter { + display: grid; + gap: var(--sp-1); +} +.capmeter-figures { + display: flex; + justify-content: space-between; + font-size: 12px; + color: var(--text-muted); +} +.capmeter-figures .mono { + color: var(--text); +} +.capmeter-figures .tone-warning { + color: var(--warning); +} +.capmeter-figures .tone-danger { + color: var(--danger); +} +.capmeter .budget-meter { + width: 100%; +} +.capmeter-none { + font-size: 12px; + color: var(--text-muted); +} diff --git a/gateway/internal/adminapi/users.go b/gateway/internal/adminapi/users.go index 4c5ab0bc6..da731e4dc 100644 --- a/gateway/internal/adminapi/users.go +++ b/gateway/internal/adminapi/users.go @@ -1,10 +1,14 @@ package adminapi import ( + "errors" "net/http" "sort" "strings" "time" + + "github.com/stakwork/stakgraph/gateway/internal/auth" + "github.com/stakwork/stakgraph/gateway/internal/pluginlog" ) // UserAgentUsage is one row in `UserDetailResponse.AgentsUsed` — @@ -56,12 +60,73 @@ type UserDetailResponse struct { LastSeen string `json:"last_seen,omitempty"` } -// userDetail handles `GET /_plugin/users/:user_id`. +// UserSpendResponse is the envelope for /_plugin/users/:id/spend — +// the user's totals over the window, straight from Bifrost's +// SearchStats (one limit=1 call, no row paging). +type UserSpendResponse struct { + UserID string `json:"user_id"` + Window string `json:"window"` + TotalCost float64 `json:"total_cost"` + TotalTokens int64 `json:"total_tokens"` + RequestCount int64 `json:"request_count"` +} + +// UserQuotaResponse is the envelope for /_plugin/users/:id/quota: +// the user's Bifrost Customer budget (Hive's reconciler provisions +// one per workspace × user; llm-governance-v2.md "Hive as credential +// broker") blended with the runs the accumulator has indexed for +// them in Redis. +// +// Two independent degradations, both non-fatal: +// - no Customer for this id ⇒ customer_found=false, budget fields +// null (pre-reconciler traffic, or an id that was only ever a +// dim value). +// - Redis unavailable ⇒ redis_available=false, inflight_runs null. +// +// The budget window is whatever Bifrost's reset_duration says +// (Bifrost duration vocabulary); phase 7's sketch called it +// "daily", but the reconciler decides that, not the plugin. +type UserQuotaResponse struct { + UserID string `json:"user_id"` + CustomerFound bool `json:"customer_found"` + BudgetUSD *float64 `json:"budget_usd"` + BudgetWindow string `json:"budget_window,omitempty"` + SpentUSD float64 `json:"spent_usd"` + RemainingUSD *float64 `json:"remaining_usd"` + // BudgetLastReset is Bifrost's last_reset (RFC3339); the next + // reset is last_reset + budget_window. + BudgetLastReset string `json:"budget_last_reset,omitempty"` + RedisAvailable bool `json:"redis_available"` + InflightRuns []InflightRun `json:"inflight_runs"` +} + +// InflightRun is one run in a user's quota view: the live +// accumulators and caps from Redis, as /runs/:id/state would report +// them. "In flight" means the macaroon layer hasn't expired; a run +// that finished early still lists until its exp passes. +type InflightRun struct { + RunID string `json:"run_id"` + AgentName string `json:"agent_name,omitempty"` + CostUSD float64 `json:"cost_usd"` + Steps int64 `json:"steps"` + MaxCostUSD *float64 `json:"max_cost_usd"` + MaxSteps *int64 `json:"max_steps"` + Exp string `json:"exp,omitempty"` + Killed bool `json:"killed"` +} + +// maxInflightRuns caps the quota view's run list. The index is +// pruned by exp on read, so this only matters for a user driving an +// unusual number of concurrent runs. +const maxInflightRuns = 50 + +// userDetail dispatches the /_plugin/users/ subtree: // -// Path parsing follows the same subtree convention runDetail uses — -// the trailing segment is the user_id, anything deeper 404s so -// future endpoints like `/_plugin/users/:id/quota` can land -// alongside without ambiguity. +// /_plugin/users/{user_id} → the phase-8 rollup below +// /_plugin/users/{user_id}/spend → userSpend +// /_plugin/users/{user_id}/quota → userQuota +// +// Anything else 404s. func (h *observabilityHandlers) userDetail(w http.ResponseWriter, r *http.Request) { if r.Method != http.MethodGet { methodNotAllowed(w, http.MethodGet) @@ -69,12 +134,109 @@ func (h *observabilityHandlers) userDetail(w http.ResponseWriter, r *http.Reques } const prefix = "/_plugin/users/" rest := strings.TrimPrefix(r.URL.Path, prefix) - if rest == "" || strings.ContainsRune(rest, '/') { + parts := strings.Split(rest, "/") + switch { + case len(parts) == 1 && parts[0] != "": + h.userRollup(w, r, parts[0]) + case len(parts) == 2 && parts[0] != "" && parts[1] == "spend": + h.userSpend(w, r, parts[0]) + case len(parts) == 2 && parts[0] != "" && parts[1] == "quota": + h.userQuota(w, r, parts[0]) + default: http.NotFound(w, r) + } +} + +// userSpend handles `GET /_plugin/users/:user_id/spend`. Filters on +// metadata.user-id, not Bifrost's customer_id column — see the +// source-of-truth note on spendByUser. +func (h *observabilityHandlers) userSpend(w http.ResponseWriter, r *http.Request, userID string) { + window, start, end, ok := parseWindow(w, r) + if !ok { + return + } + res, err := h.logs.search(r.Context(), searchOpts{ + StartTime: &start, + EndTime: &end, + Metadata: map[string]string{"user-id": userID}, + Limit: 1, + }) + if err != nil { + writeUpstreamError(w, err, "users.spend") + return + } + writeJSON(w, http.StatusOK, UserSpendResponse{ + UserID: userID, + Window: window, + TotalCost: res.Stats.TotalCost, + TotalTokens: res.Stats.TotalTokens, + RequestCount: res.Stats.TotalRequests, + }) +} + +// userQuota handles `GET /_plugin/users/:user_id/quota`. +func (h *observabilityHandlers) userQuota(w http.ResponseWriter, r *http.Request, userID string) { + cust, err := h.logs.customer(r.Context(), userID) + if err != nil { + writeUpstreamError(w, err, "users.quota") return } - userID := rest + out := UserQuotaResponse{UserID: userID} + if cust != nil { + out.CustomerFound = true + // Hive provisions one budget per customer; if there are + // several, the first is the one the reconciler wrote. + if len(cust.Budgets) > 0 { + b := cust.Budgets[0] + cap := b.MaxLimit + remaining := b.MaxLimit - b.CurrentUsage + if remaining < 0 { + remaining = 0 + } + out.BudgetUSD = &cap + out.BudgetWindow = b.ResetDuration + out.SpentUSD = b.CurrentUsage + out.RemainingUSD = &remaining + out.BudgetLastReset = b.LastReset + } + } + // Live portion. Any Redis failure degrades to "unavailable" + // rather than failing the budget half — the spec's contract for + // this endpoint is "return the logs-derived portion". + now := time.Now().UTC() + ids, err := auth.ListUserRuns(r.Context(), userID, now, maxInflightRuns) + if err != nil { + if !errors.Is(err, auth.ErrRedisUnavailable) { + pluginlog.Warnf("adminapi: users.quota: run index for %s: %v", userID, err) + } + writeJSON(w, http.StatusOK, out) + return + } + out.RedisAvailable = true + out.InflightRuns = make([]InflightRun, 0, len(ids)) + for _, id := range ids { + st, err := auth.GetRunState(r.Context(), id) + if err != nil { + pluginlog.Warnf("adminapi: users.quota: run state %s: %v", id, err) + continue + } + out.InflightRuns = append(out.InflightRuns, InflightRun{ + RunID: st.RunID, + AgentName: st.AgentName, + CostUSD: st.CostUSD, + Steps: st.Steps, + MaxCostUSD: capUSD(st), + MaxSteps: capSteps(st), + Exp: st.Exp, + Killed: st.Killed, + }) + } + writeJSON(w, http.StatusOK, out) +} + +// userRollup is the phase-8 `GET /_plugin/users/:user_id` body. +func (h *observabilityHandlers) userRollup(w http.ResponseWriter, r *http.Request, userID string) { window, start, end, ok := parseWindow(w, r) if !ok { return @@ -210,12 +372,5 @@ func (h *observabilityHandlers) userDetail(w http.ResponseWriter, r *http.Reques out.RecentRuns = out.RecentRuns[:50] } - // Used-but-not-needed: time package import is for sort - // determinism on equal timestamps. Not currently relied on, - // but keeping the import slot pinned so future enhancements - // (e.g. parsing FirstSeen for "active duration") don't shuffle - // the file diff. - _ = time.Time{} - writeJSON(w, http.StatusOK, out) } diff --git a/gateway/internal/auth/accumulator.go b/gateway/internal/auth/accumulator.go index 671652e31..96d517145 100644 --- a/gateway/internal/auth/accumulator.go +++ b/gateway/internal/auth/accumulator.go @@ -4,6 +4,8 @@ import ( "context" "time" + "github.com/redis/go-redis/v9" + macaroon "github.com/stakwork/stakgraph/gateway/auth/go" "github.com/stakwork/stakgraph/gateway/internal/duration" "github.com/stakwork/stakgraph/gateway/internal/pluginlog" @@ -19,8 +21,32 @@ const ( toolsRunPrefix = "tools:run:" // LIST, capped at toolHistoryLen costUAPrefix = "cost:ua:" // HASH { total: float } costAgentPrefix = "cost:agent:" // HASH { total: float }, key + ":" + bucket + + // metaRunPrefix + is the run's static shape as the + // macaroon chain declared it: the caps the cap walk compares + // against, the layer's expiry, and the parent run (the next + // layer outward, "" for the invocation). Written alongside the + // accumulators so the operator UI can render "spent $X of $Y" and + // walk ancestors without a macaroon in hand (phase 9 "cap + // meters"). `agent` / `user` are only stamped by the run's own + // calls (the leaf layer) — an ancestor's agent isn't in the + // child's chain. HASH { max_cost_usd, max_steps, exp, parent, + // agent, user }; same TTL axis as cost:run. + metaRunPrefix = "meta:run:" + + // runsUserPrefix + indexes the runs a user has driven + // through this gateway, for /_plugin/users/:id/quota's in-flight + // list (phase 7). ZSET member = leaf run_id, score = that layer's + // exp (unix seconds) so readers can prune expired runs by score. + // The key itself lives runsUserTTL past the last write. + runsUserPrefix = "runs:user:" ) +// runsUserTTL is the per-user run index's key expiry, refreshed on +// every write. Matches the 7d ceiling on the per-run keys: after a +// week of silence there is no run state left to point at anyway. +const runsUserTTL = 7 * 24 * time.Hour + // toolHistoryLen is the tool-loop detection window: tools:run keeps // the last N tool names (LPUSH + LTRIM 0 N-1). const toolHistoryLen = 10 @@ -38,6 +64,11 @@ const toolHistoryLen = 10 // the leaf agent has a configured windowed budget. // - LPUSH/LTRIM tools:run: + EXPIRE when the response // contained tool calls. +// - HSET meta:run: (caps, exp, parent link; agent + user on the +// leaf) + EXPIRE for every layer, so the operator UI can render +// cap meters and walk ancestors from Redis alone. +// - ZADD runs:user: + EXPIRE, the +// per-user in-flight run index behind /_plugin/users/:id/quota. // // Fire-and-forget per the phase-6 failure-mode contract: accounting // fails open. The pipeline runs on a goroutine with its own timeout; @@ -92,20 +123,53 @@ func accumulate( // accumulator must outlive the short-lived leaf that wrote it. // TTL refreshes on every write, so an actively-spending run // keeps its keys alive for its whole lifetime. - seen := make(map[string]bool, len(claims.Chain)) + // + // The same pass stamps meta:run: for each layer: caps, exp + // and the parent link (the previous distinct layer). Idempotent — + // a layer's caveats never change for a given run_id, so re-HSET + // on every call just refreshes the value. The leaf additionally + // records its agent and user, which is what the per-user run + // index and the quota view key on. + layers := chainLayers(claims) leafTTL := runKeyTTL(parseRFC3339(claims.EffectiveCaveats.Exp), now) - for _, layer := range claims.Chain { - if layer.RunID == "" || seen[layer.RunID] { - continue - } - seen[layer.RunID] = true + parent := "" + for i, layer := range layers { ttl := runKeyTTL(parseRFC3339(layer.Exp), now) costKey := redisclient.Key(costRunPrefix + layer.RunID) stepsKey := redisclient.Key(stepsRunPrefix + layer.RunID) + metaKey := redisclient.Key(metaRunPrefix + layer.RunID) pipe.HIncrByFloat(ctx, costKey, "total", costUSD) pipe.HIncrBy(ctx, stepsKey, "total", 1) pipe.Expire(ctx, costKey, ttl) pipe.Expire(ctx, stepsKey, ttl) + + meta := []any{ + "max_cost_usd", layer.MaxCostUSD, + "max_steps", layer.MaxSteps, + "exp", layer.Exp, + "parent", parent, + } + if i == len(layers)-1 { + meta = append(meta, "agent", claims.AgentName, "user", claims.UserID) + } + pipe.HSet(ctx, metaKey, meta...) + pipe.Expire(ctx, metaKey, ttl) + parent = layer.RunID + } + + // Per-user run index (leaf only). Score is the leaf's exp so a + // reader can drop runs whose macaroon has lapsed without a + // second lookup; a missing / unparseable exp scores as "now + + // key TTL" so the run still shows up until its state expires. + if n := len(layers); n > 0 && claims.UserID != "" { + leaf := layers[n-1] + score := parseRFC3339(leaf.Exp) + if score.IsZero() { + score = now.Add(leafTTL) + } + userKey := redisclient.Key(runsUserPrefix + claims.UserID) + pipe.ZAdd(ctx, userKey, redis.Z{Score: float64(score.Unix()), Member: leaf.RunID}) + pipe.Expire(ctx, userKey, runsUserTTL) } // UA cumulative envelope — only when the org actually set one, @@ -155,3 +219,33 @@ func accumulate( _, err := pipe.Exec(ctx) return err } + +// chainLayers returns the chain's distinct run layers outermost-first +// (invocation, then each attenuation inward; the last element is the +// leaf). Mirrors capLayers' dedup and its fallback: when Chain is +// empty (older callers, tests) the leaf is synthesized from +// Claims.RunID + EffectiveCaveats so accounting and the cap walk see +// the same set of runs. +func chainLayers(claims *macaroon.Claims) []macaroon.ChainLayer { + if len(claims.Chain) == 0 { + if claims.RunID == "" { + return nil + } + return []macaroon.ChainLayer{{ + RunID: claims.RunID, + MaxCostUSD: claims.EffectiveCaveats.MaxCostUSD, + MaxSteps: claims.EffectiveCaveats.MaxSteps, + Exp: claims.EffectiveCaveats.Exp, + }} + } + out := make([]macaroon.ChainLayer, 0, len(claims.Chain)) + seen := make(map[string]bool, len(claims.Chain)) + for _, l := range claims.Chain { + if l.RunID == "" || seen[l.RunID] { + continue + } + seen[l.RunID] = true + out = append(out, l) + } + return out +} diff --git a/gateway/internal/auth/accumulator_test.go b/gateway/internal/auth/accumulator_test.go index b65ef2f6a..ed315f120 100644 --- a/gateway/internal/auth/accumulator_test.go +++ b/gateway/internal/auth/accumulator_test.go @@ -5,6 +5,8 @@ import ( "testing" "time" + "github.com/alicebob/miniredis/v2" + macaroon "github.com/stakwork/stakgraph/gateway/auth/go" ) @@ -154,7 +156,8 @@ func TestAccumulate_AgentBucket_OnlyWhenConfigured(t *testing.T) { if err := accumulate(context.Background(), claims, 0.30, nil, testNow()); err != nil { t.Fatal(err) } - if keys := mr.Keys(); len(keys) != 2 { // cost:run + steps:run only + // cost:run + steps:run + meta:run + runs:user — no cost:agent. + if keys := mr.Keys(); len(keys) != 4 || mr.Exists("bifrost:cost:agent:coder:2026-05-14") { t.Fatalf("unexpected keys without agent budget: %v", keys) } @@ -255,3 +258,139 @@ func TestAccumulate_UAEnvelope_WrittenForRealmCap(t *testing.T) { t.Fatal("cost:ua written for a realm cap that isn't this swarm's") } } + +func TestAccumulate_RunMetaAndUserIndex(t *testing.T) { + mr := newMiniRedis(t) + claims := chainClaims( + []string{"r_parent", "r_child"}, + []string{"2026-05-14T18:00:00Z", "2026-05-14T10:30:00Z"}, + ) + claims.Chain[0].MaxCostUSD = 20 + claims.Chain[0].MaxSteps = 500 + claims.Chain[1].MaxCostUSD = 2.5 + claims.Chain[1].MaxSteps = 40 + + if err := accumulate(context.Background(), claims, 0.10, nil, testNow()); err != nil { + t.Fatal(err) + } + + // Leaf: caps, exp, parent link, and — only here — agent/user. + leaf := hgetall(mr, "bifrost:meta:run:r_child") + want := map[string]string{ + "max_cost_usd": "2.5", "max_steps": "40", + "exp": "2026-05-14T10:30:00Z", "parent": "r_parent", + "agent": "coder", "user": testUserID, + } + for k, v := range want { + if leaf[k] != v { + t.Errorf("meta:run:r_child[%s] = %q, want %q", k, leaf[k], v) + } + } + + // Ancestor: its own caps, no parent (invocation), and no agent — + // the child's chain doesn't know what agent the parent runs as. + par := hgetall(mr, "bifrost:meta:run:r_parent") + if par["max_cost_usd"] != "20" || par["max_steps"] != "500" || par["parent"] != "" { + t.Errorf("meta:run:r_parent = %v", par) + } + if _, ok := par["agent"]; ok { + t.Errorf("ancestor meta must not carry the leaf's agent: %v", par) + } + + // Meta shares the accumulator's TTL axis per layer. + if got, want := mr.TTL("bifrost:meta:run:r_child"), mr.TTL("bifrost:cost:run:r_child"); got != want { + t.Errorf("meta:run:r_child ttl %v != cost:run ttl %v", got, want) + } + + // Per-user index: leaf run scored by its exp, key TTL'd at 7d. + members, err := mr.ZMembers("bifrost:runs:user:" + testUserID) + if err != nil || len(members) != 1 || members[0] != "r_child" { + t.Fatalf("runs:user members = %v (%v)", members, err) + } + score, err := mr.ZScore("bifrost:runs:user:"+testUserID, "r_child") + if err != nil || int64(score) != parseRFC3339("2026-05-14T10:30:00Z").Unix() { + t.Errorf("runs:user score = %v (%v)", score, err) + } + if got := mr.TTL("bifrost:runs:user:" + testUserID); got != runsUserTTL { + t.Errorf("runs:user ttl = %v, want %v", got, runsUserTTL) + } +} + +func TestAccumulate_NoChainSynthesizesLeafMeta(t *testing.T) { + mr := newMiniRedis(t) + claims := chainClaims([]string{"r_solo"}, []string{"2026-05-14T12:00:00Z"}) + claims.Chain = nil // older callers: leaf comes from RunID + EffectiveCaveats + + if err := accumulate(context.Background(), claims, 0.25, nil, testNow()); err != nil { + t.Fatal(err) + } + if got := mr.HGet("bifrost:cost:run:r_solo", "total"); got != "0.25" { + t.Errorf("cost:run:r_solo = %q", got) + } + meta := hgetall(mr, "bifrost:meta:run:r_solo") + if meta["max_cost_usd"] != "5" || meta["max_steps"] != "100" || meta["agent"] != "coder" { + t.Errorf("meta:run:r_solo = %v", meta) + } +} + +func TestGetRunState_ReadsMeta_And_ListUserRuns(t *testing.T) { + mr := newMiniRedis(t) + claims := chainClaims( + []string{"r_parent", "r_child"}, + []string{"2026-05-14T18:00:00Z", "2026-05-14T10:30:00Z"}, + ) + if err := accumulate(context.Background(), claims, 0.4, nil, testNow()); err != nil { + t.Fatal(err) + } + + st, err := GetRunState(context.Background(), "r_child") + if err != nil { + t.Fatal(err) + } + if !st.HasMeta || st.MaxCostUSD != 5 || st.MaxSteps != 100 || st.Parent != "r_parent" || + st.AgentName != "coder" || st.UserID != testUserID || st.Exp != "2026-05-14T10:30:00Z" { + t.Errorf("run state: %+v", st) + } + par, err := GetRunState(context.Background(), "r_parent") + if err != nil { + t.Fatal(err) + } + if !par.HasMeta || par.Parent != "" || par.AgentName != "" || par.CostUSD != 0.4 { + t.Errorf("parent state: %+v", par) + } + + // Never-seen run: no meta, zero caps. + none, err := GetRunState(context.Background(), "r_never") + if err != nil || none.HasMeta { + t.Errorf("unknown run: %+v (%v)", none, err) + } + + // Index lists the leaf while its exp is in the future ... + ids, err := ListUserRuns(context.Background(), testUserID, testNow(), 10) + if err != nil || len(ids) != 1 || ids[0] != "r_child" { + t.Fatalf("ListUserRuns = %v (%v)", ids, err) + } + // ... prunes it once the exp has passed ... + ids, err = ListUserRuns(context.Background(), testUserID, testNow().Add(2*time.Hour), 10) + if err != nil || len(ids) != 0 { + t.Fatalf("ListUserRuns after exp = %v (%v)", ids, err) + } + if members, _ := mr.ZMembers("bifrost:runs:user:" + testUserID); len(members) != 0 { + t.Errorf("expired member not pruned: %v", members) + } + // ... and an unknown user is an empty list, not an error. + ids, err = ListUserRuns(context.Background(), "u_nobody", testNow(), 10) + if err != nil || ids == nil || len(ids) != 0 { + t.Fatalf("ListUserRuns unknown user = %v (%v)", ids, err) + } +} + +// hgetall is the HGETALL miniredis doesn't expose on its handle. +func hgetall(mr *miniredis.Miniredis, key string) map[string]string { + out := map[string]string{} + keys, _ := mr.HKeys(key) + for _, k := range keys { + out[k] = mr.HGet(key, k) + } + return out +} diff --git a/gateway/internal/auth/kill.go b/gateway/internal/auth/kill.go index 7f94e0d91..6b89ef003 100644 --- a/gateway/internal/auth/kill.go +++ b/gateway/internal/auth/kill.go @@ -4,6 +4,7 @@ import ( "context" "errors" "fmt" + "strconv" "strings" "time" @@ -34,7 +35,12 @@ const ( killAgentTTL = 24 * time.Hour ) -// RunState is a snapshot of one run's phase-6 Redis accumulators. +// RunState is a snapshot of one run's phase-6 Redis accumulators, +// plus the static shape the accumulator recorded in meta:run: +// (caps, expiry, parent, and — for the run's own calls — agent and +// user). HasMeta is false when the run has never been accounted +// through this gateway; the cap fields are then zero and mean +// "unknown", not "uncapped". type RunState struct { RunID string CostUSD float64 // HGET cost:run: total (0 when absent) @@ -42,6 +48,14 @@ type RunState struct { Tools []string // LRANGE tools:run: 0 9, most recent first Killed bool // EXISTS kill: TTLSeconds int64 // TTL cost:run:; -2 when the key is absent, -1 when no expiry + + HasMeta bool // meta:run: present + MaxCostUSD float64 // macaroon layer cap; 0 = no cap + MaxSteps int64 // macaroon layer cap; 0 = no cap + Exp string // layer exp, RFC3339 + Parent string // next layer outward; "" for the invocation + AgentName string // leaf agent (only stamped by the run's own calls) + UserID string // user the macaroon was issued to } // AgentState is a snapshot of one agent's current-bucket spend and @@ -136,10 +150,21 @@ func GetRunState(ctx context.Context, runID string) (RunState, error) { toolsCmd := pipe.LRange(octx, redisclient.Key(toolsRunPrefix+runID), 0, toolHistoryLen-1) killCmd := pipe.Exists(octx, redisclient.Key(killRunPrefix+runID)) ttlCmd := pipe.TTL(octx, costKey) + metaCmd := pipe.HGetAll(octx, redisclient.Key(metaRunPrefix+runID)) if _, err := pipe.Exec(octx); err != nil && !errors.Is(err, redis.Nil) { return st, fmt.Errorf("redis pipeline: %w", err) } + if meta, err := metaCmd.Result(); err == nil && len(meta) > 0 { + st.HasMeta = true + st.MaxCostUSD, _ = strconv.ParseFloat(meta["max_cost_usd"], 64) + st.MaxSteps, _ = strconv.ParseInt(meta["max_steps"], 10, 64) + st.Exp = meta["exp"] + st.Parent = meta["parent"] + st.AgentName = meta["agent"] + st.UserID = meta["user"] + } + if v, err := costCmd.Float64(); err == nil { st.CostUSD = v } else if !errors.Is(err, redis.Nil) { @@ -233,3 +258,42 @@ func validateKillID(field, v string) error { } return nil } + +// ListUserRuns returns the run_ids the accumulator indexed under +// runs:user: whose macaroon layer had not expired at `now`, +// newest-expiring first, capped at `limit`. Expired members are +// pruned on the way out so the index stays bounded without a +// separate sweeper. Absent key ⇒ empty list, not an error. +func ListUserRuns(ctx context.Context, userID string, now time.Time, limit int64) ([]string, error) { + if err := validateKillID("user_id", userID); err != nil { + return nil, err + } + rdb := redisclient.Client() + if rdb == nil { + return nil, ErrRedisUnavailable + } + if limit <= 0 { + limit = 50 + } + octx, cancel := context.WithTimeout(ctx, adminTimeout) + defer cancel() + + key := redisclient.Key(runsUserPrefix + userID) + nowScore := strconv.FormatInt(now.Unix(), 10) + pipe := rdb.Pipeline() + pipe.ZRemRangeByScore(octx, key, "-inf", "("+nowScore) + listCmd := pipe.ZRevRangeByScore(octx, key, &redis.ZRangeBy{ + Min: nowScore, Max: "+inf", Offset: 0, Count: limit, + }) + if _, err := pipe.Exec(octx); err != nil && !errors.Is(err, redis.Nil) { + return nil, fmt.Errorf("redis pipeline: %w", err) + } + ids, err := listCmd.Result() + if err != nil && !errors.Is(err, redis.Nil) { + return nil, fmt.Errorf("runs:user: %w", err) + } + if ids == nil { + ids = []string{} + } + return ids, nil +} diff --git a/gateway/internal/duration/duration.go b/gateway/internal/duration/duration.go index 77a69133a..0f16cb555 100644 --- a/gateway/internal/duration/duration.go +++ b/gateway/internal/duration/duration.go @@ -136,6 +136,12 @@ func (w Window) TTL() time.Duration { return 2 * w.length() } +// Length is the window's span as a plain duration, using the same +// 30d / 365d approximations for months and years as TTL. Rolling +// analytics windows ("the last 1d of calls") subtract this from now; +// calendar bucketing goes through Bounds instead. +func (w Window) Length() time.Duration { return w.length() } + func (w Window) length() time.Duration { n := time.Duration(w.N) switch w.Unit { diff --git a/gateway/plans/phases/phase-7-observability.md b/gateway/plans/phases/phase-7-observability.md index df60bb3d3..5a091e40f 100644 --- a/gateway/plans/phases/phase-7-observability.md +++ b/gateway/plans/phases/phase-7-observability.md @@ -1,7 +1,7 @@ # Phase 7 — Observability: Per-Dim Analytics over `logs.db` > Read-only HTTP surface that exposes per-agent / per-session / -> per-user / per-realm spend and usage analytics out of Bifrost's +> per-user / per-model spend and usage analytics out of Bifrost's > own log store, via the dim headers the plugin canonicalizes in > `PreLLMHook`. Companion to `phase-6-plugin-enforcement.md` (which > defines the kill/state admin endpoints over the same `/_plugin/*` @@ -122,17 +122,34 @@ GET /_plugin/spend/by-agent?window=24h results: [ { agent_name, total_cost, total_tokens, request_count }, ... ] } -GET /_plugin/spend/by-realm?window=24h - Same shape, dim = realm-id - GET /_plugin/spend/by-session?window=24h - Same shape, dim = session-id + Same strategy, dim = session-id. A session spans runs, so each row + also carries the user, a run count, and the first/last timestamps + for a Sessions list page. + returns: { + window, + results: [ { session_id, user_id, total_cost, total_tokens, request_count, + run_count, first_seen, last_seen }, ... ] + } GET /_plugin/spend/by-model?window=24h - Bifrost API: GetModelRankings — first-class on the LogStore interface - returns: { window, results: [ { model, provider, total_cost, request_count }, ... ] } + Keyed on (provider, model) — both are first-class columns, so unlike + the dim rollups nothing is excluded; these totals reconcile with + Bifrost's own dashboard. (Bifrost's GetModelRankings applies a + trend calculation we don't want surfaced; same reasoning as + by-user.) + returns: { window, results: [ { model, provider, total_cost, total_tokens, request_count }, ... ] } + +GET /_plugin/spend/by-agent-user?window=24h + The (agent × user) crossing the Canvas page renders, with a + per-provider breakdown per pairing. Not in the original sketch; + added by phase 8. ``` +Phase 11 removed `by-realm`: every row in a swarm's `logs.db` is +implicitly for that swarm's realm, so a per-swarm realm rollup is a +single number the central aggregator already has. + ### Histograms (time-series) ``` @@ -148,14 +165,23 @@ GET /_plugin/histogram/cost?window=24h&bucket=1h&dimension=agent-name } GET /_plugin/histogram/tokens?window=24h&bucket=1h&dimension=user-id - Bifrost API: GetDimensionTokenHistogram - Same shape as above with `tokens` instead of `cost`. + Same shape; points are { ts, prompt_tokens, completion_tokens, total_tokens }. + Rows without token usage (errors) contribute nothing. GET /_plugin/histogram/latency?window=24h&bucket=1h&dimension=agent-name - Bifrost API: GetDimensionLatencyHistogram - Returns percentile series (p50, p95, p99) per dimension value. + Same shape; points are { ts, p50, p95, p99, count } — nearest-rank + percentiles (ms) over the bucket's successful calls, plus the + sample size they came from. Rows with no latency (errored / + in-flight) are excluded. ``` +All three histograms bucket in Go. Bifrost's +`GetDimension{Cost,Token,Latency}Histogram` only group by its +column-bound dimensions (provider / team / customer / user / +business-unit), not `metadata.*`, so the plugin pages the window +out of `/api/logs` and folds rows into epoch-aligned buckets itself +(200k-row ceiling, same as the rollups). + ### Drill-down ``` @@ -166,31 +192,47 @@ GET /_plugin/runs/:run_id Combine with phase-6 /_plugin/runs/:run_id/state for live numbers. GET /_plugin/sessions/:session_id - Bifrost API: GetSessionLogs(session_id) — first-class on LogStore; - uses metadata.session-id under the hood. - returns: + MetadataFilters: {"session-id": }, paginated like /runs/:id + returns: { session_id, logs: [...], stats: , total_count } GET /_plugin/sessions/:session_id/summary - Bifrost API: GetSessionSummary(session_id) - returns: (cost, tokens, started_at, latest_at, duration_ms) + Scans the whole session (no window — a session is finite). + returns: { session_id, user_id, total_cost, total_tokens, request_count, + started_at, latest_at, duration_ms, agents: [...], runs: [...] } + + NOT Bifrost's native /api/logs/sessions/{id}: Bifrost's "session_id" + aliases parent_request_id (its own multi-turn linkage), which is a + different thing from the x-bf-dim-session-id dim Hive stamps. Both + plugin routes filter on metadata.session-id, exactly like /runs/:id + filters on metadata.run-id. GET /_plugin/users/:user_id/spend?window=24h - MetadataFilters: none; SearchFilters.CustomerIDs = [user_id] - Bifrost API: GET /api/logs with stats + MetadataFilters: {"user-id": }; Bifrost's SearchStats over the + filtered set, so one limit=1 call — no row paging. Filters on the + dim, not the customer_id column, for the reasons on spendByUser + (customer_id is the VK's Hive UUID; metadata.user-id is what phase + 6 canonicalises from the verified claim). returns: { user_id, window, total_cost, total_tokens, request_count } GET /_plugin/users/:user_id/quota - Combines: Bifrost customer cap (GET /api/governance/customers/:id) - + live Redis aggregate of in-flight runs the user owns + Combines: Bifrost customer budget (GET /api/governance/customers/:id; + Hive's reconciler provisions one per workspace × user) + + the runs the accumulator indexed for the user in Redis + (bifrost:runs:user:, pruned by macaroon exp on read) returns: { user_id, - daily_budget_usd, spent_today_usd, remaining_today_usd, - inflight_runs: [ { run_id, current_spend, max_cost_usd, exp } ] + customer_found, // false ⇒ budget fields null + budget_usd, budget_window, // Bifrost max_limit + reset_duration + spent_usd, remaining_usd, budget_last_reset, + redis_available, // false ⇒ inflight_runs null + inflight_runs: [ { run_id, agent_name, cost_usd, steps, + max_cost_usd, max_steps, exp, killed } ] } + The window is whatever Bifrost's reset_duration says; the sketch + above assumed "daily", but that is the reconciler's decision. GET /_plugin/agents/:name/spend?window=24h - MetadataFilters: {"agent-name": } - Bifrost API: GET /api/logs with stats + MetadataFilters: {"agent-name": }; SearchStats, one limit=1 call returns: { agent_name, window, total_cost, total_tokens, request_count } ``` @@ -198,19 +240,21 @@ GET /_plugin/agents/:name/spend?window=24h | Param | Type | Default | Notes | |---|---|---|---| -| `window` | duration | `24h` | Accepts Bifrost duration vocabulary (`1h`, `24h`, `1d`, `1w`, `1M`, `1Y`). Translated to `StartTime` / `EndTime` server-side. | -| `bucket` | duration | required for histogram endpoints | Time-bucket granularity. Must be `≤ window`. | -| `dimension` | enum | required for histogram endpoints | One of `agent-name`, `user-id`, `realm-id`, `session-id`, `model`, `provider`. | +| `window` | duration | `24h` | Any Bifrost duration (`1h`, `6h`, `24h`, `1d`, `7d`, `1w`, `30d`, `1M`, `1Y`; see `internal/duration`), at most 1Y. Rolling: `end = now`, `start = now − window` — `1d` is the last 24 hours, not the calendar day. The SPA's picker offers a four-option subset. | +| `bucket` | duration | `1h` for histogram endpoints | Any Bifrost duration ≥ `1m` and `≤ window`. Buckets are aligned to the unix epoch. | +| `dimension` | enum | `agent-name` for histogram endpoints | One of `agent-name`, `user-id`, `session-id`, `run-id`. (`realm-id` removed by phase 11; `model` / `provider` are served by `/spend/by-model` instead.) | | `limit` | int | `100` | Pagination for drill-down endpoints. | | `offset` | int | `0` | Pagination. | | `sort_by` | enum | `timestamp` | For drill-down: `timestamp`, `latency`, `tokens`, `cost`. | | `order` | enum | `desc` | `asc` or `desc`. | The window→`StartTime`/`EndTime` translation uses the request's -arrival time as `now`. Bucket alignment follows Bifrost's -convention (the `1d` bucket is UTC midnight; `1M` is UTC -month-boundary; etc.) — see phase 6 "Duration vocabulary" for the -table. +arrival time as `now`. Calendar alignment (`1d` = UTC midnight, `1M` += month boundary — phase 6 "Duration vocabulary") is the agent +budget bucket's concern, not the analytics window's: an operator +asking for "1d" of spend wants the last day, and the histogram +buckets are epoch-aligned so the same window/bucket pair produces +identical bucket edges across polls. ## Response shape contract @@ -267,48 +311,50 @@ Error codes: **Plugin observability HTTP (`gateway/internal/adminapi/`):** -- [ ] `logstore_client.go`: HTTP client to - `http://127.0.0.1:8080/api/logs` (and rankings / histogram - sub-paths). Owns `SearchFilters` serialization, retry, - timeout. Reusable across all handlers below. -- [ ] `spend.go`: `/_plugin/spend/by-{user,agent,realm,session,model}` - handlers. -- [ ] `histogram.go`: `/_plugin/histogram/{cost,tokens,latency}` - handlers calling `GetDimensionCostHistogram` / - `GetDimensionTokenHistogram` / `GetDimensionLatencyHistogram` - through the client. -- [ ] `sessions.go`: `/_plugin/sessions/:id` and +- [x] `logstore_client.go`: HTTP client to + `http://127.0.0.1:8080/api/logs` (+ `/api/logs/{id}` and + `/api/governance/customers/{id}`). Owns the query-string + composition, Basic auth, timeout, paging (`searchAll`) and + the `upstreamError` mapping. Shared by every handler below. +- [x] `spend.go`: `by-session`, `by-model`, `agents/:name/spend` + (`by-user`, `by-agent`, `by-agent-user` live in + `observability.go` from phase 8). +- [x] `histogram.go`: `tokens` and `latency` (`cost` in + `observability.go`). Bucketed in Go — see the note under + "Histograms". +- [x] `sessions.go`: `/_plugin/sessions/:id` and `/_plugin/sessions/:id/summary`. -- [ ] `users.go`: `/_plugin/users/:id/spend` and - `/_plugin/users/:id/quota`. Quota blends Bifrost customer API - + Redis live state. -- [ ] Extend `runs.go` (from phase 6) with `GET /_plugin/runs/:id` - drill-down (logs.db slice). The phase-6 `state` and `kill` - siblings already live here. -- [ ] Extend `agents.go` (from phase 6) with - `GET /_plugin/agents/:name/spend`. -- [ ] Route registration in `server.go` — extend the existing - `routeDeps` with a `logstore *LogstoreClient` and register - the new route families behind the same bearer middleware. +- [x] `users.go`: `/_plugin/users/:id/spend` and + `/_plugin/users/:id/quota`. Quota blends Bifrost's customer + budget with the accumulator's per-user run index in Redis. +- [x] `GET /_plugin/runs/:id` drill-down (`observability.go`) and + `/runs/:id/calls/:call_id`; `state` / `kill` siblings in + `hotstate.go`. +- [x] `GET /_plugin/agents/:name/spend` (dispatched from the shared + `/_plugin/agents/` subtree in `server.go`). +- [x] Route registration in `server.go` — `routeDeps.logstore`; + every read route is cookie-or-bearer. **Tests:** -- [ ] `logstore_client_test.go`: hits a `miniredis`-equivalent - fake Bifrost (`httptest.Server` mocking `/api/logs`). -- [ ] Per-handler tests: window/bucket/dimension parsing, - MetadataFilters composition, upstream error mapping. +- [x] `logstore_client_test.go`: query composition, paging + row + cap, 404 → nil, upstream error mapping, customer decode, + against an `httptest.Server`. +- [x] Per-handler tests (`observability_test.go`, + `observability_phase7_test.go`): window/bucket/dimension + vocabulary, MetadataFilters scoping, percentile math, quota + degradation with and without Redis, upstream 502 mapping. - [ ] One end-to-end test that stamps dims via PreHook → drives a real LLM call (Bifrost mocker plugin) → queries `/_plugin/spend/by-agent` → asserts the call shows up under - the right agent. + the right agent. `scripts/smoke-test.sh` does this by hand + against a compose stack; it is not in CI. **Documentation:** -- [ ] Update `gateway/internal/adminapi/server.go` doc to note the - new route families (spend, histogram, sessions, users - drill-down). -- [ ] Update `llm-governance-v2.md` §"Observability" forward-pointer - to this phase if it's still sketchy at the time phase 7 ships. +- [x] `gateway/internal/adminapi/server.go` route table notes the + phase-7 families. +- [x] `llm-governance-v2.md` §"Observability" points here. **Gate:** phase 7 ships once the v1 endpoint surface returns correct results for the canonical "alice spent $X on coder yesterday" diff --git a/gateway/plans/phases/phase-8-observability-dashboard.md b/gateway/plans/phases/phase-8-observability-dashboard.md index 7e760eec8..f53741329 100644 --- a/gateway/plans/phases/phase-8-observability-dashboard.md +++ b/gateway/plans/phases/phase-8-observability-dashboard.md @@ -583,21 +583,21 @@ GET /_plugin/runs/:run_id Returns: { run_id, logs: [ ... ], stats: } ``` -Two endpoints from phase 7's full inventory are intentionally -**not** included in phase 8 even though they could technically -ship: `GET /_plugin/spend/by-realm` and -`GET /_plugin/spend/by-session`. They're easy to add when needed, -but the four pages phase 8 ships don't need them yet, and shipping -unused endpoints invites premature use. +`GET /_plugin/spend/by-session` was intentionally **not** included +in phase 8 even though it could technically ship: the four pages +phase 8 ships didn't need it, and shipping unused endpoints invites +premature use. (`by-realm` was dropped outright by phase 11.) ### What's deferred -Phase 9 adds the rest of phase 7 plus the mutations: -- `/_plugin/spend/by-{realm,session,model}` +The rest of phase 7 has since shipped (see its wire-up checklist): +- `/_plugin/spend/by-{session,model}` - `/_plugin/histogram/{tokens,latency}` - `/_plugin/sessions/:id`, `/_plugin/sessions/:id/summary` - `/_plugin/users/:id/spend`, `/_plugin/users/:id/quota` - `/_plugin/agents/:name/spend` + +Phase 9 adds the mutations and the pages over them. - All phase 6 kill/state mutations - All phase 9 config mutations diff --git a/gateway/plans/phases/phase-9-operator-ui.md b/gateway/plans/phases/phase-9-operator-ui.md index 5f809f489..551627738 100644 --- a/gateway/plans/phases/phase-9-operator-ui.md +++ b/gateway/plans/phases/phase-9-operator-ui.md @@ -683,12 +683,18 @@ packages: time.Time: "string" ``` -`make tygo` runs the codegen; CI runs `make tygo && git diff --exit-code` -to catch drift. - -Types that are not response shapes (internal-only structs) are -either kept out of the codegen by living in non-exported packages or -explicitly excluded in `tygo.yaml`. +`make tygo` runs the codegen; CI (`.github/workflows/gateway-check.yml`, +job `tygo-check`) installs the pinned tygo, regenerates, and fails on +any diff. Only the `adminapi` package is in `tygo.yaml` — the `auth` +entry above never shipped; the SPA reads `auth`'s state through +`adminapi`'s named response structs. Types with no Go struct behind +them (the `Window` / `Bucket` / `Dimension` unions, the error +envelope, the `trust` mirrors, Bifrost's chat / usage shapes that Go +carries as `json.RawMessage`) are hand-maintained in +`ui/src/api/manual.ts`. + +Types that are not response shapes (internal-only structs) are kept +out of the codegen by being unexported. The handlers in `adminapi/` declare named response types (`type RunStateResponse struct { ... }`) rather than returning diff --git a/gateway/tygo.yaml b/gateway/tygo.yaml index 6e104565c..1ba6f51f2 100644 --- a/gateway/tygo.yaml +++ b/gateway/tygo.yaml @@ -11,21 +11,27 @@ # Source-of-truth: every type the SPA decodes is declared in # gateway/internal/adminapi/ as an exported Go struct with `json:` # tags. tygo translates those into TS interfaces; no hand-editing -# of types.ts. +# of types.ts. Types with no adminapi Go struct behind them +# (string-literal unions, the trust-package mirrors, Bifrost +# pass-through shapes Go carries as json.RawMessage) are +# hand-maintained in ui/src/api/manual.ts instead. # -# Install tygo (once, per dev machine): -# go install github.com/gzuidhof/tygo@latest +# Install tygo (once, per dev machine). CI pins the same version in +# .github/workflows/gateway-check.yml — bump both together, since a +# tygo release that changes its formatting would fail every PR: +# go install github.com/gzuidhof/tygo@v0.2.21 packages: - path: "github.com/stakwork/stakgraph/gateway/internal/adminapi" output_path: "internal/adminapi/ui/src/api/types.ts" - # frontmatter is prepended to the generated file. We mark it - # generated so editors / reviewers know not to hand-edit. + # frontmatter follows tygo's own "Code generated … DO NOT EDIT" + # banner, so editors / reviewers know not to hand-edit. frontmatter: | - // Code generated by tygo. DO NOT EDIT. // // Source: github.com/stakwork/stakgraph/gateway/internal/adminapi - // Regenerate with `make tygo`. + // Regenerate with `make tygo`; CI enforces via `make tygo-check`. + // Hand-maintained companions (unions, trust mirrors, Bifrost + // pass-through shapes) live in manual.ts. /* eslint-disable */ type_mappings: "time.Time": "string"