diff --git a/CLAUDE.md b/CLAUDE.md index 4499720..a46ddf0 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -62,7 +62,7 @@ core ← db ← cli - **@agentops/core** — Domain types, policy engine, scoring algorithm, and builder functions for all entities. No external dependencies. All types use `readonly` and branded ID types (RunId, JobId, SessionId, etc.) for type safety. - **@agentops/db** — SQLite persistence via Drizzle ORM + better-sqlite3. Sixteen tables: `runs`, `policies`, `policy_results`, `run_metrics`, `jobs`, `sessions`, `events`, `locks`, `users`, `api_tokens`, `auth_sessions`, `device_codes`, `webhooks`, `webhook_deliveries`, `audit_log`, `user_budgets`. Complex fields stored as JSON columns. DB defaults to `~/.agentops/agentops.db` (override with `AGENTOPS_DB_PATH`). WAL mode with `foreign_keys = ON` — deletes of parent rows must cascade children first (see `deletePolicy`, `deleteOldRuns`). - **@agentops/cli** — CLI entry point (`agentops`). Commands: `init`, `serve`, `setup`, `hook`, `login`, `doctor`, `user`, `admin`, `cleanup`, `run`, `policy`, `report`, `wrap`, `watch`, `link`, `pr`, `job`, `session`, `events`, `lock`, `dispatch`. Supports `--json` output and `--db-path` override. `init` bootstraps the DB (`--seed` for sample data, `--seed-policies` for the starter policy set, `--clean` to reset). `serve` starts the dashboard server (`--port` to override 3000). `setup` configures Claude Code hooks (`--global`, `--uninstall`, `--dry-run`). `hook` handles Claude Code hook events (session-start, pre-tool-use, post-tool-use, user-prompt-submit, stop, subagent-stop, session-end) — reads JSON from stdin, manages state under `~/.agentops/state/`, evaluates policies in real-time, can block risky tool calls (exit code 2), and reads cost/token usage from the Claude Code transcript (deduped by `message.id`; unknown models warn on stderr instead of pricing at $0). `login` runs the device-flow auth against a dashboard; `doctor` diagnoses a local install. Helper modules: `format.ts` (output formatting), `git.ts` (git integration), `github.ts` (GitHub API), `transcript.ts` (usage/cost from Claude Code transcripts), `pricing` lives in core. Note: `job`, `lock`, `dispatch`, `wrap`, and `watch` operate on orchestration machinery that Claude Code hooks never populate — treat them as experimental/vestigial. -- **@agentops/sdk** — Lightweight HTTP client for agent runtimes to talk to the AgentOps server. Depends only on `@agentops/core` for types. Uses native `fetch`. Provides `AgentOpsClient` class (via `createClient()` factory) with methods: `createSession`, `startRun`, `reportAction`, `reportArtifact`, `reportMetrics`, `checkPolicy`, `heartbeat`, `completeRun`, `failRun`. Also exports `PolicyMiddleware` for pre-flight policy checks before actions. Throws typed `AgentOpsError` with status codes. +- **@agentops/sdk** — Lightweight HTTP client for agent runtimes to talk to the AgentOps server. Depends only on `@agentops/core` for types. Uses native `fetch`. Provides `AgentOpsClient` class (via `createClient()` factory) with methods: `createSession`, `startRun`, `reportAction`, `reportArtifact`, `reportMetrics`, `checkPolicy`, `heartbeat`, `completeRun`, `failRun`, `terminateSession`. Also exports `PolicyMiddleware` for pre-flight policy checks before actions. Throws typed `AgentOpsError` with status codes. - **@agentops/web** — Next.js 16 App Router dashboard with React 19, Tailwind CSS 4. API routes under `src/app/api/` organized by resource (runs, sessions, policies, events, analytics, admin, stats, sdk, auth, budgets, webhooks). Jobs, Locks, and Coordination pages were removed from the dashboard because hooks do not populate this data. Sidebar nav: Runs | Sessions | Events | Analytics | Usage | Policies | Settings. **Every API route is authenticated** (`src/lib/auth.ts`): bearer tokens or session cookies, `admin`/`member` roles, members are scoped to their own runs/sessions (`resolveViewScope`), non-owners get 404 (not 403) to avoid ID enumeration, mutations additionally pass `checkSameOrigin` CSRF checks — new routes must follow this pattern (regression tests live in `api/__tests__/auth-gaps.test.ts`). Inbound SDK routes under `/api/sdk/` use bearer auth + per-token rate limits. The Usage page (`/usage`) always shows the local hook-captured rollup (incl. Bedrock-vs-direct backend split and per-user attribution); when `ANTHROPIC_ADMIN_API_KEY` is set it additionally shows org-wide cost/tokens via `/api/admin/{status,cost,analytics}`, which normalize the Anthropic Admin API's reports (RFC 3339 `starting_at`, paginated, amounts are decimal-string cents) server-side. Server components access SQLite directly via a singleton lazy-loaded DB instance (`src/lib/db.ts`). Uses `serverExternalPackages: ["better-sqlite3"]` in next.config.ts. Path alias: `@/*` → `./src/*`. Dark theme by default. ### Core Domain Concepts diff --git a/packages/sdk/README.md b/packages/sdk/README.md index 86155fe..262cef8 100644 --- a/packages/sdk/README.md +++ b/packages/sdk/README.md @@ -20,9 +20,33 @@ npm install @agentops/sdk ```ts import { AgentOpsClient } from "@agentops/sdk"; - -const client = new AgentOpsClient({ baseUrl: "http://localhost:3000" }); -await client.startSession({ agentId: "my-agent" }); +import { createAgentId, AgentRole } from "@agentops/core"; + +const client = new AgentOpsClient({ baseUrl: "http://localhost:3000", apiKey: "ao_..." }); + +const { sessionId } = await client.createSession({ agentId: "my-agent" }); +const { runId } = await client.startRun({ + goal: { humanReadable: "Fix the bug", structured: { type: "bugfix", description: "", parameters: {} } }, + agents: [{ id: createAgentId("my-agent"), model: "claude-opus-4-6", role: AgentRole.Implementer }], + environment: { repo: "acme/api", branch: "main", permissions: [], sandbox: { enabled: false, isolationLevel: "none" } }, + sessionId, +}); + +// Metrics reports are partial updates — omitted fields keep their stored values. +await client.reportMetrics(runId, { + tokenUsage: { input: 100, output: 50, total: 150 }, + costUsd: 0.05, + backend: "bedrock", // optional Bedrock/Anthropic spend attribution +}); + +// Test results reported at completion drive the correctness score and +// merge recommendation. +const { recommendation } = await client.completeRun(runId, { + testResults: [{ name: "unit", passed: true, duration: 12, message: "" }], + confidenceScore: 0.9, +}); + +await client.terminateSession(sessionId); ``` ## License diff --git a/packages/sdk/src/__tests__/client.test.ts b/packages/sdk/src/__tests__/client.test.ts index 481d620..41ba70e 100644 --- a/packages/sdk/src/__tests__/client.test.ts +++ b/packages/sdk/src/__tests__/client.test.ts @@ -3,6 +3,7 @@ import { createRunId, createSessionId, createAgentId, + createArtifactId, AgentRole, } from "@agentops/core"; import { AgentOpsClient, AgentOpsError, createClient } from "../client.js"; @@ -123,6 +124,13 @@ describe("AgentOpsClient", () => { expect(result.runId).toBe(testRunId); expect(result.status).toBe("running"); + + // The declared agents must actually go over the wire — the server + // persists them onto the run (request-side contract). + const sentBody = JSON.parse(mockFetch.mock.calls[0]![1].body as string); + expect(sentBody.agents).toEqual([ + { id: "agent-1", model: "claude-opus-4-6", role: "implementer" }, + ]); }); }); @@ -175,6 +183,20 @@ describe("AgentOpsClient", () => { expect(result.runId).toBe(testRunId); }); + + it("sends backend and byModel for Bedrock spend attribution", async () => { + mockFetch.mockResolvedValueOnce(jsonResponse({ runId: testRunId })); + + await client.reportMetrics(testRunId, { + costUsd: 2.5, + backend: "bedrock", + byModel: { "us.anthropic.claude-opus-4-7-v1:0": 2.5 }, + }); + + const sentBody = JSON.parse(mockFetch.mock.calls[0]![1].body as string); + expect(sentBody.backend).toBe("bedrock"); + expect(sentBody.byModel).toEqual({ "us.anthropic.claude-opus-4-7-v1:0": 2.5 }); + }); }); describe("checkPolicy", () => { @@ -248,7 +270,7 @@ describe("AgentOpsClient", () => { expect(result.runId).toBe(testRunId); }); - it("sends result when provided", async () => { + it("sends result when provided (deprecated, still transmitted)", async () => { mockFetch.mockResolvedValueOnce( jsonResponse({ runId: testRunId } as unknown as CompleteRunResponse), ); @@ -258,6 +280,50 @@ describe("AgentOpsClient", () => { const sentBody = JSON.parse(mockFetch.mock.calls[0]![1].body as string); expect(sentBody.result).toBe("Bug fixed successfully"); }); + + it("sends testResults / policyChecks / confidenceScore / artifacts", async () => { + mockFetch.mockResolvedValueOnce( + jsonResponse({ runId: testRunId } as unknown as CompleteRunResponse), + ); + + await client.completeRun(testRunId, { + testResults: [{ name: "unit", passed: false, duration: 5, message: "boom" }], + policyChecks: [], + confidenceScore: 0.8, + artifacts: [ + { + id: createArtifactId("artifact_1"), + diffs: [], + logs: [], + testOutputs: ["1 failed"], + reports: [], + }, + ], + }); + + const sentBody = JSON.parse(mockFetch.mock.calls[0]![1].body as string); + expect(sentBody.testResults).toEqual([ + { name: "unit", passed: false, duration: 5, message: "boom" }, + ]); + expect(sentBody.policyChecks).toEqual([]); + expect(sentBody.confidenceScore).toBe(0.8); + expect(sentBody.artifacts).toHaveLength(1); + expect(sentBody.artifacts[0].id).toBe("artifact_1"); + }); + }); + + describe("terminateSession", () => { + it("posts to /api/sdk/sessions/:id/terminate and returns { status }", async () => { + mockFetch.mockResolvedValueOnce(jsonResponse({ status: "terminated" })); + + const result = await client.terminateSession(testSessionId); + + expect(result.status).toBe("terminated"); + expect(mockFetch).toHaveBeenCalledWith( + `${BASE_URL}/api/sdk/sessions/${testSessionId}/terminate`, + expect.objectContaining({ method: "POST" }), + ); + }); }); describe("failRun", () => { diff --git a/packages/sdk/src/client.ts b/packages/sdk/src/client.ts index 291bcfc..f464451 100644 --- a/packages/sdk/src/client.ts +++ b/packages/sdk/src/client.ts @@ -15,6 +15,7 @@ import type { CheckPolicyRequest, CheckPolicyResponse, HeartbeatResponse, + TerminateSessionResponse, CompleteRunRequest, CompleteRunResponse, FailRunRequest, @@ -158,6 +159,20 @@ export class AgentOpsClient { ); } + /** + * Gracefully end a session: archives its current run and marks it + * terminated. Without this, SDK-created sessions only end when the + * server's staleness reaper gives up on their heartbeats. + */ + async terminateSession( + sessionId: SessionId, + ): Promise { + return this.request( + `/api/sdk/sessions/${sessionId}/terminate`, + {}, + ); + } + async completeRun( runId: RunId, result?: CompleteRunRequest, diff --git a/packages/sdk/src/index.ts b/packages/sdk/src/index.ts index b68be3e..ec98360 100644 --- a/packages/sdk/src/index.ts +++ b/packages/sdk/src/index.ts @@ -21,6 +21,7 @@ export type { CheckPolicyResponse, PolicyViolation, HeartbeatResponse, + TerminateSessionResponse, CompleteRunRequest, CompleteRunResponse, FailRunRequest, diff --git a/packages/sdk/src/types.ts b/packages/sdk/src/types.ts index 7218734..7cebbc8 100644 --- a/packages/sdk/src/types.ts +++ b/packages/sdk/src/types.ts @@ -6,6 +6,7 @@ import type { Metrics, Agent, Environment, + Evaluation, Goal, ScoreCard, MergeRecommendation, @@ -47,11 +48,18 @@ export interface ReportArtifactRequest { readonly reports?: Artifact["reports"]; } +// Partial-update semantics: every field is optional, and the server MERGES +// each report into the run's stored metrics — an omitted field preserves the +// previously reported value (it is never reset to zero). export interface ReportMetricsRequest { readonly tokenUsage?: Metrics["tokenUsage"]; readonly wallTimeMs?: number; readonly costUsd?: number; readonly flakeRate?: number; + /** Which API backend served the traffic ("anthropic" | "bedrock"). */ + readonly backend?: Metrics["backend"]; + /** Per-model cost in USD, keyed by raw model id (Bedrock ids stay namespaced). */ + readonly byModel?: Metrics["byModel"]; } export interface CheckPolicyRequest { @@ -62,8 +70,25 @@ export interface CheckPolicyRequest { readonly editedFiles?: ReadonlyArray; } +// Mirrors what /api/sdk/runs/[id]/complete actually consumes: the evaluation +// (test results, policy checks, confidence) plus any final artifacts to +// append. Without testResults the scorer has nothing to grade — correctness +// reads "not scored" and mutating runs can reach Merge with no tests run. export interface CompleteRunRequest { + /** + * @deprecated The server has never read this field; it is silently + * dropped. Report outcome data via testResults / policyChecks / + * confidenceScore instead. Kept only so existing callers keep compiling. + */ readonly result?: string; + /** Test outcomes for the run — these drive the correctness score. */ + readonly testResults?: Evaluation["testResults"]; + /** Client-side policy check outcomes to record on the evaluation. */ + readonly policyChecks?: Evaluation["policyChecks"]; + /** Agent self-reported confidence, 0-1. Defaults to 0 server-side. */ + readonly confidenceScore?: number; + /** Final artifacts to append to the run before scoring. */ + readonly artifacts?: ReadonlyArray; } export interface FailRunRequest { @@ -118,6 +143,10 @@ export interface HeartbeatResponse { readonly commands: ReadonlyArray; } +export interface TerminateSessionResponse { + readonly status: string; +} + export interface CompleteRunResponse { readonly runId: RunId; readonly score: ScoreCard; diff --git a/packages/web/src/app/api/sdk/__tests__/sdk-contract.test.ts b/packages/web/src/app/api/sdk/__tests__/sdk-contract.test.ts index 1a27d1d..5a3303f 100644 --- a/packages/web/src/app/api/sdk/__tests__/sdk-contract.test.ts +++ b/packages/web/src/app/api/sdk/__tests__/sdk-contract.test.ts @@ -50,6 +50,7 @@ import { POST as reportMetricsRoute } from "@/app/api/sdk/runs/[id]/metrics/rout import { POST as completeRunRoute } from "@/app/api/sdk/runs/[id]/complete/route"; import { POST as failRunRoute } from "@/app/api/sdk/runs/[id]/fail/route"; import { POST as heartbeatRoute } from "@/app/api/sdk/sessions/[id]/heartbeat/route"; +import { POST as terminateSessionRoute } from "@/app/api/sdk/sessions/[id]/terminate/route"; import { POST as policyCheckRoute } from "@/app/api/sdk/policy/check/route"; let db: AgentOpsDb; @@ -177,6 +178,21 @@ describe("SDK route ⇄ type contract", () => { expect(Array.isArray(body.commands)).toBe(true); }); + it("terminateSession → { status } (TerminateSessionResponse)", async () => { + const { insertSession } = await import("@agentops/db"); + const { createSession, activateSession } = await import("@agentops/core"); + const session = { ...activateSession(createSession("agent", {})), userId: alice.user.id }; + insertSession(db, session); + const res = await terminateSessionRoute( + reqFor(`/api/sdk/sessions/${session.id}/terminate`, {}), + withParams({ id: session.id as string }), + ); + expect(res.status).toBe(200); + const body = await bodyOf(res); + expect(Object.keys(body)).toEqual(["status"]); + expect(body.status).toBe("terminated"); + }); + it("policy check (allow) → { decision, violations, warnings } (CheckPolicyResponse)", async () => { const runId = seedRun(); const res = await policyCheckRoute(reqFor("/api/sdk/policy/check", { runId, toolName: "Read", toolInput: {} })); @@ -212,3 +228,101 @@ describe("SDK route ⇄ type contract", () => { expect(Array.isArray(body.warnings)).toBe(true); }); }); + +// Request-side contract: bodies shaped exactly like the SDK's REQUEST types +// must be consumed by the routes, not silently dropped. This is the drift the +// response-only tests above couldn't see (StartRunRequest.agents was ignored, +// CompleteRunRequest couldn't carry testResults, ReportMetricsRequest +// replaced instead of merged). + +describe("SDK request type ⇄ route contract", () => { + it("StartRunRequest.agents is persisted onto the run", async () => { + // Body shaped like StartRunRequest, agents included. + const res = await createRunRoute( + reqFor("/api/sdk/runs", { + goal: { humanReadable: "t", structured: { type: "t", description: "t", parameters: {} } }, + agents: [ + { id: "agent-1", model: "claude-opus-4-6", role: "implementer" }, + { id: "agent-2", model: "claude-haiku-4-5", role: "reviewer" }, + ], + environment: { repo: "acme/test", branch: "main", permissions: [], sandbox: { enabled: false, isolationLevel: "none" } }, + }), + ); + expect(res.status).toBe(201); + const { runId } = (await bodyOf(res)) as { runId: string }; + const { getRun } = await import("@agentops/db"); + const { createRunId } = await import("@agentops/core"); + const saved = getRun(db, createRunId(runId))!; + expect(saved.agents).toEqual([ + { id: "agent-1", model: "claude-opus-4-6", role: "implementer" }, + { id: "agent-2", model: "claude-haiku-4-5", role: "reviewer" }, + ]); + }); + + it("CompleteRunRequest.testResults reach the stored run AND drive scoring", async () => { + const runId = seedRun(); + // Body shaped like CompleteRunRequest (post-fix): failing test included. + const res = await completeRunRoute( + reqFor(`/api/sdk/runs/${runId}/complete`, { + testResults: [ + { name: "passes", passed: true, duration: 10, message: "" }, + { name: "fails", passed: false, duration: 12, message: "assertion failed" }, + ], + policyChecks: [], + confidenceScore: 0.7, + artifacts: [ + { id: "artifact_final", diffs: ["+x"], logs: [], testOutputs: ["1 failed"], reports: [] }, + ], + }), + withParams({ id: runId }), + ); + expect(res.status).toBe(200); + const body = await bodyOf(res); + + // Scoring saw the tests: correctness is 1/2, not the "not scored" 1.0 + // that let mutating runs reach Merge with zero tests run. + const score = body.score as { correctness: { score: number; rationale: string } }; + expect(score.correctness.score).toBe(0.5); + expect(score.correctness.rationale).toContain("1/2 tests passing"); + + // Round-trip: the evaluation + artifacts landed on the stored run. + const { getRun } = await import("@agentops/db"); + const { createRunId } = await import("@agentops/core"); + const saved = getRun(db, createRunId(runId))!; + expect(saved.evaluations).toHaveLength(1); + expect(saved.evaluations[0]!.testResults).toHaveLength(2); + expect(saved.evaluations[0]!.confidenceScore).toBe(0.7); + expect(saved.artifacts.map((a) => a.id as string)).toContain("artifact_final"); + }); + + it("ReportMetricsRequest is a partial update: omitted fields keep stored values", async () => { + const runId = seedRun(); + // First report: tokens only. + await reportMetricsRoute( + reqFor(`/api/sdk/runs/${runId}/metrics`, { + tokenUsage: { input: 100, output: 50, total: 150 }, + }), + withParams({ id: runId }), + ); + // Second report: cost + backend attribution only (ReportMetricsRequest + // now declares backend/byModel so SDK clients can set Bedrock spend). + const res = await reportMetricsRoute( + reqFor(`/api/sdk/runs/${runId}/metrics`, { + costUsd: 2.5, + backend: "bedrock", + byModel: { "us.anthropic.claude-opus-4-7-v1:0": 2.5 }, + }), + withParams({ id: runId }), + ); + expect(res.status).toBe(200); + + const { getRun } = await import("@agentops/db"); + const { createRunId } = await import("@agentops/core"); + const saved = getRun(db, createRunId(runId))!; + // Tokens from report #1 survived report #2 (previously zero-filled). + expect(saved.metrics.tokenUsage).toEqual({ input: 100, output: 50, total: 150 }); + expect(saved.metrics.costUsd).toBe(2.5); + expect(saved.metrics.backend).toBe("bedrock"); + expect(saved.metrics.byModel).toEqual({ "us.anthropic.claude-opus-4-7-v1:0": 2.5 }); + }); +}); diff --git a/packages/web/src/app/api/sdk/__tests__/sdk-routes.test.ts b/packages/web/src/app/api/sdk/__tests__/sdk-routes.test.ts index 43f1902..665f3c8 100644 --- a/packages/web/src/app/api/sdk/__tests__/sdk-routes.test.ts +++ b/packages/web/src/app/api/sdk/__tests__/sdk-routes.test.ts @@ -231,6 +231,50 @@ describe("POST /api/sdk/runs", () => { const res = await createRunRoute(req); expect(res.status).toBe(400); }); + + it("persists reported agents on the created run", async () => { + const req = authedRequest("http://localhost/api/sdk/runs", { + token: alice.token, + body: { + ...validBody, + agents: [{ id: "claude-code", model: "claude-opus-4-6", role: "lead" }], + }, + }); + const res = await createRunRoute(req); + expect(res.status).toBe(201); + const body = (await jsonOf(res)) as { runId: string }; + const { createRunId } = await import("@agentops/core"); + const saved = getRun(db, createRunId(body.runId))!; + expect(saved.agents).toEqual([ + { id: "claude-code", model: "claude-opus-4-6", role: "lead" }, + ]); + }); + + it("defaults agents to [] when omitted (minimal clients keep working)", async () => { + const req = authedRequest("http://localhost/api/sdk/runs", { + token: alice.token, + body: validBody, + }); + const res = await createRunRoute(req); + expect(res.status).toBe(201); + const body = (await jsonOf(res)) as { runId: string }; + const { createRunId } = await import("@agentops/core"); + expect(getRun(db, createRunId(body.runId))!.agents).toEqual([]); + }); + + it.each([ + { label: "non-array", agents: { id: "x", model: "m", role: "lead" } }, + { label: "missing model", agents: [{ id: "x", role: "lead" }] }, + { label: "empty id", agents: [{ id: "", model: "m", role: "lead" }] }, + { label: "unknown role", agents: [{ id: "x", model: "m", role: "supervisor" }] }, + ])("400 on malformed agents ($label)", async ({ agents }) => { + const req = authedRequest("http://localhost/api/sdk/runs", { + token: alice.token, + body: { ...validBody, agents }, + }); + const res = await createRunRoute(req); + expect(res.status).toBe(400); + }); }); // ─── POST /api/sdk/runs/[id]/actions ────────────────────────────────────── @@ -392,6 +436,57 @@ describe("POST /api/sdk/runs/[id]/metrics", () => { const res = await reportMetricsRoute(req, withParams({ id: runId })); expect(res.status).toBe(400); }); + + it("merges partial reports: a later costUsd-only report preserves earlier tokenUsage", async () => { + const { runId } = seedRun(alice); + const post = (body: Record) => + reportMetricsRoute( + authedRequest(`http://localhost/api/sdk/runs/${runId}/metrics`, { + token: alice.token, + body, + }), + withParams({ id: runId }), + ); + + await post({ tokenUsage: { input: 100, output: 50, total: 150 }, wallTimeMs: 5000 }); + const res = await post({ costUsd: 1.5 }); + expect(res.status).toBe(200); + + const { createRunId } = await import("@agentops/core"); + const run = getRun(db, createRunId(runId))!; + expect(run.metrics.tokenUsage).toEqual({ input: 100, output: 50, total: 150 }); + expect(run.metrics.wallTimeMs).toBe(5000); + expect(run.metrics.costUsd).toBe(1.5); + + // The run_metrics table row is updated with the same MERGED values, + // not zero-filled ones — both stores must agree. + const { getRunMetrics } = await import("@agentops/db"); + const row = getRunMetrics(db, createRunId(runId))!; + expect(row.tokenUsage).toEqual({ input: 100, output: 50, total: 150 }); + expect(row.wallTimeMs).toBe(5000); + expect(row.costUsd).toBe(1.5); + }); + + it("merge preserves previously reported backend/byModel when omitted", async () => { + const { runId } = seedRun(alice); + const post = (body: Record) => + reportMetricsRoute( + authedRequest(`http://localhost/api/sdk/runs/${runId}/metrics`, { + token: alice.token, + body, + }), + withParams({ id: runId }), + ); + + await post({ backend: "bedrock", byModel: { "us.anthropic.claude-opus-4-7-v1:0": 1 } }); + await post({ flakeRate: 0.1 }); + + const { createRunId } = await import("@agentops/core"); + const run = getRun(db, createRunId(runId))!; + expect(run.metrics.backend).toBe("bedrock"); + expect(run.metrics.byModel).toEqual({ "us.anthropic.claude-opus-4-7-v1:0": 1 }); + expect(run.metrics.flakeRate).toBe(0.1); + }); }); // ─── POST /api/sdk/runs/[id]/complete ───────────────────────────────────── @@ -457,6 +552,61 @@ describe("POST /api/sdk/runs/[id]/complete", () => { ); expect(rollupSources.every((s) => s === "run-complete")).toBe(true); }); + + it("stores reported testResults on the run's evaluation", async () => { + const { runId } = seedRun(alice); + const req = authedRequest(`http://localhost/api/sdk/runs/${runId}/complete`, { + token: alice.token, + body: { + testResults: [{ name: "unit", passed: true, duration: 3, message: "" }], + policyChecks: [], + confidenceScore: 0.9, + }, + }); + const res = await completeRunRoute(req, withParams({ id: runId })); + expect(res.status).toBe(200); + + const { createRunId } = await import("@agentops/core"); + const saved = getRun(db, createRunId(runId))!; + expect(saved.evaluations).toHaveLength(1); + expect(saved.evaluations[0]!.testResults).toEqual([ + { name: "unit", passed: true, duration: 3, message: "" }, + ]); + expect(saved.evaluations[0]!.confidenceScore).toBe(0.9); + }); + + it("appends reported artifacts to the run", async () => { + const { runId } = seedRun(alice); + const req = authedRequest(`http://localhost/api/sdk/runs/${runId}/complete`, { + token: alice.token, + body: { + artifacts: [ + { id: "artifact_done", diffs: [], logs: ["done"], testOutputs: [], reports: [] }, + ], + }, + }); + const res = await completeRunRoute(req, withParams({ id: runId })); + expect(res.status).toBe(200); + + const { createRunId } = await import("@agentops/core"); + const saved = getRun(db, createRunId(runId))!; + expect(saved.artifacts.map((a) => a.id as string)).toContain("artifact_done"); + }); + + it.each([ + { field: "testResults", value: "not-an-array" }, + { field: "policyChecks", value: 42 }, + { field: "confidenceScore", value: "high" }, + { field: "artifacts", value: { id: "not-an-array" } }, + ])("400 on non-conforming $field", async ({ field, value }) => { + const { runId } = seedRun(alice); + const req = authedRequest(`http://localhost/api/sdk/runs/${runId}/complete`, { + token: alice.token, + body: { [field]: value }, + }); + const res = await completeRunRoute(req, withParams({ id: runId })); + expect(res.status).toBe(400); + }); }); // ─── POST /api/sdk/runs/[id]/fail ───────────────────────────────────────── diff --git a/packages/web/src/app/api/sdk/runs/[id]/complete/route.ts b/packages/web/src/app/api/sdk/runs/[id]/complete/route.ts index 5694acd..d30fbf7 100644 --- a/packages/web/src/app/api/sdk/runs/[id]/complete/route.ts +++ b/packages/web/src/app/api/sdk/runs/[id]/complete/route.ts @@ -61,6 +61,13 @@ export async function POST( ); } + if (body.artifacts !== undefined && !Array.isArray(body.artifacts)) { + return NextResponse.json( + { error: "artifacts must be an array" }, + { status: 400 }, + ); + } + const evaluation: Evaluation = { testResults: (body.testResults as Evaluation["testResults"]) ?? [], policyChecks: (body.policyChecks as Evaluation["policyChecks"]) ?? [], diff --git a/packages/web/src/app/api/sdk/runs/[id]/metrics/route.ts b/packages/web/src/app/api/sdk/runs/[id]/metrics/route.ts index fe2d16f..2cc3620 100644 --- a/packages/web/src/app/api/sdk/runs/[id]/metrics/route.ts +++ b/packages/web/src/app/api/sdk/runs/[id]/metrics/route.ts @@ -103,45 +103,48 @@ export async function POST( } } - const tokenUsage = (body.tokenUsage as Record) ?? { - input: 0, - output: 0, - total: 0, - }; - const costUsd = (body.costUsd as number) ?? 0; - const wallTimeMs = (body.wallTimeMs as number) ?? 0; - const flakeRate = (body.flakeRate as number) ?? 0; - + const tokenUsage = body.tokenUsage as + | { input: number; output: number; total: number } + | undefined; + const costUsd = body.costUsd as number | undefined; + const wallTimeMs = body.wallTimeMs as number | undefined; + const flakeRate = body.flakeRate as number | undefined; const backend = body.backend as "anthropic" | "bedrock" | undefined; const byModel = body.byModel as Record | undefined; - // Update the run's embedded metrics. backend/byModel live only in this - // JSON blob (no run_metrics column yet) — nothing queries run_metrics for + // MERGE into the run's embedded metrics rather than replace: every + // request field is optional (partial-update semantics), so an omitted + // field must preserve the previously reported value — replacing would + // zero-fill it (e.g. reporting tokenUsage, then later only costUsd, + // used to wipe the tokens). backend/byModel live only in this JSON + // blob (no run_metrics column yet) — nothing queries run_metrics for // backend, so a column would be speculative denormalization for now. const metrics = { - tokenUsage: tokenUsage as { input: number; output: number; total: number }, - wallTimeMs, - costUsd, - flakeRate, - ...(backend ? { backend } : {}), - ...(byModel ? { byModel } : {}), + ...run.metrics, + ...(tokenUsage !== undefined ? { tokenUsage } : {}), + ...(wallTimeMs !== undefined ? { wallTimeMs } : {}), + ...(costUsd !== undefined ? { costUsd } : {}), + ...(flakeRate !== undefined ? { flakeRate } : {}), + ...(backend !== undefined ? { backend } : {}), + ...(byModel !== undefined ? { byModel } : {}), }; updateRun(db(), run.id, { metrics, updatedAt: new Date().toISOString(), }); - // Upsert into run_metrics table + // Upsert into run_metrics table using the same merged values so both + // stores stay consistent. const existing = getRunMetrics(db(), run.id); const now = new Date().toISOString(); if (existing) { db() .update(runMetrics) .set({ - tokenUsage: tokenUsage as Record, - wallTimeMs, - costCents: costUsd * 100, - flakeRate, + tokenUsage: metrics.tokenUsage as unknown as Record, + wallTimeMs: metrics.wallTimeMs, + costCents: metrics.costUsd * 100, + flakeRate: metrics.flakeRate, recordedAt: now, }) .where(eq(runMetrics.runId, id)) @@ -152,17 +155,18 @@ export async function POST( .values({ id: `rm_${Date.now()}`, runId: id, - tokenUsage: tokenUsage as Record, - wallTimeMs, - costCents: costUsd * 100, - flakeRate, + tokenUsage: metrics.tokenUsage as unknown as Record, + wallTimeMs: metrics.wallTimeMs, + costCents: metrics.costUsd * 100, + flakeRate: metrics.flakeRate, recordedAt: now, }) .run(); } - // Emit cost threshold event when cost is reported - if (costUsd > 0) { + // Emit cost threshold event when cost is reported in THIS request + // (merged totals must not re-emit for a previously reported cost). + if (costUsd !== undefined && costUsd > 0) { const event = createEvent( EventCategory.Cost, EVENT_TYPES["cost.threshold"], diff --git a/packages/web/src/app/api/sdk/runs/route.ts b/packages/web/src/app/api/sdk/runs/route.ts index 5025b29..b04a8ff 100644 --- a/packages/web/src/app/api/sdk/runs/route.ts +++ b/packages/web/src/app/api/sdk/runs/route.ts @@ -1,14 +1,16 @@ import { NextRequest, NextResponse } from "next/server"; -import type { Goal, Environment } from "@agentops/core"; +import type { Goal, Environment, Agent } from "@agentops/core"; import { createRun, startRun, assignRun, + createAgentId, createSessionId, createEvent, EventCategory, EVENT_TYPES, normalizeRepo, + AgentRole, } from "@agentops/core"; import { insertRun, insertEvent, getSession, updateSession } from "@agentops/db"; import { db } from "@/lib/db"; @@ -76,6 +78,44 @@ export async function POST(request: NextRequest) { ); } + // agents: validate-if-present and persist. StartRunRequest declares + // agents, and dropping them here left run.agents forever [] — the + // dashboard could never attribute an SDK run to the model/role that + // performed it. Optional so minimal clients keep working. + const validRoles = new Set(Object.values(AgentRole)); + const rawAgents = body.agents; + if (rawAgents !== undefined) { + const shapeOk = + Array.isArray(rawAgents) && + rawAgents.every((a: unknown) => { + if (a === null || typeof a !== "object") return false; + const agent = a as Record; + return ( + typeof agent.id === "string" && + agent.id.length > 0 && + typeof agent.model === "string" && + agent.model.length > 0 && + typeof agent.role === "string" && + validRoles.has(agent.role) + ); + }); + if (!shapeOk) { + return NextResponse.json( + { + error: `agents must be an array of { id, model, role } objects with role one of: ${[...validRoles].join(", ")}`, + }, + { status: 400 }, + ); + } + } + const agents: ReadonlyArray = ( + (rawAgents as Array<{ id: string; model: string; role: string }> | undefined) ?? [] + ).map((a) => ({ + id: createAgentId(a.id), + model: a.model, + role: a.role as AgentRole, + })); + // Canonicalize the repo identity at the write boundary so SDK-supplied // values bucket the same way as CLI-produced ones (lowercase owner/name); // otherwise the same repo fragments across the dashboard's analytics. @@ -95,8 +135,9 @@ export async function POST(request: NextRequest) { const baseRun = startRun(createRun(goal, normalizedEnvironment)); // Tag the run with the authenticated user so the dashboard can scope - // by owner. Both insertRun and rowToRun round-trip this field. - const run = { ...baseRun, userId: user.id }; + // by owner, and attach the reported agents at creation (runs are + // immutable — spread into a new object, never mutate). + const run = { ...baseRun, agents, userId: user.id }; insertRun(db(), run);