diff --git a/.gitignore b/.gitignore index 6cd644c..a1a363f 100644 --- a/.gitignore +++ b/.gitignore @@ -14,3 +14,4 @@ dist/ # Debugging screenshots and scratch artifacts captured during dev *.png +packages/cli/src/build-info.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index b2d71de..a99f400 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -15,6 +15,45 @@ This section accumulates work going into the first tagged customer release. Phase tags (A, B, C) match the customer-trial roadmap and the commit prefix convention used in the git log. +### June 2026 audit sweep (PRs #1–#16) + +A 14-agent review (2026-06-02) produced 28 findings; all shipped by 2026-06-09: +SDK contract + rate-limit fixes, low-tier hardening, a green Playwright smoke +test in CI, a repo-wide lint pass + CI Lint job, `eslint-plugin-react-hooks` +7.1.x upgrade, `checkSameOrigin` CSRF guard rolled out to all cookie-auth +mutation routes, repo-identity normalization and user attribution fixed at +write time, SSRF hardening for 6to4/NAT64 IPv6 wrappers, Usage-page backend +segmentation (Bedrock vs direct) with a per-user spend table, and +better-sqlite3 12.11.1 + CI Node 22 for Node 26 support. + +### July 2026 revival audit (PRs #17–#23) + +A follow-up 4-agent review (2026-07-11) found the cost pipeline and several +security gaps; all critical/high findings shipped the same day: + +- **Cost pipeline repaired (#17).** The model pricing table had gone stale — + current models (Opus 4.8, Sonnet 5, Fable 5) silently priced at $0 and + Opus 4.6/4.7 at 3× their real rate; the transcript reader double-counted + usage 2–7× (Claude Code writes one line per content block; now deduped by + `message.id`); and transcript paths were mis-derived for project dirs with + non-alphanumeric characters (hooks now use the payload's `transcript_path`). + Unknown models now warn on stderr instead of failing open at $0. +- **Four unauthenticated GET routes gated (#18).** `runs/[id]/{agents,metrics, + policies}` and `policies/[id]/results` leaked cross-tenant data; they now + enforce the same owner-or-admin + 404-on-non-owner rules as their siblings, + and policy results are scoped to the member's own runs. +- **Dependency advisories cleared (#19, #21).** Next.js 16.1.6 → 16.2.10 and + drizzle-orm 0.39 → 0.45.2 clear every high advisory in `npm audit`. +- **Usage page org-wide section works for the first time (#22).** The Admin + API proxies sent parameters the API doesn't accept and the page parsed + fields that don't exist in the responses; the routes now send RFC 3339 + `starting_at`, follow pagination, and normalize amounts (decimal-string + cents) server-side. The Activity Summary card (which could only render + zeros) was removed pending Claude Code Analytics API integration. +- **Policies are deletable again (#23).** Deleting any evaluated policy hit a + foreign-key constraint and 500'd; `deletePolicy` now cascades its + `policy_results` in a transaction and reports the count in the audit log. + ### Added #### Phase A — trial-blocking foundations diff --git a/CLAUDE.md b/CLAUDE.md index e50a65f..4499720 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -60,24 +60,24 @@ core ← db ← cli ``` - **@agentops/core** — Domain types, policy engine, scoring algorithm, and builder functions for all entities. No external dependencies. All types use `readonly` and branded ID types (RunId, JobId, SessionId, etc.) for type safety. -- **@agentops/db** — SQLite persistence via Drizzle ORM + better-sqlite3. Eight tables: `runs`, `policies`, `policy_results`, `run_metrics`, `jobs`, `sessions`, `events`, `locks`. Complex fields stored as JSON columns. DB defaults to `~/.agentops/agentops.db` (override with `AGENTOPS_DB_PATH`). -- **@agentops/cli** — CLI entry point (`agentops`). Commands: `init`, `serve`, `setup`, `hook`, `run`, `policy`, `report`, `wrap`, `watch`, `link`, `pr`, `job`, `session`, `events`, `lock`, `dispatch`. Supports `--json` output and `--db-path` override. `init` bootstraps the DB (`--seed` for sample data, `--clean` to reset). `serve` starts the dashboard server (`--port` to override 3000). `setup` configures Claude Code hooks (`--global`, `--uninstall`, `--dry-run`). `hook` handles Claude Code hook events (session-start, pre-tool-use, post-tool-use, session-end) — reads JSON from stdin, manages state via temp files, evaluates policies in real-time, can block risky tool calls (exit code 2). `wrap` emits real-time events during execution. Helper modules: `format.ts` (output formatting), `git.ts` (git integration), `github.ts` (GitHub API). +- **@agentops/db** — SQLite persistence via Drizzle ORM + better-sqlite3. Sixteen tables: `runs`, `policies`, `policy_results`, `run_metrics`, `jobs`, `sessions`, `events`, `locks`, `users`, `api_tokens`, `auth_sessions`, `device_codes`, `webhooks`, `webhook_deliveries`, `audit_log`, `user_budgets`. Complex fields stored as JSON columns. DB defaults to `~/.agentops/agentops.db` (override with `AGENTOPS_DB_PATH`). WAL mode with `foreign_keys = ON` — deletes of parent rows must cascade children first (see `deletePolicy`, `deleteOldRuns`). +- **@agentops/cli** — CLI entry point (`agentops`). Commands: `init`, `serve`, `setup`, `hook`, `login`, `doctor`, `user`, `admin`, `cleanup`, `run`, `policy`, `report`, `wrap`, `watch`, `link`, `pr`, `job`, `session`, `events`, `lock`, `dispatch`. Supports `--json` output and `--db-path` override. `init` bootstraps the DB (`--seed` for sample data, `--seed-policies` for the starter policy set, `--clean` to reset). `serve` starts the dashboard server (`--port` to override 3000). `setup` configures Claude Code hooks (`--global`, `--uninstall`, `--dry-run`). `hook` handles Claude Code hook events (session-start, pre-tool-use, post-tool-use, user-prompt-submit, stop, subagent-stop, session-end) — reads JSON from stdin, manages state under `~/.agentops/state/`, evaluates policies in real-time, can block risky tool calls (exit code 2), and reads cost/token usage from the Claude Code transcript (deduped by `message.id`; unknown models warn on stderr instead of pricing at $0). `login` runs the device-flow auth against a dashboard; `doctor` diagnoses a local install. Helper modules: `format.ts` (output formatting), `git.ts` (git integration), `github.ts` (GitHub API), `transcript.ts` (usage/cost from Claude Code transcripts), `pricing` lives in core. Note: `job`, `lock`, `dispatch`, `wrap`, and `watch` operate on orchestration machinery that Claude Code hooks never populate — treat them as experimental/vestigial. - **@agentops/sdk** — Lightweight HTTP client for agent runtimes to talk to the AgentOps server. Depends only on `@agentops/core` for types. Uses native `fetch`. Provides `AgentOpsClient` class (via `createClient()` factory) with methods: `createSession`, `startRun`, `reportAction`, `reportArtifact`, `reportMetrics`, `checkPolicy`, `heartbeat`, `completeRun`, `failRun`. Also exports `PolicyMiddleware` for pre-flight policy checks before actions. Throws typed `AgentOpsError` with status codes. -- **@agentops/web** — Next.js 16 App Router dashboard with React 19, Tailwind CSS 4. API routes under `src/app/api/` organized by resource (runs, sessions, policies, events, analytics, admin, stats, sdk). Jobs, Locks, and Coordination pages were removed from the dashboard because hooks do not populate this data. Sidebar nav: Runs | Sessions | Events | Analytics | Usage | Policies | Settings. Includes inbound SDK routes under `/api/sdk/` for agent runtime communication and mutation routes for dashboard controls (approve/block runs). The Usage page (`/usage`) proxies to the Anthropic Admin API for cost/usage data; requires the `ANTHROPIC_ADMIN_API_KEY` env var. Admin API routes live at `/api/admin/{status,cost,analytics}`. Server components access SQLite directly via a singleton lazy-loaded DB instance (`src/lib/db.ts`). Uses `serverExternalPackages: ["better-sqlite3"]` in next.config.ts. Path alias: `@/*` → `./src/*`. Dark theme by default. +- **@agentops/web** — Next.js 16 App Router dashboard with React 19, Tailwind CSS 4. API routes under `src/app/api/` organized by resource (runs, sessions, policies, events, analytics, admin, stats, sdk, auth, budgets, webhooks). Jobs, Locks, and Coordination pages were removed from the dashboard because hooks do not populate this data. Sidebar nav: Runs | Sessions | Events | Analytics | Usage | Policies | Settings. **Every API route is authenticated** (`src/lib/auth.ts`): bearer tokens or session cookies, `admin`/`member` roles, members are scoped to their own runs/sessions (`resolveViewScope`), non-owners get 404 (not 403) to avoid ID enumeration, mutations additionally pass `checkSameOrigin` CSRF checks — new routes must follow this pattern (regression tests live in `api/__tests__/auth-gaps.test.ts`). Inbound SDK routes under `/api/sdk/` use bearer auth + per-token rate limits. The Usage page (`/usage`) always shows the local hook-captured rollup (incl. Bedrock-vs-direct backend split and per-user attribution); when `ANTHROPIC_ADMIN_API_KEY` is set it additionally shows org-wide cost/tokens via `/api/admin/{status,cost,analytics}`, which normalize the Anthropic Admin API's reports (RFC 3339 `starting_at`, paginated, amounts are decimal-string cents) server-side. Server components access SQLite directly via a singleton lazy-loaded DB instance (`src/lib/db.ts`). Uses `serverExternalPackages: ["better-sqlite3"]` in next.config.ts. Path alias: `@/*` → `./src/*`. Dark theme by default. ### Core Domain Concepts **Run** is the central abstraction — an immutable, timestamped record of one autonomous execution containing: Goal, Agents (with roles: Lead/Implementer/Reviewer/CI/Policy), Environment, Actions, Artifacts, Metrics, Evaluations, Decisions, and optional GitHub links. Run modifications always return new objects; never mutate in-place. -**Job** (`core/src/job.ts`) is a dispatchable unit of work. Jobs have priority (Critical/High/Normal/Low), retry policies, and concurrency limits. A Job produces one or more Runs and tracks lifecycle: Queued → Dispatched → Running → Completed/Failed. The **Dispatcher** (`core/src/dispatcher.ts`) provides pure functions for queue ordering, dispatch decisions based on concurrency limits, and session matching. +**Job** (`core/src/job.ts`) is a dispatchable unit of work. Jobs have priority (Critical/High/Normal/Low), retry policies, and concurrency limits. A Job produces one or more Runs and tracks lifecycle: Queued → Dispatched → Running → Completed/Failed. The **Dispatcher** (`core/src/dispatcher.ts`) provides pure functions for queue ordering, dispatch decisions based on concurrency limits, and session matching. ⚠️ **Vestigial:** nothing in the hook-driven product path creates Jobs or dispatches them (hook-created sessions always carry a `currentRunId`, so `matchSession` can never select one); this layer is exercised only by its own tests and the experimental `job`/`dispatch` CLI commands. **Session** (`core/src/session.ts`) represents an agent runtime's lifecycle: Provisioning → Active → Paused → Terminated. Sessions track the current Run being executed, completed Runs, resource usage (memory, CPU, token/cost budgets), and heartbeats for liveness detection. **Event System** (`core/src/events.ts`) provides a typed event bus. `EVENT_TYPES` defines all events (job.queued, run.started, session.terminated, policy.violated, cost.threshold, etc.). `EventBus` class supports subscribe/unsubscribe/publish with wildcard `"*"` subscriptions. Events are persisted in the `events` table for audit trail. The web SSE endpoint reads from the events table. -**Coordination** (`core/src/coordination.ts`) handles multi-agent resource management. Lock lifecycle (createLock, releaseLock, isLockExpired, isLockHeld), conflict detection (checkConflicts checks for overlapping repo/path/branch locks), branch isolation (generateWorkBranch), and work partitioning (partitionByPath). +**Coordination** (`core/src/coordination.ts`) handles multi-agent resource management. Lock lifecycle (createLock, releaseLock, isLockExpired, isLockHeld), conflict detection (checkConflicts checks for overlapping repo/path/branch locks), branch isolation (generateWorkBranch), and work partitioning (partitionByPath). ⚠️ **Vestigial:** locks are only ever created via the experimental `lock` CLI command; no runtime path uses them. -**Policy Engine** (`core/src/policy.ts`) evaluates 6 policy types: PathRestriction, FileLimitCount, CostCeiling, RequiredApproval, TestEnforcement, RiskyOpFlag. New policy types must extend the `PolicyConfig` union type. +**Policy Engine** (`core/src/policy.ts`) evaluates 8 policy types: PathRestriction, FileLimitCount, TestEnforcement, RiskyOpFlag, SecretDetection, BranchProtection, ToolRestriction, CostCeiling. Policies run in two modes: **guard** (`evaluatePreToolPolicies` — real-time, called from the PreToolUse hook, can block a tool call) and **check** (post-hoc evaluation of a completed Run). New policy types must extend the `PolicyConfig` union type and usually need both a guard and a check implementation. **Scoring** (`core/src/scoring.ts`) computes a ScoreCard across 5 dimensions (Correctness, RegressionRisk, ScopeRisk, PolicyCompliance, Unknowns) as 0–1 ratios, producing a MergeRecommendation of Merge | Block | Review. @@ -93,6 +93,8 @@ core ← db ← cli - Repository pattern: all DB functions take `(db: AgentOpsDb, ...)` as first arg, named `insertX`, `getX`, `listX`, `updateX` - Tests colocated in `src/__tests__/` directories, pattern `*.test.ts` - Web API routes use `force-dynamic` to ensure fresh data on every request +- Every web API route authenticates via `src/lib/auth.ts` helpers (`requireUser`/`requireAdmin`/`requireOwnedRun`); mutations also call `checkSameOrigin`. Add coverage to `auth-gaps.test.ts` when adding routes +- Web tests import `@agentops/core`/`db` from their built `dist/` — after switching branches or editing core/db, run root `npm run build` first or web tests will exercise stale code - CLI commands registered via `registerXCommands(program)` pattern in `cli/src/index.ts` - CLI auto-detects git repo/branch; can override with `--repo` and `--branch` flags - DB defaults to `~/.agentops/agentops.db`; override with `AGENTOPS_DB_PATH` env var or `--db-path` CLI flag @@ -102,4 +104,6 @@ core ← db ← cli ### Environment Variables - `AGENTOPS_DB_PATH` — Override default SQLite database location (`~/.agentops/agentops.db`) -- `ANTHROPIC_ADMIN_API_KEY` — Anthropic Admin API key for the Usage page. Set before starting the dashboard (`agentops serve`) to enable cost/usage tracking from the Anthropic API. Obtain from the Anthropic Console under Organization Settings. +- `AGENTOPS_FAIL_CLOSED` — Set to `1`/`true` to make hooks block tool calls when enforcement can't be verified (offline dashboard, rejected token). Default is fail-open so AgentOps never bricks a working Claude Code session. +- `CLAUDE_CODE_USE_BEDROCK` — Read (not set) by the hooks to tag a session's spend as `bedrock` vs direct `anthropic` for the Usage page's backend split. +- `ANTHROPIC_ADMIN_API_KEY` — Anthropic Admin API key for the org-wide half of the Usage page. Set before starting the dashboard (`agentops serve`). Obtain from the Anthropic Console under Organization Settings. Model pricing for the local rollup lives in `core/src/pricing.ts` (`ANTHROPIC_PRICING`) — **it must be refreshed when new Claude models ship**, or their sessions warn on stderr and record $0. diff --git a/docker-compose.yml b/docker-compose.yml index afbdd25..adea856 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -77,9 +77,11 @@ services: depends_on: - dashboard volumes: - # Read-only mount of the SQLite db. Litestream reads the WAL frames - # but never writes back to the live database. - - ./agentops-data:/data:ro + # Read-write mount: Litestream must be able to open the database with + # write access — it takes the WAL checkpoint lock and maintains its own + # -litestream sidecar files next to the db. A :ro mount here breaks + # replication outright (it fails at startup, it doesn't degrade). + - ./agentops-data:/data - ./litestream.yml:/etc/litestream.yml:ro environment: AGENTOPS_S3_BUCKET: ${AGENTOPS_S3_BUCKET} diff --git a/litestream.yml b/litestream.yml index 3de20b3..e24279b 100644 --- a/litestream.yml +++ b/litestream.yml @@ -25,7 +25,11 @@ dbs: # Prefix within the bucket. Lets one bucket back up multiple DBs. path: ${AGENTOPS_S3_PREFIX:-agentops} region: ${AWS_REGION:-us-east-1} - # Optional endpoint override for S3-compatible stores. + # Using an S3-compatible store (MinIO, R2, Wasabi)? Set + # AGENTOPS_S3_ENDPOINT in .env AND uncomment the next line — the + # env var alone does nothing while this stays commented. Leave it + # commented for real AWS S3 (an empty endpoint can break the AWS + # default resolution). # endpoint: ${AGENTOPS_S3_ENDPOINT} # 14 days of point-in-time recovery. Tune for your RPO needs; # the cost of WAL frames at AgentOps scale is sub-cent / day. diff --git a/packages/cli/package.json b/packages/cli/package.json index 66b3d97..2d64005 100644 --- a/packages/cli/package.json +++ b/packages/cli/package.json @@ -34,6 +34,7 @@ ], "scripts": { "prebuild": "node scripts/write-build-info.mjs", + "pretest": "node scripts/write-build-info.mjs", "build": "tsc", "lint": "eslint src", "test": "vitest run" diff --git a/packages/cli/src/build-info.ts b/packages/cli/src/build-info.ts deleted file mode 100644 index a0c924e..0000000 --- a/packages/cli/src/build-info.ts +++ /dev/null @@ -1,8 +0,0 @@ -// Auto-generated by scripts/write-build-info.mjs at build time. -// Do not edit by hand — `git status` is expected to show this file as -// modified after every build, and the next prebuild regenerates it. - -export const VERSION = "0.1.0"; -export const GIT_SHA = "9d4df2a"; -export const DIRTY = true; -export const BUILT_AT = "2026-05-27T15:57:50.301Z"; diff --git a/packages/cli/src/commands/serve.ts b/packages/cli/src/commands/serve.ts index b70f44d..aaa64f5 100644 --- a/packages/cli/src/commands/serve.ts +++ b/packages/cli/src/commands/serve.ts @@ -60,9 +60,10 @@ export function registerServeCommand(program: Command): void { console.log(`AgentOps dashboard starting at http://${displayHost}:${port}`); if (host !== "127.0.0.1" && host !== "localhost") { console.warn( - `WARNING: dashboard is binding to ${host}. The API is currently unauthenticated — ` - + `anyone with network reach can read and mutate data. Bind to 127.0.0.1 unless ` - + `you have added auth.`, + `Note: dashboard is binding to ${host} and reachable from the network. ` + + `The API requires authentication (run 'agentops login' to create the first ` + + `admin), but traffic is plain HTTP — put it behind TLS for anything beyond ` + + `a trusted LAN (see docker-compose.yml's caddy profile).`, ); } diff --git a/scripts/demo-seed.mjs b/scripts/demo-seed.mjs new file mode 100644 index 0000000..fd78592 --- /dev/null +++ b/scripts/demo-seed.mjs @@ -0,0 +1,260 @@ +#!/usr/bin/env node +// Wipes the local AgentOps DB's runs / events / policy_results and +// regenerates a team-shaped demo distribution: 6 users with realistic +// monthly usage, one user in the 80% budget warning band, last 30 +// days of activity. Keeps users / policies / budgets / webhooks +// (other than the budget upserts below) untouched. +// +// DESTRUCTIVE — requires --yes. Usage: +// npm run build # ensure dist/ is current +// node scripts/demo-seed.mjs --yes # against ~/.agentops/agentops.db +// AGENTOPS_DB_PATH=/path/to.db node scripts/demo-seed.mjs --yes + +import { + getDb, + listUsers, + insertRun, + upsertBudget, +} from "../packages/db/dist/index.js"; +import { + createRunId, + createActionId, + RunStatus, +} from "../packages/core/dist/index.js"; +import Database from "better-sqlite3"; +import { resolve } from "node:path"; +import { homedir } from "node:os"; + +function resolveDbPath() { + const fromEnv = process.env["AGENTOPS_DB_PATH"]?.trim(); + if (fromEnv) return fromEnv; + return resolve(homedir(), ".agentops", "agentops.db"); +} + +// Target distribution. Sarah lands in the 80–99% warning band against +// a $300 cap — useful for showing the budget enforcement UX in action. +const PROFILES = [ + { email: "sarah@example.com", runs: 18, total: 280, budget: 300 }, + { email: "marcus@example.com", runs: 22, total: 145, budget: 250 }, + { email: "ian@example.com", runs: 25, total: 115, budget: 250 }, + { email: "priya@example.com", runs: 15, total: 97, budget: 250 }, + { email: "diego@example.com", runs: 12, total: 52, budget: 200 }, + { email: "teammate@acme.com", runs: 5, total: 12, budget: 200 }, +]; + +const REPOS = [ + { repo: "acme/backend", weight: 4 }, + { repo: "acme/frontend", weight: 3 }, + { repo: "acme/api", weight: 2 }, + { repo: "acme/mobile", weight: 2 }, + { repo: "acme/infra", weight: 1 }, +]; + +const BRANCHES = [ + "main", + "feature/auth-rewrite", + "feature/checkout-v2", + "bugfix/migration-rollback", + "feature/dashboard-redesign", + "feature/sso-integration", + "feature/budget-alerts", +]; + +const GOALS = [ + { text: "Refactor session middleware for new token strategy", type: "refactor" }, + { text: "Fix race condition in webhook retry loop", type: "bugfix" }, + { text: "Add per-user budget alerts on dashboard home", type: "feature" }, + { text: "Migrate user_orders to partitioned schema", type: "migration" }, + { text: "Investigate flaky checkout E2E suite", type: "investigation" }, + { text: "Add bulk-action support to runs list", type: "feature" }, + { text: "Tighten policy_result query plan (slow on >50k rows)", type: "perf" }, + { text: "Update README with self-host docker instructions", type: "docs" }, + { text: "Strip dead code from coordination module", type: "cleanup" }, + { text: "Wire up Stop-hook budget warning", type: "feature" }, + { text: "Add audit log entry for policy toggle", type: "feature" }, + { text: "Fix transcript path validation on Windows", type: "bugfix" }, + { text: "Add retry-after honor to outbound webhook dispatcher", type: "feature" }, + { text: "Reduce flake in hook-integration tests", type: "test" }, +]; + +const TOOL_NAMES = ["Bash", "Edit", "Read", "Write", "Grep"]; + +function pick(arr) { return arr[Math.floor(Math.random() * arr.length)]; } +function pickWeighted(weighted) { + const total = weighted.reduce((s, w) => s + w.weight, 0); + let r = Math.random() * total; + for (const w of weighted) { + r -= w.weight; + if (r <= 0) return w.repo; + } + return weighted[0].repo; +} +function randInt(min, max) { return Math.floor(Math.random() * (max - min + 1)) + min; } +function randFloat(min, max) { return Math.random() * (max - min) + min; } +function uid() { return Math.random().toString(36).slice(2, 10); } + +function daysAgo(days, jitterHours = 12) { + const t = Date.now() - days * 86400000 - randInt(0, jitterHours * 3600000); + return new Date(t); +} + +function generateRun(profile, user, idx) { + // 70% of runs in the last 14 days, 30% spread across days 14–30. + // Makes "recent activity" feel right while still showing history. + const day = Math.random() < 0.7 ? randInt(0, 14) : randInt(14, 30); + const created = daysAgo(day); + const updated = new Date(created.getTime() + randInt(5, 90) * 60_000); + + // Per-run cost varies around the profile average, with ~10% outliers. + const avgCost = profile.total / profile.runs; + const cost = Math.random() < 0.9 + ? Math.max(0.10, randFloat(avgCost * 0.3, avgCost * 1.8)) + : randFloat(avgCost * 2, avgCost * 4); + + // Status mix: 85% completed, 10% blocked (good for showing the + // enforcement story), 5% failed. + const r = randInt(0, 99); + const status = r < 85 ? RunStatus.Completed : r < 95 ? RunStatus.Blocked : RunStatus.Failed; + + const goal = pick(GOALS); + const actionCount = randInt(2, 12); + + // Rough Opus 4.7 base-input rate: ~$0.000015/token. Reverse-engineer + // a token count from cost so the dashboard tokens column looks coherent. + const tokens = Math.round(cost / 0.000015); + + const actions = Array.from({ length: actionCount }, (_, i) => ({ + id: createActionId(`action_${uid()}_${i}`), + toolCalls: [{ + name: pick(TOOL_NAMES), + input: {}, + output: "", + timestamp: created.toISOString(), + }], + fileEdits: [], + commands: [], + timestamp: created.toISOString(), + })); + + return { + id: createRunId(`run_demo_${uid()}_${idx}`), + status, + goal: { + humanReadable: goal.text, + structured: { type: goal.type, description: goal.text, parameters: {} }, + }, + agents: [], + environment: { + repo: pickWeighted(REPOS), + branch: pick(BRANCHES), + permissions: ["read", "write", "execute"], + sandbox: { enabled: true, isolationLevel: "container" }, + }, + actions, + artifacts: [], + metrics: { + tokenUsage: { + input: Math.round(tokens * 0.6), + output: Math.round(tokens * 0.4), + total: tokens, + }, + wallTimeMs: randInt(60_000, 1_800_000), + costUsd: Number(cost.toFixed(2)), + flakeRate: 0, + }, + evaluations: [], + decisions: [], + userId: user.id, + createdAt: created.toISOString(), + updatedAt: updated.toISOString(), + }; +} + +function main() { + // Destructive: wipes all runs/events/policy_results/run_metrics before + // regenerating the demo distribution. Require an explicit opt-in so a + // stray invocation can't erase real captured history. + if (!process.argv.includes("--yes")) { + console.error( + "demo-seed WIPES all runs, events, policy results, and run metrics in the target DB\n" + + `(${resolveDbPath()}) and replaces them with fabricated demo data.\n\n` + + "Re-run with --yes to confirm:\n" + + " node scripts/demo-seed.mjs --yes", + ); + process.exit(1); + } + + const db = getDb(); + const allUsers = listUsers(db); + const byEmail = new Map(allUsers.map((u) => [u.email, u])); + + const missing = PROFILES.filter((p) => !byEmail.has(p.email)); + if (missing.length > 0) { + console.error(`Missing users for: ${missing.map((m) => m.email).join(", ")}`); + console.error("Run `agentops init --seed` first to create the seed users."); + process.exit(1); + } + + // Bypass drizzle + foreign-keys for the wipe. drizzle's row-by-row + // DELETE chokes on FK constraints (run_metrics → runs, policy_results + // → runs) even when we delete children first. A raw exec with FK + // temporarily off is reliable and unambiguous. + console.log("Wiping runs / events / policy_results / run_metrics …"); + const raw = new Database(resolveDbPath()); + try { + raw.pragma("foreign_keys = OFF"); + raw.exec(` + DELETE FROM events; + DELETE FROM policy_results; + DELETE FROM run_metrics; + DELETE FROM runs; + `); + raw.pragma("foreign_keys = ON"); + } finally { + raw.close(); + } + + let runCount = 0; + let totalCost = 0; + console.log("Generating distribution:"); + for (const profile of PROFILES) { + const user = byEmail.get(profile.email); + + upsertBudget(db, { + userId: user.id, + amountUsd: profile.budget, + period: "month", + warnAtPct: 80, + }); + + // Generate all runs first, then scale per-run costs proportionally + // so the user's total lands exactly on profile.total. Pinning only + // the last run leaves wide variance — Sarah might be at 93% one + // run and 107% the next, blowing the warning-band demo. Scaling + // keeps relative variance between runs but makes totals deterministic. + const userRuns = Array.from({ length: profile.runs }, (_, i) => + generateRun(profile, user, i), + ); + const rawTotal = userRuns.reduce((s, r) => s + r.metrics.costUsd, 0); + const scale = profile.total / rawTotal; + let userCost = 0; + for (const run of userRuns) { + run.metrics.costUsd = Number((run.metrics.costUsd * scale).toFixed(2)); + userCost += run.metrics.costUsd; + insertRun(db, run); + runCount++; + } + totalCost += userCost; + const pct = ((userCost / profile.budget) * 100).toFixed(0); + console.log( + ` ${profile.email.padEnd(24)} ${String(profile.runs).padStart(2)} runs ` + + `$${userCost.toFixed(2).padStart(7)} / $${profile.budget} (${pct}%)`, + ); + } + + console.log(""); + console.log(`Team total: ${runCount} runs across last 30 days, $${totalCost.toFixed(2)} spend.`); + console.log("Open the dashboard to view."); +} + +main();