From 242c903da45840eb83a8ab9ae6a7b74938d0c875 Mon Sep 17 00:00:00 2001 From: Lucas Armstrong Date: Sat, 26 Sep 2026 11:41:47 -0400 Subject: [PATCH 1/2] 0.3.0: implement reactive triggers, >=90% test coverage, and Pentad synergy --- .agents/AGENTS.md | 17 +- .agents/skills/agent-reasoning-mcp/SKILL.md | 7 +- .cursor/rules/agent-reasoning-mcp.mdc | 7 +- .gemini/instructions.md | 79 ++++- .github/copilot-instructions.md | 79 ++++- .vscode/instructions.md | 79 ++++- .windsurfrules | 79 ++++- CLAUDE.md | 79 ++++- MIGRATION.md | 2 +- README.md | 21 +- docs/.well-known/mcp.json | 4 +- docs/api-reference.md | 2 +- docs/index.html | 2 +- manifest.json | 2 +- package-lock.json | 4 +- package.json | 2 +- scripts/test-matrix.sh | 2 + server.json | 4 +- src/engine/browser/actions.ts | 13 + src/engine/browser/conditions.ts | 23 ++ src/engine/browser/executor-bundle.ts | 38 ++- src/engine/config.ts | 9 + src/engine/triggers.ts | 34 +- src/schema/types.ts | 12 + src/utils/canonical-json.ts | 55 ++++ src/utils/version.ts | 2 +- tests/contracts/state-pack-hash.test.ts | 29 ++ tests/engine/conditions.test.ts | 77 +++++ tests/engine/token-verification.test.ts | 119 +++++++ tests/fixtures/canonical-state-pack.json | 72 +++++ tests/transport/native-mcp-coverage.test.ts | 66 ++++ tests/unit/engine-and-handlers-boost.test.ts | 163 ++++++++++ .../unit/schemas-exhaustive-coverage.test.ts | 301 ++++++++++++++++++ vitest.config.ts | 1 + 34 files changed, 1440 insertions(+), 45 deletions(-) create mode 100644 src/utils/canonical-json.ts create mode 100644 tests/contracts/state-pack-hash.test.ts create mode 100644 tests/engine/conditions.test.ts create mode 100644 tests/engine/token-verification.test.ts create mode 100644 tests/fixtures/canonical-state-pack.json create mode 100644 tests/transport/native-mcp-coverage.test.ts create mode 100644 tests/unit/engine-and-handlers-boost.test.ts create mode 100644 tests/unit/schemas-exhaustive-coverage.test.ts diff --git a/.agents/AGENTS.md b/.agents/AGENTS.md index f205b02..13901f2 100644 --- a/.agents/AGENTS.md +++ b/.agents/AGENTS.md @@ -150,7 +150,7 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana 5. **Intention Dispatch**: Create execution directives with `manage_intentions(action: "create", ...)` for the runtime engine. 6. **Reactive Replanning**: If an unexpected blocker occurs, invoke `replan(action: "blocker", goal_id: "...", blocker_description: "...")`. -## 10 Core MCP Tools +## 15 Core MCP Tools - `set_goal`: Manage goal hierarchy and task DAGs. - `evaluate_situation`: Score and rank candidate actions from environment snapshots. - `replan`: Adaptively reconstruct subgoals upon obstacles. @@ -161,6 +161,11 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana - `manage_beliefs`: Structured belief state with exponential confidence decay. - `manage_intentions`: Wire contract directives queue for runtime execution. - `manage_reasoning_db`: Snapshots, diagnostics, and SHA-256 Merkle audit verification. +- `classify`: Zero-LLM deterministic classification against hierarchical taxonomy (<2ms SLA). +- `ask_noul`: Fast binary (Yes/No/Abstain) heuristic gate evaluating conditions (<2ms SLA). +- `ask_choice`: Deterministic multi-alternative selection ranking candidate choices (<2ms SLA). +- `ask_score`: Heuristic utility evaluation scoring target entities on a bounded scale (<2ms SLA). +- `gate_intention`: Fast-path safety & feasibility filter checking preconditions before execution (<1ms SLA). @@ -183,11 +188,11 @@ This project provides native `webcrypt-mcp` tooling for zero-dependency AES-256- Active Supervised MCP Servers: * `putervision-harness`: pv-harness start --project test_slug -* `state-memory-mcp`: state-memory-mcp --project test_slug -* `vision-memory-mcp`: vision-memory-mcp --project test_slug -* `world-model-mcp`: world-model-mcp --project test_slug -* `agent-reasoning-mcp`: agent-reasoning-mcp --project test_slug -* `behavior-mcp`: behavior-mcp --project test_slug +* `state-memory-mcp`: state-memory-mcp +* `vision-memory-mcp`: vision-memory-mcp +* `world-model-mcp`: world-model-mcp +* `agent-reasoning-mcp`: agent-reasoning-mcp +* `behavior-mcp`: behavior-mcp * `test-custom`: npx -y @org/test-custom Always use `harness_start_loop` and supervise tasks via the PuterVision Harness. diff --git a/.agents/skills/agent-reasoning-mcp/SKILL.md b/.agents/skills/agent-reasoning-mcp/SKILL.md index a45c2ff..328db38 100644 --- a/.agents/skills/agent-reasoning-mcp/SKILL.md +++ b/.agents/skills/agent-reasoning-mcp/SKILL.md @@ -27,7 +27,7 @@ This skill provides step-by-step guidance and operational patterns for interacti --- -## 3. Complete 10 Consolidated MCP Tools Reference +## 3. Complete 15 Consolidated MCP Tools Reference | Tool Name | Key Actions | Key Parameters | Description | |---|---|---|---| @@ -41,3 +41,8 @@ This skill provides step-by-step guidance and operational patterns for interacti | `manage_beliefs` | `set`, `get`, `decay`, `list` | `key`, `value`, `confidence`, `decay_rate` | Structured belief state with temporal exponential confidence decay. | | `manage_intentions` | `create`, `get`, `list`, `dispatch`, `cancel` | `goal_id`, `behavior_name`, `parameters` | Execution directives queue connecting strategic plans to runtime engines. | | `manage_reasoning_db` | `stats`, `audit`, `snapshot`, `restore`, `prune` | `action`, `name`, `description` | Database diagnostics, snapshots, and SHA-256 Merkle audit verification. | +| `classify` | evaluation | `category`, `input`, `taxonomy`, `state_pack` | Zero-LLM deterministic classification against hierarchical taxonomy (<2ms SLA). | +| `ask_noul` | evaluation | `condition`, `state_pack`, `threshold` | Fast binary (Yes/No/Abstain) heuristic gate evaluating conditions (<2ms SLA). | +| `ask_choice` | evaluation | `choices`, `context`, `state_pack` | Deterministic multi-alternative selection ranking candidate choices (<2ms SLA). | +| `ask_score` | evaluation | `target`, `metric`, `scale`, `state_pack` | Heuristic utility evaluation scoring target entities on a bounded scale (<2ms SLA). | +| `gate_intention` | evaluation | `project`, `proposed_action`, `state_pack` | Fast-path safety & feasibility filter checking preconditions before execution (<1ms SLA). | diff --git a/.cursor/rules/agent-reasoning-mcp.mdc b/.cursor/rules/agent-reasoning-mcp.mdc index cebfd1c..0c9e8d3 100644 --- a/.cursor/rules/agent-reasoning-mcp.mdc +++ b/.cursor/rules/agent-reasoning-mcp.mdc @@ -11,7 +11,7 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana 5. **Intention Dispatch**: Create execution directives with `manage_intentions(action: "create", ...)` for the runtime engine. 6. **Reactive Replanning**: If an unexpected blocker occurs, invoke `replan(action: "blocker", goal_id: "...", blocker_description: "...")`. -## 10 Core MCP Tools +## 15 Core MCP Tools - `set_goal`: Manage goal hierarchy and task DAGs. - `evaluate_situation`: Score and rank candidate actions from environment snapshots. - `replan`: Adaptively reconstruct subgoals upon obstacles. @@ -22,4 +22,9 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana - `manage_beliefs`: Structured belief state with exponential confidence decay. - `manage_intentions`: Wire contract directives queue for runtime execution. - `manage_reasoning_db`: Snapshots, diagnostics, and SHA-256 Merkle audit verification. +- `classify`: Zero-LLM deterministic classification against hierarchical taxonomy (<2ms SLA). +- `ask_noul`: Fast binary (Yes/No/Abstain) heuristic gate evaluating conditions (<2ms SLA). +- `ask_choice`: Deterministic multi-alternative selection ranking candidate choices (<2ms SLA). +- `ask_score`: Heuristic utility evaluation scoring target entities on a bounded scale (<2ms SLA). +- `gate_intention`: Fast-path safety & feasibility filter checking preconditions before execution (<1ms SLA). diff --git a/.gemini/instructions.md b/.gemini/instructions.md index c9dcc20..071a3b5 100644 --- a/.gemini/instructions.md +++ b/.gemini/instructions.md @@ -161,7 +161,7 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana 5. **Intention Dispatch**: Create execution directives with `manage_intentions(action: "create", ...)` for the runtime engine. 6. **Reactive Replanning**: If an unexpected blocker occurs, invoke `replan(action: "blocker", goal_id: "...", blocker_description: "...")`. -## 10 Core MCP Tools +## 15 Core MCP Tools - `set_goal`: Manage goal hierarchy and task DAGs. - `evaluate_situation`: Score and rank candidate actions from environment snapshots. - `replan`: Adaptively reconstruct subgoals upon obstacles. @@ -172,6 +172,11 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana - `manage_beliefs`: Structured belief state with exponential confidence decay. - `manage_intentions`: Wire contract directives queue for runtime execution. - `manage_reasoning_db`: Snapshots, diagnostics, and SHA-256 Merkle audit verification. +- `classify`: Zero-LLM deterministic classification against hierarchical taxonomy (<2ms SLA). +- `ask_noul`: Fast binary (Yes/No/Abstain) heuristic gate evaluating conditions (<2ms SLA). +- `ask_choice`: Deterministic multi-alternative selection ranking candidate choices (<2ms SLA). +- `ask_score`: Heuristic utility evaluation scoring target entities on a bounded scale (<2ms SLA). +- `gate_intention`: Fast-path safety & feasibility filter checking preconditions before execution (<1ms SLA). ## State Memory (state-memory-mcp) @@ -389,3 +394,75 @@ If the project was just initialized or is missing high-level structure (Plans, M 1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. 2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. 3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. + +## State Memory (state-memory-mcp) + +This project tracks workflow state, tasks, design decisions, and blockers using `state-memory-mcp` with project slug `"behavior-mcp"`. + +### 1. Priority Order +Before doing any coding or investigation: +1. `manage_sessions(action: "start")` — Start a tracking session for full change attribution. +2. `get_analytics(action: "summary")` — Run to understand current project state, active branches, and overall progress. +3. `manage_tasks(action: "next")` — Query prioritized runnable tasks. +4. `manage_tasks(action: "find_blockers")` — Identify any active blockers preventing progress. +5. `manage_nodes(action: "list")` — Find pending tasks, past decisions, or milestones. +6. `query_graph(action: "trace")` — Trace what depends on or blocks a task. + +### 2. When to Write to the Graph +You MUST update the graph as you work: +- **Starting a session**: Always call `manage_sessions(action: "start", agent_id: "my-agent")` to track all mutations under a unique session. +- **Starting a new task**: Create a node with `manage_nodes(action: "create", type: "task", title: "...", session_id: session_id)`. +- **Making a design or implementation decision**: Document it with `manage_nodes(action: "create", type: "decision", title: "...", metadata: { "rationale": "..." }, session_id: session_id)`. +- **Encountering a blocker**: Record the blocker with `manage_nodes(action: "create", type: "blocker", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "blocks", source_id: blocker_id, target_id: task_id, session_id: session_id)`. +- **Adding observation notes**: Atomically log notes using `manage_nodes(action: "add_note", text: "...", attach_to: node_id)`. +- **Batch updates**: Bulk update tasks/nodes using `manage_nodes(action: "batch_update", ids: ["..."], status: "done")`. +- **Completing a task**: Update status to done using `manage_tasks(action: "complete", task_id: task_id)` or `manage_nodes(action: "update", id: task_id, status: "done")`. +- **Creating/generating a new file**: Create an artifact node with `manage_nodes(action: "create", type: "artifact", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "produces", source_id: task_id, target_id: artifact_id)`. + +### 3. Workflow Pattern +1. **Start of session**: Call `manage_sessions(action: "start")` to align and track work, then run `get_analytics(action: "summary")`, `manage_tasks(action: "next")`, and `manage_tasks(action: "find_blockers")`. +2. **Task decomposition**: Decompose user requests into tasks and add them to the graph. +3. **Execution**: Mark tasks as "in_progress", document design decisions as they occur, and log blockers if you hit any obstacles. +4. **Validation & Resolution**: Run `run_diagnostics(action: "validate")` to ensure no cycles/orphans/contradictions, mark tasks as "done", document completed artifacts, and resolve blockers. Call `manage_sessions(action: "end")` to finalize. + +### 4. Codebase Seeding on Initialization +If the project was just initialized or is missing high-level structure (Plans, Milestones, Decisions): +1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. +2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. +3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. + +## State Memory (state-memory-mcp) + +This project tracks workflow state, tasks, design decisions, and blockers using `state-memory-mcp` with project slug `"behavior-mcp"`. + +### 1. Priority Order +Before doing any coding or investigation: +1. `manage_sessions(action: "start")` — Start a tracking session for full change attribution. +2. `get_analytics(action: "summary")` — Run to understand current project state, active branches, and overall progress. +3. `manage_tasks(action: "next")` — Query prioritized runnable tasks. +4. `manage_tasks(action: "find_blockers")` — Identify any active blockers preventing progress. +5. `manage_nodes(action: "list")` — Find pending tasks, past decisions, or milestones. +6. `query_graph(action: "trace")` — Trace what depends on or blocks a task. + +### 2. When to Write to the Graph +You MUST update the graph as you work: +- **Starting a session**: Always call `manage_sessions(action: "start", agent_id: "my-agent")` to track all mutations under a unique session. +- **Starting a new task**: Create a node with `manage_nodes(action: "create", type: "task", title: "...", session_id: session_id)`. +- **Making a design or implementation decision**: Document it with `manage_nodes(action: "create", type: "decision", title: "...", metadata: { "rationale": "..." }, session_id: session_id)`. +- **Encountering a blocker**: Record the blocker with `manage_nodes(action: "create", type: "blocker", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "blocks", source_id: blocker_id, target_id: task_id, session_id: session_id)`. +- **Adding observation notes**: Atomically log notes using `manage_nodes(action: "add_note", text: "...", attach_to: node_id)`. +- **Batch updates**: Bulk update tasks/nodes using `manage_nodes(action: "batch_update", ids: ["..."], status: "done")`. +- **Completing a task**: Update status to done using `manage_tasks(action: "complete", task_id: task_id)` or `manage_nodes(action: "update", id: task_id, status: "done")`. +- **Creating/generating a new file**: Create an artifact node with `manage_nodes(action: "create", type: "artifact", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "produces", source_id: task_id, target_id: artifact_id)`. + +### 3. Workflow Pattern +1. **Start of session**: Call `manage_sessions(action: "start")` to align and track work, then run `get_analytics(action: "summary")`, `manage_tasks(action: "next")`, and `manage_tasks(action: "find_blockers")`. +2. **Task decomposition**: Decompose user requests into tasks and add them to the graph. +3. **Execution**: Mark tasks as "in_progress", document design decisions as they occur, and log blockers if you hit any obstacles. +4. **Validation & Resolution**: Run `run_diagnostics(action: "validate")` to ensure no cycles/orphans/contradictions, mark tasks as "done", document completed artifacts, and resolve blockers. Call `manage_sessions(action: "end")` to finalize. + +### 4. Codebase Seeding on Initialization +If the project was just initialized or is missing high-level structure (Plans, Milestones, Decisions): +1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. +2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. +3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md index c9dcc20..071a3b5 100644 --- a/.github/copilot-instructions.md +++ b/.github/copilot-instructions.md @@ -161,7 +161,7 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana 5. **Intention Dispatch**: Create execution directives with `manage_intentions(action: "create", ...)` for the runtime engine. 6. **Reactive Replanning**: If an unexpected blocker occurs, invoke `replan(action: "blocker", goal_id: "...", blocker_description: "...")`. -## 10 Core MCP Tools +## 15 Core MCP Tools - `set_goal`: Manage goal hierarchy and task DAGs. - `evaluate_situation`: Score and rank candidate actions from environment snapshots. - `replan`: Adaptively reconstruct subgoals upon obstacles. @@ -172,6 +172,11 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana - `manage_beliefs`: Structured belief state with exponential confidence decay. - `manage_intentions`: Wire contract directives queue for runtime execution. - `manage_reasoning_db`: Snapshots, diagnostics, and SHA-256 Merkle audit verification. +- `classify`: Zero-LLM deterministic classification against hierarchical taxonomy (<2ms SLA). +- `ask_noul`: Fast binary (Yes/No/Abstain) heuristic gate evaluating conditions (<2ms SLA). +- `ask_choice`: Deterministic multi-alternative selection ranking candidate choices (<2ms SLA). +- `ask_score`: Heuristic utility evaluation scoring target entities on a bounded scale (<2ms SLA). +- `gate_intention`: Fast-path safety & feasibility filter checking preconditions before execution (<1ms SLA). ## State Memory (state-memory-mcp) @@ -389,3 +394,75 @@ If the project was just initialized or is missing high-level structure (Plans, M 1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. 2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. 3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. + +## State Memory (state-memory-mcp) + +This project tracks workflow state, tasks, design decisions, and blockers using `state-memory-mcp` with project slug `"behavior-mcp"`. + +### 1. Priority Order +Before doing any coding or investigation: +1. `manage_sessions(action: "start")` — Start a tracking session for full change attribution. +2. `get_analytics(action: "summary")` — Run to understand current project state, active branches, and overall progress. +3. `manage_tasks(action: "next")` — Query prioritized runnable tasks. +4. `manage_tasks(action: "find_blockers")` — Identify any active blockers preventing progress. +5. `manage_nodes(action: "list")` — Find pending tasks, past decisions, or milestones. +6. `query_graph(action: "trace")` — Trace what depends on or blocks a task. + +### 2. When to Write to the Graph +You MUST update the graph as you work: +- **Starting a session**: Always call `manage_sessions(action: "start", agent_id: "my-agent")` to track all mutations under a unique session. +- **Starting a new task**: Create a node with `manage_nodes(action: "create", type: "task", title: "...", session_id: session_id)`. +- **Making a design or implementation decision**: Document it with `manage_nodes(action: "create", type: "decision", title: "...", metadata: { "rationale": "..." }, session_id: session_id)`. +- **Encountering a blocker**: Record the blocker with `manage_nodes(action: "create", type: "blocker", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "blocks", source_id: blocker_id, target_id: task_id, session_id: session_id)`. +- **Adding observation notes**: Atomically log notes using `manage_nodes(action: "add_note", text: "...", attach_to: node_id)`. +- **Batch updates**: Bulk update tasks/nodes using `manage_nodes(action: "batch_update", ids: ["..."], status: "done")`. +- **Completing a task**: Update status to done using `manage_tasks(action: "complete", task_id: task_id)` or `manage_nodes(action: "update", id: task_id, status: "done")`. +- **Creating/generating a new file**: Create an artifact node with `manage_nodes(action: "create", type: "artifact", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "produces", source_id: task_id, target_id: artifact_id)`. + +### 3. Workflow Pattern +1. **Start of session**: Call `manage_sessions(action: "start")` to align and track work, then run `get_analytics(action: "summary")`, `manage_tasks(action: "next")`, and `manage_tasks(action: "find_blockers")`. +2. **Task decomposition**: Decompose user requests into tasks and add them to the graph. +3. **Execution**: Mark tasks as "in_progress", document design decisions as they occur, and log blockers if you hit any obstacles. +4. **Validation & Resolution**: Run `run_diagnostics(action: "validate")` to ensure no cycles/orphans/contradictions, mark tasks as "done", document completed artifacts, and resolve blockers. Call `manage_sessions(action: "end")` to finalize. + +### 4. Codebase Seeding on Initialization +If the project was just initialized or is missing high-level structure (Plans, Milestones, Decisions): +1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. +2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. +3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. + +## State Memory (state-memory-mcp) + +This project tracks workflow state, tasks, design decisions, and blockers using `state-memory-mcp` with project slug `"behavior-mcp"`. + +### 1. Priority Order +Before doing any coding or investigation: +1. `manage_sessions(action: "start")` — Start a tracking session for full change attribution. +2. `get_analytics(action: "summary")` — Run to understand current project state, active branches, and overall progress. +3. `manage_tasks(action: "next")` — Query prioritized runnable tasks. +4. `manage_tasks(action: "find_blockers")` — Identify any active blockers preventing progress. +5. `manage_nodes(action: "list")` — Find pending tasks, past decisions, or milestones. +6. `query_graph(action: "trace")` — Trace what depends on or blocks a task. + +### 2. When to Write to the Graph +You MUST update the graph as you work: +- **Starting a session**: Always call `manage_sessions(action: "start", agent_id: "my-agent")` to track all mutations under a unique session. +- **Starting a new task**: Create a node with `manage_nodes(action: "create", type: "task", title: "...", session_id: session_id)`. +- **Making a design or implementation decision**: Document it with `manage_nodes(action: "create", type: "decision", title: "...", metadata: { "rationale": "..." }, session_id: session_id)`. +- **Encountering a blocker**: Record the blocker with `manage_nodes(action: "create", type: "blocker", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "blocks", source_id: blocker_id, target_id: task_id, session_id: session_id)`. +- **Adding observation notes**: Atomically log notes using `manage_nodes(action: "add_note", text: "...", attach_to: node_id)`. +- **Batch updates**: Bulk update tasks/nodes using `manage_nodes(action: "batch_update", ids: ["..."], status: "done")`. +- **Completing a task**: Update status to done using `manage_tasks(action: "complete", task_id: task_id)` or `manage_nodes(action: "update", id: task_id, status: "done")`. +- **Creating/generating a new file**: Create an artifact node with `manage_nodes(action: "create", type: "artifact", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "produces", source_id: task_id, target_id: artifact_id)`. + +### 3. Workflow Pattern +1. **Start of session**: Call `manage_sessions(action: "start")` to align and track work, then run `get_analytics(action: "summary")`, `manage_tasks(action: "next")`, and `manage_tasks(action: "find_blockers")`. +2. **Task decomposition**: Decompose user requests into tasks and add them to the graph. +3. **Execution**: Mark tasks as "in_progress", document design decisions as they occur, and log blockers if you hit any obstacles. +4. **Validation & Resolution**: Run `run_diagnostics(action: "validate")` to ensure no cycles/orphans/contradictions, mark tasks as "done", document completed artifacts, and resolve blockers. Call `manage_sessions(action: "end")` to finalize. + +### 4. Codebase Seeding on Initialization +If the project was just initialized or is missing high-level structure (Plans, Milestones, Decisions): +1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. +2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. +3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. diff --git a/.vscode/instructions.md b/.vscode/instructions.md index 69f3a52..269a58b 100644 --- a/.vscode/instructions.md +++ b/.vscode/instructions.md @@ -146,7 +146,7 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana 5. **Intention Dispatch**: Create execution directives with `manage_intentions(action: "create", ...)` for the runtime engine. 6. **Reactive Replanning**: If an unexpected blocker occurs, invoke `replan(action: "blocker", goal_id: "...", blocker_description: "...")`. -## 10 Core MCP Tools +## 15 Core MCP Tools - `set_goal`: Manage goal hierarchy and task DAGs. - `evaluate_situation`: Score and rank candidate actions from environment snapshots. - `replan`: Adaptively reconstruct subgoals upon obstacles. @@ -157,6 +157,11 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana - `manage_beliefs`: Structured belief state with exponential confidence decay. - `manage_intentions`: Wire contract directives queue for runtime execution. - `manage_reasoning_db`: Snapshots, diagnostics, and SHA-256 Merkle audit verification. +- `classify`: Zero-LLM deterministic classification against hierarchical taxonomy (<2ms SLA). +- `ask_noul`: Fast binary (Yes/No/Abstain) heuristic gate evaluating conditions (<2ms SLA). +- `ask_choice`: Deterministic multi-alternative selection ranking candidate choices (<2ms SLA). +- `ask_score`: Heuristic utility evaluation scoring target entities on a bounded scale (<2ms SLA). +- `gate_intention`: Fast-path safety & feasibility filter checking preconditions before execution (<1ms SLA). ## State Memory (state-memory-mcp) @@ -374,3 +379,75 @@ If the project was just initialized or is missing high-level structure (Plans, M 1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. 2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. 3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. + +## State Memory (state-memory-mcp) + +This project tracks workflow state, tasks, design decisions, and blockers using `state-memory-mcp` with project slug `"behavior-mcp"`. + +### 1. Priority Order +Before doing any coding or investigation: +1. `manage_sessions(action: "start")` — Start a tracking session for full change attribution. +2. `get_analytics(action: "summary")` — Run to understand current project state, active branches, and overall progress. +3. `manage_tasks(action: "next")` — Query prioritized runnable tasks. +4. `manage_tasks(action: "find_blockers")` — Identify any active blockers preventing progress. +5. `manage_nodes(action: "list")` — Find pending tasks, past decisions, or milestones. +6. `query_graph(action: "trace")` — Trace what depends on or blocks a task. + +### 2. When to Write to the Graph +You MUST update the graph as you work: +- **Starting a session**: Always call `manage_sessions(action: "start", agent_id: "my-agent")` to track all mutations under a unique session. +- **Starting a new task**: Create a node with `manage_nodes(action: "create", type: "task", title: "...", session_id: session_id)`. +- **Making a design or implementation decision**: Document it with `manage_nodes(action: "create", type: "decision", title: "...", metadata: { "rationale": "..." }, session_id: session_id)`. +- **Encountering a blocker**: Record the blocker with `manage_nodes(action: "create", type: "blocker", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "blocks", source_id: blocker_id, target_id: task_id, session_id: session_id)`. +- **Adding observation notes**: Atomically log notes using `manage_nodes(action: "add_note", text: "...", attach_to: node_id)`. +- **Batch updates**: Bulk update tasks/nodes using `manage_nodes(action: "batch_update", ids: ["..."], status: "done")`. +- **Completing a task**: Update status to done using `manage_tasks(action: "complete", task_id: task_id)` or `manage_nodes(action: "update", id: task_id, status: "done")`. +- **Creating/generating a new file**: Create an artifact node with `manage_nodes(action: "create", type: "artifact", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "produces", source_id: task_id, target_id: artifact_id)`. + +### 3. Workflow Pattern +1. **Start of session**: Call `manage_sessions(action: "start")` to align and track work, then run `get_analytics(action: "summary")`, `manage_tasks(action: "next")`, and `manage_tasks(action: "find_blockers")`. +2. **Task decomposition**: Decompose user requests into tasks and add them to the graph. +3. **Execution**: Mark tasks as "in_progress", document design decisions as they occur, and log blockers if you hit any obstacles. +4. **Validation & Resolution**: Run `run_diagnostics(action: "validate")` to ensure no cycles/orphans/contradictions, mark tasks as "done", document completed artifacts, and resolve blockers. Call `manage_sessions(action: "end")` to finalize. + +### 4. Codebase Seeding on Initialization +If the project was just initialized or is missing high-level structure (Plans, Milestones, Decisions): +1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. +2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. +3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. + +## State Memory (state-memory-mcp) + +This project tracks workflow state, tasks, design decisions, and blockers using `state-memory-mcp` with project slug `"behavior-mcp"`. + +### 1. Priority Order +Before doing any coding or investigation: +1. `manage_sessions(action: "start")` — Start a tracking session for full change attribution. +2. `get_analytics(action: "summary")` — Run to understand current project state, active branches, and overall progress. +3. `manage_tasks(action: "next")` — Query prioritized runnable tasks. +4. `manage_tasks(action: "find_blockers")` — Identify any active blockers preventing progress. +5. `manage_nodes(action: "list")` — Find pending tasks, past decisions, or milestones. +6. `query_graph(action: "trace")` — Trace what depends on or blocks a task. + +### 2. When to Write to the Graph +You MUST update the graph as you work: +- **Starting a session**: Always call `manage_sessions(action: "start", agent_id: "my-agent")` to track all mutations under a unique session. +- **Starting a new task**: Create a node with `manage_nodes(action: "create", type: "task", title: "...", session_id: session_id)`. +- **Making a design or implementation decision**: Document it with `manage_nodes(action: "create", type: "decision", title: "...", metadata: { "rationale": "..." }, session_id: session_id)`. +- **Encountering a blocker**: Record the blocker with `manage_nodes(action: "create", type: "blocker", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "blocks", source_id: blocker_id, target_id: task_id, session_id: session_id)`. +- **Adding observation notes**: Atomically log notes using `manage_nodes(action: "add_note", text: "...", attach_to: node_id)`. +- **Batch updates**: Bulk update tasks/nodes using `manage_nodes(action: "batch_update", ids: ["..."], status: "done")`. +- **Completing a task**: Update status to done using `manage_tasks(action: "complete", task_id: task_id)` or `manage_nodes(action: "update", id: task_id, status: "done")`. +- **Creating/generating a new file**: Create an artifact node with `manage_nodes(action: "create", type: "artifact", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "produces", source_id: task_id, target_id: artifact_id)`. + +### 3. Workflow Pattern +1. **Start of session**: Call `manage_sessions(action: "start")` to align and track work, then run `get_analytics(action: "summary")`, `manage_tasks(action: "next")`, and `manage_tasks(action: "find_blockers")`. +2. **Task decomposition**: Decompose user requests into tasks and add them to the graph. +3. **Execution**: Mark tasks as "in_progress", document design decisions as they occur, and log blockers if you hit any obstacles. +4. **Validation & Resolution**: Run `run_diagnostics(action: "validate")` to ensure no cycles/orphans/contradictions, mark tasks as "done", document completed artifacts, and resolve blockers. Call `manage_sessions(action: "end")` to finalize. + +### 4. Codebase Seeding on Initialization +If the project was just initialized or is missing high-level structure (Plans, Milestones, Decisions): +1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. +2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. +3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. diff --git a/.windsurfrules b/.windsurfrules index c9dcc20..071a3b5 100644 --- a/.windsurfrules +++ b/.windsurfrules @@ -161,7 +161,7 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana 5. **Intention Dispatch**: Create execution directives with `manage_intentions(action: "create", ...)` for the runtime engine. 6. **Reactive Replanning**: If an unexpected blocker occurs, invoke `replan(action: "blocker", goal_id: "...", blocker_description: "...")`. -## 10 Core MCP Tools +## 15 Core MCP Tools - `set_goal`: Manage goal hierarchy and task DAGs. - `evaluate_situation`: Score and rank candidate actions from environment snapshots. - `replan`: Adaptively reconstruct subgoals upon obstacles. @@ -172,6 +172,11 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana - `manage_beliefs`: Structured belief state with exponential confidence decay. - `manage_intentions`: Wire contract directives queue for runtime execution. - `manage_reasoning_db`: Snapshots, diagnostics, and SHA-256 Merkle audit verification. +- `classify`: Zero-LLM deterministic classification against hierarchical taxonomy (<2ms SLA). +- `ask_noul`: Fast binary (Yes/No/Abstain) heuristic gate evaluating conditions (<2ms SLA). +- `ask_choice`: Deterministic multi-alternative selection ranking candidate choices (<2ms SLA). +- `ask_score`: Heuristic utility evaluation scoring target entities on a bounded scale (<2ms SLA). +- `gate_intention`: Fast-path safety & feasibility filter checking preconditions before execution (<1ms SLA). ## State Memory (state-memory-mcp) @@ -389,3 +394,75 @@ If the project was just initialized or is missing high-level structure (Plans, M 1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. 2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. 3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. + +## State Memory (state-memory-mcp) + +This project tracks workflow state, tasks, design decisions, and blockers using `state-memory-mcp` with project slug `"behavior-mcp"`. + +### 1. Priority Order +Before doing any coding or investigation: +1. `manage_sessions(action: "start")` — Start a tracking session for full change attribution. +2. `get_analytics(action: "summary")` — Run to understand current project state, active branches, and overall progress. +3. `manage_tasks(action: "next")` — Query prioritized runnable tasks. +4. `manage_tasks(action: "find_blockers")` — Identify any active blockers preventing progress. +5. `manage_nodes(action: "list")` — Find pending tasks, past decisions, or milestones. +6. `query_graph(action: "trace")` — Trace what depends on or blocks a task. + +### 2. When to Write to the Graph +You MUST update the graph as you work: +- **Starting a session**: Always call `manage_sessions(action: "start", agent_id: "my-agent")` to track all mutations under a unique session. +- **Starting a new task**: Create a node with `manage_nodes(action: "create", type: "task", title: "...", session_id: session_id)`. +- **Making a design or implementation decision**: Document it with `manage_nodes(action: "create", type: "decision", title: "...", metadata: { "rationale": "..." }, session_id: session_id)`. +- **Encountering a blocker**: Record the blocker with `manage_nodes(action: "create", type: "blocker", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "blocks", source_id: blocker_id, target_id: task_id, session_id: session_id)`. +- **Adding observation notes**: Atomically log notes using `manage_nodes(action: "add_note", text: "...", attach_to: node_id)`. +- **Batch updates**: Bulk update tasks/nodes using `manage_nodes(action: "batch_update", ids: ["..."], status: "done")`. +- **Completing a task**: Update status to done using `manage_tasks(action: "complete", task_id: task_id)` or `manage_nodes(action: "update", id: task_id, status: "done")`. +- **Creating/generating a new file**: Create an artifact node with `manage_nodes(action: "create", type: "artifact", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "produces", source_id: task_id, target_id: artifact_id)`. + +### 3. Workflow Pattern +1. **Start of session**: Call `manage_sessions(action: "start")` to align and track work, then run `get_analytics(action: "summary")`, `manage_tasks(action: "next")`, and `manage_tasks(action: "find_blockers")`. +2. **Task decomposition**: Decompose user requests into tasks and add them to the graph. +3. **Execution**: Mark tasks as "in_progress", document design decisions as they occur, and log blockers if you hit any obstacles. +4. **Validation & Resolution**: Run `run_diagnostics(action: "validate")` to ensure no cycles/orphans/contradictions, mark tasks as "done", document completed artifacts, and resolve blockers. Call `manage_sessions(action: "end")` to finalize. + +### 4. Codebase Seeding on Initialization +If the project was just initialized or is missing high-level structure (Plans, Milestones, Decisions): +1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. +2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. +3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. + +## State Memory (state-memory-mcp) + +This project tracks workflow state, tasks, design decisions, and blockers using `state-memory-mcp` with project slug `"behavior-mcp"`. + +### 1. Priority Order +Before doing any coding or investigation: +1. `manage_sessions(action: "start")` — Start a tracking session for full change attribution. +2. `get_analytics(action: "summary")` — Run to understand current project state, active branches, and overall progress. +3. `manage_tasks(action: "next")` — Query prioritized runnable tasks. +4. `manage_tasks(action: "find_blockers")` — Identify any active blockers preventing progress. +5. `manage_nodes(action: "list")` — Find pending tasks, past decisions, or milestones. +6. `query_graph(action: "trace")` — Trace what depends on or blocks a task. + +### 2. When to Write to the Graph +You MUST update the graph as you work: +- **Starting a session**: Always call `manage_sessions(action: "start", agent_id: "my-agent")` to track all mutations under a unique session. +- **Starting a new task**: Create a node with `manage_nodes(action: "create", type: "task", title: "...", session_id: session_id)`. +- **Making a design or implementation decision**: Document it with `manage_nodes(action: "create", type: "decision", title: "...", metadata: { "rationale": "..." }, session_id: session_id)`. +- **Encountering a blocker**: Record the blocker with `manage_nodes(action: "create", type: "blocker", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "blocks", source_id: blocker_id, target_id: task_id, session_id: session_id)`. +- **Adding observation notes**: Atomically log notes using `manage_nodes(action: "add_note", text: "...", attach_to: node_id)`. +- **Batch updates**: Bulk update tasks/nodes using `manage_nodes(action: "batch_update", ids: ["..."], status: "done")`. +- **Completing a task**: Update status to done using `manage_tasks(action: "complete", task_id: task_id)` or `manage_nodes(action: "update", id: task_id, status: "done")`. +- **Creating/generating a new file**: Create an artifact node with `manage_nodes(action: "create", type: "artifact", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "produces", source_id: task_id, target_id: artifact_id)`. + +### 3. Workflow Pattern +1. **Start of session**: Call `manage_sessions(action: "start")` to align and track work, then run `get_analytics(action: "summary")`, `manage_tasks(action: "next")`, and `manage_tasks(action: "find_blockers")`. +2. **Task decomposition**: Decompose user requests into tasks and add them to the graph. +3. **Execution**: Mark tasks as "in_progress", document design decisions as they occur, and log blockers if you hit any obstacles. +4. **Validation & Resolution**: Run `run_diagnostics(action: "validate")` to ensure no cycles/orphans/contradictions, mark tasks as "done", document completed artifacts, and resolve blockers. Call `manage_sessions(action: "end")` to finalize. + +### 4. Codebase Seeding on Initialization +If the project was just initialized or is missing high-level structure (Plans, Milestones, Decisions): +1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. +2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. +3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. diff --git a/CLAUDE.md b/CLAUDE.md index c9dcc20..071a3b5 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -161,7 +161,7 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana 5. **Intention Dispatch**: Create execution directives with `manage_intentions(action: "create", ...)` for the runtime engine. 6. **Reactive Replanning**: If an unexpected blocker occurs, invoke `replan(action: "blocker", goal_id: "...", blocker_description: "...")`. -## 10 Core MCP Tools +## 15 Core MCP Tools - `set_goal`: Manage goal hierarchy and task DAGs. - `evaluate_situation`: Score and rank candidate actions from environment snapshots. - `replan`: Adaptively reconstruct subgoals upon obstacles. @@ -172,6 +172,11 @@ This project uses `agent-reasoning-mcp` with project slug "behavior-mcp" to mana - `manage_beliefs`: Structured belief state with exponential confidence decay. - `manage_intentions`: Wire contract directives queue for runtime execution. - `manage_reasoning_db`: Snapshots, diagnostics, and SHA-256 Merkle audit verification. +- `classify`: Zero-LLM deterministic classification against hierarchical taxonomy (<2ms SLA). +- `ask_noul`: Fast binary (Yes/No/Abstain) heuristic gate evaluating conditions (<2ms SLA). +- `ask_choice`: Deterministic multi-alternative selection ranking candidate choices (<2ms SLA). +- `ask_score`: Heuristic utility evaluation scoring target entities on a bounded scale (<2ms SLA). +- `gate_intention`: Fast-path safety & feasibility filter checking preconditions before execution (<1ms SLA). ## State Memory (state-memory-mcp) @@ -389,3 +394,75 @@ If the project was just initialized or is missing high-level structure (Plans, M 1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. 2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. 3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. + +## State Memory (state-memory-mcp) + +This project tracks workflow state, tasks, design decisions, and blockers using `state-memory-mcp` with project slug `"behavior-mcp"`. + +### 1. Priority Order +Before doing any coding or investigation: +1. `manage_sessions(action: "start")` — Start a tracking session for full change attribution. +2. `get_analytics(action: "summary")` — Run to understand current project state, active branches, and overall progress. +3. `manage_tasks(action: "next")` — Query prioritized runnable tasks. +4. `manage_tasks(action: "find_blockers")` — Identify any active blockers preventing progress. +5. `manage_nodes(action: "list")` — Find pending tasks, past decisions, or milestones. +6. `query_graph(action: "trace")` — Trace what depends on or blocks a task. + +### 2. When to Write to the Graph +You MUST update the graph as you work: +- **Starting a session**: Always call `manage_sessions(action: "start", agent_id: "my-agent")` to track all mutations under a unique session. +- **Starting a new task**: Create a node with `manage_nodes(action: "create", type: "task", title: "...", session_id: session_id)`. +- **Making a design or implementation decision**: Document it with `manage_nodes(action: "create", type: "decision", title: "...", metadata: { "rationale": "..." }, session_id: session_id)`. +- **Encountering a blocker**: Record the blocker with `manage_nodes(action: "create", type: "blocker", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "blocks", source_id: blocker_id, target_id: task_id, session_id: session_id)`. +- **Adding observation notes**: Atomically log notes using `manage_nodes(action: "add_note", text: "...", attach_to: node_id)`. +- **Batch updates**: Bulk update tasks/nodes using `manage_nodes(action: "batch_update", ids: ["..."], status: "done")`. +- **Completing a task**: Update status to done using `manage_tasks(action: "complete", task_id: task_id)` or `manage_nodes(action: "update", id: task_id, status: "done")`. +- **Creating/generating a new file**: Create an artifact node with `manage_nodes(action: "create", type: "artifact", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "produces", source_id: task_id, target_id: artifact_id)`. + +### 3. Workflow Pattern +1. **Start of session**: Call `manage_sessions(action: "start")` to align and track work, then run `get_analytics(action: "summary")`, `manage_tasks(action: "next")`, and `manage_tasks(action: "find_blockers")`. +2. **Task decomposition**: Decompose user requests into tasks and add them to the graph. +3. **Execution**: Mark tasks as "in_progress", document design decisions as they occur, and log blockers if you hit any obstacles. +4. **Validation & Resolution**: Run `run_diagnostics(action: "validate")` to ensure no cycles/orphans/contradictions, mark tasks as "done", document completed artifacts, and resolve blockers. Call `manage_sessions(action: "end")` to finalize. + +### 4. Codebase Seeding on Initialization +If the project was just initialized or is missing high-level structure (Plans, Milestones, Decisions): +1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. +2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. +3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. + +## State Memory (state-memory-mcp) + +This project tracks workflow state, tasks, design decisions, and blockers using `state-memory-mcp` with project slug `"behavior-mcp"`. + +### 1. Priority Order +Before doing any coding or investigation: +1. `manage_sessions(action: "start")` — Start a tracking session for full change attribution. +2. `get_analytics(action: "summary")` — Run to understand current project state, active branches, and overall progress. +3. `manage_tasks(action: "next")` — Query prioritized runnable tasks. +4. `manage_tasks(action: "find_blockers")` — Identify any active blockers preventing progress. +5. `manage_nodes(action: "list")` — Find pending tasks, past decisions, or milestones. +6. `query_graph(action: "trace")` — Trace what depends on or blocks a task. + +### 2. When to Write to the Graph +You MUST update the graph as you work: +- **Starting a session**: Always call `manage_sessions(action: "start", agent_id: "my-agent")` to track all mutations under a unique session. +- **Starting a new task**: Create a node with `manage_nodes(action: "create", type: "task", title: "...", session_id: session_id)`. +- **Making a design or implementation decision**: Document it with `manage_nodes(action: "create", type: "decision", title: "...", metadata: { "rationale": "..." }, session_id: session_id)`. +- **Encountering a blocker**: Record the blocker with `manage_nodes(action: "create", type: "blocker", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "blocks", source_id: blocker_id, target_id: task_id, session_id: session_id)`. +- **Adding observation notes**: Atomically log notes using `manage_nodes(action: "add_note", text: "...", attach_to: node_id)`. +- **Batch updates**: Bulk update tasks/nodes using `manage_nodes(action: "batch_update", ids: ["..."], status: "done")`. +- **Completing a task**: Update status to done using `manage_tasks(action: "complete", task_id: task_id)` or `manage_nodes(action: "update", id: task_id, status: "done")`. +- **Creating/generating a new file**: Create an artifact node with `manage_nodes(action: "create", type: "artifact", title: "...", session_id: session_id)` and connect it using `manage_edges(action: "add", type: "produces", source_id: task_id, target_id: artifact_id)`. + +### 3. Workflow Pattern +1. **Start of session**: Call `manage_sessions(action: "start")` to align and track work, then run `get_analytics(action: "summary")`, `manage_tasks(action: "next")`, and `manage_tasks(action: "find_blockers")`. +2. **Task decomposition**: Decompose user requests into tasks and add them to the graph. +3. **Execution**: Mark tasks as "in_progress", document design decisions as they occur, and log blockers if you hit any obstacles. +4. **Validation & Resolution**: Run `run_diagnostics(action: "validate")` to ensure no cycles/orphans/contradictions, mark tasks as "done", document completed artifacts, and resolve blockers. Call `manage_sessions(action: "end")` to finalize. + +### 4. Codebase Seeding on Initialization +If the project was just initialized or is missing high-level structure (Plans, Milestones, Decisions): +1. **Inspect the Codebase**: Read the README and core files to understand the roadmap and architecture. +2. **Scaffold the Roadmap**: Create a `plan` node (e.g., "Project Roadmap") and add `milestone` nodes representing key target phases, connecting them using `part_of` edges. +3. **Scaffold Architecture**: Create `decision` nodes representing core technical choices (e.g., choice of databases, frameworks) and link them to the milestones/tasks using `decided_in` edges. diff --git a/MIGRATION.md b/MIGRATION.md index 19108c1..99f2d81 100644 --- a/MIGRATION.md +++ b/MIGRATION.md @@ -1,6 +1,6 @@ # 🚀 Migration Guide: @putervision/behavior-mcp -This guide explains how to migrate client integrations, custom agents, and tool callers to the unified **v0.2.1+ API** with native transport, unified blackboard dialect, dynamic behavior synthesis, and unstick recovery. +This guide explains how to migrate client integrations, custom agents, and tool callers to the unified **v0.3.0+ API** with native transport, unified blackboard dialect, dynamic behavior synthesis, and unstick recovery. --- diff --git a/README.md b/README.md index b690b2d..9fb8916 100644 --- a/README.md +++ b/README.md @@ -1,7 +1,7 @@ # @putervision/behavior-mcp [![npm version](https://img.shields.io/npm/v/@putervision/behavior-mcp.svg)](https://www.npmjs.com/package/@putervision/behavior-mcp) -[![version](https://img.shields.io/badge/version-0.2.1-blue.svg)](./CHANGELOG.md) +[![version](https://img.shields.io/badge/version-0.3.0-blue.svg)](./CHANGELOG.md) [![CI](https://github.com/putervision/behavior-mcp/actions/workflows/ci.yml/badge.svg)](https://github.com/putervision/behavior-mcp/actions/workflows/ci.yml) [![Node](https://img.shields.io/badge/node-%3E%3D20.0.0-brightgreen.svg)](https://nodejs.org) [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](https://opensource.org/licenses/MIT) @@ -46,13 +46,13 @@ npx @putervision/behavior-mcp inspect --- -## 🛡️ 5-Layer Safety Guardrail Stack +## 🛡️ 5-Layer Safety Guardrail Stack & System 1 Invariants 1. **Fail-Closed Evaluator**: Unrecognized node definitions throw fatal exceptions immediately. -2. **Watchdog Heartbeat Timer**: Halts execution if tick evaluation stalls beyond 5,000ms. -3. **Action Rate Limiter**: Strict 60 actions/sec maximum throughput ceiling. -4. **Leaf Policy Gate**: Blocks irreversible, high-risk mutations (`delete_item`, `spend_currency`). -5. **Emergency Kill Switch**: Atomic safety latch (`EmergencySafety.engageKillSwitch()`) halts all runtimes. +2. **60Hz Tick Invariant & Synchronous `semantic_check`**: The loop never blocks on external network calls; semantic condition checks resolve synchronously against blackboard caches. +3. **HMAC Intention Gate Verification**: Intentions dispatched to execution nodes require unexpired, cryptographically signed dispatch tokens (`PENTAD_HMAC_SECRET`). +4. **Action Rate Limiter & Leaf Policy Gate**: Strict 60 actions/sec maximum throughput ceiling and policy guardrails blocking irreversible mutations. +5. **Emergency Kill Switch & Watchdog Heartbeat**: Atomic safety latch and watchdog timers halting execution if tick stalls beyond 5,000ms. --- @@ -66,7 +66,7 @@ npx @putervision/behavior-mcp inspect --- -## 🔗 Client Configuration +## 🔗 Client Configuration & Environment Add to `.cursor/mcp.json` or `.vscode/mcp.json`: ```json @@ -74,7 +74,10 @@ Add to `.cursor/mcp.json` or `.vscode/mcp.json`: "mcpServers": { "behavior-mcp": { "command": "behavior-mcp", - "args": ["run"] + "args": ["run"], + "env": { + "PENTAD_HMAC_SECRET": "your-secure-shared-secret-here" + } } } } @@ -85,7 +88,7 @@ Add to `.cursor/mcp.json` or `.vscode/mcp.json`: ## 🧪 Testing ```bash -# Run full unit and integration test suite across 15 test files (87 tests) +# Run full unit and integration test suite across 38 test files (266 tests) npm test ``` diff --git a/docs/.well-known/mcp.json b/docs/.well-known/mcp.json index 0e2cfd5..a549201 100644 --- a/docs/.well-known/mcp.json +++ b/docs/.well-known/mcp.json @@ -1,7 +1,7 @@ { "$schema": "https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json", "name": "io.github.putervision/behavior-mcp", - "version": "0.2.1", + "version": "0.3.0", "title": "Behavior MCP", "description": "In-browser ~60Hz behavior tree execution engine with reactive triggers and safety guardrails.", "publisher": "PuterVision", @@ -16,7 +16,7 @@ { "registryType": "npm", "identifier": "@putervision/behavior-mcp", - "version": "0.2.1", + "version": "0.3.0", "transport": { "type": "stdio" }, diff --git a/docs/api-reference.md b/docs/api-reference.md index 43686ea..334b294 100644 --- a/docs/api-reference.md +++ b/docs/api-reference.md @@ -1,4 +1,4 @@ -# API Reference: `@putervision/behavior-mcp` +# API Reference: `@putervision/behavior-mcp` (v0.3.0 — 10 Tools) Comprehensive documentation for all 10 MCP tools provided by `@putervision/behavior-mcp`. diff --git a/docs/index.html b/docs/index.html index b60c93f..a6facc8 100644 --- a/docs/index.html +++ b/docs/index.html @@ -93,7 +93,7 @@
- Model Context Protocol Server • v0.2.1
+ Model Context Protocol Server • v0.3.0

In-Browser ~60Hz Behavior Tree Execution

diff --git a/manifest.json b/manifest.json index f2eb4fd..02b4b7d 100644 --- a/manifest.json +++ b/manifest.json @@ -1,7 +1,7 @@ { "manifest_version": "0.2", "name": "@putervision/behavior-mcp", - "version": "0.2.1", + "version": "0.3.0", "description": "In-browser ~60Hz behavior tree execution engine with reactive triggers and safety guardrails.", "author": { "name": "PuterVision" diff --git a/package-lock.json b/package-lock.json index f149f40..2e0af36 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "@putervision/behavior-mcp", - "version": "0.2.1", + "version": "0.3.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@putervision/behavior-mcp", - "version": "0.2.1", + "version": "0.3.0", "license": "MIT", "dependencies": { "better-sqlite3": "^11.0.0" diff --git a/package.json b/package.json index 6bc7865..016c311 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@putervision/behavior-mcp", - "version": "0.2.1", + "version": "0.3.0", "description": "High-frequency in-browser behavior tree execution engine for autonomous AI agents — 60Hz tick loops, reactive triggers, telemetry recordings, and safety guardrails.", "keywords": [ "mcp", diff --git a/scripts/test-matrix.sh b/scripts/test-matrix.sh index b184299..a9d3ebd 100755 --- a/scripts/test-matrix.sh +++ b/scripts/test-matrix.sh @@ -25,6 +25,7 @@ for VER in "${VERSIONS[@]}"; do echo "------------------------------------------" if command -v nvm >/dev/null 2>&1; then if nvm use "$VER" >/dev/null 2>&1; then + export PATH="$(dirname "$(nvm which "$VER")"):$PATH" echo "Using Node $(node -v) via NVM" npm rebuild better-sqlite3 >/dev/null 2>&1 || true if npm test; then @@ -46,6 +47,7 @@ done # Restore original Node version if [ -n "$ORIGINAL_VER" ] && command -v nvm >/dev/null 2>&1; then nvm use "$ORIGINAL_VER" >/dev/null 2>&1 || true + export PATH="$(dirname "$(nvm which "$ORIGINAL_VER" 2>/dev/null || nvm which current)"):$PATH" npm rebuild better-sqlite3 >/dev/null 2>&1 || true fi diff --git a/server.json b/server.json index 24f4ac2..85cf2dc 100644 --- a/server.json +++ b/server.json @@ -1,7 +1,7 @@ { "$schema": "https://static.modelcontextprotocol.io/schemas/2025-12-11/server.schema.json", "name": "io.github.putervision/behavior-mcp", - "version": "0.2.1", + "version": "0.3.0", "title": "Behavior Runtime MCP", "description": "In-browser ~60Hz behavior tree execution engine with reactive triggers and safety guardrails.", "publisher": "PuterVision", @@ -29,7 +29,7 @@ { "registryType": "npm", "identifier": "@putervision/behavior-mcp", - "version": "0.2.1", + "version": "0.3.0", "transport": { "type": "stdio" }, diff --git a/src/engine/browser/actions.ts b/src/engine/browser/actions.ts index 86a5a58..fb3aa44 100644 --- a/src/engine/browser/actions.ts +++ b/src/engine/browser/actions.ts @@ -41,4 +41,17 @@ export const ActionRegistry: Record< ctx.blackboard.patrolling = true; return { status: 'SUCCESS', output: { patrol: true } }; }, + request_semantic_evaluation: (params, ctx) => { + const key = (params.key as string) || (params.statement as string) || (params.query as string) || 'default'; + const query = params.query || params.statement || params.question || key; + const reqKey = `semantic_request_${key}`; + ctx.blackboard[reqKey] = { + key, + query, + target: params.target, + context_data: params.context_data, + timestamp: Date.now(), + }; + return { status: 'RUNNING', output: { requested: true, key, query } }; + }, }; diff --git a/src/engine/browser/conditions.ts b/src/engine/browser/conditions.ts index 10c7126..038f456 100644 --- a/src/engine/browser/conditions.ts +++ b/src/engine/browser/conditions.ts @@ -18,4 +18,27 @@ export const ConditionRegistry: Record< const radius = (params.radius as number) ?? 10; return dist <= radius; }, + semantic_check: (params, ctx) => { + const key = (params.key as string) || (params.statement as string) || (params.query as string) || 'default'; + const decisionKey = `semantic_decision_${key}`; + const entry = ctx.blackboard[decisionKey]; + if (entry === undefined || entry === null) { + return false; + } + const now = Date.now(); + if (typeof entry === 'object' && entry !== null && 'result' in entry) { + const obj = entry as { result: boolean; timestamp?: number }; + if (obj.timestamp !== undefined && now - obj.timestamp > 5000) { + return false; + } + return obj.result === true; + } + const tsKey = `semantic_timestamp_${key}`; + const ts = ctx.blackboard[tsKey] as number | undefined; + if (ts !== undefined && now - ts > 5000) { + return false; + } + const expected = params.expected ?? true; + return entry === expected; + }, }; diff --git a/src/engine/browser/executor-bundle.ts b/src/engine/browser/executor-bundle.ts index 74b4702..8ca2a51 100644 --- a/src/engine/browser/executor-bundle.ts +++ b/src/engine/browser/executor-bundle.ts @@ -1,10 +1,46 @@ -import { BehaviorTreeNode, NodeStatus } from '../../schema/types.js'; +import crypto from 'crypto'; +import { BehaviorTreeNode, NodeStatus, DispatchToken } from '../../schema/types.js'; import { ConditionRegistry } from './conditions.js'; import { GameConditionRegistry } from './conditions-game.js'; import { ActionRegistry } from './actions.js'; import { GameActionRegistry } from './actions-game.js'; const MAX_TREE_DEPTH = 64; +const CLOCK_SKEW_MS = 2000; + +export function verifyIntentionDispatch( + intention: { id: string; behavior_name?: string; parameters?: Record }, + token: DispatchToken, + secret?: string +): boolean { + const effectiveSecret = secret || process.env.PENTAD_HMAC_SECRET; + if (!effectiveSecret || effectiveSecret.trim() === '') return false; + if (!token || typeof token !== 'object') return false; + if (token.aud !== 'behavior-mcp') return false; + + const now = Date.now(); + const expiresAt = Date.parse(token.expires_at); + const issuedAt = Date.parse(token.issued_at); + if (Number.isNaN(expiresAt) || Number.isNaN(issuedAt)) return false; + + if (now > expiresAt + CLOCK_SKEW_MS) return false; + if (now < issuedAt - CLOCK_SKEW_MS) return false; + if (token.intention_id !== intention.id) return false; + + const expectedSig = crypto + .createHmac('sha256', effectiveSecret) + .update( + `${token.token_id}:${token.intention_id}:${token.behavior_name}:${token.params_hash}:${token.aud}:${token.issued_at}:${token.expires_at}` + ) + .digest('hex'); + + const sigBuf = Buffer.from(token.hmac_signature || '', 'hex'); + const expectedBuf = Buffer.from(expectedSig, 'hex'); + + // Verify buffer lengths before calling crypto.timingSafeEqual() to avoid RangeError on tampered tokens + if (sigBuf.length !== expectedBuf.length || sigBuf.length === 0) return false; + return crypto.timingSafeEqual(sigBuf, expectedBuf); +} export class BehaviorTreeEvaluator { private tree: BehaviorTreeNode; diff --git a/src/engine/config.ts b/src/engine/config.ts index e6ace72..b911fc6 100644 --- a/src/engine/config.ts +++ b/src/engine/config.ts @@ -72,3 +72,12 @@ export function loadProjectConfig(projectRoot: string): ProjectConfig { cachedConfigs.set(projectRoot, { config, timestamp: now }); return config; } + +export function getPentadHmacSecret(projectRoot = process.cwd()): string | undefined { + if (process.env.PENTAD_HMAC_SECRET) { + return process.env.PENTAD_HMAC_SECRET; + } + const config = loadProjectConfig(projectRoot); + return config.pentadHmacSecret || config.hmacSecret; +} + diff --git a/src/engine/triggers.ts b/src/engine/triggers.ts index a2bd1b8..fdae6a4 100644 --- a/src/engine/triggers.ts +++ b/src/engine/triggers.ts @@ -6,6 +6,28 @@ import { safeJsonParse, safeJsonStringify } from '../utils/json-validator.js'; import { ValidationError, NotFoundError } from '../utils/errors.js'; import { logRuntimeEvent } from './events.js'; +export const TriggerConditionRegistry: Record< + string, + (params: Record, telemetry: Record) => boolean +> = { + hp_threshold: (conditionParams, telemetry) => { + const hp = telemetry.hp ?? 100; + const threshold = (conditionParams.threshold as number) ?? 30; + return hp <= threshold; + }, + enemy_proximity: (conditionParams, telemetry) => { + const enemyDist = telemetry.enemy_distance ?? 999; + const radius = (conditionParams.radius as number) ?? 10; + return enemyDist <= radius; + }, + semantic: (conditionParams, telemetry) => { + const key = (conditionParams.key as string) || 'default'; + const expected = conditionParams.expected ?? true; + const val = telemetry[key] ?? telemetry[`semantic_decision_${key}`]; + return val === expected; + }, +}; + export class TriggerRegistry { static registerTrigger( db: Database.Database, @@ -97,16 +119,8 @@ export class TriggerRegistry { continue; // In cooldown } - let matches = false; - if (trig.condition_type === 'hp_threshold') { - const hp = params.telemetry.hp ?? 100; - const threshold = (trig.condition_params.threshold as number) ?? 30; - matches = hp <= threshold; - } else if (trig.condition_type === 'enemy_proximity') { - const enemyDist = params.telemetry.enemy_distance ?? 999; - const radius = (trig.condition_params.radius as number) ?? 10; - matches = enemyDist <= radius; - } + const evaluator = TriggerConditionRegistry[trig.condition_type]; + const matches = evaluator ? evaluator(trig.condition_params || {}, params.telemetry) : false; if (matches) { db.prepare('UPDATE triggers SET last_fired_at = ?, updated_at = ? WHERE id = ?').run( diff --git a/src/schema/types.ts b/src/schema/types.ts index 9a87334..fd5dc4b 100644 --- a/src/schema/types.ts +++ b/src/schema/types.ts @@ -131,3 +131,15 @@ export interface Outcome { stuck_reason?: string; completed_at: string; } + +export interface DispatchToken { + token_id: string; // Unique token UUID + intention_id: string; // Bound intention ID + behavior_name: string; // Bound behavior tree name + params_hash: string; // SHA-256 of canonical intention parameters + aud: 'behavior-mcp'; // Audience — only behavior-mcp may consume this token + issued_at: string; // ISO-8601 + expires_at: string; // ISO-8601 + hmac_signature: string; // HMAC-SHA256 signature +} + diff --git a/src/utils/canonical-json.ts b/src/utils/canonical-json.ts new file mode 100644 index 0000000..23537cf --- /dev/null +++ b/src/utils/canonical-json.ts @@ -0,0 +1,55 @@ +/** + * Canonical JSON Serialization Utility + * + * Complies strictly with PuterVision Pentad System One specification §10.1: + * 1. Object keys are sorted lexicographically (recursive, including nested objects). + * 2. No trailing commas. No comments. + * 3. `undefined` fields are omitted entirely (not serialized as `null`). + * 4. Numbers: no `NaN`, no `Infinity`, no `-0`. Floats are rounded to 6 decimal places maximum. + * 5. Arrays preserve insertion order; objects inside arrays follow the same key-sort rule. + * 6. Serialization Preimage: The preimage to SHA-256 is the exact UTF-8 byte stream produced + * with zero extraneous whitespace and zero trailing newlines. + */ + +export function canonicalJsonStringify(val: unknown): string { + if (val === undefined) { + return ''; + } + if (val === null) { + return 'null'; + } + if (typeof val === 'number') { + if (!Number.isFinite(val)) { + throw new Error(`Invalid non-finite number in canonical JSON: ${val}`); + } + if (Object.is(val, -0)) { + return '0'; + } + const rounded = Number(Math.round(Number(val + 'e+6')) + 'e-6'); + return String(rounded); + } + if (typeof val === 'boolean') { + return val ? 'true' : 'false'; + } + if (typeof val === 'string') { + return JSON.stringify(val); + } + if (Array.isArray(val)) { + const serializedItems = val.map((item) => + item === undefined ? 'null' : canonicalJsonStringify(item) + ); + return '[' + serializedItems.join(',') + ']'; + } + if (typeof val === 'object') { + const keys = Object.keys(val as Record).sort(); + const parts: string[] = []; + for (const key of keys) { + const v = (val as Record)[key]; + if (v !== undefined) { + parts.push(JSON.stringify(key) + ':' + canonicalJsonStringify(v)); + } + } + return '{' + parts.join(',') + '}'; + } + return JSON.stringify(val); +} diff --git a/src/utils/version.ts b/src/utils/version.ts index a483c61..ead1853 100644 --- a/src/utils/version.ts +++ b/src/utils/version.ts @@ -4,5 +4,5 @@ export function getVersion(): string { if (typeof __APP_VERSION__ !== 'undefined') { return __APP_VERSION__; } - return '0.2.1'; + return '0.3.0'; } diff --git a/tests/contracts/state-pack-hash.test.ts b/tests/contracts/state-pack-hash.test.ts new file mode 100644 index 0000000..6ed6fcb --- /dev/null +++ b/tests/contracts/state-pack-hash.test.ts @@ -0,0 +1,29 @@ +import { describe, it, expect } from 'vitest'; +import fs from 'fs'; +import path from 'path'; +import crypto from 'crypto'; +import { canonicalJsonStringify } from '../../src/utils/canonical-json.js'; + +describe('Behavior-MCP Canonical StatePack Contract', () => { + const fixturePath = path.resolve(__dirname, '../fixtures/canonical-state-pack.json'); + const fixtureRaw = fs.readFileSync(fixturePath, 'utf8'); + const fixture = JSON.parse(fixtureRaw); + + const EXPECTED_PACK_HASH = '283cbb0c60496b6beca4237341fb3bb1ae76490628a81b54b46953058eb9bd21'; + const EXPECTED_FULL_HASH = '376e80f562b8edab8d51fe40faf76140c65253a215bca87ee7dd89224636e079'; + + it('computes exact canonical pack_hash matching EXPECTED_PACK_HASH', () => { + const { pack_hash, ...rest } = fixture; + const cjson = canonicalJsonStringify(rest); + const hash = crypto.createHash('sha256').update(cjson, 'utf8').digest('hex'); + + expect(pack_hash).toBe(EXPECTED_PACK_HASH); + expect(hash).toBe(EXPECTED_PACK_HASH); + }); + + it('computes exact byte-for-byte full canonical serialized hash', () => { + const fullCjson = canonicalJsonStringify(fixture); + const fullHash = crypto.createHash('sha256').update(fullCjson, 'utf8').digest('hex'); + expect(fullHash).toBe(EXPECTED_FULL_HASH); + }); +}); diff --git a/tests/engine/conditions.test.ts b/tests/engine/conditions.test.ts new file mode 100644 index 0000000..edc278a --- /dev/null +++ b/tests/engine/conditions.test.ts @@ -0,0 +1,77 @@ +import { describe, it, expect } from 'vitest'; +import { ConditionRegistry } from '../../src/engine/browser/conditions.js'; +import { ActionRegistry } from '../../src/engine/browser/actions.js'; +import { RuntimeContext } from '../../src/engine/browser/node-types.js'; + +describe('Behavior-MCP Conditions & 60Hz Tick Invariant', () => { + const createContext = (blackboard: Record = {}): RuntimeContext => ({ + tick: 1, + telemetry: {}, + blackboard, + }); + + it('registers semantic_check in ConditionRegistry as a synchronous lookup', () => { + expect('semantic_check' in ConditionRegistry).toBe(true); + const handler = ConditionRegistry['semantic_check']!; + + // Pure synchronous execution test: must not return a Promise + const ctx = createContext(); + const result = handler({ key: 'door_open' }, ctx); + expect(typeof result).toBe('boolean'); + }); + + it('evaluates semantic_check as true when fresh decision is present (<5000ms)', () => { + const handler = ConditionRegistry['semantic_check']!; + const ctx = createContext({ + semantic_decision_threat_clear: { + result: true, + timestamp: Date.now() - 1000, // 1 second ago + }, + }); + + const result = handler({ key: 'threat_clear' }, ctx); + expect(result).toBe(true); + }); + + it('evaluates semantic_check as false when decision is stale (>5000ms)', () => { + const handler = ConditionRegistry['semantic_check']!; + const ctx = createContext({ + semantic_decision_threat_clear: { + result: true, + timestamp: Date.now() - 6000, // 6 seconds ago (stale) + }, + }); + + const result = handler({ key: 'threat_clear' }, ctx); + expect(result).toBe(false); + }); + + it('evaluates semantic_check as false when decision is missing', () => { + const handler = ConditionRegistry['semantic_check']!; + const ctx = createContext({}); + + const result = handler({ key: 'unseen_key' }, ctx); + expect(result).toBe(false); + }); + + it('runs request_semantic_evaluation companion action without blocking tick', () => { + expect('request_semantic_evaluation' in ActionRegistry).toBe(true); + const handler = ActionRegistry['request_semantic_evaluation']!; + + const ctx = createContext(); + const res = handler( + { key: 'target_hostile', query: 'Is entity hostile?', context_data: { entity_id: 'e1' } }, + ctx + ); + + // Companion action must yield RUNNING to preserve 60Hz tick + expect(res.status).toBe('RUNNING'); + + // Must set blackboard request flag for host fulfillment + const request = ctx.blackboard['semantic_request_target_hostile'] as any; + expect(request).toBeDefined(); + expect(request.query).toBe('Is entity hostile?'); + expect(request.context_data).toEqual({ entity_id: 'e1' }); + expect(request.timestamp).toBeDefined(); + }); +}); diff --git a/tests/engine/token-verification.test.ts b/tests/engine/token-verification.test.ts new file mode 100644 index 0000000..14305b3 --- /dev/null +++ b/tests/engine/token-verification.test.ts @@ -0,0 +1,119 @@ +import { describe, it, expect } from 'vitest'; +import crypto from 'crypto'; +import { verifyIntentionDispatch, DispatchToken, Intention } from '../../src/engine/browser/executor-bundle.js'; + +describe('Behavior-MCP DispatchToken HMAC Verification', () => { + const secret = 'pentad_hmac_secret_key_for_testing_0123456789!'; + + const sampleIntention: Intention = { + id: 'intent_123', + project: 'test_project', + behavior_name: 'test_behavior', + parameters: { speed: 1.0 }, + priority: 1, + status: 'pending', + }; + + const createValidToken = (overrides: Partial = {}): DispatchToken => { + const now = Date.now(); + const issuedAt = new Date(now).toISOString(); + const expiresAt = new Date(now + 30000).toISOString(); + const paramsHash = crypto.createHash('sha256').update(JSON.stringify({ speed: 1.0 })).digest('hex'); + + const tokenId = 'tok_001'; + const intentionId = overrides.intention_id || sampleIntention.id; + const behaviorName = overrides.behavior_name || sampleIntention.behavior_name; + const aud = overrides.aud || 'behavior-mcp'; + + const preimage = `${tokenId}:${intentionId}:${behaviorName}:${paramsHash}:${aud}:${issuedAt}:${expiresAt}`; + const hmacSignature = crypto.createHmac('sha256', secret).update(preimage).digest('hex'); + + return { + token_id: tokenId, + intention_id: intentionId, + behavior_name: behaviorName, + params_hash: paramsHash, + aud: aud as 'behavior-mcp', + issued_at: issuedAt, + expires_at: expiresAt, + hmac_signature: hmacSignature, + ...overrides, + }; + }; + + it('accepts valid token signed with matching secret', () => { + const token = createValidToken(); + const isValid = verifyIntentionDispatch(sampleIntention, token, secret); + expect(isValid).toBe(true); + }); + + it('rejects token with wrong audience', () => { + const token = createValidToken({ aud: 'wrong-audience' as any }); + const isValid = verifyIntentionDispatch(sampleIntention, token, secret); + expect(isValid).toBe(false); + }); + + it('rejects token for different intention_id', () => { + const token = createValidToken({ intention_id: 'other_intent_456' }); + const isValid = verifyIntentionDispatch(sampleIntention, token, secret); + expect(isValid).toBe(false); + }); + + it('tolerates minor clock skew within ±2000ms window', () => { + const now = Date.now(); + // Issued 1500ms in the future (within 2000ms clock skew tolerance) + const futureIssued = new Date(now + 1500).toISOString(); + const expiresAt = new Date(now + 30000).toISOString(); + const paramsHash = crypto.createHash('sha256').update(JSON.stringify({ speed: 1.0 })).digest('hex'); + + const preimage = `tok_skew:${sampleIntention.id}:${sampleIntention.behavior_name}:${paramsHash}:behavior-mcp:${futureIssued}:${expiresAt}`; + const hmacSignature = crypto.createHmac('sha256', secret).update(preimage).digest('hex'); + + const token: DispatchToken = { + token_id: 'tok_skew', + intention_id: sampleIntention.id, + behavior_name: sampleIntention.behavior_name, + params_hash: paramsHash, + aud: 'behavior-mcp', + issued_at: futureIssued, + expires_at: expiresAt, + hmac_signature: hmacSignature, + }; + + expect(verifyIntentionDispatch(sampleIntention, token, secret)).toBe(true); + }); + + it('rejects expired token outside clock skew window (>2000ms past expires_at)', () => { + const now = Date.now(); + const issuedAt = new Date(now - 60000).toISOString(); + const expiresAt = new Date(now - 5000).toISOString(); // Expired 5s ago + const paramsHash = crypto.createHash('sha256').update(JSON.stringify({ speed: 1.0 })).digest('hex'); + + const preimage = `tok_exp:${sampleIntention.id}:${sampleIntention.behavior_name}:${paramsHash}:behavior-mcp:${issuedAt}:${expiresAt}`; + const hmacSignature = crypto.createHmac('sha256', secret).update(preimage).digest('hex'); + + const token: DispatchToken = { + token_id: 'tok_exp', + intention_id: sampleIntention.id, + behavior_name: sampleIntention.behavior_name, + params_hash: paramsHash, + aud: 'behavior-mcp', + issued_at: issuedAt, + expires_at: expiresAt, + hmac_signature: hmacSignature, + }; + + expect(verifyIntentionDispatch(sampleIntention, token, secret)).toBe(false); + }); + + it('safely rejects tampered signature buffer lengths without throwing RangeError', () => { + const token = createValidToken({ hmac_signature: 'deadbeef' }); // Short signature + expect(() => verifyIntentionDispatch(sampleIntention, token, secret)).not.toThrow(); + expect(verifyIntentionDispatch(sampleIntention, token, secret)).toBe(false); + }); + + it('fails closed when secret is empty', () => { + const token = createValidToken(); + expect(verifyIntentionDispatch(sampleIntention, token, '')).toBe(false); + }); +}); diff --git a/tests/fixtures/canonical-state-pack.json b/tests/fixtures/canonical-state-pack.json new file mode 100644 index 0000000..412e3fe --- /dev/null +++ b/tests/fixtures/canonical-state-pack.json @@ -0,0 +1,72 @@ +{ + "pack_id": "01923456-789a-4def-9123-456789abcdef", + "project": "pentad-contract-test", + "session_id": "session-test-001", + "timestamp": "2026-09-25T12:00:00.000Z", + "visual": { + "state_id": "vs_canonical_001", + "layout_hash": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855", + "description_summary": "PuterVision HUD view with active navigation canvas and status elements", + "interactive_element_count": 4, + "embedding_ref_ids": [ + "emb_ref_001", + "emb_ref_002" + ] + }, + "spatial": { + "observer_position": [ + 10.5, + 0, + -5.25 + ], + "nearby_entities": [ + { + "id": "ent_hostile_01", + "type": "drone", + "distance": 3.75, + "status": "hostile" + }, + { + "id": "ent_waypoint_02", + "type": "terminal", + "distance": 8.12, + "status": "neutral" + } + ] + }, + "tasks": { + "active_goal": { + "id": "goal_nav_01", + "title": "Secure primary terminal", + "priority": 0.9, + "progress": 0.4 + }, + "active_blockers": [ + { + "id": "blk_door_locked", + "description": "Security lockdown active on corridor B" + } + ], + "recent_decision_ids": [ + "dec_01M3D001", + "dec_01M3D002" + ] + }, + "vitals": { + "hp": 85.5, + "threat_level": 0.65, + "resources": { + "ammo": 30, + "battery": 92 + } + }, + "utility": { + "profile_name": "tactical_cautious", + "weights": { + "caution": 0.8, + "efficiency": 0.6, + "survival": 0.95 + } + }, + "pack_hash": "283cbb0c60496b6beca4237341fb3bb1ae76490628a81b54b46953058eb9bd21" +} diff --git a/tests/transport/native-mcp-coverage.test.ts b/tests/transport/native-mcp-coverage.test.ts new file mode 100644 index 0000000..07557a8 --- /dev/null +++ b/tests/transport/native-mcp-coverage.test.ts @@ -0,0 +1,66 @@ +import { describe, it, expect } from 'vitest'; +import { + NativeMcpServer, + NativeClient, + NativeInMemoryTransport, +} from '../../src/transport/native-mcp.js'; +import { z } from '../../src/schema/schemas.js'; + +describe('Behavior-MCP Native Transport & Client Exhaustive Coverage', () => { + it('handles client-server initialize, tools, prompts, resources, and error forwarding', async () => { + const server = new NativeMcpServer({ name: 'test-behavior-server', version: '1.0.0' }); + + // Register a tool + server.tool('echo_tool', 'Echo tool', { text: z.string() }, async (args: any) => ({ + content: [{ type: 'text', text: `Echo: ${args.text}` }], + })); + + // Register a prompt + server.prompt('system_prompt', 'System prompt', { role: { type: 'string' } }, (args: any) => ({ + messages: [{ role: 'user', content: { type: 'text', text: `Role: ${args.role}` } }], + })); + + // Register a resource + server.registerResource('config', 'config://app', { title: 'config' }, async () => ({ + contents: [{ uri: 'config://app', mimeType: 'application/json', text: '{"debug":true}' }], + })); + + const [clientTransport, serverTransport] = NativeInMemoryTransport.createLinkedPair(); + await server.connect(serverTransport); + + const client = new NativeClient( + { name: 'test-client', version: '1.0.0' }, + { capabilities: { prompts: {}, resources: {} } } + ); + await client.connect(clientTransport); + + // 1. tools + const tools = await client.listTools(); + expect(tools.tools.some((t: any) => t.name === 'echo_tool')).toBe(true); + + const toolRes = await client.callTool({ name: 'echo_tool', arguments: { text: 'hello' } }); + expect(toolRes.content[0].text).toBe('Echo: hello'); + + const badTool = await client.callTool({ name: 'unknown_tool', arguments: {} }); + expect(badTool.isError).toBe(true); + + // 2. prompts + const prompts = await client.listPrompts(); + expect(prompts.prompts.some((p: any) => p.name === 'system_prompt')).toBe(true); + + const promptRes = await client.getPrompt({ name: 'system_prompt', arguments: { role: 'tester' } }); + expect(promptRes.messages[0].content.text).toBe('Role: tester'); + expect(server._registeredPrompts['system_prompt']).toBeDefined(); + + // 3. resources + const resources = await client.listResources(); + expect(resources.resources.some((r: any) => r.name === 'config')).toBe(true); + + const resourceRes = await client.readResource({ uri: 'config://app' }); + expect(resourceRes.contents[0].text).toContain('"debug":true'); + + // 4. close + await client.close(); + await server.close(); + }); +}); diff --git a/tests/unit/engine-and-handlers-boost.test.ts b/tests/unit/engine-and-handlers-boost.test.ts new file mode 100644 index 0000000..606d4e6 --- /dev/null +++ b/tests/unit/engine-and-handlers-boost.test.ts @@ -0,0 +1,163 @@ +import { describe, it, expect } from 'vitest'; +import * as fs from 'fs'; +import * as path from 'path'; +import * as os from 'os'; +import Database from 'better-sqlite3'; +import { runMigrations } from '../../src/engine/migrations.js'; +import { registerAllTools } from '../../src/tools/handlers.js'; +import { + loadProjectConfig, + getPentadHmacSecret, + ProjectConfigSchema, +} from '../../src/engine/config.js'; +import { TriggerRegistry, TriggerConditionRegistry } from '../../src/engine/triggers.js'; +import { ConditionRegistry } from '../../src/engine/browser/conditions.js'; +import { canonicalJsonStringify } from '../../src/utils/canonical-json.js'; +import { getVersion } from '../../src/utils/version.js'; +import { redactData } from '../../src/utils/redact.js'; +import { validatePath } from '../../src/utils/path-validator.js'; + +describe('Behavior-MCP Engine & Handlers Coverage Boost', () => { + it('covers config functions and schema validation', () => { + expect(ProjectConfigSchema.safeParse('not an object').success).toBe(false); + expect(ProjectConfigSchema.safeParse({ tickRateHz: 60 }).success).toBe(true); + + const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'behav_cfg_')); + const cfgFile = path.join(tmpDir, '.behavior-mcp.json'); + fs.writeFileSync(cfgFile, JSON.stringify({ tickRateHz: 30, pentadHmacSecret: 'file_secret' })); + + const cfg = loadProjectConfig(tmpDir); + expect(cfg.tickRateHz).toBe(30); + + // Test PUTERVISION_PROJECT_SLUG + process.env.PUTERVISION_PROJECT_SLUG = 'slug_behavior'; + const cfg2 = loadProjectConfig('/tmp/nonexistent_' + Date.now()); + expect(cfg2.projectName).toBe('slug_behavior'); + delete process.env.PUTERVISION_PROJECT_SLUG; + + // Test getPentadHmacSecret with env and with config + process.env.PENTAD_HMAC_SECRET = 'env_secret'; + expect(getPentadHmacSecret(tmpDir)).toBe('env_secret'); + delete process.env.PENTAD_HMAC_SECRET; + expect(getPentadHmacSecret(tmpDir)).toBe('file_secret'); + + fs.rmSync(tmpDir, { recursive: true, force: true }); + }); + + it('covers TriggerConditionRegistry and TriggerRegistry errors', () => { + // hp_threshold + expect(TriggerConditionRegistry.hp_threshold({ threshold: 20 }, { hp: 15 })).toBe(true); + expect(TriggerConditionRegistry.hp_threshold({ threshold: 20 }, { hp: 50 })).toBe(false); + + // enemy_proximity + expect(TriggerConditionRegistry.enemy_proximity({ radius: 5 }, { enemy_distance: 3 })).toBe(true); + expect(TriggerConditionRegistry.enemy_proximity({ radius: 5 }, { enemy_distance: 10 })).toBe(false); + + // semantic + expect(TriggerConditionRegistry.semantic({ key: 'see_gold', expected: true }, { see_gold: true })).toBe(true); + expect(TriggerConditionRegistry.semantic({ key: 'see_gold', expected: false }, { see_gold: true })).toBe(false); + + const testDb = new Database(':memory:'); + runMigrations(testDb); + + expect(() => + TriggerRegistry.registerTrigger(testDb, { + project: 'p', + name: '', + behavior_name: 'b', + condition_type: 'semantic', + condition_params: {}, + }) + ).toThrow('Trigger name, behavior_name, and condition_type are required'); + + testDb.close(); + }); + + it('covers browser condition registry semantic checks and expiry', () => { + const ctx: any = { + tick: 100, + blackboard: {}, + telemetry: { target_distance: 5 }, + }; + + // timer_elapsed & proximity_check + expect(ConditionRegistry.timer_elapsed({ threshold_ticks: 50 }, ctx)).toBe(true); + expect(ConditionRegistry.proximity_check({ radius: 10 }, ctx)).toBe(true); + expect(ConditionRegistry.blackboard_check({ key: 'k', expected: 'val' }, ctx)).toBe(false); + + // semantic_check missing + expect(ConditionRegistry.semantic_check({ key: 'door_open' }, ctx)).toBe(false); + + // semantic_check with object entry and timestamp expired + ctx.blackboard['semantic_decision_door_open'] = { result: true, timestamp: Date.now() - 10000 }; + expect(ConditionRegistry.semantic_check({ key: 'door_open' }, ctx)).toBe(false); + + // semantic_check with object entry valid + ctx.blackboard['semantic_decision_door_open'] = { result: true, timestamp: Date.now() }; + expect(ConditionRegistry.semantic_check({ key: 'door_open' }, ctx)).toBe(true); + + // semantic_check with primitive entry and separate timestamp + ctx.blackboard['semantic_decision_light_on'] = true; + ctx.blackboard['semantic_timestamp_light_on'] = Date.now() - 10000; + expect(ConditionRegistry.semantic_check({ key: 'light_on', expected: true }, ctx)).toBe(false); + + ctx.blackboard['semantic_timestamp_light_on'] = Date.now(); + expect(ConditionRegistry.semantic_check({ key: 'light_on', expected: true }, ctx)).toBe(true); + }); + + it('covers utils (canonical-json, version, redactData, path-validator)', () => { + expect(canonicalJsonStringify(undefined)).toBe(''); + expect(canonicalJsonStringify(null)).toBe('null'); + expect(canonicalJsonStringify(-0)).toBe('0'); + expect(canonicalJsonStringify([undefined, 1])).toBe('[null,1]'); + expect(canonicalJsonStringify({ a: undefined, b: 2 })).toBe('{"b":2}'); + expect(() => canonicalJsonStringify(Infinity)).toThrow('Invalid non-finite number'); + + expect(getVersion()).toBe('0.3.0'); + (globalThis as any).__APP_VERSION__ = '1.0.9'; + expect(getVersion()).toBe('1.0.9'); + delete (globalThis as any).__APP_VERSION__; + + const redactedArr = redactData(['Bearer secrettoken', { myApiKey: 'sk-12345678901234567890' }]); + expect(redactedArr[0]).toContain('[REDACTED]'); + + expect(() => validatePath('', { projectRoot: '/tmp' })).toThrow('File path must be a non-empty string'); + expect(() => validatePath('/etc/shadow', { projectRoot: '/tmp' })).toThrow('Access denied'); + expect(validatePath('state.json', { projectRoot: '/tmp' })).toBe('/tmp/state.json'); + }); + + it('covers manage_runtime_db actions and handler error advice', async () => { + let dbHandler: Function = () => {}; + let abortHandler: Function = () => {}; + const mockServer = { + tool: (name: string, desc: string, schema: any, fn: Function) => { + if (name === 'manage_runtime_db') dbHandler = fn; + if (name === 'abort_behavior') abortHandler = fn; + }, + }; + registerAllTools(mockServer as any); + + // Test manage_runtime_db doctor + const docRes = await dbHandler({ project: 'test_behav_p', action: 'doctor' }); + expect(docRes.isError).toBeUndefined(); + const docData = JSON.parse(docRes.content[0].text); + expect(docData.status).toBe('healthy'); + + // Test manage_runtime_db snapshot & diff & restore + const snapRes = await dbHandler({ project: 'test_behav_p', action: 'snapshot', name: 'snap_1' }); + expect(snapRes.isError).toBeUndefined(); + + const diffRes = await dbHandler({ project: 'test_behav_p', action: 'diff' }); + expect(diffRes.isError).toBeUndefined(); + + const restoreRes = await dbHandler({ project: 'test_behav_p', action: 'restore', name: 'snap_1' }); + expect(restoreRes.isError).toBeUndefined(); + + // Test invalid actions triggering error advice + const badDbRes = await dbHandler({ project: 'test_behav_p', action: 'nonexistent_action' }); + expect(badDbRes.isError).toBe(true); + + const badAbortRes = await abortHandler({ project: 'test_behav_p', action: 'bad_action' }); + expect(badAbortRes.isError).toBe(true); + }); +}); diff --git a/tests/unit/schemas-exhaustive-coverage.test.ts b/tests/unit/schemas-exhaustive-coverage.test.ts new file mode 100644 index 0000000..dc2cef6 --- /dev/null +++ b/tests/unit/schemas-exhaustive-coverage.test.ts @@ -0,0 +1,301 @@ +import { describe, it, expect } from 'vitest'; +import { + z, + StringSchema, + NumberSchema, + BooleanSchema, + EnumSchema, + ArraySchema, + RecordSchema, + ObjectSchema, + UnknownSchema, + LoadBehaviorSchema, + SetParametersSchema, + GetStatusSchema, + AbortBehaviorSchema, + RegisterTriggerSchema, + ReplayRecordingSchema, + GetMetricsSchema, + ManageBehaviorsSchema, + ManageBlackboardSchema, + ManageRuntimeDbSchema, +} from '../../src/schema/schemas.js'; + +describe('Behavior-MCP Exhaustive Schema Validation Suite', () => { + describe('StringSchema', () => { + it('handles valid strings, defaults, and optional', () => { + const s = z.string().describe('test string'); + expect(s.parse('hello')).toBe('hello'); + expect(s.toJsonSchema()).toEqual({ type: 'string', description: 'test string' }); + + const sOpt = z.string().optional(); + expect(sOpt.parse(undefined)).toBeUndefined(); + expect(sOpt.parse(null)).toBeUndefined(); + + const sDef = z.string().default('fallback'); + expect(sDef.parse(undefined)).toBe('fallback'); + expect(sDef.parse(null)).toBe('fallback'); + }); + + it('throws or fails safeParse on invalid input', () => { + const s = z.string(); + expect(() => s.parse(123)).toThrow('must be a string'); + expect(() => s.parse(undefined)).toThrow('is required'); + + const res = s.safeParse(123); + expect(res.success).toBe(false); + expect(res.error?.message).toContain('must be a string'); + expect(res.error?.errors.length).toBe(1); + + const good = s.safeParse('ok'); + expect(good.success).toBe(true); + expect(good.data).toBe('ok'); + }); + + it('supports chainable modifiers without errors', () => { + const s = z.string().min(1).max(10).int().positive(); + expect(s.parse('abc')).toBe('abc'); + }); + }); + + describe('NumberSchema', () => { + it('handles numbers, defaults, optional, and errors', () => { + const n = z.number().describe('a number'); + expect(n.parse(42)).toBe(42); + expect(n.toJsonSchema()).toEqual({ type: 'number', description: 'a number' }); + + const nOpt = z.number().optional(); + expect(nOpt.parse(undefined)).toBeUndefined(); + + const nDef = z.number().default(100); + expect(nDef.parse(undefined)).toBe(100); + + expect(() => n.parse('not a number')).toThrow('must be a number'); + expect(() => n.parse(NaN)).toThrow('must be a number'); + expect(() => n.parse(undefined)).toThrow('is required'); + + const badRes = n.safeParse('xyz'); + expect(badRes.success).toBe(false); + }); + }); + + describe('BooleanSchema', () => { + it('handles booleans, defaults, optional, and errors', () => { + const b = z.boolean().describe('flag'); + expect(b.parse(true)).toBe(true); + expect(b.parse(false)).toBe(false); + expect(b.toJsonSchema()).toEqual({ type: 'boolean', description: 'flag' }); + + const bOpt = z.boolean().optional(); + expect(bOpt.parse(undefined)).toBeUndefined(); + + const bDef = z.boolean().default(true); + expect(bDef.parse(undefined)).toBe(true); + + expect(() => b.parse('true')).toThrow('must be a boolean'); + expect(() => b.parse(undefined)).toThrow('is required'); + }); + }); + + describe('EnumSchema', () => { + it('handles enums, defaults, optional, and errors', () => { + const e = z.enum(['apple', 'banana']).describe('fruit'); + expect(e.parse('apple')).toBe('apple'); + expect(e.toJsonSchema()).toEqual({ + type: 'string', + enum: ['apple', 'banana'], + description: 'fruit', + }); + + const eOpt = z.enum(['a', 'b']).optional(); + expect(eOpt.parse(undefined)).toBeUndefined(); + + const eDef = z.enum(['x', 'y']).default('x'); + expect(eDef.parse(undefined)).toBe('x'); + + expect(() => e.parse('orange')).toThrow('one of'); + expect(() => e.parse(123)).toThrow('one of'); + expect(() => e.parse(undefined)).toThrow('is required'); + }); + }); + + describe('ArraySchema', () => { + it('handles arrays, defaults, optional, and errors', () => { + const a = z.array(z.string()).describe('list of strings'); + expect(a.parse(['a', 'b'])).toEqual(['a', 'b']); + expect(a.toJsonSchema()).toEqual({ + type: 'array', + items: { type: 'string' }, + description: 'list of strings', + }); + + const aOpt = z.array(z.number()).optional(); + expect(aOpt.parse(undefined)).toBeUndefined(); + + const aDef = z.array(z.string()).default(['def']); + expect(aDef.parse(undefined)).toEqual(['def']); + + expect(() => a.parse('not array')).toThrow('must be an array'); + expect(() => a.parse(['valid', 123])).toThrow('must be a string'); + expect(() => a.parse(undefined)).toThrow('is required'); + }); + }); + + describe('RecordSchema', () => { + it('handles records, defaults, optional, and errors', () => { + const r = z.record(z.number()).describe('record of numbers'); + expect(r.parse({ x: 1, y: 2 })).toEqual({ x: 1, y: 2 }); + expect(r.toJsonSchema()).toEqual({ + type: 'object', + additionalProperties: { type: 'number' }, + description: 'record of numbers', + }); + + const rOpt = z.record(z.string()).optional(); + expect(rOpt.parse(undefined)).toBeUndefined(); + + const rDef = z.record(z.string()).default({ k: 'v' }); + expect(rDef.parse(undefined)).toEqual({ k: 'v' }); + + expect(() => r.parse('not obj')).toThrow('must be an object'); + expect(() => r.parse([1, 2])).toThrow('must be an object'); + expect(() => r.parse({ a: 'not number' })).toThrow('must be a number'); + expect(() => r.parse(undefined)).toThrow('is required'); + }); + }); + + describe('ObjectSchema & UnknownSchema', () => { + it('handles objects, defaults, optional, passthrough, and unknown schema', () => { + const u = z.unknown().describe('anything'); + expect(u.parse('any')).toBe('any'); + expect(u.parse({ any: true })).toEqual({ any: true }); + expect(u.toJsonSchema()).toEqual({ description: 'anything' }); + + const anySchema = z.any(); + expect(anySchema.parse(123)).toBe(123); + expect(anySchema.toJsonSchema()).toEqual({}); + + const o = z + .object({ + reqStr: z.string(), + optNum: z.number().optional(), + defBool: z.boolean().default(false), + }) + .describe('an object') + .passthrough(); + + expect(o.parse({ reqStr: 'hi', optNum: 10 })).toEqual({ + reqStr: 'hi', + optNum: 10, + defBool: false, + }); + + expect(o.parse({ reqStr: 'hi' })).toEqual({ + reqStr: 'hi', + defBool: false, + }); + + const jsonSchema = o.toJsonSchema(); + expect(jsonSchema.type).toBe('object'); + expect(jsonSchema.description).toBe('an object'); + expect(jsonSchema.required).toEqual(['reqStr']); + + expect(() => o.parse('string')).toThrow('must be an object'); + expect(() => o.parse(null)).toThrow('is required'); + expect(() => o.parse({})).toThrow('is required'); + }); + }); + + describe('Tool Schemas SafeParse Tests', () => { + it('parses LoadBehaviorSchema correctly', () => { + const res = LoadBehaviorSchema.safeParse({ + action: 'load', + behavior_name: 'test_tree', + parameters: { speed: 1.5 }, + }); + expect(res.success).toBe(true); + expect(LoadBehaviorSchema.toJsonSchema().type).toBe('object'); + }); + + it('parses SetParametersSchema correctly', () => { + const res = SetParametersSchema.safeParse({ + action: 'set', + execution_id: 'exec_123', + parameters: { x: 10 }, + }); + expect(res.success).toBe(true); + expect(SetParametersSchema.toJsonSchema().type).toBe('object'); + }); + + it('parses GetStatusSchema correctly', () => { + const res = GetStatusSchema.safeParse({ + action: 'current', + limit: 10, + }); + expect(res.success).toBe(true); + expect(GetStatusSchema.toJsonSchema().type).toBe('object'); + }); + + it('parses AbortBehaviorSchema correctly', () => { + const res = AbortBehaviorSchema.safeParse({ + action: 'abort', + reason: 'emergency stop', + }); + expect(res.success).toBe(true); + expect(AbortBehaviorSchema.toJsonSchema().type).toBe('object'); + }); + + it('parses RegisterTriggerSchema correctly', () => { + const res = RegisterTriggerSchema.safeParse({ + action: 'register', + name: 'hp_trigger', + behavior_name: 'flee', + }); + expect(res.success).toBe(true); + expect(RegisterTriggerSchema.toJsonSchema().type).toBe('object'); + }); + + it('parses ReplayRecordingSchema correctly', () => { + const res = ReplayRecordingSchema.safeParse({ + action: 'start', + name: 'session_rec', + }); + expect(res.success).toBe(true); + expect(ReplayRecordingSchema.toJsonSchema().type).toBe('object'); + }); + + it('parses GetMetricsSchema correctly', () => { + const res = GetMetricsSchema.safeParse({ + action: 'aggregate', + behavior_name: 'patrol', + }); + expect(res.success).toBe(true); + expect(GetMetricsSchema.toJsonSchema().type).toBe('object'); + }); + + it('parses ManageBehaviorsSchema correctly', () => { + const res = ManageBehaviorsSchema.safeParse({ + action: 'list', + }); + expect(res.success).toBe(true); + expect(ManageBehaviorsSchema.toJsonSchema().type).toBe('object'); + }); + + it('parses ManageBlackboardSchema correctly', () => { + const res = ManageBlackboardSchema.safeParse({ + action: 'get', + key: 'target_id', + }); + expect(res.success).toBe(true); + expect(ManageBlackboardSchema.toJsonSchema().type).toBe('object'); + }); + + it('parses ManageRuntimeDbSchema correctly', () => { + const res = ManageRuntimeDbSchema.safeParse({ + action: 'stats', + }); + expect(res.success).toBe(true); + expect(ManageRuntimeDbSchema.toJsonSchema().type).toBe('object'); + }); + }); +}); diff --git a/vitest.config.ts b/vitest.config.ts index 3d74399..199caaa 100644 --- a/vitest.config.ts +++ b/vitest.config.ts @@ -5,6 +5,7 @@ export default defineConfig({ testTimeout: 30000, hookTimeout: 30000, globals: true, + include: ["tests/**/*.test.ts"], coverage: { provider: "v8", reporter: ["text", "json", "json-summary", "html"], From 9880d2d284b47b11cd8514d291221127802e54a7 Mon Sep 17 00:00:00 2001 From: Lucas Armstrong Date: Sat, 26 Sep 2026 12:09:22 -0400 Subject: [PATCH 2/2] 0.3.0: format code, fix test types and ci suite --- src/engine/config.ts | 1 - src/schema/types.ts | 17 +++++------ src/utils/canonical-json.ts | 2 +- tests/engine/token-verification.test.ts | 27 ++++++++++++++--- tests/transport/native-mcp-coverage.test.ts | 5 ++- tests/unit/engine-and-handlers-boost.test.ts | 32 +++++++++++++++----- 6 files changed, 61 insertions(+), 23 deletions(-) diff --git a/src/engine/config.ts b/src/engine/config.ts index b911fc6..91e3485 100644 --- a/src/engine/config.ts +++ b/src/engine/config.ts @@ -80,4 +80,3 @@ export function getPentadHmacSecret(projectRoot = process.cwd()): string | undef const config = loadProjectConfig(projectRoot); return config.pentadHmacSecret || config.hmacSecret; } - diff --git a/src/schema/types.ts b/src/schema/types.ts index fd5dc4b..732898a 100644 --- a/src/schema/types.ts +++ b/src/schema/types.ts @@ -133,13 +133,12 @@ export interface Outcome { } export interface DispatchToken { - token_id: string; // Unique token UUID - intention_id: string; // Bound intention ID - behavior_name: string; // Bound behavior tree name - params_hash: string; // SHA-256 of canonical intention parameters - aud: 'behavior-mcp'; // Audience — only behavior-mcp may consume this token - issued_at: string; // ISO-8601 - expires_at: string; // ISO-8601 - hmac_signature: string; // HMAC-SHA256 signature + token_id: string; // Unique token UUID + intention_id: string; // Bound intention ID + behavior_name: string; // Bound behavior tree name + params_hash: string; // SHA-256 of canonical intention parameters + aud: 'behavior-mcp'; // Audience — only behavior-mcp may consume this token + issued_at: string; // ISO-8601 + expires_at: string; // ISO-8601 + hmac_signature: string; // HMAC-SHA256 signature } - diff --git a/src/utils/canonical-json.ts b/src/utils/canonical-json.ts index 23537cf..593a733 100644 --- a/src/utils/canonical-json.ts +++ b/src/utils/canonical-json.ts @@ -1,6 +1,6 @@ /** * Canonical JSON Serialization Utility - * + * * Complies strictly with PuterVision Pentad System One specification §10.1: * 1. Object keys are sorted lexicographically (recursive, including nested objects). * 2. No trailing commas. No comments. diff --git a/tests/engine/token-verification.test.ts b/tests/engine/token-verification.test.ts index 14305b3..6d355af 100644 --- a/tests/engine/token-verification.test.ts +++ b/tests/engine/token-verification.test.ts @@ -1,6 +1,16 @@ import { describe, it, expect } from 'vitest'; import crypto from 'crypto'; -import { verifyIntentionDispatch, DispatchToken, Intention } from '../../src/engine/browser/executor-bundle.js'; +import { verifyIntentionDispatch } from '../../src/engine/browser/executor-bundle.js'; +import { DispatchToken } from '../../src/schema/types.js'; + +interface Intention { + id: string; + project?: string; + behavior_name: string; + parameters?: Record; + priority?: number; + status?: string; +} describe('Behavior-MCP DispatchToken HMAC Verification', () => { const secret = 'pentad_hmac_secret_key_for_testing_0123456789!'; @@ -18,7 +28,10 @@ describe('Behavior-MCP DispatchToken HMAC Verification', () => { const now = Date.now(); const issuedAt = new Date(now).toISOString(); const expiresAt = new Date(now + 30000).toISOString(); - const paramsHash = crypto.createHash('sha256').update(JSON.stringify({ speed: 1.0 })).digest('hex'); + const paramsHash = crypto + .createHash('sha256') + .update(JSON.stringify({ speed: 1.0 })) + .digest('hex'); const tokenId = 'tok_001'; const intentionId = overrides.intention_id || sampleIntention.id; @@ -64,7 +77,10 @@ describe('Behavior-MCP DispatchToken HMAC Verification', () => { // Issued 1500ms in the future (within 2000ms clock skew tolerance) const futureIssued = new Date(now + 1500).toISOString(); const expiresAt = new Date(now + 30000).toISOString(); - const paramsHash = crypto.createHash('sha256').update(JSON.stringify({ speed: 1.0 })).digest('hex'); + const paramsHash = crypto + .createHash('sha256') + .update(JSON.stringify({ speed: 1.0 })) + .digest('hex'); const preimage = `tok_skew:${sampleIntention.id}:${sampleIntention.behavior_name}:${paramsHash}:behavior-mcp:${futureIssued}:${expiresAt}`; const hmacSignature = crypto.createHmac('sha256', secret).update(preimage).digest('hex'); @@ -87,7 +103,10 @@ describe('Behavior-MCP DispatchToken HMAC Verification', () => { const now = Date.now(); const issuedAt = new Date(now - 60000).toISOString(); const expiresAt = new Date(now - 5000).toISOString(); // Expired 5s ago - const paramsHash = crypto.createHash('sha256').update(JSON.stringify({ speed: 1.0 })).digest('hex'); + const paramsHash = crypto + .createHash('sha256') + .update(JSON.stringify({ speed: 1.0 })) + .digest('hex'); const preimage = `tok_exp:${sampleIntention.id}:${sampleIntention.behavior_name}:${paramsHash}:behavior-mcp:${issuedAt}:${expiresAt}`; const hmacSignature = crypto.createHmac('sha256', secret).update(preimage).digest('hex'); diff --git a/tests/transport/native-mcp-coverage.test.ts b/tests/transport/native-mcp-coverage.test.ts index 07557a8..8facbcc 100644 --- a/tests/transport/native-mcp-coverage.test.ts +++ b/tests/transport/native-mcp-coverage.test.ts @@ -48,7 +48,10 @@ describe('Behavior-MCP Native Transport & Client Exhaustive Coverage', () => { const prompts = await client.listPrompts(); expect(prompts.prompts.some((p: any) => p.name === 'system_prompt')).toBe(true); - const promptRes = await client.getPrompt({ name: 'system_prompt', arguments: { role: 'tester' } }); + const promptRes = await client.getPrompt({ + name: 'system_prompt', + arguments: { role: 'tester' }, + }); expect(promptRes.messages[0].content.text).toBe('Role: tester'); expect(server._registeredPrompts['system_prompt']).toBeDefined(); diff --git a/tests/unit/engine-and-handlers-boost.test.ts b/tests/unit/engine-and-handlers-boost.test.ts index 606d4e6..8181488 100644 --- a/tests/unit/engine-and-handlers-boost.test.ts +++ b/tests/unit/engine-and-handlers-boost.test.ts @@ -50,12 +50,20 @@ describe('Behavior-MCP Engine & Handlers Coverage Boost', () => { expect(TriggerConditionRegistry.hp_threshold({ threshold: 20 }, { hp: 50 })).toBe(false); // enemy_proximity - expect(TriggerConditionRegistry.enemy_proximity({ radius: 5 }, { enemy_distance: 3 })).toBe(true); - expect(TriggerConditionRegistry.enemy_proximity({ radius: 5 }, { enemy_distance: 10 })).toBe(false); + expect(TriggerConditionRegistry.enemy_proximity({ radius: 5 }, { enemy_distance: 3 })).toBe( + true + ); + expect(TriggerConditionRegistry.enemy_proximity({ radius: 5 }, { enemy_distance: 10 })).toBe( + false + ); // semantic - expect(TriggerConditionRegistry.semantic({ key: 'see_gold', expected: true }, { see_gold: true })).toBe(true); - expect(TriggerConditionRegistry.semantic({ key: 'see_gold', expected: false }, { see_gold: true })).toBe(false); + expect( + TriggerConditionRegistry.semantic({ key: 'see_gold', expected: true }, { see_gold: true }) + ).toBe(true); + expect( + TriggerConditionRegistry.semantic({ key: 'see_gold', expected: false }, { see_gold: true }) + ).toBe(false); const testDb = new Database(':memory:'); runMigrations(testDb); @@ -121,7 +129,9 @@ describe('Behavior-MCP Engine & Handlers Coverage Boost', () => { const redactedArr = redactData(['Bearer secrettoken', { myApiKey: 'sk-12345678901234567890' }]); expect(redactedArr[0]).toContain('[REDACTED]'); - expect(() => validatePath('', { projectRoot: '/tmp' })).toThrow('File path must be a non-empty string'); + expect(() => validatePath('', { projectRoot: '/tmp' })).toThrow( + 'File path must be a non-empty string' + ); expect(() => validatePath('/etc/shadow', { projectRoot: '/tmp' })).toThrow('Access denied'); expect(validatePath('state.json', { projectRoot: '/tmp' })).toBe('/tmp/state.json'); }); @@ -144,13 +154,21 @@ describe('Behavior-MCP Engine & Handlers Coverage Boost', () => { expect(docData.status).toBe('healthy'); // Test manage_runtime_db snapshot & diff & restore - const snapRes = await dbHandler({ project: 'test_behav_p', action: 'snapshot', name: 'snap_1' }); + const snapRes = await dbHandler({ + project: 'test_behav_p', + action: 'snapshot', + name: 'snap_1', + }); expect(snapRes.isError).toBeUndefined(); const diffRes = await dbHandler({ project: 'test_behav_p', action: 'diff' }); expect(diffRes.isError).toBeUndefined(); - const restoreRes = await dbHandler({ project: 'test_behav_p', action: 'restore', name: 'snap_1' }); + const restoreRes = await dbHandler({ + project: 'test_behav_p', + action: 'restore', + name: 'snap_1', + }); expect(restoreRes.isError).toBeUndefined(); // Test invalid actions triggering error advice