diff --git a/.config/nextest.toml b/.config/nextest.toml index 7a40f4eb2..411b4bf0c 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -20,14 +20,50 @@ platform = 'cfg(windows)' filter = 'package(labby-codemode) & test(git::provider::tests::)' threads-required = 'num-test-threads' -# Shared lifecycle harness tests are compiled into several integration binaries. -# Windows CI observed four concurrent instances exhausting their unchanged 45s -# readiness deadline during database initialization; later instances finished in -# 2-7s. Reserve slots for these cases to avoid competing daemon startups. Their -# own multi-process scenarios still run concurrently inside the test. +# Shared lifecycle harness tests are compiled into many integration binaries. +# They launch real daemons, process groups, listeners, and cleanup guardians. +# Windows first exposed concurrent copies exhausting readiness budgets; hosted +# Linux shards and macOS reproduction showed the same interference class. +# Reserve every nextest slot for each live_labby case on every platform while +# preserving each case's intended internal multi-process concurrency. [[profile.ci.overrides]] -platform = 'cfg(windows)' -filter = 'package(labby) & test(live_labby::tests::)' +filter = 'package(labby) & test(live_labby::)' +threads-required = 'num-test-threads' + +# Integration binaries that launch real Labby daemons, browser/process fixtures, +# or the shared lifecycle harness must not overlap on one runner. Their product +# deadlines are intentional oracles; runner contention must not consume them. +# Reserve every nextest slot for each test in these binaries instead of inflating +# deadlines. The target shards still execute on separate CI runners in parallel. +[[profile.ci.overrides]] +filter = '''package(labby) & ( + binary(=browser_runtime_e2e) + | binary(=codemode_qualification) + | binary(=e2e_coverage_report) + | binary(=gateway_auto_reconnect) + | binary(=lifecycle_conformance) + | binary(=live_api_actions) + | binary(=live_browser_bridge) + | binary(=live_cli_actions) + | binary(=live_http_ipv6) + | binary(=live_http_observability) + | binary(=live_http_routes) + | binary(=live_integration_identity) + | binary(=live_mcp_actions) + | binary(=live_process_harness) + | binary(=live_protected_routes) + | binary(=live_restart_persistence) + | binary(=live_surface_parity) + | binary(=mcp_apps_host_qualification) + | binary(=mcp_primitives_qualification) + | binary(=mcp_spec_http_contracts) + | binary(=mcp_spec_wire_compliance) + | binary(=mcp_tools_transport_qualification) + | binary(=oauth_qualification) + | binary(=proxy_qualification) + | binary(=skills_mcp_e2e) + | binary(=webmcp_browser_qualification) +)''' threads-required = 'num-test-threads' # These three process-inventory controls hit their unchanged 3s deadlines when diff --git a/crates/labby-runtime/src/agent_runtime.rs b/crates/labby-runtime/src/agent_runtime.rs index 45ae96591..3f0c9c60c 100644 --- a/crates/labby-runtime/src/agent_runtime.rs +++ b/crates/labby-runtime/src/agent_runtime.rs @@ -34,7 +34,7 @@ pub struct AgentResourceBounds { impl AgentResourceBounds { pub fn validate(self) -> Result { if self.max_runtime_millis == 0 - || self.max_runtime_millis > 86_400_000 + || self.max_runtime_millis > AGENT_MAX_RUNTIME_MILLIS || self.max_output_bytes == 0 || self.max_output_bytes > 64 * 1024 * 1024 || self.max_external_effects > 10_000 @@ -619,6 +619,25 @@ mod tests { ); } + #[test] + fn resource_bounds_cannot_outlive_the_authority_lease_contract() { + let valid = AgentResourceBounds { + max_runtime_millis: AGENT_MAX_RUNTIME_MILLIS, + max_output_bytes: 1, + max_external_effects: 0, + }; + assert_eq!(valid.validate().unwrap(), valid); + + let overlong = AgentResourceBounds { + max_runtime_millis: AGENT_MAX_RUNTIME_MILLIS + 1, + ..valid + }; + assert_eq!( + overlong.validate().unwrap_err(), + AgentRuntimeError::InvalidBounds + ); + } + #[tokio::test] async fn runtime_bound_cancels_an_overrunning_executor_and_requests_cleanup() { let initial = epochs(1); diff --git a/crates/labby-runtime/src/gateway_config.rs b/crates/labby-runtime/src/gateway_config.rs index 12dc15537..a24aff90d 100644 --- a/crates/labby-runtime/src/gateway_config.rs +++ b/crates/labby-runtime/src/gateway_config.rs @@ -111,6 +111,7 @@ fn default_mcp_scopes() -> Vec { pub struct McpAppsConfig { /// Attach MCP App metadata to the always-available `mcp_app` control tool and advertise its UI resources. /// The control tool itself remains available when this is false. + /// Fresh installs expose the manager UI by default. #[serde(default = "default_true")] pub manager: bool, /// Advertise the synthetic Add Server app tool and its UI resources. diff --git a/crates/labby/src/mcp/handlers_resources.rs b/crates/labby/src/mcp/handlers_resources.rs index 80d139a28..4bc805978 100644 --- a/crates/labby/src/mcp/handlers_resources.rs +++ b/crates/labby/src/mcp/handlers_resources.rs @@ -1584,8 +1584,9 @@ impl LabMcpServer { // Branch 0: MCP Apps UI resources. This must precede all lab:// // fallbacks so ui:// has its own exact lookup semantics. // - // The `mcp_app` control tool is always locally available, but its own - // Labby-owned UI is opt-in like every other Labby-owned MCP App. + // The `mcp_app` control tool is always locally available. Its UI follows + // `mcp_apps.manager`: fresh installs enable it, while an operator can + // disable the resource surface without removing the control tool. #[cfg(feature = "gateway")] if uri.starts_with(MCP_APPS_APP_URI) { if !self.mcp_apps_config().await.manager { diff --git a/crates/labby/tests/action_matrix_completeness.rs b/crates/labby/tests/action_matrix_completeness.rs index a73f7a8e7..0137bb907 100644 --- a/crates/labby/tests/action_matrix_completeness.rs +++ b/crates/labby/tests/action_matrix_completeness.rs @@ -1,22 +1,26 @@ #![allow(clippy::panic)] -#[path = "support/lib.rs"] -mod support; +#[allow(dead_code)] +#[path = "support/action_matrix.rs"] +mod action_matrix; +#[allow(dead_code)] +#[path = "support/authority_matrix.rs"] +mod authority_matrix; use std::collections::{BTreeMap, BTreeSet}; -use labby_primitives::access::{Capability, CapabilitySchemaVersion, RoleTemplate}; -use serde_json::Value; -use support::action_matrix::{ +use action_matrix::{ CatalogAction, EXPECTED_ACTIONS, EXPECTED_API_ACTIONS, EXPECTED_CLI_ACTIONS, EXPECTED_MCP_ACTIONS, EXPECTED_SHARED_CLI_MCP_API_ACTIONS, EXPECTED_WEB_ACTIONS, EvidenceLevel, PersistenceClass, ScenarioKind, ScenarioOwner, Surface, catalog_map, intent_map, intent_map_from, intents, validate_intent_shape, }; -use support::authority_matrix::{ +use authority_matrix::{ DEPOT_OPERATIONS, OperationClass, OwnerKind, ResourceFamily, classify_labby, depot_fixture_operation_names, depot_snapshot_operation_names, duplicate_depot_operations, }; +use labby_primitives::access::{Capability, CapabilitySchemaVersion, RoleTemplate}; +use serde_json::Value; const ACTION_CATALOG: &str = include_str!("../../../docs/generated/action-catalog.json"); const AUTHORITY_MATRIX: &str = @@ -994,7 +998,7 @@ services = ["stash"] #[test] fn generated_outcomes_cannot_downgrade_required_intent() { - use support::action_matrix::{CaseOutcome, OutcomeStatus, outcome_satisfies}; + use action_matrix::{CaseOutcome, OutcomeStatus, outcome_satisfies}; let intent = intents().iter().find(|intent| intent.required).unwrap(); let skipped = CaseOutcome { key: intent.key(), diff --git a/crates/labby/tests/ci_changed_paths.rs b/crates/labby/tests/ci_changed_paths.rs index 22003a883..d63d56bcf 100644 --- a/crates/labby/tests/ci_changed_paths.rs +++ b/crates/labby/tests/ci_changed_paths.rs @@ -412,6 +412,15 @@ fn rust_manifests_lockfiles_and_toolchains_run_full_tests() { } } +#[test] +fn nextest_policy_changes_run_the_full_rust_test_path() { + let out = classify("pull_request", &[".config/nextest.toml"]); + assert_eq!(out["rust_compile"], "true"); + assert_eq!(out["rust_test"], "true"); + assert_eq!(out["docker"], "true"); + assert_eq!(out["release"], "true"); +} + #[test] fn frontend_changes_enable_web_release_and_container_without_rust_tests() { let out = classify("pull_request", &["apps/gateway-admin/app/page.tsx"]); @@ -1210,6 +1219,74 @@ fn ci_gate_aggregates_every_non_advisory_job() { } } +/// Process-owning integration fixtures are compiled into many test binaries. +/// The CI profile must serialize both duplicated harness-unit cases and every +/// integration binary that launches real process state. +#[test] +fn nextest_ci_isolates_process_harnesses_and_timing_oracles() { + let root = repo_root(); + let nextest = + fs::read_to_string(root.join(".config/nextest.toml")).expect("read .config/nextest.toml"); + let lifecycle_filter = "filter = 'package(labby) & test(live_labby::)'"; + assert!( + nextest.contains(lifecycle_filter), + "CI must serialize every embedded live_labby harness copy" + ); + let lifecycle = nextest + .split(lifecycle_filter) + .nth(1) + .and_then(|section| section.split("[[").next()) + .expect("live_labby override"); + assert!( + lifecycle.contains("threads-required = 'num-test-threads'"), + "live_labby override must reserve every nextest slot" + ); + assert!( + !lifecycle.contains("platform ="), + "process-harness isolation must apply on Linux, macOS, and Windows" + ); + + let process_override = nextest + .split("filter = '''package(labby) & (") + .nth(1) + .and_then(|section| section.split("[[").next()) + .expect("process-owning integration override"); + assert!( + process_override.contains("threads-required = 'num-test-threads'"), + "process-owning integration tests must reserve every nextest slot" + ); + let test_dir = root.join("crates/labby/tests"); + let mut process_binaries = Vec::new(); + for entry in fs::read_dir(&test_dir).expect("read integration test directory") { + let path = entry.expect("test entry").path(); + if path.extension().and_then(|value| value.to_str()) != Some("rs") { + continue; + } + let source = fs::read_to_string(&path).expect("read integration test source"); + let path_marker = ["support", "/live_labby.rs"].concat(); + let type_marker = ["support::", "LiveLabby"].concat(); + let module_marker = ["support::", "live_labby"].concat(); + if source.contains(&path_marker) + || source.contains(&type_marker) + || source.contains(&module_marker) + { + process_binaries.push( + path.file_stem() + .and_then(|value| value.to_str()) + .expect("UTF-8 test name") + .to_owned(), + ); + } + } + assert!(!process_binaries.is_empty()); + for binary in process_binaries { + assert!( + process_override.contains(&format!("binary(={binary})")), + "process-owning integration binary {binary} must be isolated in nextest CI" + ); + } +} + /// The merge gate must finish in about ten minutes. That budget is kept by /// fanning the long serial suites out across matrix shards, by moving the /// slow non-gating suites (coverage, the gateway-slice re-run, doctests) off @@ -1261,6 +1338,7 @@ fn merge_gate_shards_heavy_suites_to_stay_under_ten_minutes() { "--partition \"hash:${index}/${unit_shards}\"", "for file in crates/labby/tests/*.rs", "cargo nextest run -p labby", + "not test(live_labby::) | binary(=live_process_harness)", "--exclude labby", "cargo test --doc --workspace --all-features --locked", ] { diff --git a/crates/labby/tests/mcp_apps_host_qualification.rs b/crates/labby/tests/mcp_apps_host_qualification.rs index c12b70318..2e96cf1a6 100644 --- a/crates/labby/tests/mcp_apps_host_qualification.rs +++ b/crates/labby/tests/mcp_apps_host_qualification.rs @@ -157,9 +157,23 @@ async fn q4_real_resources_render_in_distinct_openai_and_anthropic_emulators() { .await .expect("real Labby MCP process"); - let initially_hidden = runner.read_resource("ui://lab/apps/manage").await; + // Fresh installs intentionally expose Labby-owned MCP Apps by default. + // Exercise the policy transition explicitly so this live oracle verifies + // that direct resources/read cannot bypass a disabled surface. + let initially_disabled = runner + .call_raw( + "mcp_app", + serde_json::json!({"action":"disable", "params":{"target":"manager"}}), + ) + .await + .expect("disable manager through real tools/call"); + assert_ne!( + initially_disabled.is_error, + Some(true), + "initial disable failed: {initially_disabled:?}" + ); assert!( - initially_hidden.is_err(), + runner.read_resource("ui://lab/apps/manage").await.is_err(), "disabled app resource must not bypass policy" ); @@ -263,10 +277,12 @@ async fn q4_real_resources_render_in_distinct_openai_and_anthropic_emulators() { Some(true), "disable failed: {disabled:?}" ); - assert!( - runner.read_resource(mcp_uri).await.is_err(), - "revoked resource remained readable on the same authenticated session" - ); + for uri in [mcp_uri, openai_uri] { + assert!( + runner.read_resource(uri).await.is_err(), + "revoked resource {uri} remained readable on the same authenticated session" + ); + } let cleanup = runner.finish().await; assert!( diff --git a/crates/labby/tests/support/mcp_apps_host_qualification/runner.mjs b/crates/labby/tests/support/mcp_apps_host_qualification/runner.mjs index 5650c6466..2aee20b85 100644 --- a/crates/labby/tests/support/mcp_apps_host_qualification/runner.mjs +++ b/crates/labby/tests/support/mcp_apps_host_qualification/runner.mjs @@ -121,21 +121,38 @@ function listen() { server.listen(0, "127.0.0.1", () => ok(server.address().port)); }); } -async function wait(p, s) { +async function waitManagerSwitch(p, checked) { + const expected = String(checked); try { await p - .locator("#status") - .filter({ hasText: s }) + .locator( + `button[data-target=manager][aria-checked="${expected}"]:not([disabled])`, + ) .waitFor({ timeout: 5000 }); } catch (e) { + const button = p.locator("button[data-target=manager]"); throw new Error( - `${e.message}; actual status=${await p + `${e.message}; actual checked=${await button + .getAttribute("aria-checked") + .catch(() => "missing")}; summary=${await p + .locator("#summary") + .textContent() + .catch(() => "missing")}; status=${await p .locator("#status") .textContent() .catch(() => "missing")}`, ); } } +async function setManager(enabled) { + await call({ + name: "mcp_app", + arguments: { + action: enabled ? "enable" : "disable", + params: { target: "manager" }, + }, + }); +} async function csp(p, port) { const got = await p.evaluate(async (port) => { const violations = []; @@ -176,16 +193,10 @@ async function csp(p, port) { return got; } async function exercise(host, p, port) { - await wait(p, "up to date"); - const policy = await csp(p, port), - on = await p - .locator("button[data-target=manager]") - .getAttribute("aria-checked"); + await waitManagerSwitch(p, true); + const policy = await csp(p, port); await p.locator("button[data-target=manager]").click(); - await wait( - p, - on === "true" ? "Disabled MCP Apps Manager" : "Enabled MCP Apps Manager", - ); + await waitManagerSwitch(p, false); const auth = await p.evaluate(() => invalidMcpCall({ name: "mcp_app", @@ -214,6 +225,7 @@ try { await ctx.exposeFunction("invalidMcpCall", (x) => call(x, "invalid-q4-token"), ); + await setManager(true); const o = await ctx.newPage(); await o.addInitScript( () => @@ -242,6 +254,7 @@ try { message: "injected OpenAI transport failure", }); events.find((event) => event.host === "openai-emulator").type_error_calls = 1; + await setManager(true); const h = await ctx.newPage(); await h.goto(`http://127.0.0.1:${port}/anthropic`); await h.waitForTimeout(250); diff --git a/docs/dev/TESTING.md b/docs/dev/TESTING.md index 27cc6e6cb..b456a1da4 100644 --- a/docs/dev/TESTING.md +++ b/docs/dev/TESTING.md @@ -252,6 +252,15 @@ Preferred runner: - use `cargo test` only when nextest is unavailable or you need a narrow one-off command that nextest does not cover cleanly - for this repo, `cargo nextest run --manifest-path crates/labby/Cargo.toml --all-features` is the standard full-crate verification command +CI splits `crates/labby/tests/*.rs` across `labby-int-*` target shards. Many of +those binaries embed the same `live_labby` process-supervision harness. Shards +de-duplicate those internal support tests and execute them once through the +`live_process_harness` target. The CI nextest profile also reserves all local +test slots for process-owning integration binaries, so separate daemon/browser +fixtures cannot consume one another's readiness, cleanup, or timeout budgets. +This preserves the product deadlines the tests prove without trading flakiness +for inflated timeouts. + If tests were not run, say so explicitly. ## Command Guidance diff --git a/docs/runtime/OAUTH.md b/docs/runtime/OAUTH.md index ead95c530..0bf379f41 100644 --- a/docs/runtime/OAUTH.md +++ b/docs/runtime/OAUTH.md @@ -1242,8 +1242,9 @@ sets `readOnlyHint: true` without a contradictory `destructiveHint: true`. `codemode` and the optional `codemode_ui` require `lab` or `lab:admin` and retain full execution authority. On the root gateway, the always-available `mcp_app` control tool uses the same read/open scopes, while changing Labby-owned app -visibility requires `lab:admin`. Its own manager UI is opt-in like every other -Labby-owned app surface. The control tool is omitted from protected subset routes +visibility requires `lab:admin`. Fresh installs expose its manager UI by default; +an operator can disable that UI resource without removing the control tool. The +control tool is omitted from protected subset routes so a subset-scoped token cannot mutate gateway-global UI visibility. Gateway management actions on a protected `gateway_subset` route are bounded diff --git a/scripts/ci/changed_paths.py b/scripts/ci/changed_paths.py index 789df6bf0..8e8d4032b 100755 --- a/scripts/ci/changed_paths.py +++ b/scripts/ci/changed_paths.py @@ -211,6 +211,7 @@ def classify(event: str, paths: list[str]) -> dict[str, bool]: "clippy.toml", "deny.toml", "verification/Cargo.toml", + ".config/nextest.toml", }, ) # `verification/` is a separate Cargo workspace and deliberately matches diff --git a/scripts/ci/run-test-shard.sh b/scripts/ci/run-test-shard.sh index 2104d6c49..429032dd6 100755 --- a/scripts/ci/run-test-shard.sh +++ b/scripts/ci/run-test-shard.sh @@ -1,11 +1,11 @@ #!/usr/bin/env bash # Run one shard of the workspace test suite for the `test` job in ci.yml. # -# Shards select Cargo targets rather than nextest filters: nextest compiles -# every selected target before it filters tests, so a hash partition over the -# whole workspace would still link all of `crates/labby/tests` on every -# runner. Keeping the target set small is what lets the pool finish a shard in -# a few minutes instead of one runner spending 40 minutes on everything. +# Shards select a small Cargo target set first: nextest compiles every selected +# target before applying test filters, so a hash partition over the whole +# workspace would still link all of `crates/labby/tests` on every runner. The +# integration shards apply one additional filter only to de-duplicate shared +# support-harness tests after the target set is bounded. set -euo pipefail shard="${1:-}" @@ -56,7 +56,13 @@ case "$shard" in fi done [ "${#targets[@]}" -gt 0 ] || { echo "shard $shard selected no targets" >&2; exit 1; } - cargo nextest run -p labby "${common[@]}" "${targets[@]}" + # `support/live_labby.rs` is embedded into many integration binaries, so its + # internal harness tests would otherwise run once per binary. Execute those + # shared tests only in the dedicated live_process_harness target; every + # binary's actual product tests still run normally. This removes duplicate + # process supervisors and keeps the shard fast without weakening coverage. + cargo nextest run -p labby "${common[@]}" "${targets[@]}" \ + -E 'not test(live_labby::) | binary(=live_process_harness)' ;; crates) # Integration tests of the extracted crates, then the workspace doctests