From 2e0619b935a07ce6dd63d8eb723481e6b900ed79 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 8 Sep 2026 10:27:22 +0900 Subject: [PATCH 1/5] test(heldout): require declared item-covariate sample size Item-side language/domain evidence must fail closed without a positive sample_size, and the declared count must be the person population. --- .../test_psychometric_benchmark_boundaries.py | 34 +++++++++++++++++++ tests/test_psychometric_routing.py | 2 +- 2 files changed, 35 insertions(+), 1 deletion(-) diff --git a/tests/test_psychometric_benchmark_boundaries.py b/tests/test_psychometric_benchmark_boundaries.py index a7f2bbc92..afac15c4e 100644 --- a/tests/test_psychometric_benchmark_boundaries.py +++ b/tests/test_psychometric_benchmark_boundaries.py @@ -1,6 +1,7 @@ """Diagnostic harness denominators and startup guards, not estimator validation.""" import builtins +import inspect import json from pathlib import Path import runpy @@ -166,6 +167,39 @@ def test_observation_p95_tracks_actual_sample_count(monkeypatch, capsys, sample_ assert report["p95_observe_ms"] == (95 * sample_count + 99) // 100 + +def test_item_covariate_requires_declared_sample_size() -> None: + """Item-side language/domain evidence cannot invent a 1,200-row default.""" + parameters = inspect.signature( + heldout._validate_item_covariate_effect + ).parameters + assert parameters["sample_size"].default is None + with pytest.raises(ValueError, match="sample_size"): + heldout._validate_item_covariate_effect() + with pytest.raises(ValueError, match="sample_size"): + heldout._validate_item_covariate_effect(sample_size=True) + + +def test_item_covariate_uses_declared_sample_size(monkeypatch) -> None: + """The declared sample size is the actual item-covariate person population.""" + seen = {} + + def fake_fit(responses, *_args, **_kwargs): + """Record the person count without running the native fit.""" + seen["n"] = responses.shape[0] + return SimpleNamespace( + population={"delta": -0.8}, + convergence_status="converged", + n_iter=1, + ) + + monkeypatch.setattr(heldout.fast_mlsirm, "fit", fake_fit) + report = heldout._validate_item_covariate_effect(sample_size=40) + assert report["sample_size"] == 40 + assert seen["n"] == 40 + assert report["items"] == 12 + + def test_heldout_runtime_guard_precedes_optional_dependency_imports(monkeypatch): """Unsupported Python must receive the runnable command before loading ML libraries.""" original_import = builtins.__import__ diff --git a/tests/test_psychometric_routing.py b/tests/test_psychometric_routing.py index 8d22a09fb..805036226 100644 --- a/tests/test_psychometric_routing.py +++ b/tests/test_psychometric_routing.py @@ -722,7 +722,7 @@ def test_heldout_report_pairs_every_delta_with_its_interval(monkeypatch) -> None assert judge["severity_rmse"] == pytest.approx(0.018292059677437307) covariate = report["item_language_domain_effect_validation"] assert covariate["method"] == "multigroup_item_covariate" - assert covariate["sample_size"] == heldout_benchmark.ITEM_COVARIATE_SAMPLE_SIZE + assert covariate["sample_size"] == heldout_benchmark.DECLARED_ITEM_COVARIATE_SAMPLE_SIZE assert covariate["seed"] == heldout_benchmark.ITEM_COVARIATE_SEED assert covariate["true_delta"] == -0.8 assert covariate["estimated_delta"] == pytest.approx(-0.7896498094289646) From cb5a41e8efa98c19ab87a2dee5f4831b844e6b64 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 8 Sep 2026 10:27:23 +0900 Subject: [PATCH 2/5] fix(heldout): declare item-covariate sample size Remove the hidden 1,200-row item-covariate default. The harness run still writes 1,200 as this run's choice and records sample_size. --- scripts/benchmark_psychometric_heldout.py | 26 +++++++++++++++++------ 1 file changed, 19 insertions(+), 7 deletions(-) diff --git a/scripts/benchmark_psychometric_heldout.py b/scripts/benchmark_psychometric_heldout.py index 8dec2f8eb..a8fd9629b 100644 --- a/scripts/benchmark_psychometric_heldout.py +++ b/scripts/benchmark_psychometric_heldout.py @@ -41,7 +41,7 @@ DIF_SEED = 260_906 JUDGE_SAMPLE_SIZE = 1_000 JUDGE_SEED = 260_907 -ITEM_COVARIATE_SAMPLE_SIZE = 1_200 +DECLARED_ITEM_COVARIATE_SAMPLE_SIZE = 1_200 ITEM_COVARIATE_SEED = 260_908 ITEM_COVARIATE_MAX_ITER = 1_000 UNCERTAINTY_SAMPLE_SIZE = 1_200 @@ -1172,15 +1172,25 @@ def _validate_judge_effects() -> dict[str, object]: } -def _validate_item_covariate_effect() -> dict[str, object]: - """Estimate one known item-side context contrast without claiming invariance.""" +def _validate_item_covariate_effect( + *, + sample_size: int | None = None, +) -> dict[str, object]: + """Estimate one known item-side context contrast without claiming invariance. + + ``sample_size`` is a required declaration. ``None`` is a fail-closed + sentinel, not a statistical default. + """ + declared_sample_size = _require_declared_positive_int( + sample_size, "sample_size" + ) generator = np.random.default_rng(ITEM_COVARIATE_SEED) item_count = 12 - group_id = np.arange(ITEM_COVARIATE_SAMPLE_SIZE) % 2 + group_id = np.arange(declared_sample_size) % 2 covariate = np.vstack( (np.linspace(0.0, 1.0, item_count), np.linspace(1.0, 0.0, item_count)) ) - ability = generator.standard_normal(ITEM_COVARIATE_SAMPLE_SIZE) + ability = generator.standard_normal(declared_sample_size) intercept = np.linspace(-1.0, 1.0, item_count) true_delta = -0.8 logits = ability[:, None] + intercept[None, :] + true_delta * covariate[group_id] @@ -1206,7 +1216,7 @@ def _validate_item_covariate_effect() -> dict[str, object]: return { "method": "multigroup_item_covariate", "max_iterations": ITEM_COVARIATE_MAX_ITER, - "sample_size": ITEM_COVARIATE_SAMPLE_SIZE, + "sample_size": declared_sample_size, "seed": ITEM_COVARIATE_SEED, "contexts": len(covariate), "items": item_count, @@ -1801,7 +1811,9 @@ def run_benchmark( selection_utility = _validate_selection_utility() candidate_group_dif = _validate_candidate_group_dif() judge_effects = _validate_judge_effects() - item_covariate_effect = _validate_item_covariate_effect() + item_covariate_effect = _validate_item_covariate_effect( + sample_size=DECLARED_ITEM_COVARIATE_SAMPLE_SIZE + ) parameter_uncertainty = _validate_parameter_uncertainty() baseline_latency, candidate_latency, baseline_medians, candidate_medians = ( _measure_paired_latency(baseline_evidence, candidate_evidence) From 232502f1f3ddf61b8c2a6e5b62476168c9d5dbe6 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 8 Sep 2026 10:27:24 +0900 Subject: [PATCH 3/5] docs: record declared item-covariate sample size ADR 0053 is Proposed. Production route/conduct defaults stay locked. --- CHANGELOG.md | 4 +- .../doctoring/nim-benchmark-evidence-grade.md | 22 +++++++ ...053-declared-item-covariate-sample-size.md | 64 +++++++++++++++++++ docs/product-technical-gap-baseline.md | 15 +++++ 4 files changed, 104 insertions(+), 1 deletion(-) create mode 100644 docs/planning/adrs/0053-declared-item-covariate-sample-size.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 911455f56..b29d98dbb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,9 +10,11 @@ and this project uses [Semantic Versioning](https://semver.org/spec/v2.0.0.html) ## [0.2.0] - Unreleased - ### Changed +- Held-out item-side language/domain evidence now requires a declared sample + size. The hidden 1,200-row default is removed. The harness run still writes + 1,200 as this run's choice. - The psychometric held-out harness now requires a declared resample count, percentile coverage, and seed for paired intervals. Hidden 2,000-sample 95% defaults are removed. The script entry still writes 2,000, 0.95, and diff --git a/docs/doctoring/nim-benchmark-evidence-grade.md b/docs/doctoring/nim-benchmark-evidence-grade.md index bd53b4d59..8712ec3f5 100644 --- a/docs/doctoring/nim-benchmark-evidence-grade.md +++ b/docs/doctoring/nim-benchmark-evidence-grade.md @@ -332,6 +332,28 @@ sequenceDiagram Note over Operator,Report: Production route and conduct defaults stay locked ``` +### Declared item-covariate sample size (2026-09-08, proposed) + +The multigroup item-covariate screen used a hidden 1,200-row person sample. +That number chose Monte Carlo precision without an operator declaration. +`_validate_item_covariate_effect` now requires `sample_size`. The harness +run writes 1,200 as this run's choice and records `sample_size`. + +This slice does not change production route/conduct defaults. Other harness +sample sizes remain later work. + +```mermaid +sequenceDiagram + participant Operator as Run declaration + participant Covariate as Item-side covariate + participant Report as Held-out report + Operator->>Covariate: Sample size + Covariate->>Covariate: Fail closed on missing or non-positive declarations + Covariate->>Report: Contrast error and declared sample size + Note over Operator,Report: Production route and conduct defaults stay locked +``` + + ### Failure-inclusive comparison repair (2026-09-05, proposed) The previous paired comparison selected only jointly successful cells even diff --git a/docs/planning/adrs/0053-declared-item-covariate-sample-size.md b/docs/planning/adrs/0053-declared-item-covariate-sample-size.md new file mode 100644 index 000000000..63961bef9 --- /dev/null +++ b/docs/planning/adrs/0053-declared-item-covariate-sample-size.md @@ -0,0 +1,64 @@ +--- +id: "0053" +title: "Declare held-out item-covariate sample size" +status: proposed +proposed_date: "2026-09-08" +deciders: + - "repository maintainer" +affected_components: + - "scripts/benchmark_psychometric_heldout.py" +related: + - path: "docs/planning/adrs/0044-declared-heldout-bootstrap-coverage.md" + relation: extends +success_criteria: + - metric: "no hidden item-covariate sample default" + target: "omitted or non-positive sample_size fails closed" + source: "tests/test_psychometric_benchmark_boundaries.py::test_item_covariate_requires_declared_sample_size" + - metric: "declared count is the person population" + target: "report sample_size equals the declared count" + source: "tests/test_psychometric_benchmark_boundaries.py::test_item_covariate_uses_declared_sample_size" +--- + +# ADR 0053: Declare held-out item-covariate sample size + +- Status: Proposed +- Date: 2026-09-08 +- Doctoring record: [`docs/doctoring/nim-benchmark-evidence-grade.md`](../../doctoring/nim-benchmark-evidence-grade.md) + +## Product requirement + +Buyer-facing item-side language/domain contrast evidence must be +reconstructible. A hidden `ITEM_COVARIATE_SAMPLE_SIZE = 1_200` chose Monte +Carlo precision without an operator declaration. + +## Decision + +In the context of the held-out multigroup item-covariate screen, facing +`ITEM_COVARIATE_SAMPLE_SIZE = 1_200`, we chose a required `sample_size` +declaration and against restoring that constant, to keep the person +population reconstructible, accepting that `_validate_item_covariate_effect` +without the argument fails closed. + +The harness run writes 1,200 as this run's choice and records `sample_size`. +Optimizer max-iter stays later work. Production route/conduct defaults stay +locked. + +## Alternatives considered + +- Keep 1,200 as a module constant. Rejected: it hides the person population + from the report consumer. +- Declare max_iter in the same slice. Rejected: it is optimizer budget, not + the sample size. + +## Consequences + +Positive: item-covariate sample size is reconstructible from the report and +the helper signature. + +Negative: library callers of `_validate_item_covariate_effect` must pass the +declaration. + +## Remaining work + +Other repository-authored harness sample sizes stay open. This ADR is +Proposed until independent review and protected delivery. diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index 99f516c8c..06bd13483 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -1,5 +1,20 @@ # Contextual Orchestrator: Product & Technical Gap Baseline +## 2026-09-08 declared item-covariate sample size (proposed) + +Successor of the held-out bootstrap-coverage slice removes hidden +`ITEM_COVARIATE_SAMPLE_SIZE = 1_200` from +`scripts/benchmark_psychometric_heldout.py`. Sample size is a required +positive integer declaration. Missing, boolean, or non-positive values fail +closed. The harness run still writes 1,200 as this run's choice and records +`sample_size`. ADR 0053 is Proposed. + +Local contract tests on this working tree: item-covariate sample-size +declaration and population checks plus existing held-out key/report pins. +This is not buyer-held-out accuracy, p95 latency, or protected merge +evidence. Production route/conduct defaults stay locked. Other harness +sample sizes remain later work. + ## 2026-09-13 NIM consumed-response repair — Proposed Source `dda57de36dcbd9254f2e4215279494fbaccfc03f` closes HTTP errors consumed From 87642d7bc44aaae418d5182e28f952a6d8b2c496 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Tue, 8 Sep 2026 14:36:59 +0900 Subject: [PATCH 4/5] fix(heldout): validate two-group covariate sample boundary Signed-off-by: Seongho Bae --- ...053-declared-item-covariate-sample-size.md | 9 +++++++- docs/product-technical-gap-baseline.md | 17 +++++++++++++- scripts/benchmark_psychometric_heldout.py | 6 +++-- .../test_psychometric_benchmark_boundaries.py | 13 +++++++---- tests/test_psychometric_routing.py | 22 ++++++++++--------- 5 files changed, 49 insertions(+), 18 deletions(-) diff --git a/docs/planning/adrs/0053-declared-item-covariate-sample-size.md b/docs/planning/adrs/0053-declared-item-covariate-sample-size.md index 63961bef9..aa1e35a19 100644 --- a/docs/planning/adrs/0053-declared-item-covariate-sample-size.md +++ b/docs/planning/adrs/0053-declared-item-covariate-sample-size.md @@ -12,7 +12,7 @@ related: relation: extends success_criteria: - metric: "no hidden item-covariate sample default" - target: "omitted or non-positive sample_size fails closed" + target: "omitted, boolean, or sample_size below two fails closed" source: "tests/test_psychometric_benchmark_boundaries.py::test_item_covariate_requires_declared_sample_size" - metric: "declared count is the person population" target: "report sample_size equals the declared count" @@ -33,6 +33,13 @@ Carlo precision without an operator declaration. ## Decision +The two-group contrast requires at least two observations. A declaration of +one produces only group zero while the covariate matrix contains two groups; +reject it before fitting rather than exposing a downstream matrix-shape error. +This is a structural minimum, not evidence of adequate statistical power. +Boundary tests retain both groups for counts two, three, and forty. The +full-size report test independently pins this run's declared count to 1,200. + In the context of the held-out multigroup item-covariate screen, facing `ITEM_COVARIATE_SAMPLE_SIZE = 1_200`, we chose a required `sample_size` declaration and against restoring that constant, to keep the person diff --git a/docs/product-technical-gap-baseline.md b/docs/product-technical-gap-baseline.md index 06bd13483..9a12b0616 100644 --- a/docs/product-technical-gap-baseline.md +++ b/docs/product-technical-gap-baseline.md @@ -1,11 +1,26 @@ # Contextual Orchestrator: Product & Technical Gap Baseline +## 2026-09-08 item-covariate two-group boundary repair (proposed) + +Review of PR #1104 at `78d331451c2e9667e949d1d274dfe48708782fa9` +identified a mismatch between the positive-count contract and the two-group +covariate design. One observation reached the native fitter with only group +zero and a two-row covariate matrix; the new boundary test reproduced its +matrix-shape error (one failed test, 33 deselected). The caller now rejects +counts below two before fitting. This structural minimum does not establish +statistical power or measurement validity. Tests verify both group IDs at +counts two, three, and forty, reject zero and negative counts, and independently +pin the harness declaration to 1,200. Boundary and ADR checks passed: 37 tests +in 17.14 seconds. Expanded routing, full-size synthetic report, boundary, +and ADR regression passed: 72 tests in 883.69 seconds (exit zero). +ADR 0053 remains Proposed; production defaults remain unchanged. + ## 2026-09-08 declared item-covariate sample size (proposed) Successor of the held-out bootstrap-coverage slice removes hidden `ITEM_COVARIATE_SAMPLE_SIZE = 1_200` from `scripts/benchmark_psychometric_heldout.py`. Sample size is a required -positive integer declaration. Missing, boolean, or non-positive values fail +integer declaration of at least two. Missing, boolean, or smaller values fail closed. The harness run still writes 1,200 as this run's choice and records `sample_size`. ADR 0053 is Proposed. diff --git a/scripts/benchmark_psychometric_heldout.py b/scripts/benchmark_psychometric_heldout.py index a8fd9629b..9f7f464e4 100644 --- a/scripts/benchmark_psychometric_heldout.py +++ b/scripts/benchmark_psychometric_heldout.py @@ -1178,12 +1178,14 @@ def _validate_item_covariate_effect( ) -> dict[str, object]: """Estimate one known item-side context contrast without claiming invariance. - ``sample_size`` is a required declaration. ``None`` is a fail-closed - sentinel, not a statistical default. + ``sample_size`` must declare at least two observations for two groups. + ``None`` is a fail-closed sentinel, not a statistical default. """ declared_sample_size = _require_declared_positive_int( sample_size, "sample_size" ) + if declared_sample_size < 2: + raise ValueError("sample_size must be at least 2 for two groups") generator = np.random.default_rng(ITEM_COVARIATE_SEED) item_count = 12 group_id = np.arange(declared_sample_size) % 2 diff --git a/tests/test_psychometric_benchmark_boundaries.py b/tests/test_psychometric_benchmark_boundaries.py index afac15c4e..f8e2eb675 100644 --- a/tests/test_psychometric_benchmark_boundaries.py +++ b/tests/test_psychometric_benchmark_boundaries.py @@ -178,15 +178,20 @@ def test_item_covariate_requires_declared_sample_size() -> None: heldout._validate_item_covariate_effect() with pytest.raises(ValueError, match="sample_size"): heldout._validate_item_covariate_effect(sample_size=True) + for sample_size in (0, -1, 1): + with pytest.raises(ValueError, match="sample_size"): + heldout._validate_item_covariate_effect(sample_size=sample_size) -def test_item_covariate_uses_declared_sample_size(monkeypatch) -> None: +@pytest.mark.parametrize("sample_size", [2, 3, 40]) +def test_item_covariate_uses_declared_sample_size(monkeypatch, sample_size) -> None: """The declared sample size is the actual item-covariate person population.""" seen = {} def fake_fit(responses, *_args, **_kwargs): """Record the person count without running the native fit.""" seen["n"] = responses.shape[0] + assert set(_kwargs["group_id"]) == {0, 1} return SimpleNamespace( population={"delta": -0.8}, convergence_status="converged", @@ -194,9 +199,9 @@ def fake_fit(responses, *_args, **_kwargs): ) monkeypatch.setattr(heldout.fast_mlsirm, "fit", fake_fit) - report = heldout._validate_item_covariate_effect(sample_size=40) - assert report["sample_size"] == 40 - assert seen["n"] == 40 + report = heldout._validate_item_covariate_effect(sample_size=sample_size) + assert report["sample_size"] == sample_size + assert seen["n"] == sample_size assert report["items"] == 12 diff --git a/tests/test_psychometric_routing.py b/tests/test_psychometric_routing.py index 805036226..9a3541644 100644 --- a/tests/test_psychometric_routing.py +++ b/tests/test_psychometric_routing.py @@ -4,29 +4,30 @@ import inspect import math -import numpy as np +import threading from dataclasses import replace from pathlib import Path -import threading + +import numpy as np import pytest -from contextual_orchestrator import orchestrator as routing_module -import scripts.benchmark_psychometric_heldout as heldout_benchmark +import scripts.benchmark_psychometric_heldout as heldout_benchmark from contextual_orchestrator import ( ModelAgent, TaskOrchestrator, default_role_effort_catalog, ) +from contextual_orchestrator import orchestrator as routing_module from contextual_orchestrator.psychometric_routing import PsychometricRoutingEvidence from contextual_orchestrator.reasoning_effort_profile import EffortProfileError -from scripts.benchmark_psychometric_routing import ( - _last_context_request, - _require_runtime, -) from scripts.benchmark_psychometric_heldout import ( _expected_brier, _paired_bootstrap_mean_ci, ) +from scripts.benchmark_psychometric_routing import ( + _last_context_request, + _require_runtime, +) def test_psychometric_benchmark_requires_python_312() -> None: @@ -722,7 +723,7 @@ def test_heldout_report_pairs_every_delta_with_its_interval(monkeypatch) -> None assert judge["severity_rmse"] == pytest.approx(0.018292059677437307) covariate = report["item_language_domain_effect_validation"] assert covariate["method"] == "multigroup_item_covariate" - assert covariate["sample_size"] == heldout_benchmark.DECLARED_ITEM_COVARIATE_SAMPLE_SIZE + assert covariate["sample_size"] == 1_200 assert covariate["seed"] == heldout_benchmark.ITEM_COVARIATE_SEED assert covariate["true_delta"] == -0.8 assert covariate["estimated_delta"] == pytest.approx(-0.7896498094289646) @@ -1014,7 +1015,8 @@ def run(callable_): """Capture a worker-thread exception for the joining test to assert.""" try: callable_() - except BaseException as error: # pragma: no cover - surfaced below + # Preserve every worker failure for the joining test's assertion. + except BaseException as error: # noqa: BLE001 # pragma: no cover errors.append(error) observe = threading.Thread( From 69d5056f14de6275faed20bc154da02628381128 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 18 Sep 2026 08:32:54 +0900 Subject: [PATCH 5/5] test(heldout): restore SimpleNamespace import for covariate fixtures Co-authored-by: Cursor --- tests/test_psychometric_benchmark_boundaries.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_psychometric_benchmark_boundaries.py b/tests/test_psychometric_benchmark_boundaries.py index f8e2eb675..a5b499b26 100644 --- a/tests/test_psychometric_benchmark_boundaries.py +++ b/tests/test_psychometric_benchmark_boundaries.py @@ -6,6 +6,7 @@ from pathlib import Path import runpy import sys +from types import SimpleNamespace import pytest