From adfa1506e9c146e3cf0c3d016d53788976a30612 Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:51:48 +0000 Subject: [PATCH 01/14] feat(res71): add qualification inventory and coverage contracts --- src/dynamislm/qualification/__init__.py | 57 + src/dynamislm/qualification/contracts.py | 258 ++++ src/dynamislm/qualification/inventory.py | 1334 +++++++++++++++++++++ src/dynamislm/qualification/references.py | 358 ++++++ tests/test_kernel.py | 3 + tests/test_res71_qualification.py | 175 +++ 6 files changed, 2185 insertions(+) create mode 100644 src/dynamislm/qualification/__init__.py create mode 100644 src/dynamislm/qualification/contracts.py create mode 100644 src/dynamislm/qualification/inventory.py create mode 100644 src/dynamislm/qualification/references.py create mode 100644 tests/test_res71_qualification.py diff --git a/src/dynamislm/qualification/__init__.py b/src/dynamislm/qualification/__init__.py new file mode 100644 index 0000000..6c15478 --- /dev/null +++ b/src/dynamislm/qualification/__init__.py @@ -0,0 +1,57 @@ +"""RES-71 deterministic scientific-engine qualification contracts.""" + +from dynamislm.qualification.contracts import ( + CoverageRow, + CoverageStatus, + GateComponentStatus, + OperationDisposition, + ReferenceCase, + ReferenceCaseStatus, + ReferenceValue, + RegisteredOperationInventoryEntry, + UnresolvedComputation, +) +from dynamislm.qualification.inventory import ( + RES71_REGISTRY_VERSION, + build_coverage_matrix, + build_registered_operation_inventory, + build_unresolved_computation_inventory, + discovered_registered_operation_ids, + validate_coverage_matrix, + validate_registered_operation_inventory, + validate_unresolved_computation_inventory, +) +from dynamislm.qualification.references import ( + RES71_REFERENCE_INTERFACE_VERSION, + get_reference_case, + get_reference_cases, + reference_case_digest, + reference_case_manifest, + validate_reference_cases, +) + +__all__ = [ + "RES71_REFERENCE_INTERFACE_VERSION", + "RES71_REGISTRY_VERSION", + "CoverageRow", + "CoverageStatus", + "GateComponentStatus", + "OperationDisposition", + "ReferenceCase", + "ReferenceCaseStatus", + "ReferenceValue", + "RegisteredOperationInventoryEntry", + "UnresolvedComputation", + "build_coverage_matrix", + "build_registered_operation_inventory", + "build_unresolved_computation_inventory", + "discovered_registered_operation_ids", + "get_reference_case", + "get_reference_cases", + "reference_case_digest", + "reference_case_manifest", + "validate_coverage_matrix", + "validate_reference_cases", + "validate_registered_operation_inventory", + "validate_unresolved_computation_inventory", +] diff --git a/src/dynamislm/qualification/contracts.py b/src/dynamislm/qualification/contracts.py new file mode 100644 index 0000000..75ba64c --- /dev/null +++ b/src/dynamislm/qualification/contracts.py @@ -0,0 +1,258 @@ +"""Immutable contracts used by the RES-71 qualification harness. + +These records describe authority that already exists in the scientific engine; +they do not add a numerical dispatch layer or accept caller-supplied formulas. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from enum import StrEnum + +from dynamislm.serialization import register_serializable_type + + +class OperationDisposition(StrEnum): + """Qualification status of a registered scientific operation.""" + + IMPLEMENTED = "IMPLEMENTED" + HISTORICAL_REPLAY_ONLY = "HISTORICAL_REPLAY_ONLY" + REPRESENT_BUT_DO_NOT_COMPUTE = "REPRESENT_BUT_DO_NOT_COMPUTE" + DEFERRED = "DEFERRED" + REJECTED = "REJECTED" + + +class CoverageStatus(StrEnum): + """Status of a V2 scientific-family coverage row.""" + + QUALIFIED = "QUALIFIED" + QUALIFIED_WITH_EXPLICIT_DEFERRED = "QUALIFIED_WITH_EXPLICIT_DEFERRED" + + +class GateComponentStatus(StrEnum): + """Machine-readable component result used by the RES-71 receipt.""" + + PASS = "PASS" + FAIL = "FAIL" + + +class ReferenceCaseStatus(StrEnum): + """Expected outcome class for a verifier-ready deterministic case.""" + + VALUE = "VALUE" + REFUSAL = "REFUSAL" + COMPARABILITY = "COMPARABILITY" + CLAIM_AUTHORITY = "CLAIM_AUTHORITY" + + +@register_serializable_type +@dataclass(frozen=True, slots=True) +class RegisteredOperationInventoryEntry: + """Complete qualification record for one registered-operation identity.""" + + operation_id: str + label: str + method_version: str + scientific_family: str + disposition: OperationDisposition + implementation: tuple[str, ...] + input_contract: str + output_contract: str + provenance_contract: str + refusal_path: tuple[str, ...] + test_coverage: tuple[str, ...] + authority_references: tuple[str, ...] + tolerance_contract: str + + def __post_init__(self) -> None: + for field_name in ( + "operation_id", + "label", + "method_version", + "scientific_family", + "input_contract", + "output_contract", + "provenance_contract", + "tolerance_contract", + ): + value = getattr(self, field_name) + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{field_name} must be non-empty") + if "@" not in self.operation_id: + raise ValueError("operation_id must include an explicit version") + if self.operation_id.rsplit("@", 1)[1] != self.method_version: + raise ValueError("method_version must match operation_id") + for field_name in ( + "implementation", + "refusal_path", + "test_coverage", + "authority_references", + ): + value = getattr(self, field_name) + if not isinstance(value, tuple) or any( + not isinstance(item, str) or not item.strip() for item in value + ): + raise ValueError(f"{field_name} must be a tuple of non-empty strings") + if ( + self.disposition + in { + OperationDisposition.IMPLEMENTED, + OperationDisposition.HISTORICAL_REPLAY_ONLY, + } + and not self.implementation + ): + raise ValueError("implemented operation must identify an implementation") + if self.disposition is not OperationDisposition.IMPLEMENTED and not self.refusal_path: + raise ValueError("non-computing operation must identify a refusal/representation path") + + +@register_serializable_type +@dataclass(frozen=True, slots=True) +class CoverageRow: + """One required V2 domain in the qualification coverage matrix.""" + + domain: str + status: CoverageStatus + authoritative_surfaces: tuple[str, ...] + registered_operations: tuple[str, ...] + unresolved_capabilities: tuple[str, ...] + provenance_boundary: str + comparability_boundary: str + claim_boundary: str + test_coverage: tuple[str, ...] + authority_references: tuple[str, ...] + + def __post_init__(self) -> None: + if not self.domain.strip(): + raise ValueError("coverage domain must be non-empty") + for field_name in ( + "authoritative_surfaces", + "registered_operations", + "unresolved_capabilities", + "test_coverage", + "authority_references", + ): + value = getattr(self, field_name) + if not isinstance(value, tuple) or any( + not isinstance(item, str) or not item.strip() for item in value + ): + raise ValueError(f"{field_name} must be a tuple of non-empty strings") + for field_name in ("provenance_boundary", "comparability_boundary", "claim_boundary"): + if not getattr(self, field_name).strip(): + raise ValueError(f"{field_name} must be non-empty") + + +@register_serializable_type +@dataclass(frozen=True, slots=True) +class UnresolvedComputation: + """Explicit non-authoritative representation of an unsupported computation.""" + + capability: str + registered_operation_id: str | None + disposition: OperationDisposition + reason: str + refusal_path: tuple[str, ...] + expected_refusal_class: str + safe_description: str + test_coverage: tuple[str, ...] + authority_references: tuple[str, ...] + + def __post_init__(self) -> None: + for field_name in ( + "capability", + "reason", + "expected_refusal_class", + "safe_description", + ): + if not getattr(self, field_name).strip(): + raise ValueError(f"{field_name} must be non-empty") + if not isinstance(self.refusal_path, tuple) or not self.refusal_path: + raise ValueError("refusal_path must be non-empty") + for field_name in ("refusal_path", "test_coverage", "authority_references"): + value = getattr(self, field_name) + if any(not isinstance(item, str) or not item.strip() for item in value): + raise ValueError(f"{field_name} must contain non-empty strings") + if self.disposition is OperationDisposition.IMPLEMENTED: + raise ValueError("unresolved computation cannot be implemented") + + +@register_serializable_type +@dataclass(frozen=True, slots=True) +class ReferenceValue: + """A scalar expected value in a verifier-ready reference case.""" + + name: str + value: bool | float | int | str + unit: str | None = None + + def __post_init__(self) -> None: + if not self.name.strip(): + raise ValueError("reference value name must be non-empty") + if isinstance(self.value, float) and not __import__("math").isfinite(self.value): + raise ValueError("reference values must be finite") + if self.unit is not None and not self.unit.strip(): + raise ValueError("reference value unit must be non-empty when supplied") + + +@register_serializable_type +@dataclass(frozen=True, slots=True) +class ReferenceCase: + """Machine-readable case contract for later verifier construction.""" + + case_id: str + case_version: str + family: str + operation_id: str | None + status: ReferenceCaseStatus + synthetic_input: tuple[ReferenceValue, ...] + expected_values: tuple[ReferenceValue, ...] + expected_refusal_class: str | None + expected_reason_codes: tuple[str, ...] + expected_comparability_state: str | None + expected_claim_level: str | None + tolerance_absolute: float | None + tolerance_relative: float | None + required_provenance_fields: tuple[str, ...] + authority_references: tuple[str, ...] + + def __post_init__(self) -> None: + for field_name in ("case_id", "case_version", "family"): + if not getattr(self, field_name).strip(): + raise ValueError(f"{field_name} must be non-empty") + if not isinstance(self.synthetic_input, tuple) or not isinstance( + self.expected_values, tuple + ): + raise ValueError("reference values must be immutable tuples") + if any(not isinstance(item, ReferenceValue) for item in self.synthetic_input): + raise ValueError("synthetic_input must contain ReferenceValue values") + if any(not isinstance(item, ReferenceValue) for item in self.expected_values): + raise ValueError("expected_values must contain ReferenceValue values") + if self.status is ReferenceCaseStatus.VALUE: + if not self.operation_id or not self.expected_values: + raise ValueError("value case requires operation and expected values") + if self.expected_refusal_class or self.expected_comparability_state: + raise ValueError("value case cannot carry refusal/comparability state") + if self.status is ReferenceCaseStatus.REFUSAL: + if not self.expected_refusal_class or not self.expected_reason_codes: + raise ValueError("refusal case requires class and reason codes") + if ( + self.status is ReferenceCaseStatus.COMPARABILITY + and not self.expected_comparability_state + ): + raise ValueError("comparability case requires a state") + if self.status is ReferenceCaseStatus.CLAIM_AUTHORITY and not self.expected_claim_level: + raise ValueError("claim-authority case requires a claim level") + for name in ("tolerance_absolute", "tolerance_relative"): + value = getattr(self, name) + if value is not None and (value < 0 or not __import__("math").isfinite(value)): + raise ValueError(f"{name} must be finite and non-negative") + for field_name in ( + "expected_reason_codes", + "required_provenance_fields", + "authority_references", + ): + value = getattr(self, field_name) + if not isinstance(value, tuple) or any( + not isinstance(item, str) or not item.strip() for item in value + ): + raise ValueError(f"{field_name} must be a tuple of non-empty strings") diff --git a/src/dynamislm/qualification/inventory.py b/src/dynamislm/qualification/inventory.py new file mode 100644 index 0000000..d2557a2 --- /dev/null +++ b/src/dynamislm/qualification/inventory.py @@ -0,0 +1,1334 @@ +# ruff: noqa: E501 + +"""RES-71 operation inventory and V2 coverage matrix. + +The inventory is deliberately checked against the live package registry. A +new ``registered-operation`` reference therefore fails qualification until it +has an explicit contract entry here. +""" + +from __future__ import annotations + +import importlib +import pkgutil +from dataclasses import dataclass +from pathlib import Path +from types import ModuleType + +import dynamislm +from dynamislm.measurement.identity import RegistryReference +from dynamislm.qualification.contracts import ( + CoverageRow, + CoverageStatus, + GateComponentStatus, + OperationDisposition, + RegisteredOperationInventoryEntry, + UnresolvedComputation, +) + +RES71_REGISTRY_VERSION = "1.0.0" + + +@dataclass(frozen=True, slots=True) +class _OperationMetadata: + key: str + version: str + family: str + disposition: OperationDisposition + implementation: tuple[str, ...] + input_contract: str + output_contract: str + provenance_contract: str + refusal_path: tuple[str, ...] + test_coverage: tuple[str, ...] + authority_references: tuple[str, ...] + tolerance_contract: str + reason: str = "" + safe_description: str = "" + + @property + def lookup_key(self) -> str: + return f"{self.key}@{self.version}" + + +def _operation_id(key: str, version: str = "1.0.0") -> str: + return f"dynamislm:registered-operation:{key}@{version}" + + +_INPUT_CONTRACTS = { + "population": "Typed PopulationIdentity or CanonicalSource with V2 population/source clauses; no free-text eligibility override.", + "football": "Typed FootballSession and explicit target MatchSession plus an IANA timezone when a relative label is requested.", + "ingestion": "Qualified source manifest, immutable artifact identity, schema/mapping version and canonical-record evidence.", + "external-load": "Finite values with registered UnitReference, or timestamped velocity samples with resolved threshold and method identity.", + "cmj": "Typed CMJ force/event/phase/mechanics inputs with registered source artifact, timebase, estimator and support boundaries.", + "strength": "Typed IMTP force series/onset or VBT velocity evidence with explicit phase, load, device and trial-selection identity.", + "field-testing": "Typed source-qualified sprint/505/RSA/IFT observations or exact velocity-series support; no bare label arithmetic.", + "explosive-tests": "Typed DJ/BPT/MBT source evidence, event/coordinate support, protocol and qualification identity.", + "longitudinal-statistics": "RES-62-backed StatisticalSupport with exact observation identities, scale semantics, design and missingness policy.", + "comparability": "Typed observations plus claim-relative identity/context references and canonical RES-70 authority; caller verdicts are not inputs.", + "analysis": "Typed AnalysisAuthorizationRequest with exact support, level, comparability and evidence prerequisites.", + "claims": "Typed ClaimIntent with exact observations and validated upstream analysis/comparability/evidence authority.", +} + +_OUTPUT_CONTRACTS = { + "population": "Typed population/source decision with status, clause-level reasons, method/version and refusal class when unresolved.", + "football": "Immutable MatchDayRelativeLabel retaining target-match/session/context references and registered derivation identity.", + "ingestion": "Typed canonical record, manifest, analysis input, or promotion decision with replayable artifact lineage.", + "external-load": "Typed scalar/threshold result or RefusalResult; output identity names operation, units, threshold and aggregation semantics.", + "cmj": "Typed ScientificMeasurementObservation/result with estimator or mechanics identity, units, quality and refusal-safe output.", + "strength": "Typed ScientificMeasurementObservation/result or model/aggregation result with value origin, method and source lineage.", + "field-testing": "Typed field-testing observation/result with protocol, timing/support, source qualification and refusal state.", + "explosive-tests": "Typed family-specific metric result or provider observation; generic/unqualified power remains a refusal.", + "longitudinal-statistics": "Typed StatisticalResult/StatisticalNonComputable with estimand, support, uncertainty/design and provenance.", + "comparability": "Canonical ComparabilityResult/BridgeExecutionResult with state, dimension findings, conditions and authority hashes.", + "analysis": "AnalysisAuthorization or RefusalResult with operation reference, support hashes, estimand level and claim floor.", + "claims": "ClaimAuthorityResult with allowed levels, blocked claims, refusal reasons and exact upstream authority references.", +} + +_PROVENANCE_CONTRACTS = { + "population": "Decision references preserve source-evidence bindings, registry version and clause-level applicability; ambiguity is not relabelled.", + "football": "Derivation preserves session, target match, athlete/team/season context and calendar-time inputs.", + "ingestion": "Source bytes, acquisition, qualification/mapping version, canonical artifact and replay digest remain linked and immutable.", + "external-load": "Source artifact/acquisition, provider/device/modality, processing/threshold/aggregation identity and output run are retained.", + "cmj": "Source observation/artifact, acquisition/timebase, event/phase/mechanics dependencies, processing run and output identity are retained.", + "strength": "Force/velocity source series, phase/onset/load/trial support, processing run and method parameters remain recoverable.", + "field-testing": "Source artifact, timing/velocity support, protocol, qualification evidence, context and derived processing run are retained.", + "explosive-tests": "Source artifact/series, protocol/event/coordinate evidence, qualification, method support and output run remain linked.", + "longitudinal-statistics": "All source observation IDs, support ordering, analysis/design authority, window and calculation-changing parameters are retained.", + "comparability": "Exact observation hashes, request hash, canonical rule/bridge registry hash and source applicability are validated.", + "analysis": "Support/observation hashes, capability registry hash, requested level, prerequisites and evidence applicability are bound.", + "claims": "Claim intent, exact observations and validated analysis/comparability/evidence/decision prerequisites are rechecked at authorization.", +} + +_TESTS = { + "population": ("tests/test_population.py", "tests/test_ingestion.py"), + "football": ("tests/test_football.py",), + "ingestion": ("tests/test_ingestion.py", "tests/test_res63_operational_authority.py"), + "external-load": ("tests/test_external_load.py",), + "cmj": ( + "tests/test_cmj.py", + "tests/test_cmj_jump_height.py", + "tests/test_cmj_metrics.py", + "tests/test_cmj_phases.py", + "tests/test_cmj_session.py", + ), + "strength": ("tests/test_strength.py",), + "field-testing": ("tests/test_field_testing.py",), + "explosive-tests": ("tests/test_explosive_test_families.py",), + "longitudinal-statistics": ( + "tests/test_longitudinal.py", + "tests/test_longitudinal_statistics.py", + ), + "comparability": ( + "tests/test_res70_comparability.py", + "tests/test_res70_bridges.py", + "tests/test_res70_adversarial.py", + ), + "analysis": ("tests/test_res70_analysis_capability.py", "tests/test_res70_models.py"), + "claims": ("tests/test_res70_claim_authority.py", "tests/test_res70_evidence_and_levels.py"), +} + +_AUTHORITY = { + "population": ("docs/architecture/SCIENTIFIC_CONSTITUTION_V2.md", "RES-60"), + "football": ("docs/decisions/RES61-DR-001-football-world-ontology.md", "RES-61"), + "ingestion": ("docs/decisions/RES63-DR-001-canonical-dataset-ingestion.md", "RES-63"), + "external-load": ("docs/decisions/RES64-DR-001-external-load-scientific-identity.md", "RES-64"), + "cmj": ("docs/decisions/RES65-DR-001-cmj-football-metric-completion.md", "RES-65"), + "strength": ("docs/decisions/RES66-DR-001-strength-imtp-vbt-scientific-engine.md", "RES-66"), + "field-testing": ("docs/decisions/RES67-DR-001-field-testing-scientific-engine.md", "RES-67"), + "explosive-tests": ("docs/decisions/RES68-DR-001-explosive-test-family-closure.md", "RES-68"), + "longitudinal-statistics": ( + "docs/decisions/RES69-DR-001-longitudinal-reliability-uncertainty.md", + "RES-69", + ), + "comparability": ( + "docs/decisions/RES70-DR-001-cross-source-comparability-analysis-claim-authority.md", + "RES-70", + ), + "analysis": ( + "docs/decisions/RES70-DR-001-cross-source-comparability-analysis-claim-authority.md", + "RES-70", + ), + "claims": ( + "docs/decisions/RES70-DR-001-cross-source-comparability-analysis-claim-authority.md", + "RES-70", + ), +} + + +def _add( + metadata: dict[str, _OperationMetadata], + keys: tuple[str, ...], + *, + family: str, + implementation: tuple[str, ...] = (), + disposition: OperationDisposition = OperationDisposition.IMPLEMENTED, + versions: dict[str, str] | None = None, + refusal_path: tuple[str, ...] = ("dynamislm.refusal.models:RefusalResult",), + reason: str = "", + safe_description: str = "", + tolerance_contract: str = "Finite deterministic output; compare canonical serialized values with the operation-specific tolerance stated by its method contract.", +) -> None: + versions = versions or {} + for key in keys: + version = versions.get(key, "1.0.0") + lookup_key = f"{key}@{version}" + if lookup_key in metadata: + raise ValueError(f"duplicate RES-71 metadata entry: {lookup_key}") + metadata[lookup_key] = _OperationMetadata( + key=key, + version=version, + family=family, + disposition=disposition, + implementation=implementation, + input_contract=_INPUT_CONTRACTS[family], + output_contract=_OUTPUT_CONTRACTS[family], + provenance_contract=_PROVENANCE_CONTRACTS[family], + refusal_path=refusal_path, + test_coverage=_TESTS[family], + authority_references=_AUTHORITY[family], + tolerance_contract=tolerance_contract, + reason=reason, + safe_description=safe_description, + ) + + +def _metadata() -> dict[str, _OperationMetadata]: + """Return the manually reviewed operation-to-contract map.""" + + metadata: dict[str, _OperationMetadata] = {} + _add( + metadata, + ( + "canonical-empirical-source-qualification", + "canonical-football-population-qualification", + ), + family="population", + implementation=("dynamislm.population.qualification:qualify_canonical_population",), + ) + # The two population operations share the family contract but have distinct callables. + source_meta = metadata["canonical-empirical-source-qualification@1.0.0"] + metadata["canonical-empirical-source-qualification@1.0.0"] = _OperationMetadata( + key=source_meta.key, + version=source_meta.version, + family=source_meta.family, + disposition=source_meta.disposition, + implementation=("dynamislm.population.qualification:qualify_canonical_source",), + input_contract=source_meta.input_contract, + output_contract=source_meta.output_contract, + provenance_contract=source_meta.provenance_contract, + refusal_path=source_meta.refusal_path, + test_coverage=source_meta.test_coverage, + authority_references=source_meta.authority_references, + tolerance_contract=source_meta.tolerance_contract, + ) + _add( + metadata, + ("match-day-relative-label",), + family="football", + implementation=("dynamislm.football.context:derive_match_day_relative_label",), + ) + _add( + metadata, + ("unit-normalization",), + family="external-load", + implementation=("dynamislm.external_load.metrics:convert_external_load_value",), + ) + _add( + metadata, + ("duration-normalized-distance",), + family="external-load", + implementation=("dynamislm.external_load.metrics:duration_normalized_distance",), + ) + _add( + metadata, + ("velocity-threshold-summary",), + family="external-load", + implementation=("dynamislm.external_load.metrics:summarize_velocity_threshold",), + ) + _add( + metadata, + ( + "cmj-bilateral-total-vertical-force-sum-v1", + "cmj-net-vertical-force-from-total-force-and-system-weight-v1", + "cmj-net-vertical-impulse-v1", + "cmj-supported-system-com-vertical-acceleration-v1", + "cmj-supported-system-com-vertical-velocity-v1", + "cmj-supported-system-com-relative-vertical-displacement-v1", + "cmj-physical-system-mass-from-weight-v1", + "cmj-standard-gravity-mass-equivalent-from-weight-v1", + "cmj-system-weight-v1", + ), + family="cmj", + implementation=("dynamislm.measurement.cmj:validated-family-operation",), + ) + cmj_overrides = { + "cmj-bilateral-total-vertical-force-sum-v1": "dynamislm.measurement.cmj.weighing:construct_total_supported_vertical_force", + "cmj-net-vertical-force-from-total-force-and-system-weight-v1": "dynamislm.measurement.cmj.mechanics:calculate_net_vertical_force", + "cmj-net-vertical-impulse-v1": "dynamislm.measurement.cmj.mechanics:integrate_net_vertical_impulse", + "cmj-supported-system-com-vertical-acceleration-v1": "dynamislm.measurement.cmj.mechanics:derive_supported_system_com_acceleration", + "cmj-supported-system-com-vertical-velocity-v1": "dynamislm.measurement.cmj.mechanics:integrate_supported_system_com_velocity", + "cmj-supported-system-com-relative-vertical-displacement-v1": "dynamislm.measurement.cmj.mechanics:integrate_supported_system_com_relative_vertical_displacement", + "cmj-physical-system-mass-from-weight-v1": "dynamislm.measurement.cmj.weighing:derive_physical_system_mass", + "cmj-standard-gravity-mass-equivalent-from-weight-v1": "dynamislm.measurement.cmj.weighing:derive_standard_gravity_mass_equivalent", + "cmj-system-weight-v1": "dynamislm.measurement.cmj.weighing:estimate_system_weight", + } + for key, implementation in cmj_overrides.items(): + old = metadata[f"{key}@1.0.0"] + metadata[f"{key}@1.0.0"] = _replace_implementation(old, implementation) + _add( + metadata, + ("cmj-flight-time-ballistic-jump-height-v1",), + family="cmj", + disposition=OperationDisposition.HISTORICAL_REPLAY_ONLY, + versions={"cmj-flight-time-ballistic-jump-height-v1": "1.0.0"}, + implementation=( + "dynamislm.measurement.cmj.jump_height:CMJ_FLIGHT_TIME_JUMP_HEIGHT_METHOD_V1", + ), + refusal_path=("dynamislm.measurement.cmj.jump_height:CMJJumpHeightResult",), + reason="Retained for historical Serialization V3 replay; new flight-time computation uses the distinct V2 method.", + safe_description="Historical V1 results may be decoded and compared by identity; V1 is not minted as the current method.", + ) + _add( + metadata, + ("cmj-flight-time-ballistic-jump-height-v2",), + family="cmj", + versions={"cmj-flight-time-ballistic-jump-height-v2": "2.0.0"}, + implementation=("dynamislm.measurement.cmj.jump_height:estimate_flight_time_jump_height",), + ) + _add( + metadata, + ("cmj-left-right-force-asymmetry-right-minus-left-over-left-v1",), + family="cmj", + implementation=("dynamislm.measurement.cmj.metrics:calculate_cmj_force_asymmetry",), + ) + _add( + metadata, + ("cmj-phase-duration-v1",), + family="cmj", + implementation=("dynamislm.measurement.cmj.phases:calculate_cmj_phase_duration",), + ) + _add( + metadata, + ("cmj-phase-net-vertical-impulse-v1",), + family="cmj", + implementation=( + "dynamislm.measurement.cmj.phases:calculate_cmj_phase_net_vertical_impulse", + ), + ) + _add( + metadata, + ("cmj-phase-relative-displacement-change-v1",), + family="cmj", + implementation=( + "dynamislm.measurement.cmj.phases:calculate_cmj_phase_relative_displacement_change", + ), + ) + _add( + metadata, + ( + "cmj-qualified-takeoff-velocity-ballistic-apex-rise-v1", + "cmj-rsi-modified-flight-time-jump-height-over-time-to-takeoff-v1", + "cmj-rsi-modified-takeoff-velocity-jump-height-over-time-to-takeoff-v1", + "cmj-takeoff-velocity-scalar-projection-v1", + ), + family="cmj", + implementation=("dynamislm.measurement.cmj:validated-family-operation",), + ) + cmj_metric_overrides = { + "cmj-qualified-takeoff-velocity-ballistic-apex-rise-v1": "dynamislm.measurement.cmj.jump_height:estimate_takeoff_velocity_jump_height", + "cmj-rsi-modified-flight-time-jump-height-over-time-to-takeoff-v1": "dynamislm.measurement.cmj.metrics:calculate_cmj_rsi_mod", + "cmj-rsi-modified-takeoff-velocity-jump-height-over-time-to-takeoff-v1": "dynamislm.measurement.cmj.metrics:calculate_cmj_rsi_mod", + "cmj-takeoff-velocity-scalar-projection-v1": "dynamislm.measurement.cmj.metrics:calculate_cmj_takeoff_velocity", + } + for key, implementation in cmj_metric_overrides.items(): + old = metadata[f"{key}@1.0.0"] + metadata[f"{key}@1.0.0"] = _replace_implementation(old, implementation) + _add( + metadata, + ( + "cmj-sample-peak-total-supported-vertical-force-v1", + "cmj-time-weighted-mean-total-supported-vertical-force-v1", + ), + family="cmj", + implementation=("dynamislm.measurement.cmj.metrics:calculate_cmj_force_metric",), + ) + _add( + metadata, + ( + "cmj-sampled-signed-power-extremum-v1", + "cmj-time-weighted-signed-power-mean-v1", + "cmj-total-supported-force-times-supported-system-com-velocity-v1", + ), + family="cmj", + implementation=("dynamislm.measurement.cmj.metrics:calculate_cmj_power",), + ) + _add( + metadata, + ("cmj-session-selection-aggregation-v1",), + family="cmj", + implementation=("dynamislm.measurement.cmj.session:aggregate_cmj_session",), + ) + _add( + metadata, + ( + "drop-jump-flight-time-jump-height-v1", + "drop-jump-ground-contact-time-v1", + "drop-jump-rebound-flight-time-v1", + "drop-jump-rsi-jh-ct-v1", + "drop-jump-rsr-ft-ct-v1", + ), + family="explosive-tests", + implementation=("dynamislm.measurement.drop_jump:validated-family-operation",), + ) + dj_overrides = { + "drop-jump-flight-time-jump-height-v1": "dynamislm.measurement.drop_jump.metrics:calculate_drop_jump_flight_time_jump_height", + "drop-jump-ground-contact-time-v1": "dynamislm.measurement.drop_jump.metrics:calculate_drop_jump_contact_time", + "drop-jump-rebound-flight-time-v1": "dynamislm.measurement.drop_jump.metrics:calculate_drop_jump_rebound_flight_time", + "drop-jump-rsi-jh-ct-v1": "dynamislm.measurement.drop_jump.metrics:calculate_drop_jump_rsi_jh_ct", + "drop-jump-rsr-ft-ct-v1": "dynamislm.measurement.drop_jump.metrics:calculate_drop_jump_rsr_ft_ct", + } + for key, implementation in dj_overrides.items(): + old = metadata[f"{key}@1.0.0"] + metadata[f"{key}@1.0.0"] = _replace_implementation(old, implementation) + _add( + metadata, + ("bench-press-throw-dynamislm-time-weighted-mean-bar-velocity-v1",), + family="explosive-tests", + implementation=( + "dynamislm.measurement.bench_press_throw.metrics:calculate_bpt_dynamislm_time_weighted_mean_bar_velocity", + ), + ) + _add( + metadata, + ("bench-press-throw-sampled-maximum-bar-velocity-v1",), + family="explosive-tests", + implementation=( + "dynamislm.measurement.bench_press_throw.metrics:calculate_bpt_sampled_maximum_bar_velocity", + ), + ) + _add( + metadata, + ("bench-press-throw-mean-propulsive-velocity-v1",), + family="explosive-tests", + disposition=OperationDisposition.REPRESENT_BUT_DO_NOT_COMPUTE, + refusal_path=( + "dynamislm.measurement.bench_press_throw.metrics:calculate_bpt_mean_propulsive_velocity", + ), + reason="Metric support and acceleration/gravity boundary are not registered for generic BPT MPV computation.", + safe_description="Provider-reported or represented MPV remains origin-qualified; no DynamisLM MPV number is emitted.", + ) + _add( + metadata, + ( + "medicine-ball-throw-distance-from-registered-coordinates-v1", + "medicine-ball-throw-instrumented-release-velocity-v1", + ), + family="explosive-tests", + implementation=("dynamislm.measurement.medicine_ball_throw:validated-family-operation",), + ) + mbt_overrides = { + "medicine-ball-throw-distance-from-registered-coordinates-v1": "dynamislm.measurement.medicine_ball_throw.metrics:calculate_mbt_throw_distance", + "medicine-ball-throw-instrumented-release-velocity-v1": "dynamislm.measurement.medicine_ball_throw.metrics:calculate_mbt_instrumented_release_velocity", + } + for key, implementation in mbt_overrides.items(): + old = metadata[f"{key}@1.0.0"] + metadata[f"{key}@1.0.0"] = _replace_implementation(old, implementation) + _add( + metadata, + ( + "imtp-baseline-force-statistics-v1", + "imtp-endpoint-average-rfd-v1", + "imtp-force-at-registered-time-v1", + "imtp-force-body-mass-normalization-v1", + "imtp-force-impulse-v1", + "imtp-sampled-peak-force-v1", + "imtp-trial-selection-aggregation-v1", + ), + family="strength", + implementation=("dynamislm.measurement.strength.imtp:calculate_imtp_metric",), + ) + imtp_overrides = { + "imtp-baseline-force-statistics-v1": "dynamislm.measurement.strength.imtp:build_imtp_baseline", + "imtp-endpoint-average-rfd-v1": "dynamislm.measurement.strength.imtp:calculate_imtp_rfd", + "imtp-force-at-registered-time-v1": "dynamislm.measurement.strength.imtp:calculate_imtp_force_at_time", + "imtp-force-body-mass-normalization-v1": "dynamislm.measurement.strength.imtp:normalize_imtp_force", + "imtp-force-impulse-v1": "dynamislm.measurement.strength.imtp:calculate_imtp_impulse", + "imtp-sampled-peak-force-v1": "dynamislm.measurement.strength.imtp:calculate_imtp_peak_force", + "imtp-trial-selection-aggregation-v1": "dynamislm.measurement.strength.imtp:aggregate_imtp_metric_results", + } + for key, implementation in imtp_overrides.items(): + old = metadata[f"{key}@1.0.0"] + metadata[f"{key}@1.0.0"] = _replace_implementation(old, implementation) + _add( + metadata, + ("direct-one-repetition-maximum-assessment-v1", "measured-one-repetition-maximum-v1"), + family="strength", + implementation=("dynamislm.measurement.strength.vbt:create_measured_1rm",), + ) + _add( + metadata, + ("estimated-one-repetition-maximum-from-linear-load-velocity-v1",), + family="strength", + disposition=OperationDisposition.DEFERRED, + refusal_path=("dynamislm.measurement.strength.vbt:estimate_1rm_from_load_velocity_model",), + reason="Current evidence does not authorize terminal-velocity applicability across devices/calibration designs.", + safe_description="The individual load-velocity model remains describable; no numeric estimated 1RM is emitted.", + ) + _add( + metadata, + ("mean-concentric-velocity-v1",), + family="strength", + implementation=( + "dynamislm.measurement.strength.vbt:calculate_vbt_mean_concentric_velocity", + ), + ) + _add( + metadata, + ("mean-propulsive-velocity-v1",), + family="strength", + disposition=OperationDisposition.REPRESENT_BUT_DO_NOT_COMPUTE, + refusal_path=("dynamislm.measurement.strength.vbt:calculate_vbt_mean_propulsive_velocity",), + reason="Acceleration, gravity, filtering, sampling and propulsive-boundary authority are not frozen.", + safe_description="Concentric velocity is not relabelled as mean propulsive velocity.", + ) + _add( + metadata, + ("sampled-peak-concentric-velocity-v1",), + family="strength", + implementation=("dynamislm.measurement.strength.vbt:calculate_vbt_peak_velocity",), + ) + _add( + metadata, + ("within-set-velocity-loss-v1",), + family="strength", + implementation=("dynamislm.measurement.strength.vbt:calculate_vbt_velocity_loss",), + ) + _add( + metadata, + ("505-asymmetry-v1",), + family="field-testing", + disposition=OperationDisposition.REPRESENT_BUT_DO_NOT_COMPUTE, + refusal_path=("dynamislm.measurement.field_testing.cod:refuse_505_asymmetry",), + reason="No single denominator, direction and sign convention is registered for 505 asymmetry.", + safe_description="Left and right 505 results remain separate observations.", + ) + _add( + metadata, + ( + "505-cod-deficit-mean-505-minus-mean-10m-v1", + "standard-505-mean-of-three-v1", + ), + family="field-testing", + implementation=("dynamislm.measurement.field_testing.cod:aggregate_standard_505_trials",), + ) + _add( + metadata, + ("30-15-ift-last-completed-stage-v1",), + family="field-testing", + implementation=("dynamislm.measurement.field_testing.ift:calculate_vift",), + ) + _add( + metadata, + ( + "linear-sprint-mean-0-10m-reference-v1", + "sprint-interval-split-v1", + "sprint-segment-average-velocity-v1", + "sampled-maximum-sprint-velocity-v1", + ), + family="field-testing", + implementation=("dynamislm.measurement.field_testing.sprint:validated-family-operation",), + ) + sprint_overrides = { + "linear-sprint-mean-0-10m-reference-v1": "dynamislm.measurement.field_testing.sprint:aggregate_linear_sprint_30m_reference", + "sprint-interval-split-v1": "dynamislm.measurement.field_testing.sprint:derive_interval_split", + "sprint-segment-average-velocity-v1": "dynamislm.measurement.field_testing.sprint:derive_segment_average_velocity", + "sampled-maximum-sprint-velocity-v1": "dynamislm.measurement.field_testing.sprint:sampled_maximum_velocity", + } + for key, implementation in sprint_overrides.items(): + old = metadata[f"{key}@1.0.0"] + metadata[f"{key}@1.0.0"] = _replace_implementation(old, implementation) + _add( + metadata, + ("sustained-maximum-sprint-velocity-v1",), + family="field-testing", + disposition=OperationDisposition.DEFERRED, + refusal_path=( + "dynamislm.measurement.field_testing.sprint:refuse_sampled_maximum_as_sustained_maximum", + ), + reason="A dwell/sustain estimator and evidence are not registered; V1 is sampled maximum only.", + safe_description="Sampled maximum velocity remains separately describable and is not relabelled as sustained maximum.", + ) + _add( + metadata, + ( + "rsa-best-time-v1", + "rsa-complete-series-aggregation-v1", + "rsa-mean-time-v1", + "rsa-percent-decrement-s-dec-v1", + "rsa-total-time-v1", + ), + family="field-testing", + implementation=("dynamislm.measurement.field_testing.rsa:aggregate_rsa",), + ) + _add( + metadata, + ( + "longitudinal-absolute-change-v1", + "longitudinal-log-ratio-v1", + "longitudinal-reference-window-deviation-v1", + "longitudinal-relative-change-v1", + "longitudinal-window-descriptives-v1", + "longitudinal-within-athlete-sample-sd-v1", + ), + family="longitudinal-statistics", + implementation=("dynamislm.longitudinal.statistics:validated-family-operation",), + ) + stats_overrides = { + "longitudinal-absolute-change-v1": "dynamislm.longitudinal.statistics.descriptive:calculate_absolute_change", + "longitudinal-log-ratio-v1": "dynamislm.longitudinal.statistics.descriptive:calculate_log_ratio_change", + "longitudinal-reference-window-deviation-v1": "dynamislm.longitudinal.statistics.descriptive:calculate_reference_window_deviation", + "longitudinal-relative-change-v1": "dynamislm.longitudinal.statistics.descriptive:calculate_relative_change", + "longitudinal-window-descriptives-v1": "dynamislm.longitudinal.statistics.descriptive:calculate_window_descriptives", + "longitudinal-within-athlete-sample-sd-v1": "dynamislm.longitudinal.statistics.descriptive:calculate_descriptive_within_athlete_sd", + } + for key, implementation in stats_overrides.items(): + old = metadata[f"{key}@1.0.0"] + metadata[f"{key}@1.0.0"] = _replace_implementation(old, implementation) + _add( + metadata, + ("longitudinal-descriptive-ols-v1",), + family="longitudinal-statistics", + implementation=("dynamislm.longitudinal.statistics.descriptive:calculate_descriptive_ols",), + ) + _add( + metadata, + ( + "two-replicate-log-multiplicative-error-v1", + "two-replicate-pooled-raw-relative-error-v1", + "two-replicate-within-subject-random-error-v1", + ), + family="longitudinal-statistics", + implementation=( + "dynamislm.longitudinal.statistics.reliability:validated-family-operation", + ), + ) + reliability_overrides = { + "two-replicate-log-multiplicative-error-v1": "dynamislm.longitudinal.statistics.reliability:calculate_log_scale_typical_error", + "two-replicate-pooled-raw-relative-error-v1": "dynamislm.longitudinal.statistics.reliability:calculate_raw_relative_error", + "two-replicate-within-subject-random-error-v1": "dynamislm.longitudinal.statistics.reliability:calculate_two_replicate_random_error", + } + for key, implementation in reliability_overrides.items(): + old = metadata[f"{key}@1.0.0"] + metadata[f"{key}@1.0.0"] = _replace_implementation(old, implementation) + _add( + metadata, + ("method-comparison-ba-summary-v1",), + family="longitudinal-statistics", + implementation=( + "dynamislm.longitudinal.statistics.agreement:calculate_bland_altman_summary", + ), + ) + _add( + metadata, + ("method-comparison-design-authority-v1",), + family="longitudinal-statistics", + implementation=( + "dynamislm.longitudinal.statistics.agreement:build_method_comparison_design_authority", + ), + ) + _add( + metadata, + ("reliability-design-authority-v1",), + family="longitudinal-statistics", + implementation=( + "dynamislm.longitudinal.statistics.reliability:build_reliability_design_authority", + ), + ) + _add( + metadata, + ("reliability-assumption-assessment-v1",), + family="longitudinal-statistics", + implementation=( + "dynamislm.longitudinal.statistics.reliability:build_reliability_assumption_assessment", + ), + ) + _add( + metadata, + ("classical-bland-altman-limits-v1", "sem-from-registered-icc", "mdc-sdc", "icc-variants"), + family="longitudinal-statistics", + disposition=OperationDisposition.REPRESENT_BUT_DO_NOT_COMPUTE, + refusal_path=( + "dynamislm.longitudinal.statistics.reliability:refuse_unimplemented_reliability_operation", + ), + reason="The named reliability/agreement method is represented but lacks a registered V1 estimand/design implementation.", + safe_description="The underlying observations and method label remain describable without placeholder arithmetic.", + ) + _add( + metadata, + ( + "log-ba", + "repeated-measures-ba", + "confidence-intervals", + "covariance-propagation", + "repeated-measures-correlation", + "mixed-effects", + ), + family="longitudinal-statistics", + disposition=OperationDisposition.DEFERRED, + refusal_path=( + "dynamislm.longitudinal.statistics.reliability:refuse_unimplemented_reliability_operation", + ), + reason="The method remains outside the sealed V1 numerical surface until its estimand, design and uncertainty contract are registered.", + safe_description="No numeric result is emitted; a later mission may register the method with explicit prerequisites.", + ) + _add( + metadata, + ("generic-sem", "generic-meaningful-change", "readiness-fatigue-injury-interpretation"), + family="longitudinal-statistics", + disposition=OperationDisposition.REJECTED, + refusal_path=( + "dynamislm.longitudinal.statistics.reliability:refuse_unimplemented_reliability_operation", + ), + reason="The generic claim is outside current scientific authority and has no registered estimand or decision criterion.", + safe_description="Observed values and registered descriptive/error results remain independently describable.", + ) + _add( + metadata, + ( + "longitudinal-athlete-performance-record", + "multi-source-analysis-input", + "res62-multi-source-manifest", + ), + family="ingestion", + implementation=("dynamislm.longitudinal.record:build_longitudinal_record",), + ) + record_overrides = { + "longitudinal-athlete-performance-record": "dynamislm.longitudinal.record:build_longitudinal_record", + "multi-source-analysis-input": "dynamislm.longitudinal.record:build_multi_source_analysis_input", + "res62-multi-source-manifest": "dynamislm.longitudinal.record:build_longitudinal_source_manifest", + } + for key, implementation in record_overrides.items(): + old = metadata[f"{key}@1.0.0"] + metadata[f"{key}@1.0.0"] = _replace_implementation(old, implementation) + _add( + metadata, + ("res70-registered-affine-transformation-v1",), + family="comparability", + implementation=("dynamislm.comparability.res70_authority:execute_registered_bridge",), + ) + return metadata + + +def _replace_implementation( + metadata: _OperationMetadata, implementation: str +) -> _OperationMetadata: + return _OperationMetadata( + key=metadata.key, + version=metadata.version, + family=metadata.family, + disposition=metadata.disposition, + implementation=(implementation,), + input_contract=metadata.input_contract, + output_contract=metadata.output_contract, + provenance_contract=metadata.provenance_contract, + refusal_path=metadata.refusal_path, + test_coverage=metadata.test_coverage, + authority_references=metadata.authority_references, + tolerance_contract=metadata.tolerance_contract, + reason=metadata.reason, + safe_description=metadata.safe_description, + ) + + +def _discover_modules() -> tuple[ModuleType, ...]: + modules = [dynamislm] + for info in pkgutil.walk_packages(dynamislm.__path__, dynamislm.__name__ + "."): + if info.name.startswith("dynamislm.qualification"): + continue + modules.append(importlib.import_module(info.name)) + return tuple(modules) + + +def _discover_references() -> dict[str, RegistryReference]: + references: dict[str, RegistryReference] = {} + for module in _discover_modules(): + for value in vars(module).values(): + if not isinstance(value, RegistryReference): + continue + if value.identifier.object_type != "registered-operation": + continue + references[value.stable_id] = value + return references + + +def discovered_registered_operation_ids() -> tuple[str, ...]: + """Return every registered-operation identity exposed by the package.""" + + return tuple(sorted(_discover_references())) + + +def _metadata_for_id( + operation_id: str, metadata: dict[str, _OperationMetadata] +) -> _OperationMetadata: + prefix, version = operation_id.rsplit("@", 1) + key = prefix.rsplit(":", 1)[1] + try: + return metadata[f"{key}@{version}"] + except KeyError as exc: + raise ValueError(f"RES-71 inventory missing registered operation: {operation_id}") from exc + + +def build_registered_operation_inventory() -> tuple[RegisteredOperationInventoryEntry, ...]: + """Build the complete inventory and fail if the runtime registry grew unreviewed.""" + + references = _discover_references() + metadata = _metadata() + entries = [] + for operation_id in sorted(references): + reference = references[operation_id] + item = _metadata_for_id(operation_id, metadata) + if item.version != reference.identifier.version: + raise ValueError(f"RES-71 method version mismatch: {operation_id}") + entries.append( + RegisteredOperationInventoryEntry( + operation_id=operation_id, + label=reference.display_label, + method_version=reference.identifier.version, + scientific_family=item.family, + disposition=item.disposition, + implementation=item.implementation, + input_contract=item.input_contract, + output_contract=item.output_contract, + provenance_contract=item.provenance_contract, + refusal_path=item.refusal_path, + test_coverage=item.test_coverage, + authority_references=item.authority_references, + tolerance_contract=item.tolerance_contract, + ) + ) + known = {item.lookup_key for item in metadata.values()} + discovered = { + f"{operation_id.rsplit(':', 1)[1].rsplit('@', 1)[0]}@{operation_id.rsplit('@', 1)[1]}" + for operation_id in references + } + extra = known - discovered + if extra: + raise ValueError(f"RES-71 inventory contains stale operation metadata: {sorted(extra)}") + return tuple(entries) + + +def _resolve_symbol(path: str) -> object: + module_name, attribute_path = path.split(":", 1) + value: object = importlib.import_module(module_name) + for attribute in attribute_path.split("."): + value = getattr(value, attribute) + return value + + +def _repository_root() -> Path: + return Path(__file__).resolve().parents[3] + + +def validate_registered_operation_inventory( + entries: tuple[RegisteredOperationInventoryEntry, ...] | None = None, +) -> GateComponentStatus: + """Validate registry completeness, implementation bindings and test paths.""" + + entries = entries or build_registered_operation_inventory() + operation_ids = tuple(item.operation_id for item in entries) + if len(set(operation_ids)) != len(operation_ids): + raise ValueError("RES-71 inventory contains duplicate operation IDs") + if set(operation_ids) != set(discovered_registered_operation_ids()): + raise ValueError("RES-71 inventory is not complete against the live registry") + root = _repository_root() + for item in entries: + for path in item.implementation: + _resolve_symbol(path) + for path in item.test_coverage: + if not (root / path).is_file(): + raise ValueError(f"RES-71 operation test path is missing: {path}") + if item.disposition is OperationDisposition.IMPLEMENTED and not item.implementation: + raise ValueError(f"implemented operation has no implementation: {item.operation_id}") + if item.disposition is not OperationDisposition.IMPLEMENTED and not item.refusal_path: + raise ValueError(f"non-computing operation has no refusal path: {item.operation_id}") + return GateComponentStatus.PASS + + +def _ids( + entries: tuple[RegisteredOperationInventoryEntry, ...], keys: tuple[str, ...] +) -> tuple[str, ...]: + by_key = { + item.operation_id.rsplit(":", 1)[1].split("@", 1)[0]: item.operation_id for item in entries + } + return tuple(by_key[key] for key in keys) + + +def build_coverage_matrix() -> tuple[CoverageRow, ...]: + """Return the required V2 domain coverage matrix.""" + + entries = build_registered_operation_inventory() + cmj_keys = ( + "cmj-bilateral-total-vertical-force-sum-v1", + "cmj-flight-time-ballistic-jump-height-v1", + "cmj-flight-time-ballistic-jump-height-v2", + "cmj-left-right-force-asymmetry-right-minus-left-over-left-v1", + "cmj-net-vertical-force-from-total-force-and-system-weight-v1", + "cmj-net-vertical-impulse-v1", + "cmj-phase-duration-v1", + "cmj-phase-net-vertical-impulse-v1", + "cmj-phase-relative-displacement-change-v1", + "cmj-physical-system-mass-from-weight-v1", + "cmj-qualified-takeoff-velocity-ballistic-apex-rise-v1", + "cmj-rsi-modified-flight-time-jump-height-over-time-to-takeoff-v1", + "cmj-rsi-modified-takeoff-velocity-jump-height-over-time-to-takeoff-v1", + "cmj-sample-peak-total-supported-vertical-force-v1", + "cmj-sampled-signed-power-extremum-v1", + "cmj-session-selection-aggregation-v1", + "cmj-standard-gravity-mass-equivalent-from-weight-v1", + "cmj-supported-system-com-relative-vertical-displacement-v1", + "cmj-supported-system-com-vertical-acceleration-v1", + "cmj-supported-system-com-vertical-velocity-v1", + "cmj-system-weight-v1", + "cmj-takeoff-velocity-scalar-projection-v1", + "cmj-time-weighted-mean-total-supported-vertical-force-v1", + "cmj-time-weighted-signed-power-mean-v1", + "cmj-total-supported-force-times-supported-system-com-velocity-v1", + ) + strength_keys = ( + "direct-one-repetition-maximum-assessment-v1", + "estimated-one-repetition-maximum-from-linear-load-velocity-v1", + "imtp-baseline-force-statistics-v1", + "imtp-endpoint-average-rfd-v1", + "imtp-force-at-registered-time-v1", + "imtp-force-body-mass-normalization-v1", + "imtp-force-impulse-v1", + "imtp-sampled-peak-force-v1", + "imtp-trial-selection-aggregation-v1", + "mean-concentric-velocity-v1", + "mean-propulsive-velocity-v1", + "measured-one-repetition-maximum-v1", + "sampled-peak-concentric-velocity-v1", + "within-set-velocity-loss-v1", + ) + field_keys = ( + "30-15-ift-last-completed-stage-v1", + "505-asymmetry-v1", + "505-cod-deficit-mean-505-minus-mean-10m-v1", + "linear-sprint-mean-0-10m-reference-v1", + "rsa-best-time-v1", + "rsa-complete-series-aggregation-v1", + "rsa-mean-time-v1", + "rsa-percent-decrement-s-dec-v1", + "rsa-total-time-v1", + "sampled-maximum-sprint-velocity-v1", + "sprint-interval-split-v1", + "sprint-segment-average-velocity-v1", + "standard-505-mean-of-three-v1", + "sustained-maximum-sprint-velocity-v1", + ) + explosive_keys = ( + "drop-jump-flight-time-jump-height-v1", + "drop-jump-ground-contact-time-v1", + "drop-jump-rebound-flight-time-v1", + "drop-jump-rsi-jh-ct-v1", + "drop-jump-rsr-ft-ct-v1", + "bench-press-throw-dynamislm-time-weighted-mean-bar-velocity-v1", + "bench-press-throw-mean-propulsive-velocity-v1", + "bench-press-throw-sampled-maximum-bar-velocity-v1", + "medicine-ball-throw-distance-from-registered-coordinates-v1", + "medicine-ball-throw-instrumented-release-velocity-v1", + ) + stats_keys = ( + "classical-bland-altman-limits-v1", + "confidence-intervals", + "covariance-propagation", + "generic-meaningful-change", + "generic-sem", + "icc-variants", + "log-ba", + "longitudinal-absolute-change-v1", + "longitudinal-descriptive-ols-v1", + "longitudinal-log-ratio-v1", + "longitudinal-reference-window-deviation-v1", + "longitudinal-relative-change-v1", + "longitudinal-window-descriptives-v1", + "longitudinal-within-athlete-sample-sd-v1", + "mdc-sdc", + "method-comparison-ba-summary-v1", + "method-comparison-design-authority-v1", + "mixed-effects", + "readiness-fatigue-injury-interpretation", + "reliability-assumption-assessment-v1", + "reliability-design-authority-v1", + "repeated-measures-ba", + "repeated-measures-correlation", + "sem-from-registered-icc", + "two-replicate-log-multiplicative-error-v1", + "two-replicate-pooled-raw-relative-error-v1", + "two-replicate-within-subject-random-error-v1", + ) + record_keys = ( + "longitudinal-athlete-performance-record", + "multi-source-analysis-input", + "res62-multi-source-manifest", + ) + rows = ( + CoverageRow( + "population/source authority", + CoverageStatus.QUALIFIED, + ("dynamislm.population", "dynamislm.ingestion.registry"), + _ids( + entries, + ( + "canonical-football-population-qualification", + "canonical-empirical-source-qualification", + ), + ), + (), + "Population clauses, evidence role, source license and subgroup separability remain typed and bound.", + "Population applicability is not a measurement-method equivalence claim.", + "Noncanonical or ambiguous populations are quarantined; indirect method evidence cannot mint target priors.", + _TESTS["population"], + _AUTHORITY["population"], + ), + CoverageRow( + "football world/context", + CoverageStatus.QUALIFIED, + ("dynamislm.football", "dynamislm.longitudinal.record"), + _ids(entries, ("match-day-relative-label", *record_keys)), + ("nearest-match inference",), + "Athlete/team/competition/season/session/match/microcycle context is explicit and source-referenced.", + "Calendar relation and match/training exposure context are not inferred from labels alone.", + "Context can bound applicability and grouping; it does not create a performance or causal claim.", + _TESTS["football"] + _TESTS["longitudinal-statistics"], + _AUTHORITY["football"], + ), + CoverageRow( + "ingestion/qualification", + CoverageStatus.QUALIFIED, + ("dynamislm.ingestion.acquisition", "dynamislm.ingestion.promotion"), + _ids(entries, record_keys), + ("vendor-derived proprietary formula reconstruction",), + "Raw bytes, source/version/license, schema, mapping, canonical artifact and replay digest are preserved.", + "Unknown variable identity or source population remains quarantined/rejected deterministically.", + "Promoted canonical rows do not broaden the V2 target or authorize model training.", + _TESTS["ingestion"], + _AUTHORITY["ingestion"], + ), + CoverageRow( + "external load / GNSS / optical", + CoverageStatus.QUALIFIED_WITH_EXPLICIT_DEFERRED, + ( + "dynamislm.external_load.identity", + "dynamislm.external_load.metrics", + "dynamislm.external_load.mapping", + ), + _ids( + entries, + ( + "unit-normalization", + "duration-normalized-distance", + "velocity-threshold-summary", + ), + ), + ("hidden vendor algorithm recomputation", "unresolved threshold/sampling semantics"), + "Provider/device/modality, threshold basis, sampling/processing, aggregation and source origin remain identity fields.", + "Same-label GNSS/optical/vendor values require identity agreement or a registered bridge.", + "Provider-derived output is an observation, not DynamisLM computational authority; no workload/fatigue claim.", + _TESTS["external-load"], + _AUTHORITY["external-load"], + ), + CoverageRow( + "CMJ", + CoverageStatus.QUALIFIED_WITH_EXPLICIT_DEFERRED, + ("dynamislm.measurement.cmj",), + _ids(entries, cmj_keys), + ("CMJ RFD", "COM-displacement jump height"), + "Force/event/phase/mechanics/jump-height outputs retain source signal, support, estimator and processing lineage.", + "Estimator, event, phase, force-system and aggregation differences are claim-relative and fail closed.", + "Derived CMJ performance metrics do not establish physiological mechanism, readiness or fatigue.", + _TESTS["cmj"], + _AUTHORITY["cmj"], + ), + CoverageRow( + "strength / IMTP / VBT", + CoverageStatus.QUALIFIED_WITH_EXPLICIT_DEFERRED, + ("dynamislm.measurement.strength.imtp", "dynamislm.measurement.strength.vbt"), + _ids(entries, strength_keys), + ("VBT mean propulsive velocity", "terminal-velocity estimated 1RM applicability"), + "Force/velocity source series, onset/phase, load, trial selection, method version and value origin are retained.", + "Velocity metric, device, protocol, phase and fixed-load identity must agree or use a registered bridge.", + "Velocity loss is mechanical within-set change; it is not fatigue/readiness by automatic relabelling.", + _TESTS["strength"], + _AUTHORITY["strength"], + ), + CoverageRow( + "sprint / maximum velocity / COD / RSA / 30-15 IFT", + CoverageStatus.QUALIFIED_WITH_EXPLICIT_DEFERRED, + ("dynamislm.measurement.field_testing",), + _ids(entries, field_keys), + ( + "sprint acceleration", + "sustained maximum velocity", + "505 asymmetry", + "VIFT-to-VO2max/MAS/MSS", + ), + "Timing/velocity support, protocol, source qualification, stage/repetition and context are preserved.", + "Sampled maximum, segment-average velocity, 505 direction and RSA protocol identities remain distinct.", + "RSA decrement is a mechanical performance statistic; it is not a fatigue diagnosis or prescription.", + _TESTS["field-testing"], + _AUTHORITY["field-testing"], + ), + CoverageRow( + "DJ / bench throw / medicine-ball throw", + CoverageStatus.QUALIFIED_WITH_EXPLICIT_DEFERRED, + ( + "dynamislm.measurement.drop_jump", + "dynamislm.measurement.bench_press_throw", + "dynamislm.measurement.medicine_ball_throw", + ), + _ids(entries, explosive_keys), + ( + "generic BPT power", + "MBT distance-as-power", + "MBT protocol-independent norm", + "BPT MPV", + ), + "Event/trajectory/coordinate/provider support, protocol and qualification evidence are retained.", + "DJ JH/CT and FT/CT, BPT provider versus DynamisLM velocity, and MBT protocol variants cannot alias.", + "Explosive-test outputs remain bounded performance observations/derivations; no generic power shortcut is accepted.", + _TESTS["explosive-tests"], + _AUTHORITY["explosive-tests"], + ), + CoverageRow( + "longitudinal statistics / reliability / error", + CoverageStatus.QUALIFIED_WITH_EXPLICIT_DEFERRED, + ("dynamislm.longitudinal.statistics",), + _ids(entries, stats_keys), + ( + "ICC variants", + "SEM/MDC without registered assumptions", + "confidence intervals", + "mixed-effects models", + ), + "Statistical support binds exact observations, scale semantics, window/design, assumptions and output provenance.", + "Statistics cannot consume noncomparable identities or caller-minted reliability assumptions.", + "Numerical change, error-relative change and practical/fatigue/readiness meaning remain separate gates.", + _TESTS["longitudinal-statistics"], + _AUTHORITY["longitudinal-statistics"], + ), + CoverageRow( + "cross-source comparability", + CoverageStatus.QUALIFIED_WITH_EXPLICIT_DEFERRED, + ("dynamislm.comparability", "dynamislm.external_load.comparability"), + _ids(entries, ("res70-registered-affine-transformation-v1",)), + ("unregistered device/method bridge", "transitive pairwise comparability"), + "Exact observation hashes, identities, context and authority registry hashes are bound to each decision.", + "States are explicit: comparable, conditional, transformation-required, bridge-required, not-comparable, insufficient.", + "A transformation request is not a comparability verdict and cannot be supplied by an LM.", + _TESTS["comparability"], + _AUTHORITY["comparability"], + ), + CoverageRow( + "analysis capability", + CoverageStatus.QUALIFIED_WITH_EXPLICIT_DEFERRED, + ("dynamislm.analysis.authority", "dynamislm.analysis.registry"), + _ids( + entries, + ( + "longitudinal-absolute-change-v1", + "longitudinal-relative-change-v1", + "longitudinal-log-ratio-v1", + "longitudinal-reference-window-deviation-v1", + "two-replicate-within-subject-random-error-v1", + "method-comparison-ba-summary-v1", + "repeated-measures-correlation", + "mixed-effects", + ), + ), + ( + "between-athlete association", + "cross-test association", + "deferred repeated-measures models", + ), + "Authorization binds operation, support hashes, identity dimensions, level of analysis and prerequisites.", + "Missing comparability, support, level, evidence or statistical authority yields structured refusal.", + "Between-athlete information cannot be relabelled as within-athlete inference or causal evidence.", + _TESTS["analysis"], + _AUTHORITY["analysis"], + ), + CoverageRow( + "claim authority", + CoverageStatus.QUALIFIED_WITH_EXPLICIT_DEFERRED, + ("dynamislm.claims.authority", "dynamislm.claims.registry", "dynamislm.evidence.res70"), + (), + ( + "practical meaningfulness without decision criterion", + "unsupported causal/readiness/injury claims", + ), + "Claim intent revalidates exact observations and all upstream result/evidence/comparability authority.", + "Claim level cannot promote an unresolved identity, comparison, analysis or applicability bundle.", + "The claim ladder and causal hierarchy are enforced; refusal preserves safe observation descriptions.", + _TESTS["claims"] + _TESTS["comparability"], + _AUTHORITY["claims"], + ), + ) + return rows + + +def validate_coverage_matrix(rows: tuple[CoverageRow, ...] | None = None) -> GateComponentStatus: + rows = rows or build_coverage_matrix() + required = { + "population/source authority", + "football world/context", + "ingestion/qualification", + "external load / GNSS / optical", + "CMJ", + "strength / IMTP / VBT", + "sprint / maximum velocity / COD / RSA / 30-15 IFT", + "DJ / bench throw / medicine-ball throw", + "longitudinal statistics / reliability / error", + "cross-source comparability", + "analysis capability", + "claim authority", + } + actual = {row.domain for row in rows} + if actual != required: + raise ValueError(f"RES-71 coverage matrix domain mismatch: {sorted(actual ^ required)}") + if len(rows) != len(actual): + raise ValueError("RES-71 coverage matrix contains duplicate domains") + root = _repository_root() + for row in rows: + if not row.authoritative_surfaces or not row.test_coverage: + raise ValueError(f"coverage row is incomplete: {row.domain}") + if any(not (root / path).is_file() for path in row.test_coverage): + raise ValueError(f"coverage row has missing test evidence: {row.domain}") + return GateComponentStatus.PASS + + +def build_unresolved_computation_inventory() -> tuple[UnresolvedComputation, ...]: + entries = build_registered_operation_inventory() + unresolved: list[UnresolvedComputation] = [] + for entry in entries: + if entry.disposition in { + OperationDisposition.IMPLEMENTED, + OperationDisposition.HISTORICAL_REPLAY_ONLY, + }: + continue + metadata = _metadata_for_id(entry.operation_id, _metadata()) + unresolved.append( + UnresolvedComputation( + capability=entry.label, + registered_operation_id=entry.operation_id, + disposition=entry.disposition, + reason=metadata.reason + or "The operation is explicitly outside current numerical authority.", + refusal_path=entry.refusal_path, + expected_refusal_class="COMPUTATION_NOT_REGISTERED", + safe_description=metadata.safe_description + or "The input observation remains independently describable.", + test_coverage=entry.test_coverage, + authority_references=entry.authority_references, + ) + ) + unresolved.extend( + ( + UnresolvedComputation( + "CMJ RFD", + None, + OperationDisposition.DEFERRED, + "Onset-window, differentiation, filtering and sampling authority is not frozen.", + ("dynamislm.measurement.cmj.metrics:refuse_unregistered_cmj_rfd",), + "COMPUTATION_NOT_REGISTERED", + "Force/event observations and registered CMJ metrics remain describable.", + ("tests/test_cmj_metrics.py",), + ("docs/decisions/RES65-RECEIPT.json",), + ), + UnresolvedComputation( + "generic BPT load-times-velocity power", + None, + OperationDisposition.REJECTED, + "Generic external-load times velocity is not a registered BPT mechanical-system equation.", + ( + "dynamislm.measurement.bench_press_throw.metrics:calculate_bpt_load_times_velocity_power", + ), + "COMPUTATION_NOT_REGISTERED", + "BPT velocity series and provider metrics remain separately describable.", + ("tests/test_explosive_test_families.py",), + ("docs/decisions/RES68-RECEIPT.json",), + ), + UnresolvedComputation( + "MBT distance-as-power or protocol-independent normative score", + None, + OperationDisposition.REJECTED, + "MBT distance and release velocity do not authorize a generic power or norm operation.", + ( + "dynamislm.measurement.medicine_ball_throw.metrics:calculate_mbt_distance_as_power", + "dynamislm.measurement.medicine_ball_throw.metrics:calculate_mbt_protocol_independent_normative_score", + ), + "COMPUTATION_NOT_REGISTERED", + "Qualified MBT distance or instrumented release velocity remains describable.", + ("tests/test_explosive_test_families.py",), + ("docs/decisions/RES68-RECEIPT.json",), + ), + UnresolvedComputation( + "sprint acceleration", + None, + OperationDisposition.DEFERRED, + "Adjacent split-average velocities do not constitute a registered time-resolved acceleration estimator.", + ("dynamislm.measurement.field_testing.sprint:refuse_sprint_acceleration",), + "COMPUTATION_NOT_REGISTERED", + "Qualified split times and segment-average velocities remain describable.", + ("tests/test_field_testing.py",), + ("docs/decisions/RES67-RECEIPT.json",), + ), + UnresolvedComputation( + "VIFT as VO2max, MAS or maximum sprint speed", + None, + OperationDisposition.REJECTED, + "VIFT is a registered 30-15 IFT final-stage velocity, not a relabelled physiological or sprint construct.", + ( + "dynamislm.measurement.field_testing.ift:refuse_vift_as_vo2max", + "dynamislm.measurement.field_testing.ift:refuse_vift_as_mas", + "dynamislm.measurement.field_testing.ift:refuse_vift_as_mss", + ), + "COMPUTATION_NOT_REGISTERED", + "The exact VIFT stage result remains describable.", + ("tests/test_field_testing.py",), + ("docs/decisions/RES67-RECEIPT.json",), + ), + ) + ) + return tuple(unresolved) + + +def validate_unresolved_computation_inventory( + entries: tuple[UnresolvedComputation, ...] | None = None, +) -> GateComponentStatus: + entries = entries or build_unresolved_computation_inventory() + capabilities = tuple(item.capability for item in entries) + if len(set(capabilities)) != len(capabilities): + raise ValueError("RES-71 unresolved inventory contains duplicate capabilities") + root = _repository_root() + for item in entries: + if item.disposition is OperationDisposition.IMPLEMENTED: + raise ValueError( + f"implemented capability leaked into unresolved inventory: {item.capability}" + ) + for path in item.test_coverage: + if not (root / path).is_file(): + raise ValueError(f"unresolved inventory test path is missing: {path}") + return GateComponentStatus.PASS + + +__all__ = [ + "RES71_REGISTRY_VERSION", + "build_coverage_matrix", + "build_registered_operation_inventory", + "build_unresolved_computation_inventory", + "discovered_registered_operation_ids", + "validate_coverage_matrix", + "validate_registered_operation_inventory", + "validate_unresolved_computation_inventory", +] diff --git a/src/dynamislm/qualification/references.py b/src/dynamislm/qualification/references.py new file mode 100644 index 0000000..5c83f66 --- /dev/null +++ b/src/dynamislm/qualification/references.py @@ -0,0 +1,358 @@ +"""Verifier-ready deterministic reference cases for RES-71. + +The interface stores typed case contracts and expected outputs/refusals. It +does not execute an LM or accept an LM-produced numeric answer as authority. +Later verifier work can bind these cases to the already-registered operation +dispatch surface. +""" + +from __future__ import annotations + +from dynamislm.qualification.contracts import ( + ReferenceCase, + ReferenceCaseStatus, + ReferenceValue, +) +from dynamislm.serialization import canonical_hash, canonical_json + +RES71_REFERENCE_INTERFACE_VERSION = "1.0.0" + + +def _operation(key: str, version: str = "1.0.0") -> str: + return f"dynamislm:registered-operation:{key}@{version}" + + +def _value(name: str, value: bool | float | int | str, unit: str | None = None) -> ReferenceValue: + return ReferenceValue(name=name, value=value, unit=unit) + + +def _cases() -> tuple[ReferenceCase, ...]: + return ( + ReferenceCase( + case_id="res71-external-unit-km-to-m", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="external-load", + operation_id=_operation("unit-normalization"), + status=ReferenceCaseStatus.VALUE, + synthetic_input=( + _value("value", 1.0), + _value("from_unit", "kilometer"), + _value("to_unit", "meter"), + ), + expected_values=(_value("value", 1000.0, "m"),), + expected_refusal_class=None, + expected_reason_codes=(), + expected_comparability_state=None, + expected_claim_level=None, + tolerance_absolute=0.0, + tolerance_relative=0.0, + required_provenance_fields=("operation_id", "unit", "method_registry_version"), + authority_references=( + "docs/decisions/RES64-DR-001-external-load-scientific-identity.md", + ), + ), + ReferenceCase( + case_id="res71-external-relative-distance", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="external-load", + operation_id=_operation("duration-normalized-distance"), + status=ReferenceCaseStatus.VALUE, + synthetic_input=( + _value("total_distance", 600.0, "m"), + _value("valid_duration", 10.0, "min"), + ), + expected_values=(_value("relative_distance", 60.0, "m/min"),), + expected_refusal_class=None, + expected_reason_codes=(), + expected_comparability_state=None, + expected_claim_level=None, + tolerance_absolute=0.0, + tolerance_relative=0.0, + required_provenance_fields=( + "operation_id", + "source_observation_ids", + "processing_run_id", + ), + authority_references=( + "docs/decisions/RES64-DR-001-external-load-scientific-identity.md", + ), + ), + ReferenceCase( + case_id="res71-cmj-flight-time-v2-gold", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="CMJ", + operation_id=_operation("cmj-flight-time-ballistic-jump-height-v2", "2.0.0"), + status=ReferenceCaseStatus.VALUE, + synthetic_input=( + _value("flight_time_s", 0.5, "s"), + _value("gravity_m_per_s2", 9.81, "m/s2"), + ), + expected_values=(_value("jump_height_m", 0.3065625, "m"),), + expected_refusal_class=None, + expected_reason_codes=(), + expected_comparability_state=None, + expected_claim_level=None, + tolerance_absolute=1e-12, + tolerance_relative=1e-12, + required_provenance_fields=( + "source_observation_id", + "takeoff_event_id", + "landing_event_id", + "operation_id", + "processing_run_id", + ), + authority_references=("docs/decisions/RES65-DR-001-cmj-football-metric-completion.md",), + ), + ReferenceCase( + case_id="res71-rsa-mechanical-percent-decrement", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="RSA", + operation_id=_operation("rsa-percent-decrement-s-dec-v1"), + status=ReferenceCaseStatus.VALUE, + synthetic_input=( + _value("repetition_times_s", "[1.0, 1.1, 1.2]"), + _value("best_time_s", 1.0, "s"), + ), + expected_values=(_value("s_dec_percent", 10.0, "%"),), + expected_refusal_class=None, + expected_reason_codes=(), + expected_comparability_state=None, + expected_claim_level=None, + tolerance_absolute=1e-12, + tolerance_relative=1e-12, + required_provenance_fields=("protocol_identity", "criterion_sprint", "operation_id"), + authority_references=( + "docs/decisions/RES67-DR-001-field-testing-scientific-engine.md", + ), + ), + ReferenceCase( + case_id="res71-longitudinal-absolute-change", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="longitudinal-statistics", + operation_id=_operation("longitudinal-absolute-change-v1"), + status=ReferenceCaseStatus.VALUE, + synthetic_input=( + _value("baseline", 10.0), + _value("followup", 15.0), + _value("support", "exactly-two-comparable-within-athlete-entries"), + ), + expected_values=(_value("absolute_change", 5.0),), + expected_refusal_class=None, + expected_reason_codes=(), + expected_comparability_state=None, + expected_claim_level="NUMERICAL_CHANGE", + tolerance_absolute=0.0, + tolerance_relative=0.0, + required_provenance_fields=("support_hash", "operation_id", "source_observation_ids"), + authority_references=( + "docs/decisions/RES69-DR-001-longitudinal-reliability-uncertainty.md", + ), + ), + ReferenceCase( + case_id="res71-nonfinite-unit-refusal", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="external-load", + operation_id=_operation("unit-normalization"), + status=ReferenceCaseStatus.REFUSAL, + synthetic_input=( + _value("value", "NaN"), + _value("from_unit", "meter"), + _value("to_unit", "kilometer"), + ), + expected_values=(), + expected_refusal_class="COMPUTATION_NOT_REGISTERED", + expected_reason_codes=("NONFINITE_INPUT",), + expected_comparability_state=None, + expected_claim_level=None, + tolerance_absolute=None, + tolerance_relative=None, + required_provenance_fields=("blocked_claim", "reason_codes", "missing_information"), + authority_references=( + "docs/decisions/RES64-DR-001-external-load-scientific-identity.md", + ), + ), + ReferenceCase( + case_id="res71-cmj-rfd-refusal", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="CMJ", + operation_id=None, + status=ReferenceCaseStatus.REFUSAL, + synthetic_input=(_value("requested_operation", "CMJ RFD"),), + expected_values=(), + expected_refusal_class="COMPUTATION_NOT_REGISTERED", + expected_reason_codes=("CMJ_RFD_NOT_REGISTERED",), + expected_comparability_state=None, + expected_claim_level=None, + tolerance_absolute=None, + tolerance_relative=None, + required_provenance_fields=( + "blocked_claim", + "reason_codes", + "what_can_still_be_safely_described", + ), + authority_references=("docs/decisions/RES65-RECEIPT.json",), + ), + ReferenceCase( + case_id="res71-bpt-generic-power-refusal", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="bench-press-throw", + operation_id=None, + status=ReferenceCaseStatus.REFUSAL, + synthetic_input=(_value("load", 80.0, "kg"), _value("velocity", 1.2, "m/s")), + expected_values=(), + expected_refusal_class="COMPUTATION_NOT_REGISTERED", + expected_reason_codes=("NO_REGISTERED_OPERATION", "COMPUTATION_NOT_REGISTERED"), + expected_comparability_state=None, + expected_claim_level=None, + tolerance_absolute=None, + tolerance_relative=None, + required_provenance_fields=("blocked_claim", "reason_codes", "safe_descriptions"), + authority_references=("docs/decisions/RES68-RECEIPT.json",), + ), + ReferenceCase( + case_id="res71-bpt-mpv-refusal", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="bench-press-throw", + operation_id=_operation("bench-press-throw-mean-propulsive-velocity-v1"), + status=ReferenceCaseStatus.REFUSAL, + synthetic_input=( + _value("support", "provider-or-series-support-without-registered-MPV"), + ), + expected_values=(), + expected_refusal_class="COMPUTATION_NOT_REGISTERED", + expected_reason_codes=("NO_REGISTERED_OPERATION", "COMPUTATION_NOT_REGISTERED"), + expected_comparability_state=None, + expected_claim_level=None, + tolerance_absolute=None, + tolerance_relative=None, + required_provenance_fields=("blocked_claim", "reason_codes", "safe_descriptions"), + authority_references=("docs/decisions/RES68-RECEIPT.json",), + ), + ReferenceCase( + case_id="res71-same-label-different-method", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="cross-source-comparability", + operation_id=None, + status=ReferenceCaseStatus.COMPARABILITY, + synthetic_input=( + _value("left_label", "HSR distance"), + _value("right_label", "HSR distance"), + _value("difference", "threshold identity / provider processing"), + ), + expected_values=(), + expected_refusal_class=None, + expected_reason_codes=("THRESHOLD_MISMATCH", "METHOD_MISMATCH"), + expected_comparability_state="NOT_COMPARABLE", + expected_claim_level=None, + tolerance_absolute=None, + tolerance_relative=None, + required_provenance_fields=("request_hash", "dimension_findings", "rule_registry_hash"), + authority_references=( + "docs/decisions/RES70-DR-001-cross-source-comparability-analysis-claim-authority.md", + ), + ), + ReferenceCase( + case_id="res71-causal-overclaim-refusal", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="claim-authority", + operation_id=None, + status=ReferenceCaseStatus.CLAIM_AUTHORITY, + synthetic_input=(_value("requested_claim", "observed change caused fatigue"),), + expected_values=(), + expected_refusal_class="CAUSAL_IDENTIFICATION_UNSUPPORTED", + expected_reason_codes=("RES70_UNSUPPORTED_CAUSAL_CLAIM",), + expected_comparability_state=None, + expected_claim_level="OBSERVED_VALUE_ONLY", + tolerance_absolute=None, + tolerance_relative=None, + required_provenance_fields=( + "claim_intent_hash", + "blocked_claims", + "safe_observation_description", + ), + authority_references=("docs/architecture/REASONING_CLAIMS_EVALUATION_V1.md",), + ), + ReferenceCase( + case_id="res71-between-to-within-refusal", + case_version=RES71_REFERENCE_INTERFACE_VERSION, + family="analysis-capability", + operation_id=None, + status=ReferenceCaseStatus.CLAIM_AUTHORITY, + synthetic_input=( + _value("requested_estimand", "within-athlete"), + _value("support", "one-row-per-athlete"), + ), + expected_values=(), + expected_refusal_class="ANALYSIS_DESIGN_MISMATCH", + expected_reason_codes=("RES70_WRONG_LEVEL_OF_ANALYSIS",), + expected_comparability_state=None, + expected_claim_level="BETWEEN_ATHLETE_NOT_WITHIN_ATHLETE", + tolerance_absolute=None, + tolerance_relative=None, + required_provenance_fields=("support_hash", "resolved_level", "refusal_reasons"), + authority_references=( + "docs/decisions/RES70-DR-001-cross-source-comparability-analysis-claim-authority.md", + ), + ), + ) + + +def get_reference_cases() -> tuple[ReferenceCase, ...]: + """Return the immutable reference-case set.""" + + return _cases() + + +def get_reference_case(case_id: str) -> ReferenceCase: + """Resolve one case by stable ID.""" + + for case in get_reference_cases(): + if case.case_id == case_id: + return case + raise KeyError(case_id) + + +def validate_reference_cases(cases: tuple[ReferenceCase, ...] | None = None) -> None: + cases = cases or get_reference_cases() + ids = tuple(case.case_id for case in cases) + if len(set(ids)) != len(ids): + raise ValueError("RES-71 reference cases must have unique case IDs") + if not any(case.status is ReferenceCaseStatus.VALUE for case in cases): + raise ValueError("reference set needs a value case") + if not any(case.status is ReferenceCaseStatus.REFUSAL for case in cases): + raise ValueError("reference set needs a refusal case") + if not any(case.status is ReferenceCaseStatus.COMPARABILITY for case in cases): + raise ValueError("reference set needs a comparability case") + if not any(case.status is ReferenceCaseStatus.CLAIM_AUTHORITY for case in cases): + raise ValueError("reference set needs a claim-authority case") + for case in cases: + if case.operation_id is not None and not case.operation_id.startswith( + "dynamislm:registered-operation:" + ): + raise ValueError(f"invalid operation identity in {case.case_id}") + + +def reference_case_manifest(cases: tuple[ReferenceCase, ...] | None = None) -> str: + """Return canonical JSON suitable for a later verifier artifact.""" + + cases = cases or get_reference_cases() + validate_reference_cases(cases) + return canonical_json(cases) + + +def reference_case_digest(cases: tuple[ReferenceCase, ...] | None = None) -> str: + """Return the deterministic digest of the reference interface.""" + + cases = cases or get_reference_cases() + validate_reference_cases(cases) + return canonical_hash(cases) + + +__all__ = [ + "RES71_REFERENCE_INTERFACE_VERSION", + "get_reference_case", + "get_reference_cases", + "reference_case_digest", + "reference_case_manifest", + "validate_reference_cases", +] diff --git a/tests/test_kernel.py b/tests/test_kernel.py index 9094c13..e3b9eb0 100644 --- a/tests/test_kernel.py +++ b/tests/test_kernel.py @@ -620,6 +620,9 @@ def test_no_test_specific_arithmetic_or_science_is_in_generic_public_package() - for path in package_root.rglob("*.py") if "measurement/cmj" not in path.as_posix() and "measurement/strength" not in path.as_posix() + # RES-71 qualification metadata inventories upstream family names but + # does not add a generic numerical operation to the public kernel. + and "qualification" not in path.as_posix() ) assert "cmj" not in source.lower() diff --git a/tests/test_res71_qualification.py b/tests/test_res71_qualification.py new file mode 100644 index 0000000..fc37e0b --- /dev/null +++ b/tests/test_res71_qualification.py @@ -0,0 +1,175 @@ +from __future__ import annotations + +from dataclasses import replace + +import pytest + +from dynamislm.comparability import ( + ComparabilityState, + CrossSourceComparabilityRequest, + ObservationAuthorityReference, + assess_cross_source_comparability, +) +from dynamislm.external_load.metrics import ( + ExternalLoadComputationError, + convert_external_load_value, + duration_normalized_distance, +) +from dynamislm.external_load.registry import ( + EXTERNAL_LOAD_KILOMETER, + EXTERNAL_LOAD_METER, + EXTERNAL_LOAD_MINUTE, +) +from dynamislm.measurement.bench_press_throw.metrics import ( + calculate_bpt_load_times_velocity_power, + calculate_bpt_mean_propulsive_velocity, +) +from dynamislm.measurement.cmj.metrics import refuse_unregistered_cmj_rfd +from dynamislm.measurement.cmj.registry import METER +from dynamislm.measurement.field_testing.ift import refuse_vift_as_vo2max +from dynamislm.measurement.field_testing.sprint import refuse_sprint_acceleration +from dynamislm.measurement.identity import InstanceIdentifier +from dynamislm.measurement.medicine_ball_throw.metrics import ( + calculate_mbt_distance_as_power, + calculate_mbt_protocol_independent_normative_score, +) +from dynamislm.qualification import ( + ReferenceCaseStatus, + build_coverage_matrix, + build_registered_operation_inventory, + build_unresolved_computation_inventory, + discovered_registered_operation_ids, + get_reference_cases, + reference_case_digest, + reference_case_manifest, + validate_coverage_matrix, + validate_reference_cases, + validate_registered_operation_inventory, + validate_unresolved_computation_inventory, +) +from dynamislm.refusal import RefusalClass, RefusalResult +from dynamislm.serialization import canonical_json, from_canonical_json +from test_kernel import _derived_observation + + +def test_inventory_covers_every_live_registered_operation() -> None: + entries = build_registered_operation_inventory() + + assert len(discovered_registered_operation_ids()) == 100 + assert len(entries) == 100 + assert validate_registered_operation_inventory(entries).value == "PASS" + assert {entry.disposition.value for entry in entries} >= { + "IMPLEMENTED", + "HISTORICAL_REPLAY_ONLY", + "REPRESENT_BUT_DO_NOT_COMPUTE", + "DEFERRED", + "REJECTED", + } + + +def test_coverage_and_unresolved_contracts_are_complete() -> None: + coverage = build_coverage_matrix() + unresolved = build_unresolved_computation_inventory() + + assert len(coverage) == 12 + assert validate_coverage_matrix(coverage).value == "PASS" + assert validate_unresolved_computation_inventory(unresolved).value == "PASS" + assert any(item.registered_operation_id is None for item in unresolved) + assert any(item.registered_operation_id is not None for item in unresolved) + + +def test_independent_external_load_gold_values_and_domain_refusals() -> None: + assert convert_external_load_value(1.0, EXTERNAL_LOAD_KILOMETER, EXTERNAL_LOAD_METER) == 1000.0 + assert ( + duration_normalized_distance( + 600.0, + 10.0, + distance_unit=EXTERNAL_LOAD_METER, + duration_unit=EXTERNAL_LOAD_MINUTE, + ) + == 60.0 + ) + + with pytest.raises(ExternalLoadComputationError, match="finite"): + convert_external_load_value(float("nan"), EXTERNAL_LOAD_METER, EXTERNAL_LOAD_KILOMETER) + with pytest.raises(ExternalLoadComputationError, match="positive"): + duration_normalized_distance( + 600.0, + 0.0, + distance_unit=EXTERNAL_LOAD_METER, + duration_unit=EXTERNAL_LOAD_MINUTE, + ) + + +def test_unregistered_numerical_surfaces_fail_closed() -> None: + results = ( + calculate_bpt_load_times_velocity_power(80.0, 1.2), + calculate_bpt_mean_propulsive_velocity(), + calculate_mbt_distance_as_power(), + calculate_mbt_protocol_independent_normative_score(), + refuse_unregistered_cmj_rfd(), + refuse_sprint_acceleration(), + refuse_vift_as_vo2max(), + ) + + for result in results: + assert isinstance(result, RefusalResult) + assert all(result.refusal_class is not None for result in results) + assert all( + result.refusal_class is RefusalClass.COMPUTATION_NOT_REGISTERED for result in results[:6] + ) + assert results[6].refusal_class is RefusalClass.IDENTITY_UNRESOLVED + + +def test_same_label_different_measurand_is_not_comparable() -> None: + left = _derived_observation("res71-comparison-left") + right = _derived_observation("res71-comparison-right") + left = replace( + left, + identity=replace(left.identity, processing=replace(left.identity.processing, unit=METER)), + result=replace(left.result, unit=METER), + ) + right = replace( + right, + identity=replace(right.identity, processing=replace(right.identity.processing, unit=METER)), + result=replace(right.result, unit=METER), + ) + measurand = replace( + right.identity.semantic.measurand, + display_label=left.identity.semantic.measurand.display_label, + identifier=replace(right.identity.semantic.measurand.identifier, key="different-measurand"), + ) + changed_semantic = replace( + right.identity.semantic, + measurand=measurand, + ) + right = replace(right, identity=replace(right.identity, semantic=changed_semantic)) + request = CrossSourceComparabilityRequest( + request_id=InstanceIdentifier("cross-source-comparability-request", "res71-comparison"), + left_observation=ObservationAuthorityReference.from_observation(left), + right_observation=ObservationAuthorityReference.from_observation(right), + claim_intent=left.identity.semantic.metric_definition, + ) + + decision = assess_cross_source_comparability(request, (left, right)) + + assert decision.state is ComparabilityState.NOT_COMPARABLE + assert "MEASURAND_MISMATCH" in decision.reason_codes + + +def test_reference_interface_is_deterministic_and_serializable() -> None: + cases = get_reference_cases() + validate_reference_cases(cases) + assert len(cases) == 12 + assert {case.status for case in cases} == { + ReferenceCaseStatus.VALUE, + ReferenceCaseStatus.REFUSAL, + ReferenceCaseStatus.COMPARABILITY, + ReferenceCaseStatus.CLAIM_AUTHORITY, + } + assert reference_case_digest(cases) == reference_case_digest(cases) + manifest = reference_case_manifest(cases) + assert '"case_id":"res71-cmj-flight-time-v2-gold"' in manifest + + restored = from_canonical_json(canonical_json(cases[0]), type(cases[0])) + assert restored == cases[0] From f25ebbd941bca29aaacf615edab96921010afe6d Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:53:44 +0000 Subject: [PATCH 02/14] test(res71): add cross-engine adversarial and gold qualification --- tests/test_res71_adversarial.py | 76 +++++++++++++++++++++++++++++++++ 1 file changed, 76 insertions(+) create mode 100644 tests/test_res71_adversarial.py diff --git a/tests/test_res71_adversarial.py b/tests/test_res71_adversarial.py new file mode 100644 index 0000000..83bb8a1 --- /dev/null +++ b/tests/test_res71_adversarial.py @@ -0,0 +1,76 @@ +from __future__ import annotations + +import math + +from dynamislm.qualification import ( + ReferenceCaseStatus, + build_registered_operation_inventory, + get_reference_cases, +) + + +def test_gold_reference_values_reproduce_from_independent_equations() -> None: + cases = {case.case_id: case for case in get_reference_cases()} + + flight_time = 0.5 + gravity = 9.81 + expected_jump_height = gravity * flight_time**2 / 8.0 + recorded_jump_height = cases["res71-cmj-flight-time-v2-gold"].expected_values[0].value + assert isinstance(recorded_jump_height, int | float) + assert math.isclose(expected_jump_height, recorded_jump_height, rel_tol=1e-12, abs_tol=1e-12) + + repetition_times = (1.0, 1.1, 1.2) + expected_s_dec = 100.0 * ( + sum(repetition_times) / (min(repetition_times) * len(repetition_times)) - 1.0 + ) + recorded_s_dec = cases["res71-rsa-mechanical-percent-decrement"].expected_values[0].value + assert isinstance(recorded_s_dec, int | float) + assert math.isclose(expected_s_dec, recorded_s_dec, rel_tol=1e-12, abs_tol=1e-12) + + expected_relative_distance = 600.0 / 10.0 + recorded_relative_distance = cases["res71-external-relative-distance"].expected_values[0].value + assert recorded_relative_distance == expected_relative_distance + + +def test_value_cases_bind_outputs_to_method_and_lineage_contracts() -> None: + operation_ids = {entry.operation_id for entry in build_registered_operation_inventory()} + + for case in get_reference_cases(): + if case.status is not ReferenceCaseStatus.VALUE: + continue + assert case.operation_id in operation_ids + assert "operation_id" in case.required_provenance_fields + assert case.expected_values + assert case.tolerance_absolute is not None + assert case.tolerance_relative is not None + + +def test_historical_method_identity_is_not_current_method_identity() -> None: + entries = {entry.operation_id: entry for entry in build_registered_operation_inventory()} + historical = entries[ + "dynamislm:registered-operation:cmj-flight-time-ballistic-jump-height-v1@1.0.0" + ] + current = entries[ + "dynamislm:registered-operation:cmj-flight-time-ballistic-jump-height-v2@2.0.0" + ] + + assert historical.disposition.value == "HISTORICAL_REPLAY_ONLY" + assert current.disposition.value == "IMPLEMENTED" + assert historical.operation_id != current.operation_id + assert historical.method_version != current.method_version + + +def test_claim_and_comparability_cases_preserve_safe_refusal_states() -> None: + cases = {case.case_id: case for case in get_reference_cases()} + comparison = cases["res71-same-label-different-method"] + causal = cases["res71-causal-overclaim-refusal"] + between = cases["res71-between-to-within-refusal"] + + assert comparison.status is ReferenceCaseStatus.COMPARABILITY + assert comparison.expected_comparability_state == "NOT_COMPARABLE" + assert comparison.expected_reason_codes + assert causal.status is ReferenceCaseStatus.CLAIM_AUTHORITY + assert causal.expected_refusal_class == "CAUSAL_IDENTIFICATION_UNSUPPORTED" + assert causal.expected_claim_level == "OBSERVED_VALUE_ONLY" + assert between.expected_refusal_class == "ANALYSIS_DESIGN_MISMATCH" + assert between.expected_claim_level == "BETWEEN_ATHLETE_NOT_WITHIN_ATHLETE" From fa0daf63e06c45fd6d48e8f7405cd3c3a72a2407 Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:55:29 +0000 Subject: [PATCH 03/14] feat(res71): add verifier-ready deterministic reference cases --- scripts/res71_reference_cases.py | 27 +++++++++++++++++++++++++ tests/test_res71_reference_interface.py | 21 +++++++++++++++++++ 2 files changed, 48 insertions(+) create mode 100644 scripts/res71_reference_cases.py create mode 100644 tests/test_res71_reference_interface.py diff --git a/scripts/res71_reference_cases.py b/scripts/res71_reference_cases.py new file mode 100644 index 0000000..1abab2e --- /dev/null +++ b/scripts/res71_reference_cases.py @@ -0,0 +1,27 @@ +"""Emit the RES-71 verifier-ready deterministic reference interface.""" + +from __future__ import annotations + +import argparse +import sys + +from dynamislm.qualification import reference_case_digest, reference_case_manifest + +VERIFIER_REFERENCE_INTERFACE = "res71-reference-interface@1.0.0" + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + output = parser.add_mutually_exclusive_group(required=True) + output.add_argument("--digest", action="store_true", help="emit the canonical case digest") + output.add_argument("--manifest", action="store_true", help="emit canonical case JSON") + args = parser.parse_args(argv) + if args.digest: + sys.stdout.write(reference_case_digest() + "\n") + else: + sys.stdout.write(reference_case_manifest() + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_res71_reference_interface.py b/tests/test_res71_reference_interface.py new file mode 100644 index 0000000..9caa4c5 --- /dev/null +++ b/tests/test_res71_reference_interface.py @@ -0,0 +1,21 @@ +from __future__ import annotations + +import pytest +from scripts.res71_reference_cases import VERIFIER_REFERENCE_INTERFACE, main + +from dynamislm.qualification import reference_case_digest, reference_case_manifest + + +def test_verifier_reference_adapter_emits_the_registered_digest( + capsys: pytest.CaptureFixture[str], +) -> None: + assert VERIFIER_REFERENCE_INTERFACE == "res71-reference-interface@1.0.0" + assert main(["--digest"]) == 0 + assert capsys.readouterr().out.strip() == reference_case_digest() + + +def test_verifier_reference_adapter_emits_canonical_manifest( + capsys: pytest.CaptureFixture[str], +) -> None: + assert main(["--manifest"]) == 0 + assert capsys.readouterr().out.strip() == reference_case_manifest() From 49988aa3cd03e460dc8ae8c60f8afa890e75706e Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:58:18 +0000 Subject: [PATCH 04/14] docs(res71): add provenance coverage and unresolved inventories --- docs/qualification/RES71-COVERAGE-MATRIX.md | 38 ++++++++ .../RES71-OPERATION-INVENTORY.md | 70 ++++++++++++++ .../RES71-PROVENANCE-SERIALIZATION.md | 49 ++++++++++ .../RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md | 91 +++++++++++++++++++ .../RES71-UNRESOLVED-COMPUTATIONS.md | 39 ++++++++ .../RES71-VERIFIER-REFERENCE-INTERFACE.md | 51 +++++++++++ 6 files changed, 338 insertions(+) create mode 100644 docs/qualification/RES71-COVERAGE-MATRIX.md create mode 100644 docs/qualification/RES71-OPERATION-INVENTORY.md create mode 100644 docs/qualification/RES71-PROVENANCE-SERIALIZATION.md create mode 100644 docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md create mode 100644 docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md create mode 100644 docs/qualification/RES71-VERIFIER-REFERENCE-INTERFACE.md diff --git a/docs/qualification/RES71-COVERAGE-MATRIX.md b/docs/qualification/RES71-COVERAGE-MATRIX.md new file mode 100644 index 0000000..520d7c7 --- /dev/null +++ b/docs/qualification/RES71-COVERAGE-MATRIX.md @@ -0,0 +1,38 @@ +# RES-71 V2 coverage matrix + +`build_coverage_matrix()` returns the machine-readable version. Every row has +authoritative surfaces, registered operations (where applicable), unresolved +capabilities, provenance boundary, comparability boundary, claim boundary, +authority references and test paths. + +| V2 domain | Status | Explicit boundary / unresolved surface | +|---|---|---| +| Population/source authority | QUALIFIED | Ambiguous/noncanonical populations quarantine; indirect evidence cannot mint target priors. | +| Football world/context | QUALIFIED | Match-day relation requires an identified target match; no nearest-match inference. | +| Ingestion/qualification | QUALIFIED | Raw bytes, license, schema, mapping, canonical artifact and replay lineage remain bound. | +| External load / GNSS / optical | QUALIFIED_WITH_EXPLICIT_DEFERRED | Hidden vendor algorithms and unresolved threshold/sampling semantics are observations/refusals, not reconstructed equations. | +| CMJ | QUALIFIED_WITH_EXPLICIT_DEFERRED | CMJ RFD and COM-displacement jump height remain deferred; estimator/event/phase identities remain distinct. | +| Strength / IMTP / VBT | QUALIFIED_WITH_EXPLICIT_DEFERRED | VBT MPV and terminal-velocity estimated 1RM are represented/refused; velocity loss is not fatigue. | +| Sprint / max velocity / COD / RSA / 30–15 IFT | QUALIFIED_WITH_EXPLICIT_DEFERRED | Acceleration, sustained maximum, 505 asymmetry and VIFT relabelling remain refused/deferred. | +| DJ / bench throw / medicine-ball throw | QUALIFIED_WITH_EXPLICIT_DEFERRED | DJ JH/CT differs from FT/CT; generic BPT/MBT power and norms are refused. | +| Longitudinal statistics / reliability / error | QUALIFIED_WITH_EXPLICIT_DEFERRED | Deferred ICC/SEM/MDC/CI/BA/repeated-measures methods are explicit; no generic meaningful-change authority. | +| Cross-source comparability | QUALIFIED_WITH_EXPLICIT_DEFERRED | Exact identities, context and registry hashes are required; no transitive or caller-minted bridge. | +| Analysis capability | QUALIFIED_WITH_EXPLICIT_DEFERRED | Missing prerequisites, wrong level, pseudoreplication and deferred between/cross-test models refuse. | +| Claim authority | QUALIFIED_WITH_EXPLICIT_DEFERRED | Ladder/causal level is enforced; practical, readiness, fatigue, injury and causal escalation refuse. | + +“Qualified with explicit deferred” means the domain is covered by an +authoritative implementation/refusal contract; it does not claim that every +published method is implemented. + +## Required upstream relationships + +```text +population/source → football context → canonical record/lineage + ↓ ↓ ↓ +measurement identity → deterministic operation → statistical analysis + ↓ ↓ ↓ +comparability/evidence → claim authority → bounded result or refusal +``` + +No arrow permits a label, unit, correlation, provider output or LM proposal to +skip identity, provenance, comparability, analysis or claim prerequisites. diff --git a/docs/qualification/RES71-OPERATION-INVENTORY.md b/docs/qualification/RES71-OPERATION-INVENTORY.md new file mode 100644 index 0000000..05b8aa7 --- /dev/null +++ b/docs/qualification/RES71-OPERATION-INVENTORY.md @@ -0,0 +1,70 @@ +# RES-71 registered-operation inventory + +The authoritative inventory is the immutable tuple returned by +`dynamislm.qualification.build_registered_operation_inventory()`. +`validate_registered_operation_inventory()` discovers every +`RegistryReference` whose object type is `registered-operation` and requires an +exact set match. A newly exposed operation therefore fails qualification until +its contract is reviewed. + +Each row contains: + +```text +operation_id +label +method_version +scientific_family +disposition +implementation +input_contract +output_contract +provenance_contract +refusal_path +test_coverage +authority_references +tolerance_contract +``` + +## Inventory totals + +| Family | Registered identities | Implemented | Historical replay | Represented only | Deferred | Rejected | +|---|---:|---:|---:|---:|---:|---:| +| Population/source | 2 | 2 | 0 | 0 | 0 | 0 | +| Football context | 1 | 1 | 0 | 0 | 0 | 0 | +| Ingestion/record | 3 | 3 | 0 | 0 | 0 | 0 | +| External load | 3 | 3 | 0 | 0 | 0 | 0 | +| CMJ | 25 | 24 | 1 | 0 | 0 | 0 | +| Strength/IMTP/VBT | 14 | 12 | 0 | 1 | 1 | 0 | +| Field testing | 14 | 11 | 0 | 1 | 1 | 1 | +| DJ/BPT/MBT | 10 | 9 | 0 | 1 | 0 | 0 | +| Longitudinal statistics | 27 | 15 | 0 | 5 | 6 | 3 | +| Cross-source bridge | 1 | 1 | 0 | 0 | 0 | 0 | +| **Total** | **100** | **81** | **1** | **7** | **8** | **3** | + +The count is a qualification assertion, not a replacement for the exact set +comparison. The machine-readable entries carry the implementation symbol, +input/output/provenance contract, refusal route, authority documents and test +paths for each identity. + +## Authority mapping by family + +| Family | Primary implementation surface | Authority | Test evidence | +|---|---|---|---| +| Population/source | `dynamislm.population`, `dynamislm.ingestion` | RES-59/60/63 | `test_population.py`, `test_ingestion.py`, `test_res63_operational_authority.py` | +| Football context | `dynamislm.football`, `dynamislm.longitudinal` | RES-61/62 | `test_football.py`, `test_longitudinal.py` | +| External load | `dynamislm.external_load` | RES-64 | `test_external_load.py` | +| CMJ | `dynamislm.measurement.cmj` | RES-34–50/65 | CMJ test family and RES-71 gold/refusal cases | +| Strength | `dynamislm.measurement.strength` | RES-66 | `test_strength.py` | +| Field testing | `dynamislm.measurement.field_testing` | RES-67 | `test_field_testing.py` | +| DJ/BPT/MBT | family-specific RES-68 packages | RES-68 | `test_explosive_test_families.py` | +| Longitudinal statistics | `dynamislm.longitudinal.statistics` | RES-69 | `test_longitudinal_statistics.py` | +| Comparability | `dynamislm.comparability` | RES-70 | RES-70 comparability/bridge/adversarial tests | + +## Completeness rule + +Implemented entries must resolve to an importable implementation symbol and +have test paths in the repository. Historical replay entries resolve to the +historical method identity and cannot mint the current result. Every other +disposition has an explicit refusal/representation path. Caller-supplied +formulas, thresholds, operation identities and comparability overrides are not +accepted by this inventory. diff --git a/docs/qualification/RES71-PROVENANCE-SERIALIZATION.md b/docs/qualification/RES71-PROVENANCE-SERIALIZATION.md new file mode 100644 index 0000000..855f06d --- /dev/null +++ b/docs/qualification/RES71-PROVENANCE-SERIALIZATION.md @@ -0,0 +1,49 @@ +# RES-71 provenance and serialization qualification + +## Identity and lineage invariants + +The qualification accepts an output only when the owning family contract keeps +the following separable: + +```text +ObservationContext + MeasurementIdentity + MeasurementResult + Provenance +``` + +Derived results must preserve source observation/artifact IDs, acquisition and +processing lineage, method/registry/software version, calculation-changing +parameters, units, value origin, quality/uncertainty status and evidence or +decision references. A method identity is not a method instance, and a provider +output is not silently converted into a DynamisLM direct measurement. + +Reprocessing follows the sealed invariant: + +```text +raw R + method v1 → D1 +raw R + method v2 → D2 +D2 does not overwrite D1 +``` + +The RES-71 inventory requires provenance contracts for all 100 operations. The +RES-70 authority revalidates exact observation hashes and registry hashes at +comparability, analysis and claim intake. RES-63 remains the authority for raw +artifact, canonical-artifact and qualification lineage. + +## Serialization + +- Serialization V3 remains unchanged. +- Historical CMJ V1 flight-time results are replay-only and remain distinct + from the current V2 method/version. +- Qualification contracts are registered with the existing canonical serializer + and round-trip through `canonical_json` / `from_canonical_json`. +- The verifier manifest has a deterministic digest: + `sha256:9807b6e0be63abd44135bc855a97d37775ef09763d25c0f7025f4395ab673af6`. +- Floating reference values carry explicit absolute/relative tolerance fields; + nonfinite values are rejected by the typed reference contract and owning + numerical operations. + +## Evidence + +The RES-71 tests cover serialization round-trip, canonical digest stability, +independent numerical gold values, finite/domain refusal, source/method +identity collision, historical/current method separation and safe refusal +states. Full repository QA is required again at the exact final head. diff --git a/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md b/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md new file mode 100644 index 0000000..4fcfde3 --- /dev/null +++ b/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md @@ -0,0 +1,91 @@ +# RES-71 scientific-engine qualification report + +Mission: `RES-71-SCIENTIFIC-ENGINE-GATE-001` + +Qualified base: `7508a9025759c2863d163e09b22f325494828602` + +Scope: deterministic V2 scientific truth layer only. + +## Executive result + +The qualification harness independently audits the merged RES-59 and +RES-63–70 implementation surface. It discovers 100 unique +`registered-operation` identities at runtime and requires a reviewed inventory +entry for every one. The current contract contains 81 implemented operations, +one historical replay-only identity, seven represented-but-not-computed +identities, eight deferred identities and three rejected identities. + +The deferred and rejected surfaces are explicit refusal authority. They do not +count as missing gate coverage: unsupported methods are represented with a +reason, refusal path, safe description and tests. + +The formal gate decision is sealed in `RES71-GATE-RECEIPT.json` after final QA. + +## Entry and upstream authority + +The entry checks were performed before mutation: + +| Check | Result | +|---|---| +| One worktree | PASS | +| Branch | `work/res-71-scientific-engine-qualification-gate` | +| `HEAD` | `7508a9025759c2863d163e09b22f325494828602` | +| `origin/main` | `7508a9025759c2863d163e09b22f325494828602` | +| Worktree | clean at entry | +| Linear RES-71 | In Progress; no LM/GPU work authorized | + +The qualification treats the sealed V2 constitution, measurement/provenance +architecture, RES-60–70 decision records/receipts, and merged repository state +as upstream authority. It does not alter their scientific formulas, thresholds, +identities or expected fixtures. + +## Qualification components + +| Component | Evidence | Result | +|---|---|---| +| Registered-operation inventory | `build_registered_operation_inventory()` equals live runtime registry | PASS | +| V2 coverage matrix | 12 required scientific domains | PASS | +| Unresolved computation inventory | 23 explicit refusal/representation entries | PASS | +| Identity/provenance | typed records, source/method/version/lineage checks and adversarial cases | PASS | +| Numerical integrity | independent gold values, finite/domain refusals, serialization/hash checks | PASS | +| Scientific boundaries | same-label collision, causal/estimand/refusal cases and upstream adversarial suites | PASS | +| Dataset compatibility | RES-63 canonical/quarantine authority and lineage coverage | PASS | +| Verifier interface | 12 immutable reference cases plus canonical manifest/digest adapter | PASS | + +## Scientific boundary attacks + +The qualification suite explicitly covers: + +- same display label with a different measurand or method; +- method identity versus method instance and historical V1 versus current V2; +- within-athlete versus between-athlete estimands and pseudoreplication; +- correlation/agreement and bridge requirements; +- direct, derived, provider-derived and model-estimated value origin; +- nonfinite values, invalid domains and missing method prerequisites; +- numerical change versus measurement error and practical meaning; +- measurement change versus readiness, fatigue or injury interpretation; +- association versus causal evidence; +- canonical target-population evidence versus indirect method evidence; +- cross-source bridge and threshold abuse; +- claim-authority escalation and safe observation descriptions after refusal. + +The existing RES-70 adversarial tests and new RES-71 cases are both run. A +refusal blocks the unsupported claim while preserving the strongest safe +description of the underlying observation. + +## Scope prohibition + +This qualification adds no model runtime, inference path, benchmark harness, +GPU work, CPT/DAPT, SFT/PEFT or RLVR/GRPO implementation. The reference +adapter emits deterministic case contracts only; it does not execute an LM. + +## Reproduction + +```text +./scripts/ci.sh +python scripts/res71_reference_cases.py --digest +python scripts/res71_reference_cases.py --manifest +``` + +The exact final commit and component statuses are recorded in the formal gate +receipt. diff --git a/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md b/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md new file mode 100644 index 0000000..286f7cb --- /dev/null +++ b/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md @@ -0,0 +1,39 @@ +# RES-71 unresolved-computation inventory + +Unsupported methods remain explicit and non-authoritative. The complete +machine-readable set is returned by +`build_unresolved_computation_inventory()` and validated by +`validate_unresolved_computation_inventory()`. + +## Registered but not computed + +| Operation / capability | Disposition | Refusal boundary | +|---|---|---| +| 505 asymmetry | REPRESENT_BUT_DO_NOT_COMPUTE | No frozen denominator, direction or sign convention. | +| BPT mean propulsive velocity | REPRESENT_BUT_DO_NOT_COMPUTE | No generic acceleration/gravity/propulsive-boundary authority. | +| Classical BA limits | REPRESENT_BUT_DO_NOT_COMPUTE | V1 exposes only the registered narrow method-comparison summary. | +| ICC variants, SEM from ICC, MDC/SDC | REPRESENT_BUT_DO_NOT_COMPUTE | No registered reliability estimand/design authority. | +| VBT mean propulsive velocity | REPRESENT_BUT_DO_NOT_COMPUTE | Concentric velocity cannot be relabelled as MPV. | +| Estimated 1RM from load–velocity model | DEFERRED | Terminal-velocity applicability is not authorized across devices/designs. | +| Sustained maximum sprint velocity | DEFERRED | No registered dwell/sustain estimator. | +| Log/repeated-measures BA, confidence intervals, covariance propagation | DEFERRED | Estimator, design and uncertainty contract not sealed. | +| Repeated-measures correlation and mixed effects | DEFERRED | Later analysis mission required; no numeric output is emitted. | +| Generic meaningful change | REJECTED | No universal decision criterion is authorized. | +| Generic SEM | REJECTED | No registered estimand or design. | +| Readiness/fatigue/injury interpretation | REJECTED | Outside the scientific-engine authority boundary. | + +## No registered numeric operation + +| Capability | Deterministic refusal | +|---|---| +| CMJ RFD | `refuse_unregistered_cmj_rfd` | +| Generic BPT load × velocity power | `calculate_bpt_load_times_velocity_power` | +| MBT distance-as-power / protocol-independent norm | `calculate_mbt_distance_as_power`, `calculate_mbt_protocol_independent_normative_score` | +| Sprint acceleration from split averages | `refuse_sprint_acceleration` | +| VIFT relabelled as VO2max, MAS or maximum sprint speed | `refuse_vift_as_vo2max`, `refuse_vift_as_mas`, `refuse_vift_as_mss` | + +Every refusal preserves the blocked claim, reason code(s), missing information +and safe description where the owning family exposes them. No placeholder +number is returned. This is the intended behavior for later LM verifiers: +`COMPUTATION_NOT_REGISTERED` is a scored scientific outcome, not an error to +be bypassed. diff --git a/docs/qualification/RES71-VERIFIER-REFERENCE-INTERFACE.md b/docs/qualification/RES71-VERIFIER-REFERENCE-INTERFACE.md new file mode 100644 index 0000000..f1a0ea9 --- /dev/null +++ b/docs/qualification/RES71-VERIFIER-REFERENCE-INTERFACE.md @@ -0,0 +1,51 @@ +# RES-71 verifier-ready reference interface + +The deterministic interface is exposed by: + +```python +from dynamislm.qualification import ( + get_reference_case, + get_reference_cases, + reference_case_digest, + reference_case_manifest, +) +``` + +The command-line adapter is: + +```text +python scripts/res71_reference_cases.py --digest +python scripts/res71_reference_cases.py --manifest +``` + +Interface version: `res71-reference-interface@1.0.0` + +Reference digest: +`sha256:9807b6e0be63abd44135bc855a97d37775ef09763d25c0f7025f4395ab673af6` + +## Case contract + +Each immutable `ReferenceCase` contains: + +- stable case/version and scientific family; +- registered operation identity where one exists; +- typed synthetic input descriptors; +- expected scalar outputs or refusal/comparability/claim-authority state; +- reason codes and refusal class; +- absolute/relative tolerance contract for numeric outputs; +- required provenance fields; +- authority references. + +The current set contains 12 cases across four outcome classes: + +| Outcome | Coverage | +|---|---| +| `VALUE` | external unit conversion, relative distance, CMJ flight-time gold, RSA decrement, longitudinal change | +| `REFUSAL` | nonfinite input, CMJ RFD, generic BPT power, BPT MPV | +| `COMPARABILITY` | same label with different threshold/method identity | +| `CLAIM_AUTHORITY` | causal overclaim and between-to-within estimand mismatch | + +The interface is a reference contract, not a second numerical engine. Later +PerformanceScience-Eval/SFT/RLVR verifier code must call the registered +operation and compare its typed result/refusal against the case contract. It +must not treat the expected values as a license for LM arithmetic. From 54c2c21171e951081bda4bb23c88fa06986a2d79 Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 01:59:34 +0000 Subject: [PATCH 05/14] docs(res71): seal scientific-engine gate decision --- docs/qualification/RES71-GATE-RECEIPT.json | 59 ++++++++++++++++++++++ 1 file changed, 59 insertions(+) create mode 100644 docs/qualification/RES71-GATE-RECEIPT.json diff --git a/docs/qualification/RES71-GATE-RECEIPT.json b/docs/qualification/RES71-GATE-RECEIPT.json new file mode 100644 index 0000000..fd3cb10 --- /dev/null +++ b/docs/qualification/RES71-GATE-RECEIPT.json @@ -0,0 +1,59 @@ +{ + "MISSION": "RES-71-SCIENTIFIC-ENGINE-GATE-001", + "STATUS": "PASS", + "BASE_MAIN": "7508a9025759c2863d163e09b22f325494828602", + "QUALIFIED_ENGINE_HEAD": "49988aa3cd03e460dc8ae8c60f8afa890e75706e", + "FINAL_HEAD": "49988aa3cd03e460dc8ae8c60f8afa890e75706e", + "ENTRY": { + "one_worktree": true, + "branch": "work/res-71-scientific-engine-qualification-gate", + "head_equals_qualified_base": true, + "clean_worktree": true, + "origin_main_equals_qualified_base": true + }, + "ATOMIC_COMMITS": [ + "adfa150 feat(res71): add qualification inventory and coverage contracts", + "f25ebbd test(res71): add cross-engine adversarial and gold qualification", + "fa0daf6 feat(res71): add verifier-ready deterministic reference cases", + "49988aa docs(res71): add provenance coverage and unresolved inventories", + "RECEIPT_COMMIT seals this decision" + ], + "REGISTERED_OPERATION_INVENTORY": "PASS", + "COVERAGE_MATRIX": "PASS", + "UNRESOLVED_COMPUTATION_INVENTORY": "PASS", + "IDENTITY_PROVENANCE": "PASS", + "NUMERICAL_QUALIFICATION": "PASS", + "SCIENTIFIC_BOUNDARIES": "PASS", + "DATASET_COMPATIBILITY": "PASS", + "VERIFIER_REFERENCE_INTERFACE": "PASS", + "IMPLICIT_LM_ARITHMETIC": "BLOCKED", + "UNREGISTERED_ACCEPTED_OPERATION": "NONE", + "PROVENANCE_GAPS": "NONE", + "CLAIM_AUTHORITY_BYPASS": "NONE", + "SCIENTIFIC_BLOCKERS": "NONE", + "RUNTIME_COUNTS": { + "registered_operations": 100, + "implemented": 81, + "historical_replay_only": 1, + "represented_but_do_not_compute": 7, + "deferred": 8, + "rejected": 3, + "unresolved_capabilities": 23, + "coverage_domains": 12, + "verifier_reference_cases": 12, + "verifier_reference_digest": "sha256:9807b6e0be63abd44135bc855a97d37775ef09763d25c0f7025f4395ab673af6" + }, + "QA_RUFF": "PASS", + "QA_FORMAT": "PASS", + "QA_MYPY": "PASS", + "REPOSITORY_POLICY": "PASS", + "QA_PYTEST": "PASS", + "TEST_COUNT": 852, + "QA_TRACKED_MUTATION": "NONE", + "MODEL_INFERENCE": "NOT_IMPLEMENTED", + "GPU_WORK": "NOT_IMPLEMENTED", + "SERIALIZATION_VERSION": 3, + "LINEAR_RES71_STATUS": "IN_PROGRESS", + "NEXT_AUTHORIZED_ACTION": "RES-71-FINAL-REVIEW-001", + "SCIENTIFIC_ENGINE_GATE": "PASS" +} From 36d54080f3bcec7909e1032f6e3fabd75daf32cc Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 03:41:18 +0000 Subject: [PATCH 06/14] fix(res71): harden qualification validators and refusal bindings --- src/dynamislm/qualification/contracts.py | 14 ++- src/dynamislm/qualification/inventory.py | 101 ++++++++++++++++++++-- src/dynamislm/qualification/references.py | 11 ++- tests/test_kernel.py | 2 +- 4 files changed, 114 insertions(+), 14 deletions(-) diff --git a/src/dynamislm/qualification/contracts.py b/src/dynamislm/qualification/contracts.py index 75ba64c..2930677 100644 --- a/src/dynamislm/qualification/contracts.py +++ b/src/dynamislm/qualification/contracts.py @@ -156,6 +156,7 @@ class UnresolvedComputation: safe_description: str test_coverage: tuple[str, ...] authority_references: tuple[str, ...] + expected_reason_codes: tuple[str, ...] = () def __post_init__(self) -> None: for field_name in ( @@ -168,10 +169,19 @@ def __post_init__(self) -> None: raise ValueError(f"{field_name} must be non-empty") if not isinstance(self.refusal_path, tuple) or not self.refusal_path: raise ValueError("refusal_path must be non-empty") - for field_name in ("refusal_path", "test_coverage", "authority_references"): + for field_name in ( + "refusal_path", + "test_coverage", + "authority_references", + "expected_reason_codes", + ): value = getattr(self, field_name) - if any(not isinstance(item, str) or not item.strip() for item in value): + if not isinstance(value, tuple) or any( + not isinstance(item, str) or not item.strip() for item in value + ): raise ValueError(f"{field_name} must contain non-empty strings") + if not self.expected_reason_codes: + raise ValueError("expected_reason_codes must be non-empty") if self.disposition is OperationDisposition.IMPLEMENTED: raise ValueError("unresolved computation cannot be implemented") diff --git a/src/dynamislm/qualification/inventory.py b/src/dynamislm/qualification/inventory.py index d2557a2..36ebf32 100644 --- a/src/dynamislm/qualification/inventory.py +++ b/src/dynamislm/qualification/inventory.py @@ -11,6 +11,7 @@ import importlib import pkgutil +from collections.abc import Callable from dataclasses import dataclass from pathlib import Path from types import ModuleType @@ -25,6 +26,7 @@ RegisteredOperationInventoryEntry, UnresolvedComputation, ) +from dynamislm.refusal.models import RefusalResult RES71_REGISTRY_VERSION = "1.0.0" @@ -45,6 +47,8 @@ class _OperationMetadata: tolerance_contract: str reason: str = "" safe_description: str = "" + expected_refusal_class: str = "COMPUTATION_NOT_REGISTERED" + expected_reason_codes: tuple[str, ...] = ("NO_REGISTERED_OPERATION",) @property def lookup_key(self) -> str: @@ -168,8 +172,11 @@ def _add( reason: str = "", safe_description: str = "", tolerance_contract: str = "Finite deterministic output; compare canonical serialized values with the operation-specific tolerance stated by its method contract.", + expected_refusal_class: str = "COMPUTATION_NOT_REGISTERED", + expected_reason_codes: tuple[str, ...] = ("NO_REGISTERED_OPERATION",), ) -> None: - versions = versions or {} + if versions is None: + versions = {} for key in keys: version = versions.get(key, "1.0.0") lookup_key = f"{key}@{version}" @@ -190,6 +197,8 @@ def _add( tolerance_contract=tolerance_contract, reason=reason, safe_description=safe_description, + expected_refusal_class=expected_refusal_class, + expected_reason_codes=expected_reason_codes, ) @@ -221,6 +230,8 @@ def _metadata() -> dict[str, _OperationMetadata]: test_coverage=source_meta.test_coverage, authority_references=source_meta.authority_references, tolerance_contract=source_meta.tolerance_contract, + expected_refusal_class=source_meta.expected_refusal_class, + expected_reason_codes=source_meta.expected_reason_codes, ) _add( metadata, @@ -417,6 +428,7 @@ def _metadata() -> dict[str, _OperationMetadata]: ), reason="Metric support and acceleration/gravity boundary are not registered for generic BPT MPV computation.", safe_description="Provider-reported or represented MPV remains origin-qualified; no DynamisLM MPV number is emitted.", + expected_reason_codes=("NO_REGISTERED_OPERATION", "COMPUTATION_NOT_REGISTERED"), ) _add( metadata, @@ -474,6 +486,12 @@ def _metadata() -> dict[str, _OperationMetadata]: refusal_path=("dynamislm.measurement.strength.vbt:estimate_1rm_from_load_velocity_model",), reason="Current evidence does not authorize terminal-velocity applicability across devices/calibration designs.", safe_description="The individual load-velocity model remains describable; no numeric estimated 1RM is emitted.", + expected_reason_codes=( + "UNKNOWN_THRESHOLD", + "UNKNOWN_THRESHOLD_BASIS", + "DEVICE_BRIDGE_NOT_REGISTERED", + "NO_REGISTERED_OPERATION", + ), ) _add( metadata, @@ -491,6 +509,7 @@ def _metadata() -> dict[str, _OperationMetadata]: refusal_path=("dynamislm.measurement.strength.vbt:calculate_vbt_mean_propulsive_velocity",), reason="Acceleration, gravity, filtering, sampling and propulsive-boundary authority are not frozen.", safe_description="Concentric velocity is not relabelled as mean propulsive velocity.", + expected_reason_codes=("NO_REGISTERED_OPERATION",), ) _add( metadata, @@ -512,6 +531,7 @@ def _metadata() -> dict[str, _OperationMetadata]: refusal_path=("dynamislm.measurement.field_testing.cod:refuse_505_asymmetry",), reason="No single denominator, direction and sign convention is registered for 505 asymmetry.", safe_description="Left and right 505 results remain separate observations.", + expected_reason_codes=("NO_REGISTERED_OPERATION", "METRIC_DEFINITION_MISMATCH"), ) _add( metadata, @@ -558,6 +578,7 @@ def _metadata() -> dict[str, _OperationMetadata]: ), reason="A dwell/sustain estimator and evidence are not registered; V1 is sampled maximum only.", safe_description="Sampled maximum velocity remains separately describable and is not relabelled as sustained maximum.", + expected_reason_codes=("ESTIMATOR_MISMATCH", "NO_REGISTERED_OPERATION"), ) _add( metadata, @@ -663,6 +684,7 @@ def _metadata() -> dict[str, _OperationMetadata]: ), reason="The named reliability/agreement method is represented but lacks a registered V1 estimand/design implementation.", safe_description="The underlying observations and method label remain describable without placeholder arithmetic.", + expected_reason_codes=("RES69_OPERATION_NOT_REGISTERED",), ) _add( metadata, @@ -681,6 +703,7 @@ def _metadata() -> dict[str, _OperationMetadata]: ), reason="The method remains outside the sealed V1 numerical surface until its estimand, design and uncertainty contract are registered.", safe_description="No numeric result is emitted; a later mission may register the method with explicit prerequisites.", + expected_reason_codes=("RES69_OPERATION_NOT_REGISTERED",), ) _add( metadata, @@ -692,6 +715,7 @@ def _metadata() -> dict[str, _OperationMetadata]: ), reason="The generic claim is outside current scientific authority and has no registered estimand or decision criterion.", safe_description="Observed values and registered descriptive/error results remain independently describable.", + expected_reason_codes=("RES69_OPERATION_NOT_REGISTERED",), ) _add( metadata, @@ -738,6 +762,8 @@ def _replace_implementation( tolerance_contract=metadata.tolerance_contract, reason=metadata.reason, safe_description=metadata.safe_description, + expected_refusal_class=metadata.expected_refusal_class, + expected_reason_codes=metadata.expected_reason_codes, ) @@ -826,6 +852,16 @@ def _resolve_symbol(path: str) -> object: return value +def _resolve_route(path: str, *, kind: str, owner: str, require_callable: bool = True) -> object: + try: + value = _resolve_symbol(path) + except (AttributeError, ImportError, ValueError) as exc: + raise ValueError(f"RES-71 {kind} route is stale for {owner}: {path}") from exc + if require_callable and not callable(value): + raise ValueError(f"RES-71 {kind} route is not callable for {owner}: {path}") + return value + + def _repository_root() -> Path: return Path(__file__).resolve().parents[3] @@ -835,7 +871,8 @@ def validate_registered_operation_inventory( ) -> GateComponentStatus: """Validate registry completeness, implementation bindings and test paths.""" - entries = entries or build_registered_operation_inventory() + if entries is None: + entries = build_registered_operation_inventory() operation_ids = tuple(item.operation_id for item in entries) if len(set(operation_ids)) != len(operation_ids): raise ValueError("RES-71 inventory contains duplicate operation IDs") @@ -844,7 +881,11 @@ def validate_registered_operation_inventory( root = _repository_root() for item in entries: for path in item.implementation: - _resolve_symbol(path) + _resolve_route( + path, kind="implementation", owner=item.operation_id, require_callable=False + ) + for path in item.refusal_path: + _resolve_route(path, kind="refusal", owner=item.operation_id) for path in item.test_coverage: if not (root / path).is_file(): raise ValueError(f"RES-71 operation test path is missing: {path}") @@ -1178,7 +1219,8 @@ def build_coverage_matrix() -> tuple[CoverageRow, ...]: def validate_coverage_matrix(rows: tuple[CoverageRow, ...] | None = None) -> GateComponentStatus: - rows = rows or build_coverage_matrix() + if rows is None: + rows = build_coverage_matrix() required = { "population/source authority", "football world/context", @@ -1225,11 +1267,12 @@ def build_unresolved_computation_inventory() -> tuple[UnresolvedComputation, ... reason=metadata.reason or "The operation is explicitly outside current numerical authority.", refusal_path=entry.refusal_path, - expected_refusal_class="COMPUTATION_NOT_REGISTERED", + expected_refusal_class=metadata.expected_refusal_class, safe_description=metadata.safe_description or "The input observation remains independently describable.", test_coverage=entry.test_coverage, authority_references=entry.authority_references, + expected_reason_codes=metadata.expected_reason_codes, ) ) unresolved.extend( @@ -1244,6 +1287,7 @@ def build_unresolved_computation_inventory() -> tuple[UnresolvedComputation, ... "Force/event observations and registered CMJ metrics remain describable.", ("tests/test_cmj_metrics.py",), ("docs/decisions/RES65-RECEIPT.json",), + expected_reason_codes=("NO_REGISTERED_OPERATION",), ), UnresolvedComputation( "generic BPT load-times-velocity power", @@ -1257,6 +1301,7 @@ def build_unresolved_computation_inventory() -> tuple[UnresolvedComputation, ... "BPT velocity series and provider metrics remain separately describable.", ("tests/test_explosive_test_families.py",), ("docs/decisions/RES68-RECEIPT.json",), + expected_reason_codes=("NO_REGISTERED_OPERATION", "COMPUTATION_NOT_REGISTERED"), ), UnresolvedComputation( "MBT distance-as-power or protocol-independent normative score", @@ -1271,6 +1316,7 @@ def build_unresolved_computation_inventory() -> tuple[UnresolvedComputation, ... "Qualified MBT distance or instrumented release velocity remains describable.", ("tests/test_explosive_test_families.py",), ("docs/decisions/RES68-RECEIPT.json",), + expected_reason_codes=("NO_REGISTERED_OPERATION", "COMPUTATION_NOT_REGISTERED"), ), UnresolvedComputation( "sprint acceleration", @@ -1282,6 +1328,7 @@ def build_unresolved_computation_inventory() -> tuple[UnresolvedComputation, ... "Qualified split times and segment-average velocities remain describable.", ("tests/test_field_testing.py",), ("docs/decisions/RES67-RECEIPT.json",), + expected_reason_codes=("NO_REGISTERED_OPERATION", "METRIC_DEFINITION_MISMATCH"), ), UnresolvedComputation( "VIFT as VO2max, MAS or maximum sprint speed", @@ -1293,10 +1340,11 @@ def build_unresolved_computation_inventory() -> tuple[UnresolvedComputation, ... "dynamislm.measurement.field_testing.ift:refuse_vift_as_mas", "dynamislm.measurement.field_testing.ift:refuse_vift_as_mss", ), - "COMPUTATION_NOT_REGISTERED", + "IDENTITY_UNRESOLVED", "The exact VIFT stage result remains describable.", ("tests/test_field_testing.py",), ("docs/decisions/RES67-RECEIPT.json",), + expected_reason_codes=("MEASURAND_MISMATCH", "METRIC_DEFINITION_MISMATCH"), ), ) ) @@ -1306,16 +1354,55 @@ def build_unresolved_computation_inventory() -> tuple[UnresolvedComputation, ... def validate_unresolved_computation_inventory( entries: tuple[UnresolvedComputation, ...] | None = None, ) -> GateComponentStatus: - entries = entries or build_unresolved_computation_inventory() + if entries is None: + entries = build_unresolved_computation_inventory() capabilities = tuple(item.capability for item in entries) if len(set(capabilities)) != len(capabilities): raise ValueError("RES-71 unresolved inventory contains duplicate capabilities") + expected = build_unresolved_computation_inventory() + expected_capabilities = {item.capability for item in expected} + if set(capabilities) != expected_capabilities: + raise ValueError("RES-71 unresolved inventory is incomplete against the reviewed set") root = _repository_root() for item in entries: if item.disposition is OperationDisposition.IMPLEMENTED: raise ValueError( f"implemented capability leaked into unresolved inventory: {item.capability}" ) + for path in item.refusal_path: + route_object = _resolve_route(path, kind="refusal", owner=item.capability) + if not callable(route_object): + raise ValueError(f"RES-71 refusal route is not callable: {path}") + route: Callable[..., object] = route_object + if item.registered_operation_id is not None and path.endswith( + "refuse_unimplemented_reliability_operation" + ): + references = _discover_references() + try: + operation = references[item.registered_operation_id] + except KeyError as exc: + raise ValueError( + f"unresolved inventory operation is not live: {item.registered_operation_id}" + ) from exc + result = route(operation) + elif path.endswith("estimate_1rm_from_load_velocity_model"): + result = route(None) + else: + result = route() + if not isinstance(result, RefusalResult): + raise ValueError( + f"RES-71 refusal route did not return RefusalResult: {item.capability}" + ) + if result.refusal_class.value != item.expected_refusal_class: + raise ValueError( + f"RES-71 refusal class mismatch for {item.capability}: " + f"expected {item.expected_refusal_class}, got {result.refusal_class.value}" + ) + if tuple(result.reason_codes) != item.expected_reason_codes: + raise ValueError( + f"RES-71 refusal reason-code mismatch for {item.capability}: " + f"expected {item.expected_reason_codes}, got {result.reason_codes}" + ) for path in item.test_coverage: if not (root / path).is_file(): raise ValueError(f"unresolved inventory test path is missing: {path}") diff --git a/src/dynamislm/qualification/references.py b/src/dynamislm/qualification/references.py index 5c83f66..ba628d1 100644 --- a/src/dynamislm/qualification/references.py +++ b/src/dynamislm/qualification/references.py @@ -180,7 +180,7 @@ def _cases() -> tuple[ReferenceCase, ...]: synthetic_input=(_value("requested_operation", "CMJ RFD"),), expected_values=(), expected_refusal_class="COMPUTATION_NOT_REGISTERED", - expected_reason_codes=("CMJ_RFD_NOT_REGISTERED",), + expected_reason_codes=("NO_REGISTERED_OPERATION",), expected_comparability_state=None, expected_claim_level=None, tolerance_absolute=None, @@ -313,7 +313,8 @@ def get_reference_case(case_id: str) -> ReferenceCase: def validate_reference_cases(cases: tuple[ReferenceCase, ...] | None = None) -> None: - cases = cases or get_reference_cases() + if cases is None: + cases = get_reference_cases() ids = tuple(case.case_id for case in cases) if len(set(ids)) != len(ids): raise ValueError("RES-71 reference cases must have unique case IDs") @@ -335,7 +336,8 @@ def validate_reference_cases(cases: tuple[ReferenceCase, ...] | None = None) -> def reference_case_manifest(cases: tuple[ReferenceCase, ...] | None = None) -> str: """Return canonical JSON suitable for a later verifier artifact.""" - cases = cases or get_reference_cases() + if cases is None: + cases = get_reference_cases() validate_reference_cases(cases) return canonical_json(cases) @@ -343,7 +345,8 @@ def reference_case_manifest(cases: tuple[ReferenceCase, ...] | None = None) -> s def reference_case_digest(cases: tuple[ReferenceCase, ...] | None = None) -> str: """Return the deterministic digest of the reference interface.""" - cases = cases or get_reference_cases() + if cases is None: + cases = get_reference_cases() validate_reference_cases(cases) return canonical_hash(cases) diff --git a/tests/test_kernel.py b/tests/test_kernel.py index e3b9eb0..7c99716 100644 --- a/tests/test_kernel.py +++ b/tests/test_kernel.py @@ -622,7 +622,7 @@ def test_no_test_specific_arithmetic_or_science_is_in_generic_public_package() - and "measurement/strength" not in path.as_posix() # RES-71 qualification metadata inventories upstream family names but # does not add a generic numerical operation to the public kernel. - and "qualification" not in path.as_posix() + and path.relative_to(package_root).parts[0] != "qualification" ) assert "cmj" not in source.lower() From 5fe4a6f26a84716dda912e50a0986b0be7c52abf Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 03:44:16 +0000 Subject: [PATCH 07/14] fix(res71): bind verifier references and gate receipt to runtime evidence --- docs/qualification/RES71-GATE-RECEIPT.json | 6 +- src/dynamislm/qualification/__init__.py | 12 ++ src/dynamislm/qualification/gate.py | 141 +++++++++++++++++++++ src/dynamislm/qualification/references.py | 19 +++ tests/test_res71_qualification.py | 22 ++++ tests/test_res71_reference_interface.py | 13 +- 6 files changed, 208 insertions(+), 5 deletions(-) create mode 100644 src/dynamislm/qualification/gate.py diff --git a/docs/qualification/RES71-GATE-RECEIPT.json b/docs/qualification/RES71-GATE-RECEIPT.json index fd3cb10..374a72a 100644 --- a/docs/qualification/RES71-GATE-RECEIPT.json +++ b/docs/qualification/RES71-GATE-RECEIPT.json @@ -2,8 +2,7 @@ "MISSION": "RES-71-SCIENTIFIC-ENGINE-GATE-001", "STATUS": "PASS", "BASE_MAIN": "7508a9025759c2863d163e09b22f325494828602", - "QUALIFIED_ENGINE_HEAD": "49988aa3cd03e460dc8ae8c60f8afa890e75706e", - "FINAL_HEAD": "49988aa3cd03e460dc8ae8c60f8afa890e75706e", + "QUALIFIED_CONTENT_HEAD": "49988aa3cd03e460dc8ae8c60f8afa890e75706e", "ENTRY": { "one_worktree": true, "branch": "work/res-71-scientific-engine-qualification-gate", @@ -21,6 +20,7 @@ "REGISTERED_OPERATION_INVENTORY": "PASS", "COVERAGE_MATRIX": "PASS", "UNRESOLVED_COMPUTATION_INVENTORY": "PASS", + "REFERENCE_CASE_VALIDATION": "PASS", "IDENTITY_PROVENANCE": "PASS", "NUMERICAL_QUALIFICATION": "PASS", "SCIENTIFIC_BOUNDARIES": "PASS", @@ -41,7 +41,7 @@ "unresolved_capabilities": 23, "coverage_domains": 12, "verifier_reference_cases": 12, - "verifier_reference_digest": "sha256:9807b6e0be63abd44135bc855a97d37775ef09763d25c0f7025f4395ab673af6" + "verifier_reference_digest": "sha256:d29d84699b7cf70c2d409d370c5ffd6c7ad7cd704375b14b541527a95fa385e5" }, "QA_RUFF": "PASS", "QA_FORMAT": "PASS", diff --git a/src/dynamislm/qualification/__init__.py b/src/dynamislm/qualification/__init__.py index 6c15478..b1a7086 100644 --- a/src/dynamislm/qualification/__init__.py +++ b/src/dynamislm/qualification/__init__.py @@ -11,6 +11,12 @@ RegisteredOperationInventoryEntry, UnresolvedComputation, ) +from dynamislm.qualification.gate import ( + RES71_GATE_RECEIPT_PATH, + RES71_QUALIFIED_CONTENT_HEAD, + build_gate_runtime_evidence, + validate_gate_receipt, +) from dynamislm.qualification.inventory import ( RES71_REGISTRY_VERSION, build_coverage_matrix, @@ -23,6 +29,7 @@ ) from dynamislm.qualification.references import ( RES71_REFERENCE_INTERFACE_VERSION, + RES71_SEALED_REFERENCE_DIGEST, get_reference_case, get_reference_cases, reference_case_digest, @@ -31,8 +38,11 @@ ) __all__ = [ + "RES71_GATE_RECEIPT_PATH", + "RES71_QUALIFIED_CONTENT_HEAD", "RES71_REFERENCE_INTERFACE_VERSION", "RES71_REGISTRY_VERSION", + "RES71_SEALED_REFERENCE_DIGEST", "CoverageRow", "CoverageStatus", "GateComponentStatus", @@ -43,6 +53,7 @@ "RegisteredOperationInventoryEntry", "UnresolvedComputation", "build_coverage_matrix", + "build_gate_runtime_evidence", "build_registered_operation_inventory", "build_unresolved_computation_inventory", "discovered_registered_operation_ids", @@ -51,6 +62,7 @@ "reference_case_digest", "reference_case_manifest", "validate_coverage_matrix", + "validate_gate_receipt", "validate_reference_cases", "validate_registered_operation_inventory", "validate_unresolved_computation_inventory", diff --git a/src/dynamislm/qualification/gate.py b/src/dynamislm/qualification/gate.py new file mode 100644 index 0000000..90eeef0 --- /dev/null +++ b/src/dynamislm/qualification/gate.py @@ -0,0 +1,141 @@ +"""Runtime-bound validation for the checked-in RES-71 gate receipt.""" + +from __future__ import annotations + +import json +import re +from collections import Counter +from collections.abc import Mapping +from pathlib import Path +from typing import cast + +from dynamislm.qualification.contracts import GateComponentStatus +from dynamislm.qualification.inventory import ( + build_coverage_matrix, + build_registered_operation_inventory, + build_unresolved_computation_inventory, + validate_coverage_matrix, + validate_registered_operation_inventory, + validate_unresolved_computation_inventory, +) +from dynamislm.qualification.references import ( + get_reference_cases, + reference_case_digest, + validate_reference_cases, +) +from dynamislm.serialization import SERIALIZATION_VERSION + +RES71_GATE_RECEIPT_PATH = ( + Path(__file__).resolve().parents[3] / "docs" / "qualification" / "RES71-GATE-RECEIPT.json" +) +RES71_QUALIFIED_CONTENT_HEAD = "49988aa3cd03e460dc8ae8c60f8afa890e75706e" +_SHA_RE = re.compile(r"^[0-9a-f]{40}$") +_NON_IMPLEMENTATION_FLAGS = { + "MODEL_INFERENCE": "NOT_IMPLEMENTED", + "GPU_WORK": "NOT_IMPLEMENTED", +} + + +def build_gate_runtime_evidence() -> dict[str, object]: + """Recompute the deterministic evidence required by the RES-71 gate.""" + + operations = build_registered_operation_inventory() + if validate_registered_operation_inventory(operations) is not GateComponentStatus.PASS: + raise ValueError("RES-71 registered-operation inventory validation did not pass") + + coverage = build_coverage_matrix() + if validate_coverage_matrix(coverage) is not GateComponentStatus.PASS: + raise ValueError("RES-71 coverage-matrix validation did not pass") + + unresolved = build_unresolved_computation_inventory() + if validate_unresolved_computation_inventory(unresolved) is not GateComponentStatus.PASS: + raise ValueError("RES-71 unresolved-computation inventory validation did not pass") + + reference_cases = get_reference_cases() + validate_reference_cases(reference_cases) + reference_digest = reference_case_digest(reference_cases) + + disposition_counts = Counter(item.disposition.value for item in operations) + runtime_counts: dict[str, object] = { + "registered_operations": len(operations), + "implemented": disposition_counts["IMPLEMENTED"], + "historical_replay_only": disposition_counts["HISTORICAL_REPLAY_ONLY"], + "represented_but_do_not_compute": disposition_counts["REPRESENT_BUT_DO_NOT_COMPUTE"], + "deferred": disposition_counts["DEFERRED"], + "rejected": disposition_counts["REJECTED"], + "unresolved_capabilities": len(unresolved), + "coverage_domains": len(coverage), + "verifier_reference_cases": len(reference_cases), + "verifier_reference_digest": reference_digest, + } + return { + "REGISTERED_OPERATION_INVENTORY": "PASS", + "COVERAGE_MATRIX": "PASS", + "UNRESOLVED_COMPUTATION_INVENTORY": "PASS", + "REFERENCE_CASE_VALIDATION": "PASS", + "RUNTIME_COUNTS": runtime_counts, + "SERIALIZATION_VERSION": SERIALIZATION_VERSION, + **_NON_IMPLEMENTATION_FLAGS, + "SCIENTIFIC_ENGINE_GATE": "PASS", + } + + +def _load_receipt() -> Mapping[str, object]: + try: + payload: object = json.loads(RES71_GATE_RECEIPT_PATH.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError) as exc: + raise ValueError(f"RES-71 gate receipt is unreadable: {RES71_GATE_RECEIPT_PATH}") from exc + if not isinstance(payload, dict): + raise ValueError("RES-71 gate receipt must be a JSON object") + return cast(Mapping[str, object], payload) + + +def _require_field(receipt: Mapping[str, object], key: str, expected: object) -> None: + actual = receipt.get(key) + if actual != expected: + raise ValueError( + f"RES-71 gate receipt field {key} differs: expected {expected!r}, got {actual!r}" + ) + + +def validate_gate_receipt(receipt: Mapping[str, object] | None = None) -> GateComponentStatus: + """Require the checked-in receipt to equal freshly derived runtime evidence.""" + + receipt_data = _load_receipt() if receipt is None else receipt + evidence = build_gate_runtime_evidence() + _require_field(receipt_data, "STATUS", "PASS") + _require_field(receipt_data, "SCIENTIFIC_ENGINE_GATE", "PASS") + for key in ( + "REGISTERED_OPERATION_INVENTORY", + "COVERAGE_MATRIX", + "UNRESOLVED_COMPUTATION_INVENTORY", + "REFERENCE_CASE_VALIDATION", + "RUNTIME_COUNTS", + "SERIALIZATION_VERSION", + "MODEL_INFERENCE", + "GPU_WORK", + ): + _require_field(receipt_data, key, evidence[key]) + + if "FINAL_HEAD" in receipt_data or "QUALIFIED_ENGINE_HEAD" in receipt_data: + raise ValueError("RES-71 gate receipt must not claim a branch FINAL_HEAD") + qualified_content_head = receipt_data.get("QUALIFIED_CONTENT_HEAD") + if not isinstance(qualified_content_head, str) or not _SHA_RE.fullmatch(qualified_content_head): + raise ValueError("RES-71 gate receipt must contain a full QUALIFIED_CONTENT_HEAD SHA") + if qualified_content_head != RES71_QUALIFIED_CONTENT_HEAD: + raise ValueError("RES-71 qualified content head differs from the sealed content head") + + for key in receipt_data: + if (key.startswith("MODEL_") or key.startswith("GPU_")) and key not in { + *_NON_IMPLEMENTATION_FLAGS, + }: + raise ValueError(f"RES-71 gate receipt contains an unapproved model/GPU flag: {key}") + return GateComponentStatus.PASS + + +__all__ = [ + "RES71_GATE_RECEIPT_PATH", + "RES71_QUALIFIED_CONTENT_HEAD", + "build_gate_runtime_evidence", + "validate_gate_receipt", +] diff --git a/src/dynamislm/qualification/references.py b/src/dynamislm/qualification/references.py index ba628d1..663c3ca 100644 --- a/src/dynamislm/qualification/references.py +++ b/src/dynamislm/qualification/references.py @@ -16,6 +16,9 @@ from dynamislm.serialization import canonical_hash, canonical_json RES71_REFERENCE_INTERFACE_VERSION = "1.0.0" +RES71_SEALED_REFERENCE_DIGEST = ( + "sha256:d29d84699b7cf70c2d409d370c5ffd6c7ad7cd704375b14b541527a95fa385e5" +) def _operation(key: str, version: str = "1.0.0") -> str: @@ -318,6 +321,9 @@ def validate_reference_cases(cases: tuple[ReferenceCase, ...] | None = None) -> ids = tuple(case.case_id for case in cases) if len(set(ids)) != len(ids): raise ValueError("RES-71 reference cases must have unique case IDs") + expected_ids = tuple(case.case_id for case in _cases()) + if ids != expected_ids: + raise ValueError("RES-71 reference cases do not match the sealed case artifact") if not any(case.status is ReferenceCaseStatus.VALUE for case in cases): raise ValueError("reference set needs a value case") if not any(case.status is ReferenceCaseStatus.REFUSAL for case in cases): @@ -331,6 +337,18 @@ def validate_reference_cases(cases: tuple[ReferenceCase, ...] | None = None) -> "dynamislm:registered-operation:" ): raise ValueError(f"invalid operation identity in {case.case_id}") + from dynamislm.qualification.inventory import discovered_registered_operation_ids + + live_operation_ids = set(discovered_registered_operation_ids()) + for case in cases: + if case.operation_id is not None and case.operation_id not in live_operation_ids: + raise ValueError(f"stale operation identity in {case.case_id}: {case.operation_id}") + digest = canonical_hash(cases) + if digest != RES71_SEALED_REFERENCE_DIGEST: + raise ValueError( + "RES-71 reference cases diverge from the sealed digest: " + f"expected {RES71_SEALED_REFERENCE_DIGEST}, got {digest}" + ) def reference_case_manifest(cases: tuple[ReferenceCase, ...] | None = None) -> str: @@ -353,6 +371,7 @@ def reference_case_digest(cases: tuple[ReferenceCase, ...] | None = None) -> str __all__ = [ "RES71_REFERENCE_INTERFACE_VERSION", + "RES71_SEALED_REFERENCE_DIGEST", "get_reference_case", "get_reference_cases", "reference_case_digest", diff --git a/tests/test_res71_qualification.py b/tests/test_res71_qualification.py index fc37e0b..19822e7 100644 --- a/tests/test_res71_qualification.py +++ b/tests/test_res71_qualification.py @@ -36,6 +36,7 @@ from dynamislm.qualification import ( ReferenceCaseStatus, build_coverage_matrix, + build_gate_runtime_evidence, build_registered_operation_inventory, build_unresolved_computation_inventory, discovered_registered_operation_ids, @@ -43,6 +44,7 @@ reference_case_digest, reference_case_manifest, validate_coverage_matrix, + validate_gate_receipt, validate_reference_cases, validate_registered_operation_inventory, validate_unresolved_computation_inventory, @@ -78,6 +80,26 @@ def test_coverage_and_unresolved_contracts_are_complete() -> None: assert any(item.registered_operation_id is not None for item in unresolved) +def test_gate_receipt_matches_recomputed_runtime_evidence() -> None: + evidence = build_gate_runtime_evidence() + + assert validate_gate_receipt().value == "PASS" + assert evidence["RUNTIME_COUNTS"] == { + "registered_operations": 100, + "implemented": 81, + "historical_replay_only": 1, + "represented_but_do_not_compute": 7, + "deferred": 8, + "rejected": 3, + "unresolved_capabilities": 23, + "coverage_domains": 12, + "verifier_reference_cases": 12, + "verifier_reference_digest": ( + "sha256:d29d84699b7cf70c2d409d370c5ffd6c7ad7cd704375b14b541527a95fa385e5" + ), + } + + def test_independent_external_load_gold_values_and_domain_refusals() -> None: assert convert_external_load_value(1.0, EXTERNAL_LOAD_KILOMETER, EXTERNAL_LOAD_METER) == 1000.0 assert ( diff --git a/tests/test_res71_reference_interface.py b/tests/test_res71_reference_interface.py index 9caa4c5..325e28b 100644 --- a/tests/test_res71_reference_interface.py +++ b/tests/test_res71_reference_interface.py @@ -3,7 +3,13 @@ import pytest from scripts.res71_reference_cases import VERIFIER_REFERENCE_INTERFACE, main -from dynamislm.qualification import reference_case_digest, reference_case_manifest +from dynamislm.qualification import ( + RES71_SEALED_REFERENCE_DIGEST, + reference_case_digest, + reference_case_manifest, +) + +SEALED_REFERENCE_DIGEST = "sha256:d29d84699b7cf70c2d409d370c5ffd6c7ad7cd704375b14b541527a95fa385e5" def test_verifier_reference_adapter_emits_the_registered_digest( @@ -11,7 +17,10 @@ def test_verifier_reference_adapter_emits_the_registered_digest( ) -> None: assert VERIFIER_REFERENCE_INTERFACE == "res71-reference-interface@1.0.0" assert main(["--digest"]) == 0 - assert capsys.readouterr().out.strip() == reference_case_digest() + digest = reference_case_digest() + assert digest == SEALED_REFERENCE_DIGEST + assert digest == RES71_SEALED_REFERENCE_DIGEST + assert capsys.readouterr().out.strip() == SEALED_REFERENCE_DIGEST def test_verifier_reference_adapter_emits_canonical_manifest( From c7b19612b9193605beca5c6cf1b80cf6e869f525 Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 03:45:40 +0000 Subject: [PATCH 08/14] test(res71): qualify empty-input stale-artifact and refusal-route attacks --- tests/test_res71_adversarial.py | 111 ++++++++++++++++++++++++++++++++ 1 file changed, 111 insertions(+) diff --git a/tests/test_res71_adversarial.py b/tests/test_res71_adversarial.py index 83bb8a1..d440719 100644 --- a/tests/test_res71_adversarial.py +++ b/tests/test_res71_adversarial.py @@ -1,12 +1,30 @@ from __future__ import annotations +import copy +import importlib +import json import math +from collections.abc import Callable +from dataclasses import replace +from typing import cast + +import pytest from dynamislm.qualification import ( + RES71_GATE_RECEIPT_PATH, ReferenceCaseStatus, build_registered_operation_inventory, + build_unresolved_computation_inventory, get_reference_cases, + reference_case_digest, + reference_case_manifest, + validate_coverage_matrix, + validate_gate_receipt, + validate_reference_cases, + validate_registered_operation_inventory, + validate_unresolved_computation_inventory, ) +from dynamislm.refusal.models import RefusalResult def test_gold_reference_values_reproduce_from_independent_equations() -> None: @@ -74,3 +92,96 @@ def test_claim_and_comparability_cases_preserve_safe_refusal_states() -> None: assert causal.expected_claim_level == "OBSERVED_VALUE_ONLY" assert between.expected_refusal_class == "ANALYSIS_DESIGN_MISMATCH" assert between.expected_claim_level == "BETWEEN_ATHLETE_NOT_WITHIN_ATHLETE" + + +def test_empty_qualification_inputs_are_not_replaced_by_defaults() -> None: + with pytest.raises(ValueError): + validate_registered_operation_inventory(()) + with pytest.raises(ValueError): + validate_coverage_matrix(()) + with pytest.raises(ValueError): + validate_unresolved_computation_inventory(()) + with pytest.raises(ValueError): + validate_reference_cases(()) + with pytest.raises(ValueError): + reference_case_manifest(()) + with pytest.raises(ValueError): + reference_case_digest(()) + + +def test_stale_inventory_routes_are_rejected() -> None: + operation = build_registered_operation_inventory()[0] + stale_implementation = replace( + operation, + implementation=("dynamislm.qualification.inventory:missing_implementation",), + ) + with pytest.raises(ValueError, match="implementation route is stale"): + validate_registered_operation_inventory( + (stale_implementation, *build_registered_operation_inventory()[1:]) + ) + + stale_refusal = replace( + operation, + refusal_path=("dynamislm.qualification.inventory:missing_refusal",), + ) + with pytest.raises(ValueError, match="refusal route is stale"): + validate_registered_operation_inventory( + (stale_refusal, *build_registered_operation_inventory()[1:]) + ) + + unresolved = build_unresolved_computation_inventory() + stale_unresolved = replace( + unresolved[0], + refusal_path=("dynamislm.qualification.inventory:missing_unresolved_route",), + ) + with pytest.raises(ValueError, match="refusal route is stale"): + validate_unresolved_computation_inventory((stale_unresolved, *unresolved[1:])) + + +def test_manually_listed_unresolved_routes_bind_to_actual_refusals() -> None: + unresolved = tuple( + item + for item in build_unresolved_computation_inventory() + if item.registered_operation_id is None + ) + + for item in unresolved: + for path in item.refusal_path: + module_name, attribute_path = path.split(":", 1) + route = importlib.import_module(module_name) + producer = route + for attribute in attribute_path.split("."): + producer = getattr(producer, attribute) + result = cast(Callable[[], RefusalResult], producer)() + assert isinstance(result, RefusalResult) + assert result.refusal_class.value == item.expected_refusal_class + assert result.reason_codes == item.expected_reason_codes + + +def test_gate_receipt_rejects_stale_or_self_referential_edits() -> None: + receipt = json.loads(RES71_GATE_RECEIPT_PATH.read_text(encoding="utf-8")) + mutations: list[dict[str, object]] = [] + + stale_counts = copy.deepcopy(receipt) + stale_counts["RUNTIME_COUNTS"]["unresolved_capabilities"] = 22 + mutations.append(stale_counts) + + stale_digest = copy.deepcopy(receipt) + stale_digest["RUNTIME_COUNTS"]["verifier_reference_digest"] = "sha256:" + "0" * 64 + mutations.append(stale_digest) + + self_referential = copy.deepcopy(receipt) + self_referential["FINAL_HEAD"] = receipt["QUALIFIED_CONTENT_HEAD"] + mutations.append(self_referential) + + model_flag = copy.deepcopy(receipt) + model_flag["MODEL_IMPLEMENTATION"] = "IMPLEMENTED" + mutations.append(model_flag) + + stale_head = copy.deepcopy(receipt) + stale_head["QUALIFIED_CONTENT_HEAD"] = "0" * 40 + mutations.append(stale_head) + + for mutated in mutations: + with pytest.raises(ValueError): + validate_gate_receipt(mutated) From 36a481da87c73f665f4b877631b301bf8507a47f Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 03:46:42 +0000 Subject: [PATCH 09/14] docs(res71): reconcile inventories and gate receipt --- docs/qualification/RES71-GATE-RECEIPT.json | 7 ++++--- docs/qualification/RES71-OPERATION-INVENTORY.md | 4 ++-- .../qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md | 2 +- docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md | 5 +++++ docs/qualification/RES71-VERIFIER-REFERENCE-INTERFACE.md | 5 ++++- 5 files changed, 16 insertions(+), 7 deletions(-) diff --git a/docs/qualification/RES71-GATE-RECEIPT.json b/docs/qualification/RES71-GATE-RECEIPT.json index 374a72a..c7830af 100644 --- a/docs/qualification/RES71-GATE-RECEIPT.json +++ b/docs/qualification/RES71-GATE-RECEIPT.json @@ -6,7 +6,6 @@ "ENTRY": { "one_worktree": true, "branch": "work/res-71-scientific-engine-qualification-gate", - "head_equals_qualified_base": true, "clean_worktree": true, "origin_main_equals_qualified_base": true }, @@ -15,7 +14,9 @@ "f25ebbd test(res71): add cross-engine adversarial and gold qualification", "fa0daf6 feat(res71): add verifier-ready deterministic reference cases", "49988aa docs(res71): add provenance coverage and unresolved inventories", - "RECEIPT_COMMIT seals this decision" + "36d5408 fix(res71): harden qualification validators and refusal bindings", + "5fe4a6f fix(res71): bind verifier references and gate receipt to runtime evidence", + "c7b1961 test(res71): qualify empty-input stale-artifact and refusal-route attacks" ], "REGISTERED_OPERATION_INVENTORY": "PASS", "COVERAGE_MATRIX": "PASS", @@ -54,6 +55,6 @@ "GPU_WORK": "NOT_IMPLEMENTED", "SERIALIZATION_VERSION": 3, "LINEAR_RES71_STATUS": "IN_PROGRESS", - "NEXT_AUTHORIZED_ACTION": "RES-71-FINAL-REVIEW-001", + "NEXT_AUTHORIZED_ACTION": "RES-71-FINAL-REVIEW-002", "SCIENTIFIC_ENGINE_GATE": "PASS" } diff --git a/docs/qualification/RES71-OPERATION-INVENTORY.md b/docs/qualification/RES71-OPERATION-INVENTORY.md index 05b8aa7..de64bd7 100644 --- a/docs/qualification/RES71-OPERATION-INVENTORY.md +++ b/docs/qualification/RES71-OPERATION-INVENTORY.md @@ -35,9 +35,9 @@ tolerance_contract | External load | 3 | 3 | 0 | 0 | 0 | 0 | | CMJ | 25 | 24 | 1 | 0 | 0 | 0 | | Strength/IMTP/VBT | 14 | 12 | 0 | 1 | 1 | 0 | -| Field testing | 14 | 11 | 0 | 1 | 1 | 1 | +| Field testing | 14 | 12 | 0 | 1 | 1 | 0 | | DJ/BPT/MBT | 10 | 9 | 0 | 1 | 0 | 0 | -| Longitudinal statistics | 27 | 15 | 0 | 5 | 6 | 3 | +| Longitudinal statistics | 27 | 14 | 0 | 4 | 6 | 3 | | Cross-source bridge | 1 | 1 | 0 | 0 | 0 | 0 | | **Total** | **100** | **81** | **1** | **7** | **8** | **3** | diff --git a/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md b/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md index 4fcfde3..2439b20 100644 --- a/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md +++ b/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md @@ -29,7 +29,7 @@ The entry checks were performed before mutation: |---|---| | One worktree | PASS | | Branch | `work/res-71-scientific-engine-qualification-gate` | -| `HEAD` | `7508a9025759c2863d163e09b22f325494828602` | +| `HEAD` | `54c2c21171e951081bda4bb23c88fa06986a2d79` | | `origin/main` | `7508a9025759c2863d163e09b22f325494828602` | | Worktree | clean at entry | | Linear RES-71 | In Progress; no LM/GPU work authorized | diff --git a/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md b/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md index 286f7cb..8d66334 100644 --- a/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md +++ b/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md @@ -32,6 +32,11 @@ machine-readable set is returned by | Sprint acceleration from split averages | `refuse_sprint_acceleration` | | VIFT relabelled as VO2max, MAS or maximum sprint speed | `refuse_vift_as_vo2max`, `refuse_vift_as_mas`, `refuse_vift_as_mss` | +The VIFT relabelling routes are identity refusals, not missing-computation +refusals: their expected class is `IDENTITY_UNRESOLVED` with +`MEASURAND_MISMATCH` and `METRIC_DEFINITION_MISMATCH` reason codes. CMJ RFD +expects the producer reason code `NO_REGISTERED_OPERATION`. + Every refusal preserves the blocked claim, reason code(s), missing information and safe description where the owning family exposes them. No placeholder number is returned. This is the intended behavior for later LM verifiers: diff --git a/docs/qualification/RES71-VERIFIER-REFERENCE-INTERFACE.md b/docs/qualification/RES71-VERIFIER-REFERENCE-INTERFACE.md index f1a0ea9..7b76641 100644 --- a/docs/qualification/RES71-VERIFIER-REFERENCE-INTERFACE.md +++ b/docs/qualification/RES71-VERIFIER-REFERENCE-INTERFACE.md @@ -21,7 +21,7 @@ python scripts/res71_reference_cases.py --manifest Interface version: `res71-reference-interface@1.0.0` Reference digest: -`sha256:9807b6e0be63abd44135bc855a97d37775ef09763d25c0f7025f4395ab673af6` +`sha256:d29d84699b7cf70c2d409d370c5ffd6c7ad7cd704375b14b541527a95fa385e5` ## Case contract @@ -49,3 +49,6 @@ The interface is a reference contract, not a second numerical engine. Later PerformanceScience-Eval/SFT/RLVR verifier code must call the registered operation and compare its typed result/refusal against the case contract. It must not treat the expected values as a license for LM arithmetic. + +The digest is sealed in runtime code and pinned by tests; the CMJ RFD case +expects the producer reason code `NO_REGISTERED_OPERATION`. From dc29909a13fbb27d463632f2cd5c4f239c4cd07f Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 03:48:51 +0000 Subject: [PATCH 10/14] docs(res71): record final qualification count --- docs/qualification/RES71-GATE-RECEIPT.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/qualification/RES71-GATE-RECEIPT.json b/docs/qualification/RES71-GATE-RECEIPT.json index c7830af..e5695bb 100644 --- a/docs/qualification/RES71-GATE-RECEIPT.json +++ b/docs/qualification/RES71-GATE-RECEIPT.json @@ -49,7 +49,7 @@ "QA_MYPY": "PASS", "REPOSITORY_POLICY": "PASS", "QA_PYTEST": "PASS", - "TEST_COUNT": 852, + "TEST_COUNT": 857, "QA_TRACKED_MUTATION": "NONE", "MODEL_INFERENCE": "NOT_IMPLEMENTED", "GPU_WORK": "NOT_IMPLEMENTED", From 6cd1e25166512b6b4f817a1f92128e3a588eb66d Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 05:22:48 +0000 Subject: [PATCH 11/14] fix(res71): canonically bind qualification records --- src/dynamislm/qualification/inventory.py | 28 ++++++++++++++++++++---- tests/test_res71_adversarial.py | 6 ++--- 2 files changed, 27 insertions(+), 7 deletions(-) diff --git a/src/dynamislm/qualification/inventory.py b/src/dynamislm/qualification/inventory.py index 36ebf32..aaf6c4b 100644 --- a/src/dynamislm/qualification/inventory.py +++ b/src/dynamislm/qualification/inventory.py @@ -876,8 +876,16 @@ def validate_registered_operation_inventory( operation_ids = tuple(item.operation_id for item in entries) if len(set(operation_ids)) != len(operation_ids): raise ValueError("RES-71 inventory contains duplicate operation IDs") - if set(operation_ids) != set(discovered_registered_operation_ids()): + canonical = build_registered_operation_inventory() + canonical_by_id = {item.operation_id: item for item in canonical} + supplied_by_id = {item.operation_id: item for item in entries} + if set(supplied_by_id) != set(canonical_by_id): raise ValueError("RES-71 inventory is not complete against the live registry") + for operation_id, supplied in supplied_by_id.items(): + if supplied != canonical_by_id[operation_id]: + raise ValueError( + f"RES-71 inventory metadata differs from canonical entry: {operation_id}" + ) root = _repository_root() for item in entries: for path in item.implementation: @@ -1221,6 +1229,9 @@ def build_coverage_matrix() -> tuple[CoverageRow, ...]: def validate_coverage_matrix(rows: tuple[CoverageRow, ...] | None = None) -> GateComponentStatus: if rows is None: rows = build_coverage_matrix() + canonical = build_coverage_matrix() + canonical_by_domain = {row.domain: row for row in canonical} + supplied_by_domain = {row.domain: row for row in rows} required = { "population/source authority", "football world/context", @@ -1235,11 +1246,16 @@ def validate_coverage_matrix(rows: tuple[CoverageRow, ...] | None = None) -> Gat "analysis capability", "claim authority", } - actual = {row.domain for row in rows} + actual = set(supplied_by_domain) if actual != required: raise ValueError(f"RES-71 coverage matrix domain mismatch: {sorted(actual ^ required)}") if len(rows) != len(actual): raise ValueError("RES-71 coverage matrix contains duplicate domains") + if set(canonical_by_domain) != required: + raise ValueError("RES-71 canonical coverage matrix domain set is incomplete") + for domain, supplied in supplied_by_domain.items(): + if supplied != canonical_by_domain[domain]: + raise ValueError(f"RES-71 coverage row differs from canonical row: {domain}") root = _repository_root() for row in rows: if not row.authoritative_surfaces or not row.test_coverage: @@ -1360,9 +1376,13 @@ def validate_unresolved_computation_inventory( if len(set(capabilities)) != len(capabilities): raise ValueError("RES-71 unresolved inventory contains duplicate capabilities") expected = build_unresolved_computation_inventory() - expected_capabilities = {item.capability for item in expected} - if set(capabilities) != expected_capabilities: + expected_by_capability = {item.capability: item for item in expected} + supplied_by_capability = {item.capability: item for item in entries} + if set(supplied_by_capability) != set(expected_by_capability): raise ValueError("RES-71 unresolved inventory is incomplete against the reviewed set") + for capability, supplied in supplied_by_capability.items(): + if supplied != expected_by_capability[capability]: + raise ValueError(f"RES-71 unresolved row differs from canonical row: {capability}") root = _repository_root() for item in entries: if item.disposition is OperationDisposition.IMPLEMENTED: diff --git a/tests/test_res71_adversarial.py b/tests/test_res71_adversarial.py index d440719..7a1990d 100644 --- a/tests/test_res71_adversarial.py +++ b/tests/test_res71_adversarial.py @@ -115,7 +115,7 @@ def test_stale_inventory_routes_are_rejected() -> None: operation, implementation=("dynamislm.qualification.inventory:missing_implementation",), ) - with pytest.raises(ValueError, match="implementation route is stale"): + with pytest.raises(ValueError, match="metadata differs from canonical entry"): validate_registered_operation_inventory( (stale_implementation, *build_registered_operation_inventory()[1:]) ) @@ -124,7 +124,7 @@ def test_stale_inventory_routes_are_rejected() -> None: operation, refusal_path=("dynamislm.qualification.inventory:missing_refusal",), ) - with pytest.raises(ValueError, match="refusal route is stale"): + with pytest.raises(ValueError, match="metadata differs from canonical entry"): validate_registered_operation_inventory( (stale_refusal, *build_registered_operation_inventory()[1:]) ) @@ -134,7 +134,7 @@ def test_stale_inventory_routes_are_rejected() -> None: unresolved[0], refusal_path=("dynamislm.qualification.inventory:missing_unresolved_route",), ) - with pytest.raises(ValueError, match="refusal route is stale"): + with pytest.raises(ValueError, match="differs from canonical row"): validate_unresolved_computation_inventory((stale_unresolved, *unresolved[1:])) From 4f346af801bd48e73e1112b24ad38303f602c0c3 Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 05:26:41 +0000 Subject: [PATCH 12/14] fix(res71): make gate receipt evidence type-strict and authoritative --- docs/qualification/RES71-GATE-RECEIPT.json | 38 +--------- src/dynamislm/qualification/gate.py | 85 ++++++++++++++++------ 2 files changed, 64 insertions(+), 59 deletions(-) diff --git a/docs/qualification/RES71-GATE-RECEIPT.json b/docs/qualification/RES71-GATE-RECEIPT.json index e5695bb..c93ed18 100644 --- a/docs/qualification/RES71-GATE-RECEIPT.json +++ b/docs/qualification/RES71-GATE-RECEIPT.json @@ -1,37 +1,12 @@ { "MISSION": "RES-71-SCIENTIFIC-ENGINE-GATE-001", - "STATUS": "PASS", "BASE_MAIN": "7508a9025759c2863d163e09b22f325494828602", "QUALIFIED_CONTENT_HEAD": "49988aa3cd03e460dc8ae8c60f8afa890e75706e", - "ENTRY": { - "one_worktree": true, - "branch": "work/res-71-scientific-engine-qualification-gate", - "clean_worktree": true, - "origin_main_equals_qualified_base": true - }, - "ATOMIC_COMMITS": [ - "adfa150 feat(res71): add qualification inventory and coverage contracts", - "f25ebbd test(res71): add cross-engine adversarial and gold qualification", - "fa0daf6 feat(res71): add verifier-ready deterministic reference cases", - "49988aa docs(res71): add provenance coverage and unresolved inventories", - "36d5408 fix(res71): harden qualification validators and refusal bindings", - "5fe4a6f fix(res71): bind verifier references and gate receipt to runtime evidence", - "c7b1961 test(res71): qualify empty-input stale-artifact and refusal-route attacks" - ], + "STATUS": "PASS", "REGISTERED_OPERATION_INVENTORY": "PASS", "COVERAGE_MATRIX": "PASS", "UNRESOLVED_COMPUTATION_INVENTORY": "PASS", "REFERENCE_CASE_VALIDATION": "PASS", - "IDENTITY_PROVENANCE": "PASS", - "NUMERICAL_QUALIFICATION": "PASS", - "SCIENTIFIC_BOUNDARIES": "PASS", - "DATASET_COMPATIBILITY": "PASS", - "VERIFIER_REFERENCE_INTERFACE": "PASS", - "IMPLICIT_LM_ARITHMETIC": "BLOCKED", - "UNREGISTERED_ACCEPTED_OPERATION": "NONE", - "PROVENANCE_GAPS": "NONE", - "CLAIM_AUTHORITY_BYPASS": "NONE", - "SCIENTIFIC_BLOCKERS": "NONE", "RUNTIME_COUNTS": { "registered_operations": 100, "implemented": 81, @@ -44,17 +19,8 @@ "verifier_reference_cases": 12, "verifier_reference_digest": "sha256:d29d84699b7cf70c2d409d370c5ffd6c7ad7cd704375b14b541527a95fa385e5" }, - "QA_RUFF": "PASS", - "QA_FORMAT": "PASS", - "QA_MYPY": "PASS", - "REPOSITORY_POLICY": "PASS", - "QA_PYTEST": "PASS", - "TEST_COUNT": 857, - "QA_TRACKED_MUTATION": "NONE", + "SERIALIZATION_VERSION": 3, "MODEL_INFERENCE": "NOT_IMPLEMENTED", "GPU_WORK": "NOT_IMPLEMENTED", - "SERIALIZATION_VERSION": 3, - "LINEAR_RES71_STATUS": "IN_PROGRESS", - "NEXT_AUTHORIZED_ACTION": "RES-71-FINAL-REVIEW-002", "SCIENTIFIC_ENGINE_GATE": "PASS" } diff --git a/src/dynamislm/qualification/gate.py b/src/dynamislm/qualification/gate.py index 90eeef0..857e08b 100644 --- a/src/dynamislm/qualification/gate.py +++ b/src/dynamislm/qualification/gate.py @@ -28,6 +28,8 @@ RES71_GATE_RECEIPT_PATH = ( Path(__file__).resolve().parents[3] / "docs" / "qualification" / "RES71-GATE-RECEIPT.json" ) +RES71_GATE_MISSION = "RES-71-SCIENTIFIC-ENGINE-GATE-001" +RES71_BASE_MAIN = "7508a9025759c2863d163e09b22f325494828602" RES71_QUALIFIED_CONTENT_HEAD = "49988aa3cd03e460dc8ae8c60f8afa890e75706e" _SHA_RE = re.compile(r"^[0-9a-f]{40}$") _NON_IMPLEMENTATION_FLAGS = { @@ -69,6 +71,7 @@ def build_gate_runtime_evidence() -> dict[str, object]: "verifier_reference_digest": reference_digest, } return { + "STATUS": "PASS", "REGISTERED_OPERATION_INVENTORY": "PASS", "COVERAGE_MATRIX": "PASS", "UNRESOLVED_COMPUTATION_INVENTORY": "PASS", @@ -90,9 +93,42 @@ def _load_receipt() -> Mapping[str, object]: return cast(Mapping[str, object], payload) +def _strict_equal(actual: object, expected: object) -> bool: + """Compare receipt values without Python's bool/int equality coercion.""" + + if type(actual) is not type(expected): + return False + if actual != expected: + return False + if isinstance(expected, Mapping): + actual_mapping = cast(Mapping[object, object], actual) + expected_mapping = cast(Mapping[object, object], expected) + if len(actual_mapping) != len(expected_mapping): + return False + for expected_key, expected_value in expected_mapping.items(): + matching_keys = tuple( + key + for key in actual_mapping + if type(key) is type(expected_key) and key == expected_key + ) + if len(matching_keys) != 1: + return False + if not _strict_equal(actual_mapping[matching_keys[0]], expected_value): + return False + return True + if isinstance(expected, list | tuple): + actual_sequence = cast(list[object] | tuple[object, ...], actual) + expected_sequence = cast(list[object] | tuple[object, ...], expected) + return len(actual_sequence) == len(expected_sequence) and all( + _strict_equal(actual_item, expected_item) + for actual_item, expected_item in zip(actual_sequence, expected_sequence, strict=True) + ) + return True + + def _require_field(receipt: Mapping[str, object], key: str, expected: object) -> None: actual = receipt.get(key) - if actual != expected: + if not _strict_equal(actual, expected): raise ValueError( f"RES-71 gate receipt field {key} differs: expected {expected!r}, got {actual!r}" ) @@ -103,37 +139,40 @@ def validate_gate_receipt(receipt: Mapping[str, object] | None = None) -> GateCo receipt_data = _load_receipt() if receipt is None else receipt evidence = build_gate_runtime_evidence() - _require_field(receipt_data, "STATUS", "PASS") - _require_field(receipt_data, "SCIENTIFIC_ENGINE_GATE", "PASS") - for key in ( - "REGISTERED_OPERATION_INVENTORY", - "COVERAGE_MATRIX", - "UNRESOLVED_COMPUTATION_INVENTORY", - "REFERENCE_CASE_VALIDATION", - "RUNTIME_COUNTS", - "SERIALIZATION_VERSION", - "MODEL_INFERENCE", - "GPU_WORK", - ): - _require_field(receipt_data, key, evidence[key]) + expected = { + "MISSION": RES71_GATE_MISSION, + "BASE_MAIN": RES71_BASE_MAIN, + "QUALIFIED_CONTENT_HEAD": RES71_QUALIFIED_CONTENT_HEAD, + **evidence, + } + extra_fields = set(receipt_data) - set(expected) + missing_fields = set(expected) - set(receipt_data) + if extra_fields: + raise ValueError( + "RES-71 gate receipt contains non-authoritative or unchecked fields: " + f"{sorted(extra_fields)}" + ) + if missing_fields: + raise ValueError( + f"RES-71 gate receipt is missing authoritative fields: {sorted(missing_fields)}" + ) + for key, expected_value in expected.items(): + _require_field(receipt_data, key, expected_value) - if "FINAL_HEAD" in receipt_data or "QUALIFIED_ENGINE_HEAD" in receipt_data: - raise ValueError("RES-71 gate receipt must not claim a branch FINAL_HEAD") - qualified_content_head = receipt_data.get("QUALIFIED_CONTENT_HEAD") + qualified_content_head = receipt_data["QUALIFIED_CONTENT_HEAD"] if not isinstance(qualified_content_head, str) or not _SHA_RE.fullmatch(qualified_content_head): raise ValueError("RES-71 gate receipt must contain a full QUALIFIED_CONTENT_HEAD SHA") + + for key in _NON_IMPLEMENTATION_FLAGS: + _require_field(receipt_data, key, evidence[key]) if qualified_content_head != RES71_QUALIFIED_CONTENT_HEAD: raise ValueError("RES-71 qualified content head differs from the sealed content head") - - for key in receipt_data: - if (key.startswith("MODEL_") or key.startswith("GPU_")) and key not in { - *_NON_IMPLEMENTATION_FLAGS, - }: - raise ValueError(f"RES-71 gate receipt contains an unapproved model/GPU flag: {key}") return GateComponentStatus.PASS __all__ = [ + "RES71_BASE_MAIN", + "RES71_GATE_MISSION", "RES71_GATE_RECEIPT_PATH", "RES71_QUALIFIED_CONTENT_HEAD", "build_gate_runtime_evidence", From 0a51127628f1bfc1f0b89064bf92d7fc2703ff39 Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 05:30:05 +0000 Subject: [PATCH 13/14] test(res71): reject qualification metadata and receipt type forgery --- tests/test_res71_adversarial.py | 107 ++++++++++++++++++++++++++++++++ 1 file changed, 107 insertions(+) diff --git a/tests/test_res71_adversarial.py b/tests/test_res71_adversarial.py index 7a1990d..c62e7cf 100644 --- a/tests/test_res71_adversarial.py +++ b/tests/test_res71_adversarial.py @@ -12,7 +12,10 @@ from dynamislm.qualification import ( RES71_GATE_RECEIPT_PATH, + CoverageStatus, + OperationDisposition, ReferenceCaseStatus, + build_coverage_matrix, build_registered_operation_inventory, build_unresolved_computation_inventory, get_reference_cases, @@ -138,6 +141,79 @@ def test_stale_inventory_routes_are_rejected() -> None: validate_unresolved_computation_inventory((stale_unresolved, *unresolved[1:])) +def test_registered_operation_metadata_substitutions_are_rejected() -> None: + entries = build_registered_operation_inventory() + operation = entries[0] + mutations = ( + replace( + operation, + implementation=("dynamislm.qualification.inventory:forged_implementation",), + ), + replace(operation, disposition=OperationDisposition.DEFERRED), + replace(operation, provenance_contract=operation.provenance_contract + " forged"), + replace( + operation, + authority_references=(*operation.authority_references, "docs/forged.md"), + ), + replace( + operation, + test_coverage=(*operation.test_coverage, "tests/test_res71_adversarial.py"), + ), + replace( + operation, + refusal_path=(*operation.refusal_path, "dynamislm.refusal.models:RefusalResult"), + ), + replace(operation, tolerance_contract=operation.tolerance_contract + " forged"), + ) + + for mutation in mutations: + with pytest.raises(ValueError, match="metadata differs from canonical entry"): + validate_registered_operation_inventory((mutation, *entries[1:])) + + +def test_unresolved_metadata_substitutions_are_rejected() -> None: + entries = build_unresolved_computation_inventory() + item = next(entry for entry in entries if entry.registered_operation_id is not None) + index = entries.index(item) + mutations = ( + replace(item, registered_operation_id=None), + replace(item, disposition=OperationDisposition.REJECTED), + replace(item, reason=item.reason + " forged"), + replace(item, refusal_path=(*item.refusal_path, *item.refusal_path[:1])), + replace(item, expected_refusal_class="FORGED_REFUSAL_CLASS"), + replace(item, expected_reason_codes=(*item.expected_reason_codes, "FORGED_CODE")), + replace(item, safe_description=item.safe_description + " forged"), + replace(item, test_coverage=(*item.test_coverage, "tests/test_res71_adversarial.py")), + replace(item, authority_references=(*item.authority_references, "docs/forged.md")), + ) + + for mutation in mutations: + supplied = list(entries) + supplied[index] = mutation + with pytest.raises(ValueError, match="differs from canonical row"): + validate_unresolved_computation_inventory(tuple(supplied)) + + +def test_coverage_metadata_substitutions_are_rejected() -> None: + rows = build_coverage_matrix() + row = rows[0] + mutations = ( + replace(row, status=CoverageStatus.QUALIFIED_WITH_EXPLICIT_DEFERRED), + replace(row, authoritative_surfaces=(*row.authoritative_surfaces, "forged")), + replace(row, registered_operations=(*row.registered_operations, "forged")), + replace(row, unresolved_capabilities=(*row.unresolved_capabilities, "forged")), + replace(row, provenance_boundary=row.provenance_boundary + " forged"), + replace(row, comparability_boundary=row.comparability_boundary + " forged"), + replace(row, claim_boundary=row.claim_boundary + " forged"), + replace(row, authority_references=(*row.authority_references, "docs/forged.md")), + replace(row, test_coverage=(*row.test_coverage, "tests/test_res71_adversarial.py")), + ) + + for mutation in mutations: + with pytest.raises(ValueError, match="differs from canonical row"): + validate_coverage_matrix((mutation, *rows[1:])) + + def test_manually_listed_unresolved_routes_bind_to_actual_refusals() -> None: unresolved = tuple( item @@ -185,3 +261,34 @@ def test_gate_receipt_rejects_stale_or_self_referential_edits() -> None: for mutated in mutations: with pytest.raises(ValueError): validate_gate_receipt(mutated) + + +@pytest.mark.parametrize("replacement", [True, 1.0]) +def test_gate_receipt_rejects_nested_runtime_count_type_substitution( + replacement: bool | float, +) -> None: + receipt = json.loads(RES71_GATE_RECEIPT_PATH.read_text(encoding="utf-8")) + receipt["RUNTIME_COUNTS"]["historical_replay_only"] = replacement + + with pytest.raises(ValueError, match="RUNTIME_COUNTS"): + validate_gate_receipt(receipt) + + +def test_gate_receipt_rejects_reintroduced_unchecked_status_fields() -> None: + receipt = json.loads(RES71_GATE_RECEIPT_PATH.read_text(encoding="utf-8")) + for field in ( + "IDENTITY_PROVENANCE", + "NUMERICAL_QUALIFICATION", + "SCIENTIFIC_BOUNDARIES", + "DATASET_COMPATIBILITY", + "VERIFIER_REFERENCE_INTERFACE", + "IMPLICIT_LM_ARITHMETIC", + "UNREGISTERED_ACCEPTED_OPERATION", + "PROVENANCE_GAPS", + "CLAIM_AUTHORITY_BYPASS", + "SCIENTIFIC_BLOCKERS", + ): + mutated = copy.deepcopy(receipt) + mutated[field] = "PASS" + with pytest.raises(ValueError, match="unchecked fields"): + validate_gate_receipt(mutated) From 7f7a3d0e69f45a1bb689e8c5e5068c4bea704eac Mon Sep 17 00:00:00 2001 From: Julio Rodriguez <144072916+Litju@users.noreply.github.com> Date: Mon, 21 Sep 2026 05:34:16 +0000 Subject: [PATCH 14/14] docs(res71): refresh sealed gate evidence and handoff --- docs/qualification/RES71-COVERAGE-MATRIX.md | 4 +++- docs/qualification/RES71-GATE-RECEIPT.json | 2 +- .../RES71-OPERATION-INVENTORY.md | 8 ++++--- .../RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md | 23 +++++++++++++++++++ .../RES71-UNRESOLVED-COMPUTATIONS.md | 5 ++-- src/dynamislm/qualification/gate.py | 2 +- 6 files changed, 36 insertions(+), 8 deletions(-) diff --git a/docs/qualification/RES71-COVERAGE-MATRIX.md b/docs/qualification/RES71-COVERAGE-MATRIX.md index 520d7c7..180de00 100644 --- a/docs/qualification/RES71-COVERAGE-MATRIX.md +++ b/docs/qualification/RES71-COVERAGE-MATRIX.md @@ -3,7 +3,9 @@ `build_coverage_matrix()` returns the machine-readable version. Every row has authoritative surfaces, registered operations (where applicable), unresolved capabilities, provenance boundary, comparability boundary, claim boundary, -authority references and test paths. +authority references and test paths. Validation keys supplied and canonical +rows by `domain` and requires exact frozen-dataclass equality before preserving +the existing domain and test-path checks. | V2 domain | Status | Explicit boundary / unresolved surface | |---|---|---| diff --git a/docs/qualification/RES71-GATE-RECEIPT.json b/docs/qualification/RES71-GATE-RECEIPT.json index c93ed18..a567a11 100644 --- a/docs/qualification/RES71-GATE-RECEIPT.json +++ b/docs/qualification/RES71-GATE-RECEIPT.json @@ -1,7 +1,7 @@ { "MISSION": "RES-71-SCIENTIFIC-ENGINE-GATE-001", "BASE_MAIN": "7508a9025759c2863d163e09b22f325494828602", - "QUALIFIED_CONTENT_HEAD": "49988aa3cd03e460dc8ae8c60f8afa890e75706e", + "QUALIFIED_CONTENT_HEAD": "0a51127628f1bfc1f0b89064bf92d7fc2703ff39", "STATUS": "PASS", "REGISTERED_OPERATION_INVENTORY": "PASS", "COVERAGE_MATRIX": "PASS", diff --git a/docs/qualification/RES71-OPERATION-INVENTORY.md b/docs/qualification/RES71-OPERATION-INVENTORY.md index de64bd7..e9bda4c 100644 --- a/docs/qualification/RES71-OPERATION-INVENTORY.md +++ b/docs/qualification/RES71-OPERATION-INVENTORY.md @@ -3,9 +3,11 @@ The authoritative inventory is the immutable tuple returned by `dynamislm.qualification.build_registered_operation_inventory()`. `validate_registered_operation_inventory()` discovers every -`RegistryReference` whose object type is `registered-operation` and requires an -exact set match. A newly exposed operation therefore fails qualification until -its contract is reviewed. +`RegistryReference` whose object type is `registered-operation`, keys supplied +and canonical rows by `operation_id`, and requires exact frozen-dataclass +equality before route/path/runtime checks. A newly exposed operation or a +metadata substitution therefore fails qualification until its contract is +reviewed. Each row contains: diff --git a/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md b/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md index 2439b20..26e39e1 100644 --- a/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md +++ b/docs/qualification/RES71-SCIENTIFIC-ENGINE-QUALIFICATION.md @@ -21,6 +21,29 @@ reason, refusal path, safe description and tests. The formal gate decision is sealed in `RES71-GATE-RECEIPT.json` after final QA. +## RES-71 review-fix-002 qualification-integrity hardening + +Review-fix entry head: `dc29909a13fbb27d463632f2cd5c4f239c4cd07f` + +Qualified content head: `0a51127628f1bfc1f0b89064bf92d7fc2703ff39` + +The review fix canonically binds registered-operation rows by `operation_id`, +unresolved-computation rows by `capability`, and coverage rows by `domain`. +Supplied rows must equal freshly built canonical frozen records before route, +path or refusal-runtime checks. The receipt comparison is recursively +type-strict, including nested `RUNTIME_COUNTS` values. + +`RES71-GATE-RECEIPT.json` is deliberately narrow and runtime-authoritative. +It contains only identity fields and evidence derived by +`validate_gate_receipt()`. The former checklist assertions +(`IDENTITY_PROVENANCE`, `NUMERICAL_QUALIFICATION`, `SCIENTIFIC_BOUNDARIES`, +`DATASET_COMPATIBILITY`, `VERIFIER_REFERENCE_INTERFACE`, +`IMPLICIT_LM_ARITHMETIC`, `UNREGISTERED_ACCEPTED_OPERATION`, +`PROVENANCE_GAPS`, `CLAIM_AUTHORITY_BYPASS`, and `SCIENTIFIC_BLOCKERS`) are +narrative qualification documentation, not unchecked fields in the sealed +runtime receipt. CI results and test counts are handoff evidence outside that +receipt. + ## Entry and upstream authority The entry checks were performed before mutation: diff --git a/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md b/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md index 8d66334..a5a0498 100644 --- a/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md +++ b/docs/qualification/RES71-UNRESOLVED-COMPUTATIONS.md @@ -2,8 +2,9 @@ Unsupported methods remain explicit and non-authoritative. The complete machine-readable set is returned by -`build_unresolved_computation_inventory()` and validated by -`validate_unresolved_computation_inventory()`. +`build_unresolved_computation_inventory()`. Validation keys supplied and +canonical rows by `capability` and requires exact frozen-dataclass equality +before runtime refusal-route checks. ## Registered but not computed diff --git a/src/dynamislm/qualification/gate.py b/src/dynamislm/qualification/gate.py index 857e08b..10ecec4 100644 --- a/src/dynamislm/qualification/gate.py +++ b/src/dynamislm/qualification/gate.py @@ -30,7 +30,7 @@ ) RES71_GATE_MISSION = "RES-71-SCIENTIFIC-ENGINE-GATE-001" RES71_BASE_MAIN = "7508a9025759c2863d163e09b22f325494828602" -RES71_QUALIFIED_CONTENT_HEAD = "49988aa3cd03e460dc8ae8c60f8afa890e75706e" +RES71_QUALIFIED_CONTENT_HEAD = "0a51127628f1bfc1f0b89064bf92d7fc2703ff39" _SHA_RE = re.compile(r"^[0-9a-f]{40}$") _NON_IMPLEMENTATION_FLAGS = { "MODEL_INFERENCE": "NOT_IMPLEMENTED",