From be2925b42764fc7bf2e7d810bf2caa1447edaa8c Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 10:37:11 +0930 Subject: [PATCH 001/229] Expand OPT with contract-driven optimization catalog --- .github/workflows/catalog-integrity.yml | 22 +++++ AGENTS.md | 32 ++++--- CATALOG.md | 96 +++++++++++++++---- OPTIMIZATION-PROBLEM.md | 80 ++++++++++++++++ README.md | 67 ++++++++----- README4AI.md | 92 +++++++++++++----- ...PROX-001-contract-bounded-approximation.md | 44 +++++++++ ...DGET-001-performance-regression-budgets.md | 45 +++++++++ ...01-concurrent-duplicate-work-coalescing.md | 45 +++++++++ ...NT-001-partitioned-coordination-domains.md | 44 +++++++++ ...T-CRIT-001-critical-path-prioritization.md | 44 +++++++++ ...T-FAN-001-shared-materialization-fanout.md | 43 +++++++++ ...1-signature-bound-incremental-execution.md | 48 ++++++++++ ...E-001-bound-driven-search-space-pruning.md | 47 +++++++++ ...-REDUCE-001-early-working-set-reduction.md | 43 +++++++++ ...SEARCH-001-budget-aware-adaptive-search.md | 45 +++++++++ ...T-SET-001-density-adaptive-compact-sets.md | 48 ++++++++++ scripts/check_catalog.py | 73 ++++++++++++++ sources/JAZCO.md | 25 +++++ sources/MATHEMATICAL-OPTIMIZATION.md | 18 ++++ sources/OPTIMIZATION-LIBRARIES.md | 28 ++++++ sources/WONDERBUILD.md | 31 ++++++ sources/WPO.md | 14 +++ templates/OPTIMIZATION-RECORD.md | 54 ++++++++--- 24 files changed, 1034 insertions(+), 94 deletions(-) create mode 100644 .github/workflows/catalog-integrity.yml create mode 100644 OPTIMIZATION-PROBLEM.md create mode 100644 optimizations/OPT-APPROX-001-contract-bounded-approximation.md create mode 100644 optimizations/OPT-BUDGET-001-performance-regression-budgets.md create mode 100644 optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md create mode 100644 optimizations/OPT-CONT-001-partitioned-coordination-domains.md create mode 100644 optimizations/OPT-CRIT-001-critical-path-prioritization.md create mode 100644 optimizations/OPT-FAN-001-shared-materialization-fanout.md create mode 100644 optimizations/OPT-INC-001-signature-bound-incremental-execution.md create mode 100644 optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md create mode 100644 optimizations/OPT-REDUCE-001-early-working-set-reduction.md create mode 100644 optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md create mode 100644 optimizations/OPT-SET-001-density-adaptive-compact-sets.md create mode 100755 scripts/check_catalog.py create mode 100644 sources/JAZCO.md create mode 100644 sources/MATHEMATICAL-OPTIMIZATION.md create mode 100644 sources/OPTIMIZATION-LIBRARIES.md create mode 100644 sources/WONDERBUILD.md create mode 100644 sources/WPO.md diff --git a/.github/workflows/catalog-integrity.yml b/.github/workflows/catalog-integrity.yml new file mode 100644 index 0000000..de8ce2a --- /dev/null +++ b/.github/workflows/catalog-integrity.yml @@ -0,0 +1,22 @@ +name: catalog-integrity + +on: + pull_request: + branches: ["main"] + push: + branches: ["main"] + workflow_dispatch: + +permissions: + contents: read + +jobs: + catalog-integrity: + runs-on: ubuntu-24.04 + timeout-minutes: 5 + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + persist-credentials: false + - name: Check catalog structure + run: python3 scripts/check_catalog.py diff --git a/AGENTS.md b/AGENTS.md index e3c84a0..ef6eef8 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -2,14 +2,24 @@ Machine-facing rules for agents using this repository. -1. Read `README4AI.md` and `CATALOG.md` before applying an optimization elsewhere. -2. Treat optimization records as patterns, not universal parameter sets. -3. Preserve reference semantics and add/retain conformance tests for optimized paths. -4. Prefer deterministic, bounded reuse over opaque caches. -5. If equivalence enables reuse, encode the equivalence as a named invariant and test it directly. -6. For Lean caches, distinguish verified reuse from cold reconstruction in both implementation and claims. -7. For parallel work, retain deterministic output ordering and verify scalar/parallel equivalence. -8. For real-time/DSP work, separate slow control work from hot sample/block work when semantics allow it; avoid allocations and synchronization on the hot path. -9. `power_module.md` contains both implemented ideas and aspirational performance language. Check the corresponding code/evidence before promoting a claim. -10. `suxen.zip` is a source candidate, not validated evidence. Use `scripts/inventory_zip.py` in an environment with the archive bytes available before extracting optimization claims. -11. New records must state status, source identity, preserved contract, evidence, limitations, and rollback conditions. +1. Read `README4AI.md`, `OPTIMIZATION-PROBLEM.md`, and `CATALOG.md` before applying an optimization elsewhere. +2. Define the target optimization contract before selecting a mechanism: search space, feasible set, objective/direction, correctness constraints, evaluation budget and stopping rule. +3. Treat optimization records as patterns, not universal parameter sets. +4. Preserve reference semantics and add/retain conformance tests for optimized paths. +5. Prefer deterministic, bounded reuse over opaque caches. +6. If equivalence enables reuse, encode the equivalence as a named invariant and test it directly. +7. For incremental reuse, bind the complete effective input identity and persist a new reusable state only after successful completion. +8. For coalescing, merge only semantically equivalent in-flight work; define cancellation/error semantics explicitly. +9. For Lean caches, distinguish verified reuse from cold reconstruction in implementation and claims. +10. For parallel work, retain deterministic output ordering where required and verify scalar/parallel equivalence. Measure effective concurrency rather than assuming requested workers ran concurrently. +11. For adaptive search, preserve the trial ledger and evaluation budget. Remember that excessive parallel batch width can reduce information efficiency. +12. For approximation, state an explicit error/degradation contract and keep an exact/reference path where practical. Never silently weaken exact semantics. +13. For pruning, test bound soundness independently; never prune on a heuristic presented as proof. +14. For critical-path/speculative work, ensure speculation cannot expose side effects before commitment and does not starve the actual critical path. +15. For performance budgets, characterize benchmark noise/environment before enforcing a threshold. +16. For real-time/DSP work, separate slow control work from hot sample/block work when semantics allow it; avoid allocations and synchronization on the hot path. +17. `power_module.md` contains both implemented ideas and aspirational performance language. Check corresponding code/evidence before promoting a claim. +18. `suxen.zip` is a source candidate, not validated evidence. Inventory and read relevant source before extracting optimization claims. +19. The three pinned v1 Lean model files are immutable historical formalization. New records do not become formally proved by association; version future formal modules separately. +20. New post-v1 records must state status, source identity, optimization problem contract, preserved contract, validation, limitations and rollback conditions. +21. Run `python3 scripts/check_catalog.py` after catalog changes. diff --git a/CATALOG.md b/CATALOG.md index 21866ee..6d4f3fe 100644 --- a/CATALOG.md +++ b/CATALOG.md @@ -2,53 +2,107 @@ ## Quick decision table -| Bottleneck | First record to inspect | Core idea | +| Bottleneck / problem shape | First record to inspect | Core idea | | --- | --- | --- | -| Test suite spends most time in deterministic sweeps/simulations | `OPT-PY-001` | Reduce redundant work while keeping coverage semantics and deterministic assertions | -| Same expensive result is recomputed at a provably equivalent parameter/state | `OPT-INV-001` | Prove equivalence, then reuse the already-computed result | -| Lean CI repeatedly rebuilds an unchanged dependency closure | `OPT-LEAN-001` | Reuse only cryptographically/structurally verified dependency state; always rebuild project source | -| Independent jobs/items can execute concurrently | `OPT-PAR-001` | Bound workers, preserve deterministic ordering, prove scalar/parallel equivalence | -| Expensive state changes far slower than the sample/hot-loop rate | `OPT-DSP-001` | Move state evolution to control rate; sparse-evaluate couplings; batch/vectorize the hot path | - -## Records +| Deterministic tests/sweeps dominate runtime | `OPT-PY-001` | Reduce redundant/high-cost work while keeping coverage semantics | +| Same expensive result is recomputed at a proven-equivalent state | `OPT-INV-001` | Prove equivalence, then reuse | +| Lean dependency reconstruction dominates CI | `OPT-LEAN-001` | Verify reusable dependency state; rebuild current project source | +| Independent work can execute concurrently | `OPT-PAR-001` | Bound workers and prove scalar/parallel equivalence | +| Slow control state is inside a high-rate numerical/audio loop | `OPT-DSP-001` | Separate rates, sparse-evaluate, vectorize | +| Inputs are unchanged but pipeline stages rerun | `OPT-INC-001` | Bind work to complete input signatures and persist only successful state | +| Many simultaneous callers request identical not-yet-computed work | `OPT-COAL-001` | One in-flight computation, many waiters | +| Integer sets alternate between sparse and dense regions | `OPT-SET-001` | Density-adaptive representation with exact set algebra | +| One global lock/counter/runtime domain serializes independent work | `OPT-CONT-001` | Partition coordination while preserving the global invariant | +| Same deterministic transform is repeated for every consumer/replay | `OPT-FAN-001` | Materialize once, reuse many times | +| Expensive parameter evaluations are being guessed or exhaustively swept | `OPT-SEARCH-001` | Adaptive, budget-aware search over the declared problem contract | +| Exactness may be traded inside an explicit quality envelope | `OPT-APPROX-001` | Bound the error/degradation and the resource cost together | +| Expensive stages consume candidates later discarded | `OPT-REDUCE-001` | Reduce the working set before composition | +| Non-critical work delays the dependency chain users actually wait on | `OPT-CRIT-001` | Prioritize the critical path; speculate/defer deliberately | +| Small performance regressions accumulate unnoticed | `OPT-BUDGET-001` | Guard stable performance expectations in CI | +| Discrete search space is huge but optimistic bounds are available | `OPT-PRUNE-001` | Prune regions that provably cannot beat the incumbent | + +Before selecting a record, define the target problem using [`OPTIMIZATION-PROBLEM.md`](OPTIMIZATION-PROBLEM.md). + +## Frozen v1 records ### OPT-PY-001 — Deterministic test execution **Status:** Verified mechanism; historical performance context incomplete. -QEC combined minimal fixtures, vectorized assertions, bounded deterministic caching, convergence/cycle early exit, smaller high-cost sweeps, lower safe iteration/trial counts, and repeated-work removal. QEC v68.4.0 reports about 126 s → 46 s (~2.7×) with 3779 passed / 8 skipped; v68.4.1 reports about 40 s after hardening. The cited v68.x records do not preserve runner/CPU/Python/pytest/repetition metadata, so these are historical observations, not transferable benchmark targets. +QEC combined minimal fixtures, vectorized assertions, bounded deterministic caching, convergence/cycle early exit, smaller high-cost sweeps, lower safe iteration/trial counts, and repeated-work removal. Historical timings remain source observations, not transferable targets. ### OPT-INV-001 — Invariant-driven computation reuse **Status:** Verified mechanism; historical performance context incomplete. -QEC formalized the baseline equivalence `URW(min_sum, rho=1.0) == baseline min-sum`, tested exact equality, centralized the predicate, and reused the baseline result rather than rerunning the benchmark. The implementation commit reports about 43% speedup for that hot test, but does not preserve its runner/toolchain, exact hot-test wall times, or repetitions. Re-measure before making a target-repo speed claim. +QEC encoded a baseline equivalence, tested exact equality and reused a proven-equivalent baseline result rather than rerunning the benchmark. ### OPT-LEAN-001 — Trust-preserving Lean CI -**Status:** Verified on QSOL-GEO-REASON PR #3 source lane; timing observations are environment-scoped. +**Status:** Verified on source PR; timing observations are environment-scoped. -Separates source-state cache identity from compiled dependency artifacts, verifies both before use, rebuilds the current project source, and keeps a no-cache `cold-trust` lane for release-grade reconstruction claims. The cache policy records a 2501.52 s cold dependency build using four Lean threads on a four-CPU `ubuntu-24.04` / x86_64 lane; a later verified-cache run records an 8.47 s GeoReason project build on a four-CPU Ubuntu 24.04.4 hosted runner. These are single observations of different scopes, with no exact CPU model/repetition distribution preserved, so they must not be divided into a portable speedup. +Separates source-state identity from compiled dependency artifacts, verifies reuse, rebuilds current project source, and keeps cold reconstruction claims separate. ### OPT-PAR-001 — Bounded parallel execution -**Status:** Verified, environment-specific; performance must be re-measured before transfer. +**Status:** Verified, environment-specific. -The QEC-validated NEXUS v4.0.1 qBraid evidence compared scalar and worker-count variants, checked output invariants, and recorded observed thread behavior. Its archived seven-worker observation was made on qBraid / Ubuntu 24.04.4 / AMD EPYC 7763 with 16 logical CPUs and effective worker capacity 7. The canonical receipt does not bind the performance samples to an exact `rustc --version`, so OPT retains the numbers as historical evidence rather than a transferable performance target. +Compares scalar and worker-count variants, checks output invariants and treats measured effective parallelism as evidence rather than assuming requested workers were used. ### OPT-DSP-001 — Control-rate sparse vector DSP **Status:** Implemented reference for control-rate/sparse/vector patterns; approximation/native ideas partly proposed. -The SPECTRAL NumPy reference is pinned to commit `5265b7f130287f80b5cf0d3de5bb2953152f90cd`. It precomputes static state, evolves E8/qutrit control state at ~1 kHz, computes all root phases once per control step, uses a sparse root subset per node, and vectorizes block synthesis. The audition renderer's whole-block modulation shortcut has no defined equivalence/error contract and is therefore not promoted as a reusable correctness-preserving optimization. `power_module.md` additionally proposes block SIMD, zero-copy buffers and lock-free/native Rust structures; those native performance claims are not promoted as verified here. +Precompute static state, evolve slow control state less often, evaluate sparse couplings and batch/vectorize hot numerical work. + +The five records above are the immutable v1.0.0 formalized catalog. Their Lean model remains pinned; post-v1 records below do not silently alter it. + +## Post-v1 records + +### OPT-INC-001 — Signature-bound incremental execution +Persist complete effective-input identity after successful work and skip a stage only while that identity and required outputs remain valid. + +### OPT-COAL-001 — Concurrent duplicate-work coalescing +Merge equivalent simultaneous misses into one in-flight computation instead of letting a thundering herd duplicate upstream work. + +### OPT-SET-001 — Density-adaptive compact sets +Partition an integer domain and use sparse or bitmap-like containers according to local density while keeping exact set semantics. + +### OPT-CONT-001 — Partitioned coordination domains +Replace one hot global coordination point with independently advancing domains while preserving required cross-domain invariants. + +### OPT-FAN-001 — Shared materialization for fan-out and replay +Perform a deterministic transform once at the production boundary, persist/retain it when justified, and reuse it across consumers and replay. + +### OPT-SEARCH-001 — Budget-aware adaptive parameter search +Classify the optimization problem, maintain a trial ledger, adapt future evaluations from observations, and stop under an explicit evaluation/resource budget. + +### OPT-APPROX-001 — Contract-bounded approximation +Permit approximation only when the interface/scientific contract explicitly defines an error or degradation envelope and a reference path exists where practical. + +### OPT-REDUCE-001 — Early working-set reduction +Push semantics-preserving filtering/culling/selection ahead of joins, rendering, simulation, DSP or other expensive composition. + +### OPT-CRIT-001 — Critical-path prioritization +Prioritize work on the true latency dependency chain; prefetch/precompute likely-soon work only when justified; defer non-critical work. + +### OPT-BUDGET-001 — Performance regression budgets +Protect a stable benchmark expectation with an environment-scoped, variance-aware CI budget rather than relying on remembered performance. + +### OPT-PRUNE-001 — Bound-driven search-space pruning +Maintain a feasible incumbent, derive optimistic bounds for subregions, and discard regions that provably cannot improve the incumbent. ## Composition guidance -Optimizations compose only when their resource models do. In particular: +Optimizations compose only when their semantic and resource models compose. -- pytest process parallelism plus BLAS/NumPy threads can oversubscribe CPUs; -- Lean parallel workers plus large dependency cache restore can raise memory and I/O pressure; -- a DSP block that is vectorized but allocates an `nodes × samples` temporary may still be unsuitable for hard real-time use; -- caching an equivalent result is safe only while the equivalence invariant remains true. +- process parallelism plus BLAS/NumPy/native threads can oversubscribe CPUs; +- coalescing reduces duplicate identical work while adaptive search may instead need to diversify independent in-flight experiments; +- a compact representation may make a formerly remote problem feasible in memory, changing the architecture rather than merely reducing bytes; +- critical-path speculation can steal resources from the path it was intended to accelerate; +- approximation must never leak into an API whose callers still assume exact semantics; +- adaptive search can lose information efficiency when parallel batches are too wide; +- performance budgets require controlled environments or statistically defensible noise handling; +- pruning is valid only when the bound is sound. -Prefer one measured bottleneck removal at a time, then re-profile. +Prefer one measured bottleneck removal at a time, then re-profile and reconsider the problem contract. diff --git a/OPTIMIZATION-PROBLEM.md b/OPTIMIZATION-PROBLEM.md new file mode 100644 index 0000000..b597947 --- /dev/null +++ b/OPTIMIZATION-PROBLEM.md @@ -0,0 +1,80 @@ +# Optimization Problem Contract + +OPT separates **the problem being optimized** from **the mechanism used to search for an improvement**. + +For a target repository, define the optimization problem before selecting an optimization record. + +## Canonical contract + +Represent a problem as + +\[ +P = (X, F, f, d, C, B, S) +\] + +where: + +- `X` — search space / decision-variable domain; +- `F ⊆ X` — feasible set after hard constraints; +- `f : F → R^k` — measured objective or objective vector; +- `d` — objective direction (`minimize`, `maximize`, or explicit multi-objective ordering); +- `C` — correctness and semantic contract that may not be weakened implicitly; +- `B` — evaluation/resource budget; +- `S` — stopping rule. + +A candidate is admissible only if it lies in `F` **and** satisfies `C`. A faster candidate that violates `C` is not an optimization under the same problem definition. + +## Required classification + +Record the following before tuning: + +| Dimension | Typical values | +| --- | --- | +| Variables | continuous / integer / categorical / conditional / mixed | +| Search scope | local / global | +| Objective | deterministic / noisy / stochastic | +| Information | gradient available / derivative-free / black-box | +| Evaluation cost | cheap / moderate / expensive | +| Constraints | bounds / equality / inequality / semantic / resource | +| Parallelism | sequential / synchronous batch / asynchronous | +| Exactness | exact / approximation permitted under an explicit error contract | + +## Examples + +### Worker-count tuning + +- `X = {1, …, 32}` +- `F = X` subject to peak-memory and platform limits +- `f(x) = median wall time` +- `d = minimize` +- `C = scalar/parallel result equivalence + deterministic required ordering` +- `B = 40 benchmark trials` +- `S = budget exhausted or improvement below the predeclared threshold` + +### Approximate visualization + +- `X = {LOD policies}` +- `F = policies satisfying frame-memory limits` +- `f = (frame latency, perceptual/error metric)` +- `C = error ≤ ε and reference path remains available` +- `B = fixed benchmark fixture set` +- `S = Pareto candidate chosen under the documented priority rule` + +## Search-mechanism selection + +Use the problem classification to select a mechanism: + +- expensive black-box continuous or mixed tuning → `OPT-SEARCH-001`; +- discrete search with provable optimistic bounds → `OPT-PRUNE-001`; +- independent work that can run concurrently → `OPT-PAR-001`; +- repeated equivalent work → `OPT-INV-001`; +- simultaneous identical in-flight work → `OPT-COAL-001`; +- approximation explicitly permitted → `OPT-APPROX-001`. + +The mechanism is subordinate to the contract. Do not reshape the problem after seeing results merely to make an optimization look successful. + +## Measurement rule + +Source-project constants and historical observations are priors, not targets. Transfer requires fresh target-context measurement and validation. + +See `FORMALIZATION.md` for the frozen v1.0.0 Lean boundary. This problem-contract layer is post-v1 catalog guidance and does not mutate the pinned v1 formal model. diff --git a/README.md b/README.md index 6275e94..a9083e1 100644 --- a/README.md +++ b/README.md @@ -1,44 +1,65 @@ # OPT — QSOL Optimization Catalog -Reusable, provenance-linked optimization patterns extracted from QSOL projects. +Reusable, provenance-linked optimization patterns extracted from QSOL projects and carefully bounded external donors. -The point of this repository is simple: when a future project needs to go faster, use less memory, avoid redundant work, or shorten CI without weakening correctness, point the implementing agent here first. +The point of this repository is simple: when a future project needs to go faster, use less memory, avoid redundant work, shorten CI, or tune an expensive system, point the implementing agent here first — **without weakening correctness to make a benchmark look good**. ## Rules of the vault 1. **Correctness outranks speed.** An optimization must preserve the contract it claims to preserve. -2. **Measured and proposed work are different things.** Records say which is which. -3. **Keep the reference path.** Optimized/native/parallel paths should have a deterministic reference or conformance gate whenever practical. -4. **Do not cargo-cult constants.** Trial counts, worker caps, cache keys, tolerances, hashes, and block sizes belong to their source environment until re-measured. -5. **Provenance matters.** Every promoted optimization links back to the code, release, PR, or evidence that established it. +2. **Define the problem before choosing the trick.** Use [`OPTIMIZATION-PROBLEM.md`](OPTIMIZATION-PROBLEM.md) for the search space, feasible set, objective, constraints, budget and stopping rule. +3. **Measured and proposed work are different things.** Records say which is which. +4. **Keep the reference path.** Optimized/native/parallel/approximate paths should have a deterministic reference or conformance gate whenever practical. +5. **Do not cargo-cult constants.** Trial counts, worker caps, cache keys, tolerances, hashes, block sizes, thresholds and search parameters belong to their source environment until re-measured. +6. **Provenance matters.** Every promoted optimization links back to the code, release, PR, paper, or source that established it. ## Catalog -| ID | Optimization | Status | Source | Evidence snapshot | -| --- | --- | --- | --- | --- | -| [OPT-PY-001](optimizations/OPT-PY-001-deterministic-test-execution.md) | Deterministic test execution | **Verified mechanism; benchmark context incomplete** | QEC v68.2.0–v68.4.1 | ~126 s → ~46 s in v68.4.0 release; ~40 s after v68.4.1; original runner/toolchain/repetition context not preserved, so re-benchmark before transfer | -| [OPT-INV-001](optimizations/OPT-INV-001-invariant-driven-reuse.md) | Invariant-driven computation reuse | **Verified mechanism; benchmark context incomplete** | QEC v68.4.1 cycle | Eliminated a redundant benchmark at a proven-equivalent baseline point; ~43% source-reported hot-test improvement, with original environment/timing samples unavailable | -| [OPT-LEAN-001](optimizations/OPT-LEAN-001-trust-preserving-lean-ci.md) | Trust-preserving Lean dependency reuse | **Verified on source PR; timings environment-scoped** | QSOL-GEO-REASON PR #3 | 2501.52 s cold dependency reconstruction and 8.47 s verified-cache project rebuild are different-scope single observations on four-CPU Ubuntu lanes; see record | -| [OPT-PAR-001](optimizations/OPT-PAR-001-bounded-parallel-execution.md) | Bounded deterministic parallel execution | **Verified, environment-specific** | QEC v170.2.1 / NEXUS evidence | qBraid / AMD EPYC 7763 observation: 69.694063 ns/eval scalar median → 18.310215 ns/eval at 7 workers; re-benchmark before transfer | -| [OPT-DSP-001](optimizations/OPT-DSP-001-control-rate-sparse-vector-dsp.md) | Control-rate + sparse + vectorized DSP | **Implemented reference; approximation/native ideas proposed** | SPECTRAL commit `5265b7f…` + `power_module.md` | Control-rate decimation, sparse E8 coupling, shared phase computation and NumPy vectorization are implemented; whole-block audition approximation and native SIMD/zero-copy/lock-free claims are not promoted | +| ID | Optimization | Status | Core idea | +| --- | --- | --- | --- | +| [OPT-PY-001](optimizations/OPT-PY-001-deterministic-test-execution.md) | Deterministic test execution | **Verified mechanism; benchmark context incomplete** | Reduce repeated/high-cost test work without weakening coverage semantics | +| [OPT-INV-001](optimizations/OPT-INV-001-invariant-driven-reuse.md) | Invariant-driven computation reuse | **Verified mechanism; benchmark context incomplete** | Prove equivalence, then reuse the existing result | +| [OPT-LEAN-001](optimizations/OPT-LEAN-001-trust-preserving-lean-ci.md) | Trust-preserving Lean dependency reuse | **Verified on source PR; timings environment-scoped** | Reuse verified dependency state while rebuilding current project source | +| [OPT-PAR-001](optimizations/OPT-PAR-001-bounded-parallel-execution.md) | Bounded deterministic parallel execution | **Verified, environment-specific** | Bound concurrency and prove scalar/parallel equivalence | +| [OPT-DSP-001](optimizations/OPT-DSP-001-control-rate-sparse-vector-dsp.md) | Control-rate + sparse + vectorized DSP | **Implemented reference; approximation/native ideas proposed** | Move slow state out of the hot path; sparse/vectorize repeated numerical work | +| [OPT-INC-001](optimizations/OPT-INC-001-signature-bound-incremental-execution.md) | Signature-bound incremental execution | **Implemented historical reference** | Rerun work only when complete effective-input identity changes | +| [OPT-COAL-001](optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md) | Concurrent duplicate-work coalescing | **Implemented external reference** | Share one in-flight computation among equivalent simultaneous callers | +| [OPT-SET-001](optimizations/OPT-SET-001-density-adaptive-compact-sets.md) | Density-adaptive compact sets | **Implemented external reference** | Choose sparse/dense representation locally while retaining exact set algebra | +| [OPT-CONT-001](optimizations/OPT-CONT-001-partitioned-coordination-domains.md) | Partitioned coordination domains | **Implemented external pattern** | Split one global contention hotspot into independent domains while preserving global invariants | +| [OPT-FAN-001](optimizations/OPT-FAN-001-shared-materialization-fanout.md) | Shared materialization for fan-out/replay | **Implemented external reference** | Transform/encode once and reuse the representation for many consumers | +| [OPT-SEARCH-001](optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md) | Budget-aware adaptive parameter search | **External mechanisms; OPT synthesis proposed** | Spend expensive evaluations where they are most informative | +| [OPT-APPROX-001](optimizations/OPT-APPROX-001-contract-bounded-approximation.md) | Contract-bounded approximation | **Target-specific validation required** | Trade exactness only inside an explicit measurable error/degradation envelope | +| [OPT-REDUCE-001](optimizations/OPT-REDUCE-001-early-working-set-reduction.md) | Early working-set reduction | **Implemented external pattern** | Filter/cull/limit before expensive composition | +| [OPT-CRIT-001](optimizations/OPT-CRIT-001-critical-path-prioritization.md) | Critical-path prioritization | **Established engineering pattern** | Do critical work now, speculate carefully, defer non-critical work | +| [OPT-BUDGET-001](optimizations/OPT-BUDGET-001-performance-regression-budgets.md) | Performance regression budgets | **Established engineering pattern** | Turn performance expectations into environment-scoped regression contracts | +| [OPT-PRUNE-001](optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md) | Bound-driven search-space pruning | **Classical optimization mechanism** | Prove whole search regions cannot improve the incumbent and skip them | See [CATALOG.md](CATALOG.md) for the decision map and [README4AI.md](README4AI.md) for machine-oriented usage. -## Provenance correction: the QEC links +## Formalization boundary -The originally supplied QEC v170.2.0/v170.2.1 links are useful, but they are **not the origin of the large pytest/CI test-speed optimization**. The primary deterministic test optimization lineage is: +The immutable `v1.0.0` release and its five original records are formalized by the pinned Lean v1 model described in [`FORMALIZATION.md`](FORMALIZATION.md). This catalog expansion is **post-v1**. It does not edit the three pinned v1 Lean model files or pretend the new records are already theorem-backed. -- [QEC v68.2.0 — Deterministic Execution Engine](https://github.com/QSOLKCB/QEC/releases/tag/v68.2.0) -- [QEC v68.4.0 — Deterministic Runtime Optimization](https://github.com/QSOLKCB/QEC/releases/tag/v68.4.0) -- [QEC v68.4.1 — Invariant Hardening & Repository Cleanup](https://github.com/QSOLKCB/QEC/releases/tag/v68.4.1) +The new [`OPTIMIZATION-PROBLEM.md`](OPTIMIZATION-PROBLEM.md) supplies a canonical problem contract for future records: -The v170.2.x releases remain relevant here because v170.2.1 contains independently validated multicore benchmark evidence used by **OPT-PAR-001**. +`P = (X, F, f, direction, C, B, S)` -## Existing source material +for search space, feasible set, objective, direction, correctness/semantic constraints, evaluation budget and stopping rule. -- [`power_module.md`](power_module.md) contains the E8/qutrit DSP architecture that motivated **OPT-DSP-001**. -- [`suxen.zip`](suxen.zip) is retained as an opaque source archive. It is **not yet promoted as optimization evidence**. See [`sources/SUXEN.md`](sources/SUXEN.md) and run the bounded recursive [`scripts/inventory_zip.py`](scripts/inventory_zip.py) scanner in a normal checkout before promoting claims from its nested payloads. +## Source material + +- [`sources/WONDERBUILD.md`](sources/WONDERBUILD.md) — incremental execution, scheduling and rebuild-benchmark donor; GPL implementation boundary recorded. +- [`sources/JAZCO.md`](sources/JAZCO.md) — production systems case studies for coalescing, compact sets, contention, fan-out, approximation and reduction. +- [`sources/OPTIMIZATION-LIBRARIES.md`](sources/OPTIMIZATION-LIBRARIES.md) — BayesianOptimization, Hyperopt and NLopt mechanism/taxonomy notes. +- [`sources/WPO.md`](sources/WPO.md) — critical-path and performance-budget discovery source. +- [`sources/MATHEMATICAL-OPTIMIZATION.md`](sources/MATHEMATICAL-OPTIMIZATION.md) — mathematical/combinatorial problem vocabulary and pruning foundations. +- [`power_module.md`](power_module.md) — E8/qutrit DSP architecture that motivated **OPT-DSP-001**. +- [`suxen.zip`](suxen.zip) — opaque source archive, still **not promoted as optimization evidence**. + +## Integrity gate + +`scripts/check_catalog.py` verifies record IDs, required post-v1 sections, catalog coverage and local optimization-record links. CI runs it via `.github/workflows/catalog-integrity.yml`. ## Add the next optimization -Copy [`templates/OPTIMIZATION-RECORD.md`](templates/OPTIMIZATION-RECORD.md), assign the next ID, record before/after evidence, and state exactly what correctness property was preserved. +Copy [`templates/OPTIMIZATION-RECORD.md`](templates/OPTIMIZATION-RECORD.md), define the optimization problem contract, assign the next ID, record evidence honestly, and state exactly what correctness property is preserved. diff --git a/README4AI.md b/README4AI.md index e01461b..5b6f2be 100644 --- a/README4AI.md +++ b/README4AI.md @@ -4,53 +4,93 @@ This repository is a reusable optimization knowledge base for QSOL projects. ## Start here -1. Identify the dominant bottleneck. -2. Select the closest optimization record from `CATALOG.md`. -3. Read the source record completely before modifying another repository. -4. Preserve the target repository's semantics, invariants, determinism, evidence boundaries, and public API unless the task explicitly changes them. -5. Benchmark before and after in the target environment. -6. Record any adaptation rather than pretending source-project constants are universal. +1. Read `OPTIMIZATION-PROBLEM.md`. +2. Define `P = (X,F,f,d,C,B,S)` for the target: search space, feasible set, objective, direction, correctness/semantic contract, budget and stopping rule. +3. Identify the dominant bottleneck and choose the closest record from `CATALOG.md`. +4. Read that record and its source note completely before modifying another repository. +5. Preserve target semantics, invariants, determinism, evidence boundaries and public API unless the task explicitly changes them. +6. Benchmark before/after in the target environment and retain raw/repeated observations where practical. +7. Record adaptations rather than pretending source constants are universal. + +## Problem classification + +Before choosing an optimizer, classify: + +- continuous / integer / categorical / conditional / mixed variables; +- local / global search; +- deterministic / noisy / stochastic objective; +- gradient available / derivative-free / black-box; +- cheap / expensive evaluations; +- bound/equality/inequality/semantic/resource constraints; +- sequential / synchronous batch / asynchronous execution; +- exact / approximation explicitly permitted. ## Decision map -- **Slow pytest / deterministic numerical tests** → `OPT-PY-001` -- **Repeated work known to be mathematically or bitwise equivalent** → `OPT-INV-001` -- **Lean/mathlib dependency rebuild dominates CI** → `OPT-LEAN-001` -- **Independent work can run concurrently** → `OPT-PAR-001` -- **High-rate numerical/audio loop with slower control state** → `OPT-DSP-001` - -Patterns may be composed. Example: a numerical CI job can use `OPT-PY-001` inside the test process and `OPT-PAR-001` at a higher independent-work layer, but only after nested parallelism and memory pressure are measured. +- slow deterministic tests → `OPT-PY-001` +- proven-equivalent repeated computation → `OPT-INV-001` +- Lean dependency reconstruction → `OPT-LEAN-001` +- independent parallel work → `OPT-PAR-001` +- hot numerical/audio loop with slower control state → `OPT-DSP-001` +- unchanged-input pipeline reruns → `OPT-INC-001` +- identical simultaneous in-flight work → `OPT-COAL-001` +- sparse/dense integer-set mixture → `OPT-SET-001` +- hot global coordination point → `OPT-CONT-001` +- repeated per-consumer transformation → `OPT-FAN-001` +- expensive parameter tuning → `OPT-SEARCH-001` +- approximation explicitly allowed → `OPT-APPROX-001` +- large population filtered only after expensive work → `OPT-REDUCE-001` +- latency-critical path competes with optional work → `OPT-CRIT-001` +- gradual performance drift/regression → `OPT-BUDGET-001` +- combinatorial search with valid optimistic bounds → `OPT-PRUNE-001` + +## Important distinctions + +- **cache/reuse**: completed result already exists; +- **coalescing**: result does not exist yet, but equivalent callers share one in-flight evaluation; +- **async search diversification**: independent workers should intentionally avoid evaluating the same pending region; +- **parallelism**: improves throughput only when resource contention and information dependencies allow it; +- **approximation**: a contract choice, never a hidden optimization. ## Status vocabulary -- **Verified**: source project contains passing validation and measured/observed evidence for the optimization. +- **Verified**: source project contains passing validation and measured/observed evidence. - **Verified, environment-specific**: measured result is real but not a universal performance guarantee. -- **Implemented reference**: the optimization mechanism exists in code, but no general speedup claim is made. -- **Proposed**: architecture/design idea only. Do not report it as achieved performance. +- **Implemented reference**: the mechanism exists in code, but no general speedup claim is made. +- **Implemented external reference/pattern**: mechanism exists in an external donor; target transfer still requires local validation. +- **Proposed / OPT synthesis**: architecture/design guidance only. Do not report it as achieved performance. - **Source candidate**: material exists but has not been inspected sufficiently to promote claims. -## Non-negotiable safety rules +## Non-negotiable rules - Never remove tests merely to make CI faster. -- Never weaken an assertion, tolerance, theorem target, receipt, claim boundary, or validation rule without an explicit contract change. +- Never weaken an assertion, tolerance, theorem target, receipt, trust boundary or validation rule without an explicit contract change. - Never treat a cache hit as proof of a cold rebuild. -- Never equate a requested worker count with observed effective parallel execution. -- Never claim SIMD/zero-copy/lock-free speedups from `power_module.md` without controlled measurements. -- Never claim anything from `suxen.zip` until it has been inventoried and the relevant source has been read. +- Never equate requested workers with observed effective execution. +- Never copy historical worker counts, thresholds, search budgets, bit partitions, cache sizes or approximation limits without target measurement. +- Never prune a search region unless the bound used for pruning is sound for the declared problem. +- Never call an approximate result exact. +- Never optimize from stale workload assumptions when fresh measurements are available. +- `suxen.zip` remains unpromoted until inventoried and inspected. ## What to copy vs what to adapt -Copy the **structure** of the optimization: cache identity binding, deterministic reuse, equivalence gates, bounded workers, control-rate separation, sparse evaluation, vectorized batches. +Copy the **structure**: equivalence gates, complete signature identity, coalescing ownership, partitioned coordination, density-adaptive representation, shared materialization, adaptive trial ledgers, explicit approximation envelopes, early reduction, critical-path classification, performance budgets and sound bounds. -Adapt the **numbers**: trial counts, iteration caps, cache sizes, worker caps, sample/control rates, sparse cardinalities, timing thresholds, tolerances and environment-specific hashes. +Adapt the **numbers and policies**: trial counts, worker caps, hashes, cache sizes, shard counts, bit splits, batch widths, domain-contraction rates, acquisition parameters, tolerances, error limits, benchmark thresholds and stopping budgets. -## Evidence expected in a new OPT record +## Evidence expected in a new record At minimum record: -- source repository and immutable-enough source identity (release/commit/PR head); +- source identity and licensing/provenance boundary where relevant; +- the `OPTIMIZATION-PROBLEM.md` contract; - baseline and optimized behavior; - correctness/conformance gate; - benchmark environment or an explicit statement that no benchmark exists; - failure/rollback condition; -- whether the optimization changes latency, throughput, memory, CI time, or only architecture. +- whether the change affects latency, throughput, memory, I/O, CI time, quality or only architecture. + +## Formalization boundary + +`v1.0.0` contains five immutable Lean-formalized records. Post-v1 catalog records are not theorem-backed merely because they live in the same repository. See `FORMALIZATION.md`. diff --git a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md new file mode 100644 index 0000000..07c2a6c --- /dev/null +++ b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md @@ -0,0 +1,44 @@ +# OPT-APPROX-001 — Contract-bounded approximation + +**Status:** External production pattern + existing OPT need; target-specific proof/measurement required +**Domains:** visualization, search, streaming, telemetry, simulation, audition DSP + +## Source evidence + +- https://jazco.dev/2025/02/19/imperfection/ +- existing `OPT-DSP-001` approximation boundary +- `sources/JAZCO.md` + +## Problem + +Exact processing has unbounded or unacceptable cost even though the product/scientific contract permits a bounded loss of precision, completeness or freshness. + +## Optimization problem contract + +- Objective: reduce bounded resource/latency cost +- Constraint: declared error/degradation metric remains within `ε` or another explicit envelope +- Reference: exact path or exact fixture remains available for conformance + +## Preserved contract + +Approximation is admissible only when the contract explicitly permits it. A previously exact API cannot be silently weakened and still be called correctness-preserving. + +## Optimization + +Introduce a resource ceiling and degrade only along a declared dimension: sample/cull, lower level of detail, approximate search, bounded stale data, or reduced update frequency. Make the error surface measurable and reversible. + +## Validation + +Measure error/degradation and resource savings together across ordinary, boundary and adversarial workloads. Keep an exact reference for differential evaluation where practical. + +## Target-repo adaptation + +Define `ε`, quality metric, workload distribution, escape hatch and exact-mode availability locally. + +## Failure modes + +Unmeasured quality loss, biased sampling, hidden rare-case failures, cumulative error and callers incorrectly assuming exact semantics. + +## Rollback trigger + +Disable when error exceeds the declared envelope, reference comparisons drift, or resource savings are not material. diff --git a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md new file mode 100644 index 0000000..6f2ec98 --- /dev/null +++ b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md @@ -0,0 +1,45 @@ +# OPT-BUDGET-001 — Performance regression budgets + +**Status:** Established engineering pattern; enforcement must be environment-scoped +**Domains:** CI, web, numerical kernels, builds, services, DSP + +## Source evidence + +- `davidsonfellipe/awesome-wpo` inspected at `84f32948a6298456d6a94cff64551f39f2666e6f` +- performance-budget tooling and measurement resources catalogued upstream +- `sources/WPO.md` + +## Problem + +Small performance regressions accumulate because performance is measured occasionally but not guarded as an engineering contract. + +## Optimization problem contract + +- Metric: explicitly named latency/throughput/memory/I/O quantity +- Fixture/environment: pinned or sufficiently characterized +- Baseline distribution: repeated observations +- Budget: warning/hard boundary with justified statistical tolerance + +## Preserved contract + +A performance gate may not incentivize weakening functional tests, correctness, evidence or workload realism. + +## Optimization + +Turn a stable, reproducible performance expectation into a regression gate. Compare distributions or robust summaries where noise matters; separate machine/environment drift from code regression; keep cold/warm claims distinct. + +## Validation + +Calibrate variance before setting the threshold. Self-test the gate with known fast/slow fixtures and preserve raw samples where practical. + +## Target-repo adaptation + +Never copy another project's milliseconds, bundle sizes or thresholds. Establish the target's own baseline and noise envelope. + +## Failure modes + +Flaky gates from uncontrolled runners, benchmark gaming, stale fixtures, hardware drift and thresholds so loose they provide no protection. + +## Rollback trigger + +Temporarily disable only when the measurement environment is proven invalid; fix/recalibrate the benchmark rather than deleting the budget because code regressed. diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md new file mode 100644 index 0000000..ea0042a --- /dev/null +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -0,0 +1,45 @@ +# OPT-COAL-001 — Concurrent duplicate-work coalescing + +**Status:** Implemented external reference; target validation required +**Domains:** services, CI, artifact generation, metadata, parsing, model/data loading + +## Source evidence + +- Jazco, "Request Coalescing": https://jazco.dev/2023/09/28/request-coalescing/ +- `sources/JAZCO.md` + +## Problem + +Many callers request the same expensive computation concurrently before any caller has populated a reusable result, producing a thundering herd. + +## Optimization problem contract + +- Key: canonical identity of equivalent in-flight requests +- Objective: minimize duplicate concurrent evaluations +- Hard constraint: all joined callers must receive a result/error valid for their request semantics + +## Preserved contract + +Coalescing may merge only requests that are semantically equivalent for the shared operation. Cancellation, timeout, authorization and error semantics must remain explicit. + +## Optimization + +Make the first caller the owner of an in-flight operation. Equivalent callers subscribe to that future/result instead of starting duplicate work. Remove the in-flight entry deterministically on completion/failure. + +This differs from caching: the reusable result does not exist yet. + +## Validation + +Stress simultaneous identical and non-identical keys; inject owner failures/timeouts; prove only one upstream evaluation occurs for a coalesced key while all callers terminate correctly. + +## Target-repo adaptation + +Define key canonicalization, maximum waiter count, cancellation semantics and whether errors are shared or retried. + +## Failure modes + +Over-broad keys merge non-equivalent work; a hung owner can stall many callers; unbounded waiter lists amplify memory; shared error policy may cause correlated failure. + +## Rollback trigger + +Disable if coalescing changes request semantics, increases tail latency materially, or creates unacceptable failure amplification. diff --git a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md new file mode 100644 index 0000000..72c2b50 --- /dev/null +++ b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md @@ -0,0 +1,44 @@ +# OPT-CONT-001 — Partitioned coordination domains + +**Status:** Implemented external pattern; target validation required +**Domains:** ID allocation, runtimes, queues, counters, ingestion, schedulers + +## Source evidence + +- https://jazco.dev/2025/09/26/interning/ +- https://jazco.dev/2024/01/10/golang-and-epoll/ +- `sources/JAZCO.md` + +## Problem + +Independent workers serialize on one globally coordinated resource even though the underlying work could proceed independently. + +## Optimization problem contract + +- Variables: shard/domain count, namespace split, worker-to-domain mapping +- Objective: reduce coordination contention and tail latency +- Hard constraint: preserve the required global invariant (for example uniqueness or ordering scope) + +## Preserved contract + +Partitioning must not silently weaken uniqueness, ownership, visibility or ordering guarantees. If ordering becomes per-domain rather than global, that is a contract change and must be explicit. + +## Optimization + +Factor a global coordination space into independent domains. Encode domain identity into keys/IDs or route work so each domain can advance mostly independently. Prefer a small explicit merge/aggregation boundary to a permanently hot global lock/counter/poller. + +## Validation + +Check global invariants across all domains, collision/duplicate behavior, rebalance/restart behavior and target-scale contention profiles. + +## Target-repo adaptation + +Shard counts and bit splits are workload-specific. Measure skew, cache locality, failure domains and merge costs. + +## Failure modes + +Hot shards merely move the bottleneck; domain proliferation raises memory/management overhead; rebalancing may violate identity stability; global ordering requirements may make the pattern inadmissible. + +## Rollback trigger + +Revert if partitioning does not reduce measured contention or if any cross-domain invariant fails. diff --git a/optimizations/OPT-CRIT-001-critical-path-prioritization.md b/optimizations/OPT-CRIT-001-critical-path-prioritization.md new file mode 100644 index 0000000..6bbe498 --- /dev/null +++ b/optimizations/OPT-CRIT-001-critical-path-prioritization.md @@ -0,0 +1,44 @@ +# OPT-CRIT-001 — Critical-path prioritization + +**Status:** Established performance-engineering pattern; target validation required +**Domains:** UI, web, games, build systems, model/data loading, interactive pipelines + +## Source evidence + +- `davidsonfellipe/awesome-wpo` inspected at `84f32948a6298456d6a94cff64551f39f2666e6f` +- resource-hint, lazy-loading and prefetch references catalogued upstream +- `sources/WPO.md` + +## Problem + +Non-critical work competes with the dependency chain that determines user-visible or pipeline latency. + +## Optimization problem contract + +- Classify work: critical now / likely soon / deferrable / unnecessary +- Objective: reduce end-to-end critical-path latency +- Constraints: no starvation, stale-state or correctness violation from deferral/speculation + +## Preserved contract + +Deferred work must still complete before its semantic deadline. Speculative work must be discardable and must not create externally visible side effects before commitment. + +## Optimization + +Execute critical dependencies first; prefetch/precompute likely-soon work only when probability and spare resources justify it; lazily defer non-critical work; avoid work with no demonstrated demand. + +## Validation + +Trace the true dependency path and measure end-to-end latency, not only individual task duration. Test cold/warm, cache-hit/miss and wrong-speculation cases. + +## Target-repo adaptation + +Criticality and prediction horizons are workload-specific. Re-profile after topology or user-flow changes. + +## Failure modes + +Speculation steals resources from critical work, lazy work causes later latency cliffs, priorities become stale, or deferred tasks starve. + +## Rollback trigger + +Disable speculative/deferred policy if critical-path latency or resource pressure worsens materially. diff --git a/optimizations/OPT-FAN-001-shared-materialization-fanout.md b/optimizations/OPT-FAN-001-shared-materialization-fanout.md new file mode 100644 index 0000000..d731aa5 --- /dev/null +++ b/optimizations/OPT-FAN-001-shared-materialization-fanout.md @@ -0,0 +1,43 @@ +# OPT-FAN-001 — Shared materialization for fan-out and replay + +**Status:** Implemented external reference; target validation required +**Domains:** streaming, serialization, compression, artifact pipelines, multi-consumer services + +## Source evidence + +- Jazco, Jetstream: https://jazco.dev/2024/09/24/jetstream/ +- `sources/JAZCO.md` + +## Problem + +The same deterministic transformation is repeated independently for each consumer and again during replay. + +## Optimization problem contract + +- Variables: materialization boundary, representation format, persistence policy +- Objectives: transformation CPU, replay CPU, fan-out latency +- Hard constraint: materialized form must satisfy the consumer contract and versioning/trust requirements + +## Preserved contract + +Consumers must receive the same declared representation semantics. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. + +## Optimization + +Perform an expensive deterministic transform once near production, persist or retain the reusable representation, and fan out/replay those bytes/objects rather than reconstructing them per consumer. + +## Validation + +Compare shared materialization against per-consumer reference output, including version changes, corruption, restart/replay and mixed consumer capabilities. + +## Target-repo adaptation + +Choose representation versioning, invalidation, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. + +## Failure modes + +Materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. + +## Rollback trigger + +Disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md new file mode 100644 index 0000000..2de86cd --- /dev/null +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -0,0 +1,48 @@ +# OPT-INC-001 — Signature-bound incremental execution + +**Status:** Implemented historical reference; target validation required +**Domains:** builds, CI, generated artifacts, preprocessing, scientific pipelines + +## Source evidence + +- `psycledelics/wonderbuild` commit `021d5ed7c298c6c34b091cf5e6d9802e200028a6` +- `sources/WONDERBUILD.md` + +## Problem + +Expensive work is rerun even though every input capable of affecting its result is unchanged. + +## Optimization problem contract + +- Variables: signature definition, persistence scope, invalidation granularity +- Objective: minimize repeated work and metadata I/O +- Hard constraint: a reused result must correspond to the complete effective input identity +- Budget/stopping: target-specific + +## Preserved contract + +Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature. + +## Optimization + +Compute a deterministic signature over the effective inputs, compare it with successfully persisted prior state, and execute only when the signature differs or required outputs are missing. Persist the new signature only after success. Reuse filesystem/configuration metadata lazily when its own validity predicate still holds. + +## Evidence boundary + +Wonderbuild demonstrates the mechanism and benchmark shapes, but its historical timings and timestamp/hash choices are not transferable targets. + +## Validation + +Test unchanged, changed-input, missing-output, failed-run, and corrupted/stale-state cases against a forced-fresh reference path. + +## Target-repo adaptation + +Re-profile signature cost, hash choice, metadata granularity and persistence format. Include environment/toolchain inputs when they affect output. + +## Failure modes + +Incomplete signatures create stale reuse; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. + +## Rollback trigger + +Disable reuse if any cache/signature hit diverges from the fresh reference or if signature maintenance costs more than the avoided work. diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md new file mode 100644 index 0000000..2d6a5c6 --- /dev/null +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -0,0 +1,47 @@ +# OPT-PRUNE-001 — Bound-driven search-space pruning + +**Status:** Classical optimization mechanism; OPT adaptation guidance +**Domains:** combinatorial optimization, scheduling, assignment, configuration search, resource allocation + +## Source evidence + +- `sources/MATHEMATICAL-OPTIMIZATION.md` +- combinatorial optimization / branch-and-bound literature referenced there +- NLopt taxonomy as supporting optimizer-selection context + +## Problem + +A discrete or mixed search space is too large for exhaustive evaluation, but whole subregions can sometimes be proven unable to beat the best feasible solution already found. + +## Optimization problem contract + +- Search space `X`, feasible set `F`, objective `f` +- Incumbent: best validated feasible solution +- Bound: optimistic objective bound for each unexplored region +- Exactness: declare whether full branch-and-bound proof or anytime/budgeted search is required + +## Preserved contract + +A region may be discarded only when its bound proves it cannot improve the incumbent under the declared objective and constraints. Heuristic guesses are not proof-based pruning. + +## Optimization + +Maintain an incumbent, partition the search space, compute cheap optimistic bounds (often from relaxations), prioritize promising regions and prune any region whose best possible outcome cannot beat the incumbent. + +A relaxed solution is evidence for a bound, not automatically a feasible final answer. + +## Validation + +For small fixtures, compare with exhaustive enumeration. Test bound soundness separately from search ordering. Record the optimality gap when stopping before exact completion. + +## Target-repo adaptation + +The quality/cost of bounds determines whether pruning helps. Develop target-specific relaxations and branch ordering; do not assume one bound is universally strong. + +## Failure modes + +Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning. + +## Rollback trigger + +Disable any pruning rule that fails exhaustive small-case validation or whose bound cost exceeds the work it eliminates. diff --git a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md new file mode 100644 index 0000000..546535f --- /dev/null +++ b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md @@ -0,0 +1,43 @@ +# OPT-REDUCE-001 — Early working-set reduction + +**Status:** Implemented external pattern; broadly applicable mechanism +**Domains:** databases, graphs, simulation, DSP, rendering, data pipelines + +## Source evidence + +- https://jazco.dev/2023/08/10/query-optimization/ +- supporting sparse-evaluation pattern in `OPT-DSP-001` + +## Problem + +An expensive operation is applied to a large population even though only a small subset can affect the final result. + +## Optimization problem contract + +- Variable: where semantics-preserving filtering/culling/limiting occurs +- Objective: minimize cardinality presented to the expensive stage +- Hard constraint: early reduction must preserve every candidate required by the final result + +## Preserved contract + +Moving a reduction earlier is valid only if it is semantically equivalent to the original later reduction, including ordering/top-k/tie and join semantics where relevant. + +## Optimization + +Push selective operations toward the input boundary: filter before join, cull before render, select candidate roots before expensive DSP, prune impossible simulations before full evaluation. Prefer indexed/cheap predicates to expensive composition over the full population. + +## Validation + +Differential-test reordered pipelines against the reference, with emphasis on ties, null/missing values, boundary ordering and rare candidates. + +## Target-repo adaptation + +Measure selectivity and reduction cost. A cheap filter with low selectivity may simply add another pass. + +## Failure modes + +Illegal predicate reordering, changed top-k semantics, underestimated filtering cost, loss of vectorization and duplicated scans. + +## Rollback trigger + +Revert if outputs differ or total measured cost does not fall on representative workloads. diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md new file mode 100644 index 0000000..559591d --- /dev/null +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -0,0 +1,45 @@ +# OPT-SEARCH-001 — Budget-aware adaptive parameter search + +**Status:** Implemented external mechanisms; OPT synthesis proposed for target tuning +**Domains:** expensive black-box tuning, CI/runtime parameters, simulation, numerical kernels + +## Source evidence + +- `bayesian-optimization/BayesianOptimization` inspected at `af8b928212f0eacd1ce20c20be72c1a7b1d8d421` +- `hyperopt/hyperopt` inspected at `9834314879c09c13e0b8e93eb678408ba46441a8` +- `stevengj/nlopt` inspected at `6e6593f131ba3a38bc9edbed0a357bc01526e54b` +- `sources/OPTIMIZATION-LIBRARIES.md` + +## Problem + +Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitrary values even when each benchmark evaluation is expensive. + +## Optimization problem contract + +Define `P = (X,F,f,d,C,B,S)` from `OPTIMIZATION-PROBLEM.md`. Explicitly classify continuous/discrete/conditional variables, noise, constraints, gradient availability, local/global scope and evaluation cost. + +## Preserved contract + +Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. + +## Optimization + +Use observations to adapt future evaluations: surrogate/acquisition search for expensive black-box objectives, conditional spaces where parameters only exist under certain choices, progressive domain contraction where justified, and explicit stopping/evaluation budgets. For asynchronous workers, reserve pending regions or otherwise diversify proposals so workers do not redundantly evaluate the same neighborhood. + +Parallelism has an information cost: very wide batches receive less feedback between suggestions and can degenerate toward non-adaptive/random search. + +## Validation + +Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. + +## Target-repo adaptation + +Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. + +## Failure modes + +Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. + +## Rollback trigger + +Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, or repeated validation does not confirm the selected improvement. diff --git a/optimizations/OPT-SET-001-density-adaptive-compact-sets.md b/optimizations/OPT-SET-001-density-adaptive-compact-sets.md new file mode 100644 index 0000000..751f92f --- /dev/null +++ b/optimizations/OPT-SET-001-density-adaptive-compact-sets.md @@ -0,0 +1,48 @@ +# OPT-SET-001 — Density-adaptive compact set representation + +**Status:** Implemented external reference; target validation required +**Domains:** graphs, indexes, membership sets, telemetry, integer identifiers + +## Source evidence + +- https://jazco.dev/2024/04/20/roaring-bitmaps/ +- https://jazco.dev/2024/04/15/in-memory-graphs/ +- `sources/JAZCO.md` + +## Problem + +A single representation performs poorly across regions with very different density: sparse bitmaps waste memory, while list-like sparse structures make dense set algebra expensive. + +## Optimization problem contract + +- Variables: partition width, sparse/dense representation threshold, serialization layout +- Objectives: memory footprint and set-operation latency +- Hard constraint: exact set semantics unless approximation is explicitly introduced elsewhere + +## Preserved contract + +Membership and set operations must match the reference set exactly. + +## Optimization + +Partition the identifier space and choose a representation per partition according to local density. Keep sparse regions compact while using bitmap-like containers where dense boolean algebra is advantageous. Prefer representations that can be serialized without expanding to a larger intermediate form. + +## Evidence boundary + +Jazco reports strong production-scale graph results, but OPT treats the numbers as source observations only. The portable claim is density-adaptive representation. + +## Validation + +Differential-test membership, union, intersection, difference and persistence against a simple canonical set implementation over sparse, dense and transition-boundary fixtures. + +## Target-repo adaptation + +Benchmark partition sizes and switching thresholds on the real identifier distribution and CPU/cache hierarchy. + +## Failure modes + +Conversion churn near thresholds, pathological distributions, serialization incompatibility and hidden temporary allocations can erase the benefit. + +## Rollback trigger + +Revert when target data does not show a memory/latency win or exact set differential tests fail. diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py new file mode 100755 index 0000000..d262769 --- /dev/null +++ b/scripts/check_catalog.py @@ -0,0 +1,73 @@ +#!/usr/bin/env python3 +"""Check OPT catalog/document integrity without external dependencies.""" + +from __future__ import annotations + +import re +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +OPT_DIR = ROOT / "optimizations" +FROZEN_V1 = { + "OPT-PY-001", + "OPT-INV-001", + "OPT-LEAN-001", + "OPT-PAR-001", + "OPT-DSP-001", +} +REQUIRED_V2 = { + "## Source evidence", + "## Problem", + "## Optimization problem contract", + "## Preserved contract", + "## Optimization", + "## Validation", + "## Target-repo adaptation", + "## Failure modes", + "## Rollback trigger", +} +LINK_RE = re.compile(r"\[[^\]]+\]\((optimizations/[^)#]+\.md)\)") +ID_RE = re.compile(r"^# (OPT-[A-Z]+-\d{3}) — ") + + +def die(msg: str) -> None: + raise SystemExit(f"catalog-integrity: {msg}") + + +records: dict[str, Path] = {} +for path in sorted(OPT_DIR.glob("OPT-*.md")): + text = path.read_text(encoding="utf-8") + first = text.splitlines()[0] if text else "" + match = ID_RE.match(first) + if not match: + die(f"bad record heading: {path.relative_to(ROOT)}") + record_id = match.group(1) + if record_id in records: + die(f"duplicate record id {record_id}: {records[record_id]} and {path}") + records[record_id] = path + if "**Status:**" not in text: + die(f"missing Status in {path.relative_to(ROOT)}") + if record_id not in FROZEN_V1: + missing = sorted(section for section in REQUIRED_V2 if section not in text) + if missing: + die(f"{path.relative_to(ROOT)} missing sections: {', '.join(missing)}") + +for doc_name in ("README.md", "CATALOG.md"): + text = (ROOT / doc_name).read_text(encoding="utf-8") + links = LINK_RE.findall(text) + if not links: + die(f"{doc_name} contains no optimization-record links") + for rel in links: + if not (ROOT / rel).is_file(): + die(f"broken record link in {doc_name}: {rel}") + +catalog = (ROOT / "CATALOG.md").read_text(encoding="utf-8") +for record_id, path in records.items(): + if record_id not in catalog: + die(f"{record_id} ({path.name}) is not mentioned in CATALOG.md") + +problem_contract = ROOT / "OPTIMIZATION-PROBLEM.md" +if not problem_contract.is_file(): + die("OPTIMIZATION-PROBLEM.md is missing") + +print(f"CATALOG_INTEGRITY_OK records={len(records)} frozen_v1={len(FROZEN_V1)}") diff --git a/sources/JAZCO.md b/sources/JAZCO.md new file mode 100644 index 0000000..bd65dfc --- /dev/null +++ b/sources/JAZCO.md @@ -0,0 +1,25 @@ +# Source note — Jazco performance engineering articles + +**Source index:** https://jazco.dev/ + +OPT uses these articles as mechanism donors and case studies. Reported production numbers remain source-reported historical observations unless independently reproduced in a target repository. + +## Articles and extracted mechanisms + +- `2023/09/28/request-coalescing/` — suppress concurrent duplicate work by sharing one in-flight result. +- `2024/04/20/roaring-bitmaps/` — density-adaptive integer-set representation and fast set algebra. +- `2025/09/26/interning/` — compact identity representation and partitioning of a global allocation hotspot into independent coordination domains. +- `2024/09/24/jetstream/` — compute/encode once, persist the reusable representation, fan out/replay without repeating transformation work. +- `2024/04/15/in-memory-graphs/` — representation compression can cross an architectural threshold and turn remote-query work into local set algebra. +- `2024/01/10/golang-and-epoll/` — vertical runtime coordination can become a bottleneck; partitioning processes/resources may outperform further vertical scaling; also illustrates explicit resource trade-offs. +- `2025/02/19/imperfection/` — bounded approximation under an explicit product/semantic contract. +- `2023/08/10/query-optimization/` — reduce a working set before expensive composition/join work. +- `2023/05/20/postgres-analyze/` — optimization quality depends on current workload/statistical models. + +## Evidence boundary + +OPT does not treat blog-case constants, machine sizes, throughput figures, thresholds or schema-specific choices as transferable. Target repositories must re-measure. + +## Records informed + +`OPT-COAL-001`, `OPT-SET-001`, `OPT-CONT-001`, `OPT-FAN-001`, `OPT-APPROX-001`, and `OPT-REDUCE-001`. diff --git a/sources/MATHEMATICAL-OPTIMIZATION.md b/sources/MATHEMATICAL-OPTIMIZATION.md new file mode 100644 index 0000000..3e392ef --- /dev/null +++ b/sources/MATHEMATICAL-OPTIMIZATION.md @@ -0,0 +1,18 @@ +# Source note — mathematical and combinatorial optimization foundations + +This source note supplies vocabulary and problem structure, not benchmark evidence. + +## Sources + +- Wang, L., Zhao, J. (2023), "Mathematical Optimization", in *Architecture of Advanced Numerical Analysis Systems*, Apress. DOI: https://doi.org/10.1007/978-1-4842-8853-5_4 +- Optimization problem overview: https://en.wikipedia.org/wiki/Optimization_problem +- Combinatorial optimization overview: https://en.wikipedia.org/wiki/Combinatorial_optimization +- NLopt algorithm taxonomy: https://github.com/stevengj/nlopt + +## Extraction for OPT + +OPT models a target problem using a search space, feasible set, objective, direction, correctness constraints, evaluation budget and stopping rule. The problem is classified before selecting the mechanism: continuous/discrete/mixed, local/global, deterministic/noisy, gradient/derivative-free, exact/approximate and sequential/parallel. + +Combinatorial optimization contributes a distinct mechanism: maintain an incumbent feasible solution, compute optimistic bounds for subregions, and prune regions that provably cannot improve the incumbent. Relaxations may be used to obtain cheap bounds without treating the relaxed solution as the final answer. + +See `OPTIMIZATION-PROBLEM.md` and `OPT-PRUNE-001`. diff --git a/sources/OPTIMIZATION-LIBRARIES.md b/sources/OPTIMIZATION-LIBRARIES.md new file mode 100644 index 0000000..380258f --- /dev/null +++ b/sources/OPTIMIZATION-LIBRARIES.md @@ -0,0 +1,28 @@ +# Source note — adaptive and nonlinear optimization libraries + +These projects inform `OPT-SEARCH-001` and the problem-classification guidance. OPT does not vendor them. + +## BayesianOptimization + +- Repository: https://github.com/bayesian-optimization/BayesianOptimization +- inspected identity: `af8b928212f0eacd1ce20c20be72c1a7b1d8d421` +- license: MIT +- useful mechanisms: Gaussian-process surrogate search for expensive objectives; exploration/exploitation acquisition functions; sequential domain reduction; asynchronous diversification via `ConstantLiar`; acquisition-function portfolio selection via `GPHedge`. + +## Hyperopt + +- Repository: https://github.com/hyperopt/hyperopt +- inspected identity: `9834314879c09c13e0b8e93eb678408ba46441a8` +- license: BSD-style permissive license in `LICENSE.txt` +- useful mechanisms: search over real, discrete and conditional spaces; Tree of Parzen Estimators; distributed evaluation; explicit trade-off between adaptive information flow and parallel batch width. + +## NLopt + +- Repository: https://github.com/stevengj/nlopt +- inspected identity: `6e6593f131ba3a38bc9edbed0a357bc01526e54b` +- licensing: combined default build includes LGPL-governed components; build configurations without the Luksan code may use the documented MIT terms. See upstream `COPYING` before incorporating code. +- useful mechanisms: explicit taxonomy of global/local and gradient/derivative-free algorithms; hybrid global/local search; objective/parameter/evaluation/time stopping criteria. + +## OPT extraction + +The reusable rule is not "always use Bayesian optimization". First classify the problem, then choose a search mechanism that matches variable type, objective cost/noise, constraints, gradient availability, exactness requirement and evaluation budget. diff --git a/sources/WONDERBUILD.md b/sources/WONDERBUILD.md new file mode 100644 index 0000000..67885ad --- /dev/null +++ b/sources/WONDERBUILD.md @@ -0,0 +1,31 @@ +# Source note — Psycledelics Wonderbuild + +**Source:** https://github.com/psycledelics/wonderbuild +**Pinned source identity:** commit `021d5ed7c298c6c34b091cf5e6d9802e200028a6` +**Role in OPT:** historical implementation donor, not a dependency. + +Wonderbuild is an older Python build system from the Psycle/Psycledelics community. Its reusable value for OPT is architectural rather than its Python-2-era implementation. + +## Mechanisms extracted + +- persistent input signatures and selective invalidation; +- success-only persistence of updated task signatures; +- cached configuration checks keyed by their effective inputs; +- dependency-aware task scheduling with run-once semantics; +- filesystem metadata/state reuse to reduce repeated scans; +- benchmark separation between cold, no-op, small-partial and large-partial rebuilds; +- historical translation-unit batching as a compiler-process amortization idea. + +## Evidence boundary + +Wonderbuild's own benchmark material is historical and environment-specific. OPT does not promote those timings as modern targets. The reusable claim is the mechanism and test shape. + +## Licensing boundary + +Wonderbuild source files state GPL-2.0-or-later terms and credit Psycle project contributors including Johan Boule. OPT therefore records the design patterns and provenance without copying Wonderbuild implementation code into this Apache-2.0 repository. + +## Records informed + +- `OPT-INC-001` +- `OPT-PAR-001` (supporting scheduling evidence) +- future C/C++ batching experiments may cite this source, but no portable compiler speedup is claimed here. diff --git a/sources/WPO.md b/sources/WPO.md new file mode 100644 index 0000000..fd51c73 --- /dev/null +++ b/sources/WPO.md @@ -0,0 +1,14 @@ +# Source note — Awesome WPO + +- Repository: https://github.com/davidsonfellipe/awesome-wpo +- inspected identity: `84f32948a6298456d6a94cff64551f39f2666e6f` +- license: MIT + +Awesome WPO is a curated web-performance index, not primary benchmark evidence. OPT uses it as a discovery source for two general mechanisms that extend beyond browsers: + +1. **critical-path prioritization** — do latency-critical work now, speculate/prefetch likely-soon work when justified, lazily defer non-critical work, and avoid work with no demonstrated demand; +2. **performance budgets** — convert measured performance expectations into regression gates rather than relying on human memory of what "used to be fast". + +Examples in the source catalog include lazy loaders, viewport-driven prefetching/resource hints, performance-budget tooling, browser timing APIs, Lighthouse/WebPageTest-style measurement, and real-user monitoring. + +Any target-repository record must define its own metric, workload, environment, statistical tolerance and rollback rule. Browser-specific thresholds are not copied as universal constants. diff --git a/templates/OPTIMIZATION-RECORD.md b/templates/OPTIMIZATION-RECORD.md index 45141c7..f0d9fdb 100644 --- a/templates/OPTIMIZATION-RECORD.md +++ b/templates/OPTIMIZATION-RECORD.md @@ -1,21 +1,41 @@ # OPT-XXX-000 — Optimization Name -**Status:** Proposed / Implemented reference / Verified / Verified, environment-specific +**Status:** Proposed / Implemented reference / Implemented external reference / Verified / Verified, environment-specific **Domains:** ... ## Source evidence -- Repository: -- Release/commit/PR: -- Exact files: +- Repository / publication / article: +- Release/commit/PR/DOI/date: +- Exact files/sections where applicable: +- Licensing/provenance boundary where code reuse may matter: ## Problem -What dominates runtime, latency, memory, I/O or CI cost? +What dominates runtime, latency, memory, I/O, CI cost, quality budget or optimization-evaluation cost? + +## Optimization problem contract + +Define the target using `OPTIMIZATION-PROBLEM.md`: + +- Search space `X`: +- Feasible set `F`: +- Objective `f`: +- Direction: minimize / maximize / explicit multi-objective ordering +- Correctness / semantic contract `C`: +- Evaluation/resource budget `B`: +- Stopping rule `S`: +- Variables: continuous / integer / categorical / conditional / mixed +- Objective: deterministic / noisy / stochastic +- Search scope: local / global +- Information: gradient / derivative-free / black-box +- Exactness: exact / approximation permitted under explicit error contract ## Preserved contract -State exactly what must remain unchanged: output bytes, theorem targets, assertions, API, numerical tolerance, ordering, statistical guarantee, evidence boundary, etc. +State exactly what must remain unchanged: output bytes, theorem targets, assertions, API, numerical tolerance, ordering, statistical guarantee, evidence boundary, trust model, etc. + +If the optimization changes the contract (for example exact → approximate), state the new contract explicitly instead of claiming preservation. ## Optimization @@ -24,25 +44,33 @@ Describe the reusable mechanism, not only the source-project patch. ## Before / after evidence - Environment: -- Baseline: +- Workload/fixture: +- Cold baseline: +- Warm/no-op baseline where relevant: +- Small invalidation / partial-work case where relevant: +- Large invalidation / full-work case where relevant: - Optimized: -- Speedup / memory reduction: -- Variance / repetitions: +- Speedup / memory / I/O / quality change: +- Variance / repetitions / raw samples: If no controlled benchmark exists, say so explicitly. ## Validation -How was equivalence/correctness established? +How was equivalence, correctness, bound soundness, approximation error or other contract compliance established? ## Target-repo adaptation -Which source constants must be re-profiled rather than copied? +Which source constants, thresholds, worker counts, bit splits, cache keys, search budgets or tolerances must be re-profiled rather than copied? ## Failure modes -What can make this optimization invalid or slower? +What can make this optimization invalid, slower, less robust or misleading? ## Rollback trigger -Define the condition that disables or reverts the optimization. +Define the measured or semantic condition that disables/reverts the optimization. + +## Composition notes + +Which other OPT records compose safely, and which resource/semantic interactions must be re-measured? From 444f297651900f31609278174ed1611e6560fc8d Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 10:38:03 +0930 Subject: [PATCH 002/229] Fix catalog integrity links --- CATALOG.md | 32 ++++++++++++++++---------------- 1 file changed, 16 insertions(+), 16 deletions(-) diff --git a/CATALOG.md b/CATALOG.md index 6d4f3fe..2993fa9 100644 --- a/CATALOG.md +++ b/CATALOG.md @@ -4,22 +4,22 @@ | Bottleneck / problem shape | First record to inspect | Core idea | | --- | --- | --- | -| Deterministic tests/sweeps dominate runtime | `OPT-PY-001` | Reduce redundant/high-cost work while keeping coverage semantics | -| Same expensive result is recomputed at a proven-equivalent state | `OPT-INV-001` | Prove equivalence, then reuse | -| Lean dependency reconstruction dominates CI | `OPT-LEAN-001` | Verify reusable dependency state; rebuild current project source | -| Independent work can execute concurrently | `OPT-PAR-001` | Bound workers and prove scalar/parallel equivalence | -| Slow control state is inside a high-rate numerical/audio loop | `OPT-DSP-001` | Separate rates, sparse-evaluate, vectorize | -| Inputs are unchanged but pipeline stages rerun | `OPT-INC-001` | Bind work to complete input signatures and persist only successful state | -| Many simultaneous callers request identical not-yet-computed work | `OPT-COAL-001` | One in-flight computation, many waiters | -| Integer sets alternate between sparse and dense regions | `OPT-SET-001` | Density-adaptive representation with exact set algebra | -| One global lock/counter/runtime domain serializes independent work | `OPT-CONT-001` | Partition coordination while preserving the global invariant | -| Same deterministic transform is repeated for every consumer/replay | `OPT-FAN-001` | Materialize once, reuse many times | -| Expensive parameter evaluations are being guessed or exhaustively swept | `OPT-SEARCH-001` | Adaptive, budget-aware search over the declared problem contract | -| Exactness may be traded inside an explicit quality envelope | `OPT-APPROX-001` | Bound the error/degradation and the resource cost together | -| Expensive stages consume candidates later discarded | `OPT-REDUCE-001` | Reduce the working set before composition | -| Non-critical work delays the dependency chain users actually wait on | `OPT-CRIT-001` | Prioritize the critical path; speculate/defer deliberately | -| Small performance regressions accumulate unnoticed | `OPT-BUDGET-001` | Guard stable performance expectations in CI | -| Discrete search space is huge but optimistic bounds are available | `OPT-PRUNE-001` | Prune regions that provably cannot beat the incumbent | +| Deterministic tests/sweeps dominate runtime | [OPT-PY-001](optimizations/OPT-PY-001-deterministic-test-execution.md) | Reduce redundant/high-cost work while keeping coverage semantics | +| Same expensive result is recomputed at a proven-equivalent state | [OPT-INV-001](optimizations/OPT-INV-001-invariant-driven-reuse.md) | Prove equivalence, then reuse | +| Lean dependency reconstruction dominates CI | [OPT-LEAN-001](optimizations/OPT-LEAN-001-trust-preserving-lean-ci.md) | Verify reusable dependency state; rebuild current project source | +| Independent work can execute concurrently | [OPT-PAR-001](optimizations/OPT-PAR-001-bounded-parallel-execution.md) | Bound workers and prove scalar/parallel equivalence | +| Slow control state is inside a high-rate numerical/audio loop | [OPT-DSP-001](optimizations/OPT-DSP-001-control-rate-sparse-vector-dsp.md) | Separate rates, sparse-evaluate, vectorize | +| Inputs are unchanged but pipeline stages rerun | [OPT-INC-001](optimizations/OPT-INC-001-signature-bound-incremental-execution.md) | Bind work to complete input signatures and persist only successful state | +| Many simultaneous callers request identical not-yet-computed work | [OPT-COAL-001](optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md) | One in-flight computation, many waiters | +| Integer sets alternate between sparse and dense regions | [OPT-SET-001](optimizations/OPT-SET-001-density-adaptive-compact-sets.md) | Density-adaptive representation with exact set algebra | +| One global lock/counter/runtime domain serializes independent work | [OPT-CONT-001](optimizations/OPT-CONT-001-partitioned-coordination-domains.md) | Partition coordination while preserving the global invariant | +| Same deterministic transform is repeated for every consumer/replay | [OPT-FAN-001](optimizations/OPT-FAN-001-shared-materialization-fanout.md) | Materialize once, reuse many times | +| Expensive parameter evaluations are being guessed or exhaustively swept | [OPT-SEARCH-001](optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md) | Adaptive, budget-aware search over the declared problem contract | +| Exactness may be traded inside an explicit quality envelope | [OPT-APPROX-001](optimizations/OPT-APPROX-001-contract-bounded-approximation.md) | Bound the error/degradation and the resource cost together | +| Expensive stages consume candidates later discarded | [OPT-REDUCE-001](optimizations/OPT-REDUCE-001-early-working-set-reduction.md) | Reduce the working set before composition | +| Non-critical work delays the dependency chain users actually wait on | [OPT-CRIT-001](optimizations/OPT-CRIT-001-critical-path-prioritization.md) | Prioritize the critical path; speculate/defer deliberately | +| Small performance regressions accumulate unnoticed | [OPT-BUDGET-001](optimizations/OPT-BUDGET-001-performance-regression-budgets.md) | Guard stable performance expectations in CI | +| Discrete search space is huge but optimistic bounds are available | [OPT-PRUNE-001](optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md) | Prune regions that provably cannot beat the incumbent | Before selecting a record, define the target problem using [`OPTIMIZATION-PROBLEM.md`](OPTIMIZATION-PROBLEM.md). From fd9e84c2179552ebf350875001989ce70422c8d7 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 10:56:57 +0930 Subject: [PATCH 003/229] Enforce post-v1 evidence schema --- .../OPT-APPROX-001-contract-bounded-approximation.md | 8 ++++++++ .../OPT-BUDGET-001-performance-regression-budgets.md | 8 ++++++++ .../OPT-COAL-001-concurrent-duplicate-work-coalescing.md | 8 ++++++++ .../OPT-CONT-001-partitioned-coordination-domains.md | 8 ++++++++ .../OPT-CRIT-001-critical-path-prioritization.md | 8 ++++++++ .../OPT-FAN-001-shared-materialization-fanout.md | 8 ++++++++ .../OPT-INC-001-signature-bound-incremental-execution.md | 8 ++++++++ .../OPT-PRUNE-001-bound-driven-search-space-pruning.md | 8 ++++++++ .../OPT-REDUCE-001-early-working-set-reduction.md | 8 ++++++++ .../OPT-SEARCH-001-budget-aware-adaptive-search.md | 8 ++++++++ .../OPT-SET-001-density-adaptive-compact-sets.md | 8 ++++++++ scripts/check_catalog.py | 9 +++++++-- 12 files changed, 95 insertions(+), 2 deletions(-) diff --git a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md index 07c2a6c..c1fc704 100644 --- a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md +++ b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md @@ -27,6 +27,14 @@ Approximation is admissible only when the contract explicitly permits it. A prev Introduce a resource ceiling and degrade only along a declared dimension: sample/cull, lower level of detail, approximate search, bounded stale data, or reduced update frequency. Make the error surface measurable and reversible. +## Before / after evidence + +- Environment: No controlled target-repository benchmark has been run for this OPT record. +- Baseline: No exact target baseline has been established here. +- Optimized: No target approximation implementation has been benchmarked here. +- Speedup / memory reduction: No transferable claim; external production observations remain source evidence only. +- Variance / repetitions: Not available for a controlled OPT target benchmark. + ## Validation Measure error/degradation and resource savings together across ordinary, boundary and adversarial workloads. Keep an exact reference for differential evaluation where practical. diff --git a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md index 6f2ec98..dad7090 100644 --- a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md +++ b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md @@ -28,6 +28,14 @@ A performance gate may not incentivize weakening functional tests, correctness, Turn a stable, reproducible performance expectation into a regression gate. Compare distributions or robust summaries where noise matters; separate machine/environment drift from code regression; keep cold/warm claims distinct. +## Before / after evidence + +- Environment: No controlled target-repository budget calibration has been run for this OPT record. +- Baseline: No target baseline distribution is claimed here. +- Optimized: Not applicable until a target repository adopts and calibrates a performance budget. +- Speedup / memory reduction: This pattern protects performance; it does not itself claim a speedup. +- Variance / repetitions: Must be established in the target environment before a hard threshold is promoted. + ## Validation Calibrate variance before setting the threshold. Self-test the gate with known fast/slow fixtures and preserve raw samples where practical. diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index ea0042a..582abe3 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -28,6 +28,14 @@ Make the first caller the owner of an in-flight operation. Equivalent callers su This differs from caching: the reusable result does not exist yet. +## Before / after evidence + +- Environment: No controlled target-repository benchmark has been run for this OPT record. +- Baseline: Not established in a target repository. +- Optimized: Not established in a target repository. +- Speedup / memory reduction: No transferable claim; the Jazco implementation is source evidence for the mechanism. +- Variance / repetitions: Not available for a controlled OPT target benchmark. + ## Validation Stress simultaneous identical and non-identical keys; inject owner failures/timeouts; prove only one upstream evaluation occurs for a coalesced key while all callers terminate correctly. diff --git a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md index 72c2b50..bda16b9 100644 --- a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md +++ b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md @@ -27,6 +27,14 @@ Partitioning must not silently weaken uniqueness, ownership, visibility or order Factor a global coordination space into independent domains. Encode domain identity into keys/IDs or route work so each domain can advance mostly independently. Prefer a small explicit merge/aggregation boundary to a permanently hot global lock/counter/poller. +## Before / after evidence + +- Environment: No controlled target-repository benchmark has been run for this OPT record. +- Baseline: Not established in a target repository. +- Optimized: Not established in a target repository. +- Speedup / memory reduction: No transferable claim; donor observations motivate the pattern only. +- Variance / repetitions: Not available for a controlled OPT target benchmark. + ## Validation Check global invariants across all domains, collision/duplicate behavior, rebalance/restart behavior and target-scale contention profiles. diff --git a/optimizations/OPT-CRIT-001-critical-path-prioritization.md b/optimizations/OPT-CRIT-001-critical-path-prioritization.md index 6bbe498..80b157e 100644 --- a/optimizations/OPT-CRIT-001-critical-path-prioritization.md +++ b/optimizations/OPT-CRIT-001-critical-path-prioritization.md @@ -27,6 +27,14 @@ Deferred work must still complete before its semantic deadline. Speculative work Execute critical dependencies first; prefetch/precompute likely-soon work only when probability and spare resources justify it; lazily defer non-critical work; avoid work with no demonstrated demand. +## Before / after evidence + +- Environment: No controlled target-repository benchmark has been run for this OPT record. +- Baseline: No target critical-path profile has been established here. +- Optimized: No target prioritization/prefetch policy has been benchmarked here. +- Speedup / memory reduction: No transferable claim; upstream WPO material supplies patterns and measurement guidance. +- Variance / repetitions: Not available for a controlled OPT target benchmark. + ## Validation Trace the true dependency path and measure end-to-end latency, not only individual task duration. Test cold/warm, cache-hit/miss and wrong-speculation cases. diff --git a/optimizations/OPT-FAN-001-shared-materialization-fanout.md b/optimizations/OPT-FAN-001-shared-materialization-fanout.md index d731aa5..a08fce8 100644 --- a/optimizations/OPT-FAN-001-shared-materialization-fanout.md +++ b/optimizations/OPT-FAN-001-shared-materialization-fanout.md @@ -26,6 +26,14 @@ Consumers must receive the same declared representation semantics. Removing veri Perform an expensive deterministic transform once near production, persist or retain the reusable representation, and fan out/replay those bytes/objects rather than reconstructing them per consumer. +## Before / after evidence + +- Environment: No controlled target-repository benchmark has been run for this OPT record. +- Baseline: Not established in a target repository. +- Optimized: Not established in a target repository. +- Speedup / memory reduction: No transferable claim; Jetstream observations remain external source evidence. +- Variance / repetitions: Not available for a controlled OPT target benchmark. + ## Validation Compare shared materialization against per-consumer reference output, including version changes, corruption, restart/replay and mixed consumer capabilities. diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index 2de86cd..d6f3dea 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -31,6 +31,14 @@ Compute a deterministic signature over the effective inputs, compare it with suc Wonderbuild demonstrates the mechanism and benchmark shapes, but its historical timings and timestamp/hash choices are not transferable targets. +## Before / after evidence + +- Environment: No controlled target-repository benchmark has been run for this OPT record. +- Baseline: Not established in a target repository. +- Optimized: Not established in a target repository. +- Speedup / memory reduction: No transferable claim. Wonderbuild observations are historical source evidence only. +- Variance / repetitions: Not available for a controlled OPT target benchmark. + ## Validation Test unchanged, changed-input, missing-output, failed-run, and corrupted/stale-state cases against a forced-fresh reference path. diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index 2d6a5c6..19c0a3e 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -30,6 +30,14 @@ Maintain an incumbent, partition the search space, compute cheap optimistic boun A relaxed solution is evidence for a bound, not automatically a feasible final answer. +## Before / after evidence + +- Environment: No controlled target-repository benchmark has been run for this OPT record. +- Baseline: No target exhaustive or unpruned search baseline has been established here. +- Optimized: No target branch-and-bound/pruned search result has been established here. +- Speedup / memory reduction: No transferable claim; this record captures a classical mechanism and adaptation rules. +- Variance / repetitions: Not available for a controlled OPT target benchmark. + ## Validation For small fixtures, compare with exhaustive enumeration. Test bound soundness separately from search ordering. Record the optimality gap when stopping before exact completion. diff --git a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md index 546535f..06f2150 100644 --- a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md +++ b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md @@ -26,6 +26,14 @@ Moving a reduction earlier is valid only if it is semantically equivalent to the Push selective operations toward the input boundary: filter before join, cull before render, select candidate roots before expensive DSP, prune impossible simulations before full evaluation. Prefer indexed/cheap predicates to expensive composition over the full population. +## Before / after evidence + +- Environment: No controlled target-repository benchmark has been run for this OPT record. +- Baseline: Not established in a target repository. +- Optimized: Not established in a target repository. +- Speedup / memory reduction: No transferable claim; external query observations and existing sparse-evaluation patterns are source evidence only. +- Variance / repetitions: Not available for a controlled OPT target benchmark. + ## Validation Differential-test reordered pipelines against the reference, with emphasis on ties, null/missing values, boundary ordering and rare candidates. diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index 559591d..1b14ec5 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -28,6 +28,14 @@ Use observations to adapt future evaluations: surrogate/acquisition search for e Parallelism has an information cost: very wide batches receive less feedback between suggestions and can degenerate toward non-adaptive/random search. +## Before / after evidence + +- Environment: No controlled target-repository tuning study has been run for this OPT synthesis. +- Baseline: No target comparison against manual/exhaustive/random tuning has been established. +- Optimized: No target adaptive-search result has been established. +- Speedup / memory reduction: No transferable claim; upstream libraries establish mechanisms, not a QSOL target win. +- Variance / repetitions: Not available for a controlled OPT target benchmark. + ## Validation Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. diff --git a/optimizations/OPT-SET-001-density-adaptive-compact-sets.md b/optimizations/OPT-SET-001-density-adaptive-compact-sets.md index 751f92f..608b49c 100644 --- a/optimizations/OPT-SET-001-density-adaptive-compact-sets.md +++ b/optimizations/OPT-SET-001-density-adaptive-compact-sets.md @@ -31,6 +31,14 @@ Partition the identifier space and choose a representation per partition accordi Jazco reports strong production-scale graph results, but OPT treats the numbers as source observations only. The portable claim is density-adaptive representation. +## Before / after evidence + +- Environment: No controlled target-repository benchmark has been run for this OPT record. +- Baseline: Not established in a target repository. +- Optimized: Not established in a target repository. +- Speedup / memory reduction: No transferable claim; reported graph results remain historical external observations. +- Variance / repetitions: Not available for a controlled OPT target benchmark. + ## Validation Differential-test membership, union, intersection, difference and persistence against a simple canonical set implementation over sparse, dense and transition-boundary fixtures. diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index d262769..1450214 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -21,6 +21,7 @@ "## Optimization problem contract", "## Preserved contract", "## Optimization", + "## Before / after evidence", "## Validation", "## Target-repo adaptation", "## Failure modes", @@ -52,11 +53,15 @@ def die(msg: str) -> None: if missing: die(f"{path.relative_to(ROOT)} missing sections: {', '.join(missing)}") +# README is the human-facing record index and must contain real Markdown links. +# CATALOG may use either links or plain/backticked record IDs; any links it does +# contain are still validated below, while complete catalog coverage is enforced +# independently by record ID. for doc_name in ("README.md", "CATALOG.md"): text = (ROOT / doc_name).read_text(encoding="utf-8") links = LINK_RE.findall(text) - if not links: - die(f"{doc_name} contains no optimization-record links") + if doc_name == "README.md" and not links: + die("README.md contains no optimization-record links") for rel in links: if not (ROOT / rel).is_file(): die(f"broken record link in {doc_name}: {rel}") From d5dbd96f1fa3308ac1fa490959c4a7c84d43eead Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 11:54:06 +0930 Subject: [PATCH 004/229] Harden canonical optimization contracts --- ...PROX-001-contract-bounded-approximation.md | 10 +++-- ...DGET-001-performance-regression-budgets.md | 11 ++++-- ...01-concurrent-duplicate-work-coalescing.md | 10 +++-- ...NT-001-partitioned-coordination-domains.md | 10 +++-- ...T-CRIT-001-critical-path-prioritization.md | 10 +++-- ...T-FAN-001-shared-materialization-fanout.md | 10 +++-- ...1-signature-bound-incremental-execution.md | 11 ++++-- ...E-001-bound-driven-search-space-pruning.md | 11 ++++-- ...-REDUCE-001-early-working-set-reduction.md | 10 +++-- ...SEARCH-001-budget-aware-adaptive-search.md | 8 +++- ...T-SET-001-density-adaptive-compact-sets.md | 10 +++-- scripts/check_catalog.py | 39 ++++++++++++++++++- templates/OPTIMIZATION-RECORD.md | 23 ++++++----- 13 files changed, 127 insertions(+), 46 deletions(-) diff --git a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md index c1fc704..7045a46 100644 --- a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md +++ b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md @@ -15,9 +15,13 @@ Exact processing has unbounded or unacceptable cost even though the product/scie ## Optimization problem contract -- Objective: reduce bounded resource/latency cost -- Constraint: declared error/degradation metric remains within `ε` or another explicit envelope -- Reference: exact path or exact fixture remains available for conformance +- X: target-supported approximation policies, quality/resource ceilings, sampling/culling/LOD policies, update frequencies, and exact-mode fallback choices +- F: policies whose declared error/degradation metric remains within the target's explicit envelope and whose resource/semantic constraints are satisfied +- f: target-measured resource or latency cost, optionally paired with the declared quality/error metric +- d: minimize resource/latency cost subject to feasibility in F, or use the target's predeclared multi-objective ordering when quality is ranked rather than hard-bounded +- C: approximation is permitted only by an explicit contract; exact callers are not silently weakened, and an exact reference path or exact fixture remains available where practical +- B: target-specific benchmark/quality-evaluation budget over predeclared ordinary, boundary, and adversarial fixtures +- S: stop when the evaluation budget is exhausted or a validated policy meets the target resource objective while remaining inside the declared quality envelope ## Preserved contract diff --git a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md index dad7090..ae34bd9 100644 --- a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md +++ b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md @@ -15,10 +15,13 @@ Small performance regressions accumulate because performance is measured occasio ## Optimization problem contract -- Metric: explicitly named latency/throughput/memory/I/O quantity -- Fixture/environment: pinned or sufficiently characterized -- Baseline distribution: repeated observations -- Budget: warning/hard boundary with justified statistical tolerance +- X: target-supported metric/fixture/statistic/threshold configurations for a performance-regression gate +- F: gate configurations based on a sufficiently characterized environment and workload, with statistically justified tolerance and no weakening of functional correctness or workload realism +- f: target-measured regression-detection quality together with CI noise/false-alarm rate and measurement overhead +- d: minimize missed material regressions and flaky/false failures under the target's predeclared multi-objective ordering +- C: the performance gate must not incentivize weakening tests, assertions, evidence, semantic coverage, or representative workload inputs +- B: target-specific calibration budget specifying repetitions, environment samples, and allowable CI/runtime measurement cost +- S: stop calibration when the declared sample budget is exhausted or the baseline/noise estimate is stable enough to justify the predeclared warning and hard thresholds ## Preserved contract diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index 582abe3..b820659 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,9 +14,13 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- Key: canonical identity of equivalent in-flight requests -- Objective: minimize duplicate concurrent evaluations -- Hard constraint: all joined callers must receive a result/error valid for their request semantics +- X: target-supported request-key canonicalizations, in-flight ownership policies, waiter limits, cancellation policies, and retry/error-sharing policies +- F: policies that coalesce only semantically equivalent requests and preserve authorization, timeout, cancellation, result, and error semantics for every joined caller +- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization and waiter-memory overhead +- d: minimize under the target's predeclared scalar or lexicographic ordering +- C: every joined caller receives a result or error valid for its original request semantics; non-equivalent requests are never merged +- B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied by this record +- S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C ## Preserved contract diff --git a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md index bda16b9..37a1386 100644 --- a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md +++ b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md @@ -15,9 +15,13 @@ Independent workers serialize on one globally coordinated resource even though t ## Optimization problem contract -- Variables: shard/domain count, namespace split, worker-to-domain mapping -- Objective: reduce coordination contention and tail latency -- Hard constraint: preserve the required global invariant (for example uniqueness or ordering scope) +- X: target-supported shard/domain counts, namespace splits, worker-to-domain mappings, and merge/aggregation policies +- F: configurations that preserve the target's required uniqueness, ownership, visibility, failure-domain, and ordering guarantees +- f: measured coordination contention, tail latency, and coordination overhead under the declared workload +- d: minimize under the target's predeclared objective ordering +- C: partitioning must not silently weaken any global invariant; any intentional shift from global to per-domain ordering is a separately declared contract change +- B: target-specific contention/scale benchmark budget declared before tuning; no portable shard count or bit split is supplied here +- S: stop when the budget is exhausted or a validated partitioning materially reduces the target bottleneck without violating C ## Preserved contract diff --git a/optimizations/OPT-CRIT-001-critical-path-prioritization.md b/optimizations/OPT-CRIT-001-critical-path-prioritization.md index 80b157e..3adbd63 100644 --- a/optimizations/OPT-CRIT-001-critical-path-prioritization.md +++ b/optimizations/OPT-CRIT-001-critical-path-prioritization.md @@ -15,9 +15,13 @@ Non-critical work competes with the dependency chain that determines user-visibl ## Optimization problem contract -- Classify work: critical now / likely soon / deferrable / unnecessary -- Objective: reduce end-to-end critical-path latency -- Constraints: no starvation, stale-state or correctness violation from deferral/speculation +- X: target-supported task-priority, prefetch/precompute, lazy/deferred-work, and speculation policies +- F: policies that preserve all semantic deadlines, avoid externally visible speculative side effects before commitment, and satisfy starvation/resource constraints +- f: measured end-to-end latency of the declared critical dependency path, including resource pressure introduced by speculation/deferment +- d: minimize +- C: critical outputs and semantic deadlines are preserved; speculative work is safely discardable; deferred work completes before it becomes semantically required +- B: target-specific trace/benchmark budget covering cold/warm, hit/miss, and wrong-speculation cases; no portable prediction horizon is supplied here +- S: stop when the declared budget is exhausted or a validated policy materially reduces critical-path latency without violating C ## Preserved contract diff --git a/optimizations/OPT-FAN-001-shared-materialization-fanout.md b/optimizations/OPT-FAN-001-shared-materialization-fanout.md index a08fce8..b9c82e3 100644 --- a/optimizations/OPT-FAN-001-shared-materialization-fanout.md +++ b/optimizations/OPT-FAN-001-shared-materialization-fanout.md @@ -14,9 +14,13 @@ The same deterministic transformation is repeated independently for each consume ## Optimization problem contract -- Variables: materialization boundary, representation format, persistence policy -- Objectives: transformation CPU, replay CPU, fan-out latency -- Hard constraint: materialized form must satisfy the consumer contract and versioning/trust requirements +- X: target-supported materialization boundaries, representation formats/versions, persistence policies, and raw-versus-materialized retention policies +- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, and trust requirement +- f: measured transformation CPU, replay CPU, fan-out latency, and storage/I/O overhead under the target's declared objective ordering +- d: minimize under the target's predeclared scalar or lexicographic ordering +- C: consumers receive the declared representation semantics exactly; verification/security metadata may be removed only under an explicit contract change +- B: target-specific fan-out/replay benchmark budget declared before tuning; no portable subscriber count, replay size, or retention duration is supplied here +- S: stop when the declared budget is exhausted or a validated materialization policy materially improves the target objective without violating C ## Preserved contract diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index d6f3dea..762ba41 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -14,10 +14,13 @@ Expensive work is rerun even though every input capable of affecting its result ## Optimization problem contract -- Variables: signature definition, persistence scope, invalidation granularity -- Objective: minimize repeated work and metadata I/O -- Hard constraint: a reused result must correspond to the complete effective input identity -- Budget/stopping: target-specific +- X: target-supported signature definitions, persistence scopes, invalidation granularities, and missing-output policies +- F: configurations whose signature covers every output-affecting input, whose reuse checks required outputs, and whose failed executions never commit new reusable state +- f: measured repeated-work cost including stage runtime plus signature/metadata I/O overhead +- d: minimize +- C: every reused output is semantically equivalent to a fresh execution for the same effective inputs, with the same failure/output-validity semantics +- B: target-specific benchmark/evaluation budget declared before tuning; no portable value is supplied by this record +- S: stop when the declared budget is exhausted or a validated configuration meets the predeclared improvement threshold without violating C ## Preserved contract diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index 19c0a3e..4e0867b 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -15,10 +15,13 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who ## Optimization problem contract -- Search space `X`, feasible set `F`, objective `f` -- Incumbent: best validated feasible solution -- Bound: optimistic objective bound for each unexplored region -- Exactness: declare whether full branch-and-bound proof or anytime/budgeted search is required +- X: the target's explicitly defined discrete or mixed candidate space together with a partition of unexplored candidates into searchable subregions +- F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F +- f: the target objective evaluated on feasible candidates, plus a sound optimistic bound for each unexplored subregion +- d: the target's predeclared minimize or maximize direction, or an explicit ordering that defines when one incumbent improves another +- C: every pruning bound is sound for the declared objective/constraints and the returned incumbent satisfies the original feasibility and semantic contract +- B: for exact search, resources required until the search frontier is exhausted or optimality is proven; for anytime search, an explicit target-specific evaluation/time/compute budget +- S: exact mode stops only when optimality is proven or the frontier is exhausted; anytime mode stops on B and reports the incumbent plus the remaining optimality gap/bound ## Preserved contract diff --git a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md index 06f2150..cf428dd 100644 --- a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md +++ b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md @@ -14,9 +14,13 @@ An expensive operation is applied to a large population even though only a small ## Optimization problem contract -- Variable: where semantics-preserving filtering/culling/limiting occurs -- Objective: minimize cardinality presented to the expensive stage -- Hard constraint: early reduction must preserve every candidate required by the final result +- X: semantically legal placements and implementations of filtering, culling, limiting, candidate selection, or other working-set reductions in the target pipeline +- F: placements that preserve every candidate and ordering/tie/join semantic required by the final result and satisfy target resource constraints +- f: measured end-to-end pipeline cost and cardinality presented to the expensive stage +- d: minimize under the target's predeclared objective ordering +- C: the reordered/reduced pipeline must be semantically equivalent to the reference pipeline for all declared output, ordering, top-k, tie, null, and join semantics +- B: target-specific benchmark budget over representative and adversarial selectivity distributions; no portable selectivity threshold is supplied here +- S: stop when the declared budget is exhausted or a validated early-reduction placement materially lowers total cost without violating C ## Preserved contract diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index 1b14ec5..c32122c 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -16,7 +16,13 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra ## Optimization problem contract -Define `P = (X,F,f,d,C,B,S)` from `OPTIMIZATION-PROBLEM.md`. Explicitly classify continuous/discrete/conditional variables, noise, constraints, gradient availability, local/global scope and evaluation cost. +- X: the target's explicitly bounded continuous, integer, categorical, conditional, or mixed parameter search space +- F: candidates in X that satisfy all hard resource, platform, semantic, and correctness constraints before objective ranking +- f: the target-measured objective or objective vector for each feasible candidate, including declared noise/statistical treatment +- d: the target's predeclared minimize, maximize, lexicographic, or Pareto ordering +- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f +- B: an explicit target-specific maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts +- S: stop on the declared budget, a predeclared objective/quality target, or a predeclared stagnation/convergence rule; preserve the reason for stopping in the trial ledger ## Preserved contract diff --git a/optimizations/OPT-SET-001-density-adaptive-compact-sets.md b/optimizations/OPT-SET-001-density-adaptive-compact-sets.md index 608b49c..a5d1788 100644 --- a/optimizations/OPT-SET-001-density-adaptive-compact-sets.md +++ b/optimizations/OPT-SET-001-density-adaptive-compact-sets.md @@ -15,9 +15,13 @@ A single representation performs poorly across regions with very different densi ## Optimization problem contract -- Variables: partition width, sparse/dense representation threshold, serialization layout -- Objectives: memory footprint and set-operation latency -- Hard constraint: exact set semantics unless approximation is explicitly introduced elsewhere +- X: target-supported partition widths, sparse/dense container choices, switching thresholds, and serialization layouts +- F: representations that preserve exact membership and set-operation semantics and satisfy target memory/serialization compatibility constraints +- f: measured memory footprint plus target-relevant set-operation and serialization latency +- d: minimize under the target's predeclared scalar, lexicographic, or Pareto ordering +- C: membership, union, intersection, difference, and persistence round trips match the canonical reference set exactly +- B: target-specific benchmark budget over declared sparse, dense, mixed, and transition-boundary datasets; no portable trial count is supplied here +- S: stop when the declared budget is exhausted or a validated representation meets the target objective without violating C ## Preserved contract diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 1450214..f005368 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -27,6 +27,7 @@ "## Failure modes", "## Rollback trigger", } +REQUIRED_CONTRACT_FIELDS = ("X", "F", "f", "d", "C", "B", "S") LINK_RE = re.compile(r"\[[^\]]+\]\((optimizations/[^)#]+\.md)\)") ID_RE = re.compile(r"^# (OPT-[A-Z]+-\d{3}) — ") @@ -35,10 +36,26 @@ def die(msg: str) -> None: raise SystemExit(f"catalog-integrity: {msg}") +def section_lines(text: str, heading: str) -> list[str]: + """Return lines belonging to one exact level-2 Markdown section.""" + lines = text.splitlines() + try: + start = lines.index(heading) + 1 + except ValueError: + return [] + end = len(lines) + for i in range(start, len(lines)): + if lines[i].startswith("## "): + end = i + break + return lines[start:end] + + records: dict[str, Path] = {} for path in sorted(OPT_DIR.glob("OPT-*.md")): text = path.read_text(encoding="utf-8") - first = text.splitlines()[0] if text else "" + lines = text.splitlines() + first = lines[0] if lines else "" match = ID_RE.match(first) if not match: die(f"bad record heading: {path.relative_to(ROOT)}") @@ -48,11 +65,29 @@ def die(msg: str) -> None: records[record_id] = path if "**Status:**" not in text: die(f"missing Status in {path.relative_to(ROOT)}") + if record_id not in FROZEN_V1: - missing = sorted(section for section in REQUIRED_V2 if section not in text) + headings = {line for line in lines if line.startswith("## ")} + missing = sorted(REQUIRED_V2 - headings) if missing: die(f"{path.relative_to(ROOT)} missing sections: {', '.join(missing)}") + contract = section_lines(text, "## Optimization problem contract") + for field in REQUIRED_CONTRACT_FIELDS: + prefix = f"- {field}:" + matches = [line for line in contract if line.startswith(prefix)] + if len(matches) != 1: + die( + f"{path.relative_to(ROOT)} must contain exactly one contract field " + f"'{prefix}' in ## Optimization problem contract" + ) + if not matches[0][len(prefix) :].strip(): + die(f"{path.relative_to(ROOT)} has empty contract field {field}") + +missing_frozen = sorted(FROZEN_V1 - records.keys()) +if missing_frozen: + die(f"frozen v1 record(s) missing: {', '.join(missing_frozen)}") + # README is the human-facing record index and must contain real Markdown links. # CATALOG may use either links or plain/backticked record IDs; any links it does # contain are still validated below, while complete catalog coverage is enforced diff --git a/templates/OPTIMIZATION-RECORD.md b/templates/OPTIMIZATION-RECORD.md index f0d9fdb..7204143 100644 --- a/templates/OPTIMIZATION-RECORD.md +++ b/templates/OPTIMIZATION-RECORD.md @@ -16,17 +16,20 @@ What dominates runtime, latency, memory, I/O, CI cost, quality budget or optimiz ## Optimization problem contract -Define the target using `OPTIMIZATION-PROBLEM.md`: - -- Search space `X`: -- Feasible set `F`: -- Objective `f`: -- Direction: minimize / maximize / explicit multi-objective ordering -- Correctness / semantic contract `C`: -- Evaluation/resource budget `B`: -- Stopping rule `S`: +Define the target using `OPTIMIZATION-PROBLEM.md`. Keep these seven canonical fields as exact list prefixes so catalog integrity can verify the contract: + +- X: search space / decision-variable domain +- F: feasible set after hard constraints +- f: measured objective or objective vector +- d: minimize / maximize / explicit multi-objective ordering +- C: correctness and semantic contract that may not be weakened implicitly +- B: evaluation/resource budget +- S: stopping rule + +Then record useful classification detail: + - Variables: continuous / integer / categorical / conditional / mixed -- Objective: deterministic / noisy / stochastic +- Objective behavior: deterministic / noisy / stochastic - Search scope: local / global - Information: gradient / derivative-free / black-box - Exactness: exact / approximation permitted under explicit error contract From b190a3cb9131e402c2dd200cf79c3621978f711a Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 12:26:11 +0930 Subject: [PATCH 005/229] Harden catalog identity and evidence contracts --- OPTIMIZATION-PROBLEM.md | 9 ++- README.md | 14 ++-- README4AI.md | 9 ++- ...PROX-001-contract-bounded-approximation.md | 2 +- ...DGET-001-performance-regression-budgets.md | 2 +- ...T-CRIT-001-critical-path-prioritization.md | 8 +- ...1-signature-bound-incremental-execution.md | 2 +- ...E-001-bound-driven-search-space-pruning.md | 23 ++++-- ...SEARCH-001-budget-aware-adaptive-search.md | 2 +- scripts/check_catalog.py | 76 ++++++++++++++++--- templates/OPTIMIZATION-RECORD.md | 20 ++--- 11 files changed, 118 insertions(+), 49 deletions(-) mode change 100755 => 100644 scripts/check_catalog.py diff --git a/OPTIMIZATION-PROBLEM.md b/OPTIMIZATION-PROBLEM.md index b597947..af9a424 100644 --- a/OPTIMIZATION-PROBLEM.md +++ b/OPTIMIZATION-PROBLEM.md @@ -54,11 +54,12 @@ Record the following before tuning: ### Approximate visualization - `X = {LOD policies}` -- `F = policies satisfying frame-memory limits` +- `F = policies satisfying frame-memory limits and error ≤ ε` - `f = (frame latency, perceptual/error metric)` -- `C = error ≤ ε and reference path remains available` -- `B = fixed benchmark fixture set` -- `S = Pareto candidate chosen under the documented priority rule` +- `d = lexicographic: first require error ≤ ε through F/C, then minimize frame latency; break equal-latency ties by lower error` +- `C = reference path remains available and declared visual/semantic invariants are preserved` +- `B = at most 200 policy evaluations over the fixed benchmark fixture set` +- `S = stop when B is exhausted or no admissible policy improves frame latency by the predeclared δ for 20 consecutive evaluations` ## Search-mechanism selection diff --git a/README.md b/README.md index a9083e1..0da10ec 100644 --- a/README.md +++ b/README.md @@ -22,17 +22,17 @@ The point of this repository is simple: when a future project needs to go faster | [OPT-LEAN-001](optimizations/OPT-LEAN-001-trust-preserving-lean-ci.md) | Trust-preserving Lean dependency reuse | **Verified on source PR; timings environment-scoped** | Reuse verified dependency state while rebuilding current project source | | [OPT-PAR-001](optimizations/OPT-PAR-001-bounded-parallel-execution.md) | Bounded deterministic parallel execution | **Verified, environment-specific** | Bound concurrency and prove scalar/parallel equivalence | | [OPT-DSP-001](optimizations/OPT-DSP-001-control-rate-sparse-vector-dsp.md) | Control-rate + sparse + vectorized DSP | **Implemented reference; approximation/native ideas proposed** | Move slow state out of the hot path; sparse/vectorize repeated numerical work | -| [OPT-INC-001](optimizations/OPT-INC-001-signature-bound-incremental-execution.md) | Signature-bound incremental execution | **Implemented historical reference** | Rerun work only when complete effective-input identity changes | +| [OPT-INC-001](optimizations/OPT-INC-001-signature-bound-incremental-execution.md) | Signature-bound incremental execution | **Implemented external reference** | Rerun work only when complete effective-input identity changes | | [OPT-COAL-001](optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md) | Concurrent duplicate-work coalescing | **Implemented external reference** | Share one in-flight computation among equivalent simultaneous callers | | [OPT-SET-001](optimizations/OPT-SET-001-density-adaptive-compact-sets.md) | Density-adaptive compact sets | **Implemented external reference** | Choose sparse/dense representation locally while retaining exact set algebra | | [OPT-CONT-001](optimizations/OPT-CONT-001-partitioned-coordination-domains.md) | Partitioned coordination domains | **Implemented external pattern** | Split one global contention hotspot into independent domains while preserving global invariants | | [OPT-FAN-001](optimizations/OPT-FAN-001-shared-materialization-fanout.md) | Shared materialization for fan-out/replay | **Implemented external reference** | Transform/encode once and reuse the representation for many consumers | -| [OPT-SEARCH-001](optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md) | Budget-aware adaptive parameter search | **External mechanisms; OPT synthesis proposed** | Spend expensive evaluations where they are most informative | -| [OPT-APPROX-001](optimizations/OPT-APPROX-001-contract-bounded-approximation.md) | Contract-bounded approximation | **Target-specific validation required** | Trade exactness only inside an explicit measurable error/degradation envelope | +| [OPT-SEARCH-001](optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md) | Budget-aware adaptive parameter search | **Proposed / OPT synthesis** | Spend expensive evaluations where they are most informative | +| [OPT-APPROX-001](optimizations/OPT-APPROX-001-contract-bounded-approximation.md) | Contract-bounded approximation | **Proposed / OPT synthesis** | Trade exactness only inside an explicit measurable error/degradation envelope | | [OPT-REDUCE-001](optimizations/OPT-REDUCE-001-early-working-set-reduction.md) | Early working-set reduction | **Implemented external pattern** | Filter/cull/limit before expensive composition | -| [OPT-CRIT-001](optimizations/OPT-CRIT-001-critical-path-prioritization.md) | Critical-path prioritization | **Established engineering pattern** | Do critical work now, speculate carefully, defer non-critical work | -| [OPT-BUDGET-001](optimizations/OPT-BUDGET-001-performance-regression-budgets.md) | Performance regression budgets | **Established engineering pattern** | Turn performance expectations into environment-scoped regression contracts | -| [OPT-PRUNE-001](optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md) | Bound-driven search-space pruning | **Classical optimization mechanism** | Prove whole search regions cannot improve the incumbent and skip them | +| [OPT-CRIT-001](optimizations/OPT-CRIT-001-critical-path-prioritization.md) | Critical-path prioritization | **Proposed / OPT synthesis** | Do critical work now, speculate carefully, defer non-critical work | +| [OPT-BUDGET-001](optimizations/OPT-BUDGET-001-performance-regression-budgets.md) | Performance regression budgets | **Proposed / OPT synthesis** | Turn performance expectations into environment-scoped regression contracts | +| [OPT-PRUNE-001](optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md) | Bound-driven search-space pruning | **Proposed / OPT synthesis** | Prove whole search regions cannot improve the incumbent and skip them | See [CATALOG.md](CATALOG.md) for the decision map and [README4AI.md](README4AI.md) for machine-oriented usage. @@ -58,7 +58,7 @@ for search space, feasible set, objective, direction, correctness/semantic const ## Integrity gate -`scripts/check_catalog.py` verifies record IDs, required post-v1 sections, catalog coverage and local optimization-record links. CI runs it via `.github/workflows/catalog-integrity.yml`. +`scripts/check_catalog.py` verifies heading/filename identity, post-v1 status vocabulary, complete contracts, complete README coverage, CATALOG coverage, and optimization-record link labels/targets. CI runs it via `.github/workflows/catalog-integrity.yml`. ## Add the next optimization diff --git a/README4AI.md b/README4AI.md index 5b6f2be..a3cb786 100644 --- a/README4AI.md +++ b/README4AI.md @@ -54,13 +54,18 @@ Before choosing an optimizer, classify: ## Status vocabulary +Post-v1 records must use one of these exact status categories. A semicolon may follow the category with a short evidence-boundary caveat; `scripts/check_catalog.py` validates the category before that semicolon. + - **Verified**: source project contains passing validation and measured/observed evidence. - **Verified, environment-specific**: measured result is real but not a universal performance guarantee. -- **Implemented reference**: the mechanism exists in code, but no general speedup claim is made. -- **Implemented external reference/pattern**: mechanism exists in an external donor; target transfer still requires local validation. +- **Implemented reference**: the mechanism exists in repository code, but no general speedup claim is made. +- **Implemented external reference**: the mechanism exists in an external donor; target transfer still requires local validation. +- **Implemented external pattern**: an external donor demonstrates the pattern, but this OPT record does not claim a target implementation. - **Proposed / OPT synthesis**: architecture/design guidance only. Do not report it as achieved performance. - **Source candidate**: material exists but has not been inspected sufficiently to promote claims. +The frozen v1 records retain their historical release wording and are exempt from post-v1 status normalization. + ## Non-negotiable rules - Never remove tests merely to make CI faster. diff --git a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md index 7045a46..0cacb3e 100644 --- a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md +++ b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md @@ -1,6 +1,6 @@ # OPT-APPROX-001 — Contract-bounded approximation -**Status:** External production pattern + existing OPT need; target-specific proof/measurement required +**Status:** Proposed / OPT synthesis; external production pattern, target-specific proof/measurement required **Domains:** visualization, search, streaming, telemetry, simulation, audition DSP ## Source evidence diff --git a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md index ae34bd9..3519d67 100644 --- a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md +++ b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md @@ -1,6 +1,6 @@ # OPT-BUDGET-001 — Performance regression budgets -**Status:** Established engineering pattern; enforcement must be environment-scoped +**Status:** Proposed / OPT synthesis; target calibration required **Domains:** CI, web, numerical kernels, builds, services, DSP ## Source evidence diff --git a/optimizations/OPT-CRIT-001-critical-path-prioritization.md b/optimizations/OPT-CRIT-001-critical-path-prioritization.md index 3adbd63..022629a 100644 --- a/optimizations/OPT-CRIT-001-critical-path-prioritization.md +++ b/optimizations/OPT-CRIT-001-critical-path-prioritization.md @@ -1,6 +1,6 @@ # OPT-CRIT-001 — Critical-path prioritization -**Status:** Established performance-engineering pattern; target validation required +**Status:** Proposed / OPT synthesis; target validation required **Domains:** UI, web, games, build systems, model/data loading, interactive pipelines ## Source evidence @@ -41,7 +41,7 @@ Execute critical dependencies first; prefetch/precompute likely-soon work only w ## Validation -Trace the true dependency path and measure end-to-end latency, not only individual task duration. Test cold/warm, cache-hit/miss and wrong-speculation cases. +Trace the true dependency path and measure end-to-end latency, not only individual task duration. Test cold/warm, cache-hit/miss and wrong-speculation cases. Explicitly test semantic deadlines, starvation, cancellation, and that speculative work cannot expose side effects before commitment. ## Target-repo adaptation @@ -49,8 +49,8 @@ Criticality and prediction horizons are workload-specific. Re-profile after topo ## Failure modes -Speculation steals resources from critical work, lazy work causes later latency cliffs, priorities become stale, or deferred tasks starve. +Speculation steals resources from critical work, lazy work causes later latency cliffs, priorities become stale, deferred tasks starve, semantic deadlines are missed, or speculative side effects escape before commitment. ## Rollback trigger -Disable speculative/deferred policy if critical-path latency or resource pressure worsens materially. +Immediately disable/revert the policy on any violation of C, including a required task missing its semantic deadline or speculative work exposing an externally visible side effect before commitment. Also disable it if critical-path latency or resource pressure worsens materially. diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index 762ba41..0790ce7 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -1,6 +1,6 @@ # OPT-INC-001 — Signature-bound incremental execution -**Status:** Implemented historical reference; target validation required +**Status:** Implemented external reference; historical donor, target validation required **Domains:** builds, CI, generated artifacts, preprocessing, scientific pipelines ## Source evidence diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index 4e0867b..4741f20 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -1,6 +1,6 @@ # OPT-PRUNE-001 — Bound-driven search-space pruning -**Status:** Classical optimization mechanism; OPT adaptation guidance +**Status:** Proposed / OPT synthesis; classical mechanism, target adaptation required **Domains:** combinatorial optimization, scheduling, assignment, configuration search, resource allocation ## Source evidence @@ -17,19 +17,26 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - X: the target's explicitly defined discrete or mixed candidate space together with a partition of unexplored candidates into searchable subregions - F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F -- f: the target objective evaluated on feasible candidates, plus a sound optimistic bound for each unexplored subregion -- d: the target's predeclared minimize or maximize direction, or an explicit ordering that defines when one incumbent improves another -- C: every pruning bound is sound for the declared objective/constraints and the returned incumbent satisfies the original feasibility and semantic contract +- f: the target objective evaluated on feasible candidates only +- d: the target's predeclared minimize or maximize direction, or an explicit total/partial ordering that defines when one incumbent improves another +- C: the returned incumbent satisfies the original feasibility/semantic contract, and every pruning decision is justified by a separately defined sound region-bound function b - B: for exact search, resources required until the search frontier is exhausted or optimality is proven; for anytime search, an explicit target-specific evaluation/time/compute budget - S: exact mode stops only when optimality is proven or the frontier is exhausted; anytime mode stops on B and reports the incumbent plus the remaining optimality gap/bound +For each unexplored region `R`, define a bound `b(R)` separately from `f`: + +- minimizing: `b(R) ≤ inf { f(x) | x ∈ F ∩ R }`; prune `R` only when `b(R) ≥ f(x_incumbent)`; +- maximizing: `b(R) ≥ sup { f(x) | x ∈ F ∩ R }`; prune `R` only when `b(R) ≤ f(x_incumbent)`. + +An independently proven infeasible region may also be pruned. A heuristic estimate that does not satisfy the declared bound relation is search-ordering evidence at most, not a pruning proof. + ## Preserved contract A region may be discarded only when its bound proves it cannot improve the incumbent under the declared objective and constraints. Heuristic guesses are not proof-based pruning. ## Optimization -Maintain an incumbent, partition the search space, compute cheap optimistic bounds (often from relaxations), prioritize promising regions and prune any region whose best possible outcome cannot beat the incumbent. +Maintain an incumbent, partition the search space, compute a cheap sound `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific bound relation proves the region cannot improve the incumbent. A relaxed solution is evidence for a bound, not automatically a feasible final answer. @@ -43,7 +50,7 @@ A relaxed solution is evidence for a bound, not automatically a feasible final a ## Validation -For small fixtures, compare with exhaustive enumeration. Test bound soundness separately from search ordering. Record the optimality gap when stopping before exact completion. +For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness independently by checking the direction-specific inequality against exhaustive feasible values inside each test region. Test pruning separately from search ordering, and record the optimality gap when stopping before exact completion. ## Target-repo adaptation @@ -51,8 +58,8 @@ The quality/cost of bounds determines whether pruning helps. Develop target-spec ## Failure modes -Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning. +Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation. ## Rollback trigger -Disable any pruning rule that fails exhaustive small-case validation or whose bound cost exceeds the work it eliminates. +Disable any pruning rule that fails exhaustive small-case validation, violates the declared bound relation, or whose bound cost exceeds the work it eliminates. diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index c32122c..c4926f4 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -1,6 +1,6 @@ # OPT-SEARCH-001 — Budget-aware adaptive parameter search -**Status:** Implemented external mechanisms; OPT synthesis proposed for target tuning +**Status:** Proposed / OPT synthesis; upstream mechanisms are implemented externally **Domains:** expensive black-box tuning, CI/runtime parameters, simulation, numerical kernels ## Source evidence diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py old mode 100755 new mode 100644 index f005368..080a6b8 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -4,6 +4,7 @@ from __future__ import annotations import re +from collections import Counter from pathlib import Path ROOT = Path(__file__).resolve().parents[1] @@ -28,8 +29,19 @@ "## Rollback trigger", } REQUIRED_CONTRACT_FIELDS = ("X", "F", "f", "d", "C", "B", "S") -LINK_RE = re.compile(r"\[[^\]]+\]\((optimizations/[^)#]+\.md)\)") +ALLOWED_V2_STATUS_CATEGORIES = { + "Verified", + "Verified, environment-specific", + "Implemented reference", + "Implemented external reference", + "Implemented external pattern", + "Proposed / OPT synthesis", + "Source candidate", +} +LINK_RE = re.compile(r"\[([^\]]+)\]\((optimizations/[^)#]+\.md)\)") ID_RE = re.compile(r"^# (OPT-[A-Z]+-\d{3}) — ") +FILENAME_ID_RE = re.compile(r"^(OPT-[A-Z]+-\d{3})-") +STATUS_RE = re.compile(r"^\*\*Status:\*\*\s*(.*?)\s*$") def die(msg: str) -> None: @@ -60,13 +72,36 @@ def section_lines(text: str, heading: str) -> list[str]: if not match: die(f"bad record heading: {path.relative_to(ROOT)}") record_id = match.group(1) + + filename_match = FILENAME_ID_RE.match(path.name) + if not filename_match: + die(f"record filename does not begin with an OPT ID: {path.relative_to(ROOT)}") + filename_id = filename_match.group(1) + if filename_id != record_id: + die( + f"record ID mismatch: {path.relative_to(ROOT)} declares {record_id} " + f"but filename encodes {filename_id}" + ) + if record_id in records: die(f"duplicate record id {record_id}: {records[record_id]} and {path}") records[record_id] = path - if "**Status:**" not in text: - die(f"missing Status in {path.relative_to(ROOT)}") + + status_matches = [STATUS_RE.match(line) for line in lines] + statuses = [m.group(1).strip() for m in status_matches if m is not None] + if len(statuses) != 1: + die(f"{path.relative_to(ROOT)} must contain exactly one Status line") + if not statuses[0]: + die(f"{path.relative_to(ROOT)} has empty Status") if record_id not in FROZEN_V1: + status_category = statuses[0].split(";", 1)[0].strip() + if status_category not in ALLOWED_V2_STATUS_CATEGORIES: + die( + f"{path.relative_to(ROOT)} uses undefined status category " + f"'{status_category}'" + ) + headings = {line for line in lines if line.startswith("## ")} missing = sorted(REQUIRED_V2 - headings) if missing: @@ -88,18 +123,37 @@ def section_lines(text: str, heading: str) -> list[str]: if missing_frozen: die(f"frozen v1 record(s) missing: {', '.join(missing_frozen)}") -# README is the human-facing record index and must contain real Markdown links. -# CATALOG may use either links or plain/backticked record IDs; any links it does -# contain are still validated below, while complete catalog coverage is enforced -# independently by record ID. +record_paths = {str(path.relative_to(ROOT)): record_id for record_id, path in records.items()} + +# README is the complete human-facing record index. CATALOG may use either +# links or plain/backticked IDs, but any optimization-record link in either +# document must use the target record's stable ID as its label. for doc_name in ("README.md", "CATALOG.md"): text = (ROOT / doc_name).read_text(encoding="utf-8") links = LINK_RE.findall(text) - if doc_name == "README.md" and not links: - die("README.md contains no optimization-record links") - for rel in links: - if not (ROOT / rel).is_file(): + linked_ids: list[str] = [] + for label, rel in links: + target = ROOT / rel + if not target.is_file(): die(f"broken record link in {doc_name}: {rel}") + target_id = record_paths.get(rel) + if target_id is None: + die(f"record link in {doc_name} is not a discovered OPT record: {rel}") + if label.strip() != target_id: + die( + f"record link label mismatch in {doc_name}: '{label}' points to " + f"{target_id} ({rel})" + ) + linked_ids.append(target_id) + + if doc_name == "README.md": + counts = Counter(linked_ids) + duplicates = sorted(record_id for record_id, count in counts.items() if count != 1) + if duplicates: + die(f"README.md must index each record exactly once; bad counts for: {', '.join(duplicates)}") + missing_readme = sorted(records.keys() - counts.keys()) + if missing_readme: + die(f"README.md is missing record(s): {', '.join(missing_readme)}") catalog = (ROOT / "CATALOG.md").read_text(encoding="utf-8") for record_id, path in records.items(): diff --git a/templates/OPTIMIZATION-RECORD.md b/templates/OPTIMIZATION-RECORD.md index 7204143..9596d31 100644 --- a/templates/OPTIMIZATION-RECORD.md +++ b/templates/OPTIMIZATION-RECORD.md @@ -1,8 +1,10 @@ # OPT-XXX-000 — Optimization Name -**Status:** Proposed / Implemented reference / Implemented external reference / Verified / Verified, environment-specific +**Status:** **Domains:** ... +Choose one documented status category from `README4AI.md` and replace every placeholder below before promoting this file into `optimizations/`. + ## Source evidence - Repository / publication / article: @@ -16,15 +18,15 @@ What dominates runtime, latency, memory, I/O, CI cost, quality budget or optimiz ## Optimization problem contract -Define the target using `OPTIMIZATION-PROBLEM.md`. Keep these seven canonical fields as exact list prefixes so catalog integrity can verify the contract: +Define the target using `OPTIMIZATION-PROBLEM.md`. Keep these seven canonical fields as exact list prefixes so catalog integrity can verify the contract. **They are intentionally empty in this template and must be filled with record-specific values.** -- X: search space / decision-variable domain -- F: feasible set after hard constraints -- f: measured objective or objective vector -- d: minimize / maximize / explicit multi-objective ordering -- C: correctness and semantic contract that may not be weakened implicitly -- B: evaluation/resource budget -- S: stopping rule +- X: +- F: +- f: +- d: +- C: +- B: +- S: Then record useful classification detail: From 9f408a3e6310685a1677fc88656834689c16064a Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 12:48:12 +0930 Subject: [PATCH 006/229] Harden catalog completeness and pruning budgets --- ...E-001-bound-driven-search-space-pruning.md | 24 ++++----- scripts/check_catalog.py | 49 ++++++++++++++++++- 2 files changed, 60 insertions(+), 13 deletions(-) diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index 4741f20..f36c134 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -17,26 +17,26 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - X: the target's explicitly defined discrete or mixed candidate space together with a partition of unexplored candidates into searchable subregions - F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F -- f: the target objective evaluated on feasible candidates only -- d: the target's predeclared minimize or maximize direction, or an explicit total/partial ordering that defines when one incumbent improves another -- C: the returned incumbent satisfies the original feasibility/semantic contract, and every pruning decision is justified by a separately defined sound region-bound function b -- B: for exact search, resources required until the search frontier is exhausted or optimality is proven; for anytime search, an explicit target-specific evaluation/time/compute budget -- S: exact mode stops only when optimality is proven or the frontier is exhausted; anytime mode stops on B and reports the incumbent plus the remaining optimality gap/bound +- f: a scalar real-valued target objective `f : F → R` evaluated on feasible candidates only +- d: exactly one of scalar `minimize` or scalar `maximize`; vector, Pareto, lexicographic, or other partial-order objectives are outside this record unless a separately specified and validated frontier-bound mechanism is introduced +- C: the returned incumbent satisfies the original feasibility/semantic contract, and every pruning decision is justified by a separately defined sound scalar region-bound function `b` +- B: a finite, predeclared target-specific cap on evaluations, wall time, compute, or equivalent resource consumption; exact-mode search may prove optimality before this cap but may not run without a finite cap +- S: stop immediately when optimality is proven or the frontier is exhausted; otherwise stop when B is exhausted and return the best validated incumbent plus the remaining valid bound/optimality gap without claiming exact completion For each unexplored region `R`, define a bound `b(R)` separately from `f`: - minimizing: `b(R) ≤ inf { f(x) | x ∈ F ∩ R }`; prune `R` only when `b(R) ≥ f(x_incumbent)`; - maximizing: `b(R) ≥ sup { f(x) | x ∈ F ∩ R }`; prune `R` only when `b(R) ≤ f(x_incumbent)`. -An independently proven infeasible region may also be pruned. A heuristic estimate that does not satisfy the declared bound relation is search-ordering evidence at most, not a pruning proof. +An independently proven infeasible region may also be pruned. A heuristic estimate that does not satisfy the declared bound relation is search-ordering evidence at most, not a pruning proof. This record does not authorize scalar bounds to prune vector/Pareto or partially ordered objectives. ## Preserved contract -A region may be discarded only when its bound proves it cannot improve the incumbent under the declared objective and constraints. Heuristic guesses are not proof-based pruning. +A region may be discarded only when its scalar bound proves it cannot improve the incumbent under the declared scalar objective and constraints. Heuristic guesses are not proof-based pruning, and exhausting B without an optimality proof does not permit an exactness claim. ## Optimization -Maintain an incumbent, partition the search space, compute a cheap sound `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific bound relation proves the region cannot improve the incumbent. +Maintain an incumbent, partition the search space, compute a cheap sound `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific scalar bound relation proves the region cannot improve the incumbent. A relaxed solution is evidence for a bound, not automatically a feasible final answer. @@ -50,16 +50,16 @@ A relaxed solution is evidence for a bound, not automatically a feasible final a ## Validation -For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness independently by checking the direction-specific inequality against exhaustive feasible values inside each test region. Test pruning separately from search ordering, and record the optimality gap when stopping before exact completion. +For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness independently by checking the direction-specific inequality against exhaustive feasible values inside each test region. Test pruning separately from search ordering. Verify that budget exhaustion returns an anytime result without an exactness claim, and record the remaining valid optimality gap/bound whenever exact completion was not proven. ## Target-repo adaptation -The quality/cost of bounds determines whether pruning helps. Develop target-specific relaxations and branch ordering; do not assume one bound is universally strong. +The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. ## Failure modes -Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation. +Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; an unbounded exact-search policy can consume resources indefinitely. ## Rollback trigger -Disable any pruning rule that fails exhaustive small-case validation, violates the declared bound relation, or whose bound cost exceeds the work it eliminates. +Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar bound relation, is applied to an unsupported objective ordering, or whose bound cost exceeds the work it eliminates. Abort exact-mode claims whenever B is exhausted before optimality is proven. diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 080a6b8..ed009d1 100644 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -42,6 +42,12 @@ ID_RE = re.compile(r"^# (OPT-[A-Z]+-\d{3}) — ") FILENAME_ID_RE = re.compile(r"^(OPT-[A-Z]+-\d{3})-") STATUS_RE = re.compile(r"^\*\*Status:\*\*\s*(.*?)\s*$") +OPT_TOKEN_RE = re.compile(r"\bOPT-[A-Z]+-\d{3}\b") +README_ROW_RE = re.compile( + r"^\|\s*\[(OPT-[A-Z]+-\d{3})\]\((optimizations/[^)#]+\.md)\)" + r"\s*\|[^|]*\|\s*([^|]+?)\s*\|", + re.MULTILINE, +) def die(msg: str) -> None: @@ -63,7 +69,21 @@ def section_lines(text: str, heading: str) -> list[str]: return lines[start:end] +def section_has_content(lines: list[str]) -> bool: + """Require visible non-whitespace content, ignoring HTML comments.""" + content = "\n".join(lines) + content = re.sub(r"", "", content, flags=re.DOTALL) + return bool(content.strip()) + + +def normalized_status_category(raw: str) -> str: + """Normalize light Markdown emphasis, then return the category before ';'.""" + plain = re.sub(r"[*_`]", "", raw).strip() + return plain.split(";", 1)[0].strip() + + records: dict[str, Path] = {} +status_categories: dict[str, str] = {} for path in sorted(OPT_DIR.glob("OPT-*.md")): text = path.read_text(encoding="utf-8") lines = text.splitlines() @@ -101,12 +121,17 @@ def section_lines(text: str, heading: str) -> list[str]: f"{path.relative_to(ROOT)} uses undefined status category " f"'{status_category}'" ) + status_categories[record_id] = status_category headings = {line for line in lines if line.startswith("## ")} missing = sorted(REQUIRED_V2 - headings) if missing: die(f"{path.relative_to(ROOT)} missing sections: {', '.join(missing)}") + for heading in sorted(REQUIRED_V2): + if not section_has_content(section_lines(text, heading)): + die(f"{path.relative_to(ROOT)} has empty mandatory section {heading}") + contract = section_lines(text, "## Optimization problem contract") for field in REQUIRED_CONTRACT_FIELDS: prefix = f"- {field}:" @@ -155,9 +180,31 @@ def section_lines(text: str, heading: str) -> list[str]: if missing_readme: die(f"README.md is missing record(s): {', '.join(missing_readme)}") + row_statuses: dict[str, str] = {} + for row_id, rel, raw_status in README_ROW_RE.findall(text): + if record_paths.get(rel) != row_id: + die(f"README.md row identity mismatch for {row_id}: {rel}") + if row_id in row_statuses: + die(f"README.md has duplicate status row for {row_id}") + row_statuses[row_id] = normalized_status_category(raw_status) + + for record_id, expected_status in status_categories.items(): + observed_status = row_statuses.get(record_id) + if observed_status is None: + die(f"README.md has no catalog status cell for post-v1 record {record_id}") + if observed_status != expected_status: + die( + f"README.md status mismatch for {record_id}: " + f"record='{expected_status}' README='{observed_status}'" + ) + catalog = (ROOT / "CATALOG.md").read_text(encoding="utf-8") +catalog_ids = set(OPT_TOKEN_RE.findall(catalog)) +unknown_catalog_ids = sorted(catalog_ids - records.keys()) +if unknown_catalog_ids: + die(f"CATALOG.md references unknown record ID(s): {', '.join(unknown_catalog_ids)}") for record_id, path in records.items(): - if record_id not in catalog: + if record_id not in catalog_ids: die(f"{record_id} ({path.name}) is not mentioned in CATALOG.md") problem_contract = ROOT / "OPTIMIZATION-PROBLEM.md" From f75ecc9b9d3c9f61c0dbdc9766a4b669636d7de9 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 13:28:58 +0930 Subject: [PATCH 007/229] Harden catalog templates and async semantics --- ...01-concurrent-duplicate-work-coalescing.md | 18 +++--- ...1-signature-bound-incremental-execution.md | 22 ++++---- ...SEARCH-001-budget-aware-adaptive-search.md | 18 +++--- scripts/check_catalog.py | 55 ++++++++++++++++++- 4 files changed, 84 insertions(+), 29 deletions(-) diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index b820659..70855f0 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,21 +14,23 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- X: target-supported request-key canonicalizations, in-flight ownership policies, waiter limits, cancellation policies, and retry/error-sharing policies +- X: target-supported request-key canonicalizations, shared-operation lifetime policies, waiter limits, per-waiter cancellation policies, and retry/error-sharing policies - F: policies that coalesce only semantically equivalent requests and preserve authorization, timeout, cancellation, result, and error semantics for every joined caller - f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization and waiter-memory overhead - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives a result or error valid for its original request semantics; non-equivalent requests are never merged +- C: every joined caller receives a result or error valid for its original request semantics; non-equivalent requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller - B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied by this record - S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C ## Preserved contract -Coalescing may merge only requests that are semantically equivalent for the shared operation. Cancellation, timeout, authorization and error semantics must remain explicit. +Coalescing may merge only requests that are semantically equivalent for the shared operation. Each caller retains independent cancellation and timeout semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters, and authorization/result/error semantics must remain explicit. ## Optimization -Make the first caller the owner of an in-flight operation. Equivalent callers subscribe to that future/result instead of starting duplicate work. Remove the in-flight entry deterministically on completion/failure. +Create an in-flight registry entry for the canonical request key. The first caller starts the shared upstream operation, but **does not own its lifetime**. Every equivalent caller registers as an independent waiter on that shared operation. + +Cancellation and timeout are per waiter: when one waiter leaves, remove only that waiter. Cancel the upstream operation only when no live waiters remain, or when a separately documented target policy proves that early cancellation is safe. On success or failure, deliver the same shared terminal result/error to all waiters that are still registered, then remove the in-flight entry deterministically. Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new operation after the failed entry is removed. This differs from caching: the reusable result does not exist yet. @@ -42,16 +44,16 @@ This differs from caching: the reusable result does not exist yet. ## Validation -Stress simultaneous identical and non-identical keys; inject owner failures/timeouts; prove only one upstream evaluation occurs for a coalesced key while all callers terminate correctly. +Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; test waiter-specific deadlines; verify shared failure delivery and retry accounting; prove only one upstream evaluation occurs for a coalesced key while all surviving callers terminate correctly. ## Target-repo adaptation -Define key canonicalization, maximum waiter count, cancellation semantics and whether errors are shared or retried. +Define key canonicalization, maximum waiter count, per-waiter cancellation/deadline handling, the exact condition for canceling upstream work, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent work; a hung owner can stall many callers; unbounded waiter lists amplify memory; shared error policy may cause correlated failure. +Over-broad keys merge non-equivalent work; coupling shared lifetime to the first caller can terminate valid waiters; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; unbounded waiter lists amplify memory; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes request semantics, increases tail latency materially, or creates unacceptable failure amplification. +Disable if coalescing changes any caller's cancellation/result/error semantics, permits one caller to cancel work required by another, leaks orphaned shared operations, increases tail latency materially, or creates unacceptable failure amplification. diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index 0790ce7..6c14555 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -14,21 +14,23 @@ Expensive work is rerun even though every input capable of affecting its result ## Optimization problem contract -- X: target-supported signature definitions, persistence scopes, invalidation granularities, and missing-output policies -- F: configurations whose signature covers every output-affecting input, whose reuse checks required outputs, and whose failed executions never commit new reusable state -- f: measured repeated-work cost including stage runtime plus signature/metadata I/O overhead +- X: target-supported signature definitions, persistence scopes, invalidation granularities, and output-validity policies +- F: configurations whose signature covers every output-affecting input, whose reuse validates required outputs, and whose failed executions never commit new reusable state +- f: measured repeated-work cost including stage runtime plus signature/metadata/output-validation I/O overhead - d: minimize -- C: every reused output is semantically equivalent to a fresh execution for the same effective inputs, with the same failure/output-validity semantics +- C: every reused output is semantically equivalent to a fresh execution for the same effective inputs, with the same failure and output-validity semantics - B: target-specific benchmark/evaluation budget declared before tuning; no portable value is supplied by this record - S: stop when the declared budget is exhausted or a validated configuration meets the predeclared improvement threshold without violating C ## Preserved contract -Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature. +Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, and an unchanged input signature alone is insufficient when an existing output can be corrupted, overwritten, or otherwise invalidated externally. ## Optimization -Compute a deterministic signature over the effective inputs, compare it with successfully persisted prior state, and execute only when the signature differs or required outputs are missing. Persist the new signature only after success. Reuse filesystem/configuration metadata lazily when its own validity predicate still holds. +Compute a deterministic signature over the effective inputs and compare it with successfully persisted prior state. Reuse is allowed only when that signature still matches **and** every required output satisfies a declared validity predicate. Depending on the target, that predicate may be a content digest/version manifest, a trusted immutable/protected artifact identity, or another reproducible integrity check strong enough to detect external mutation. Mere file presence is not sufficient unless the target explicitly guarantees that reused outputs are immutable and protected from modification. Execute when the input signature differs, any required output is missing, or any output-validity check fails. Persist the new signature and output-validity metadata only after successful execution. + +Reuse filesystem/configuration metadata lazily only while its own validity predicate still holds. ## Evidence boundary @@ -44,16 +46,16 @@ Wonderbuild demonstrates the mechanism and benchmark shapes, but its historical ## Validation -Test unchanged, changed-input, missing-output, failed-run, and corrupted/stale-state cases against a forced-fresh reference path. +Test unchanged inputs with valid outputs, changed inputs, missing outputs, failed runs, corrupted persistent state, externally overwritten/corrupted outputs, and stale output-version metadata against a forced-fresh reference path. A mutated output must force reconstruction unless the target's immutable/protected-output contract proves such mutation impossible. ## Target-repo adaptation -Re-profile signature cost, hash choice, metadata granularity and persistence format. Include environment/toolchain inputs when they affect output. +Re-profile signature and output-validation cost, hash/version choice, metadata granularity and persistence format. Include environment/toolchain inputs when they affect output. Explicitly choose whether outputs are integrity-checked on reuse or are stored behind an enforceable immutable/protected boundary. ## Failure modes -Incomplete signatures create stale reuse; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. +Incomplete signatures create stale reuse; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. ## Rollback trigger -Disable reuse if any cache/signature hit diverges from the fresh reference or if signature maintenance costs more than the avoided work. +Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference, if external output mutation can bypass the declared validity predicate, or if signature/integrity maintenance costs more than the avoided work. diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index c4926f4..0ce6926 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -20,18 +20,20 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra - F: candidates in X that satisfy all hard resource, platform, semantic, and correctness constraints before objective ranking - f: the target-measured objective or objective vector for each feasible candidate, including declared noise/statistical treatment - d: the target's predeclared minimize, maximize, lexicographic, or Pareto ordering -- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f -- B: an explicit target-specific maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts -- S: stop on the declared budget, a predeclared objective/quality target, or a predeclared stagnation/convergence rule; preserve the reason for stopping in the trial ledger +- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for already reserved/in-flight trials +- B: an explicit target-specific maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit and failure/cancellation charging policy are fixed before dispatch begins +- S: stop proposing/dispatching when no additional trial can be reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger ## Preserved contract -Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. +Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard dispatch bound: pending work counts according to the predeclared accounting policy rather than being ignored until completion. ## Optimization Use observations to adapt future evaluations: surrogate/acquisition search for expensive black-box objectives, conditional spaces where parameters only exist under certain choices, progressive domain contraction where justified, and explicit stopping/evaluation budgets. For asynchronous workers, reserve pending regions or otherwise diversify proposals so workers do not redundantly evaluate the same neighborhood. +Before dispatching an asynchronous trial, atomically reserve that trial in the ledger and debit the applicable unit from B (evaluation count, money, compute quota, or the target's declared equivalent). If the reservation would exceed B, do not dispatch. A reserved trial remains budget-accounted while pending. The target must predeclare whether failed/cancelled trials consume the reservation permanently, partially, or are refunded; that rule is applied deterministically and recorded in the ledger. Completion converts the reservation into a completed trial without charging the same budget twice. + Parallelism has an information cost: very wide batches receive less feedback between suggestions and can degenerate toward non-adaptive/random search. ## Before / after evidence @@ -44,16 +46,16 @@ Parallelism has an information cost: very wide batches receive less feedback bet ## Validation -Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. +Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. For asynchronous search, test the budget boundary with multiple workers contending for the last remaining reservation (for example, 99 of 100 evaluation slots already consumed/reserved) and prove that at most one additional trial can be dispatched. Inject failures and cancellations and verify the declared charge/refund policy without double-debit or budget overshoot. ## Target-repo adaptation -Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. +Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, atomic reservation mechanism, and failure/cancellation charging policy for the target before enabling asynchronous dispatch. ## Failure modes -Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. +Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; ambiguous refund rules can make the ledger disagree with actual resource consumption. ## Rollback trigger -Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, or repeated validation does not confirm the selected improvement. +Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, or any concurrency test shows dispatch can exceed the declared budget after pending reservations are counted. diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index ed009d1..2ae1d42 100644 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -38,16 +38,41 @@ "Proposed / OPT synthesis", "Source candidate", } +TEMPLATE_PLACEHOLDER_LINES = { + "- Repository / publication / article:", + "- Release/commit/PR/DOI/date:", + "- Exact files/sections where applicable:", + "- Licensing/provenance boundary where code reuse may matter:", + "What dominates runtime, latency, memory, I/O, CI cost, quality budget or optimization-evaluation cost?", + "State exactly what must remain unchanged: output bytes, theorem targets, assertions, API, numerical tolerance, ordering, statistical guarantee, evidence boundary, trust model, etc.", + "If the optimization changes the contract (for example exact → approximate), state the new contract explicitly instead of claiming preservation.", + "Describe the reusable mechanism, not only the source-project patch.", + "If no controlled benchmark exists, say so explicitly.", + "How was equivalence, correctness, bound soundness, approximation error or other contract compliance established?", + "Which source constants, thresholds, worker counts, bit splits, cache keys, search budgets or tolerances must be re-profiled rather than copied?", + "What can make this optimization invalid, slower, less robust or misleading?", + "Define the measured or semantic condition that disables/reverts the optimization.", +} LINK_RE = re.compile(r"\[([^\]]+)\]\((optimizations/[^)#]+\.md)\)") ID_RE = re.compile(r"^# (OPT-[A-Z]+-\d{3}) — ") FILENAME_ID_RE = re.compile(r"^(OPT-[A-Z]+-\d{3})-") STATUS_RE = re.compile(r"^\*\*Status:\*\*\s*(.*?)\s*$") OPT_TOKEN_RE = re.compile(r"\bOPT-[A-Z]+-\d{3}\b") +EMPTY_LABEL_RE = re.compile(r"^-\s+[^:]+:\s*$") README_ROW_RE = re.compile( r"^\|\s*\[(OPT-[A-Z]+-\d{3})\]\((optimizations/[^)#]+\.md)\)" r"\s*\|[^|]*\|\s*([^|]+?)\s*\|", re.MULTILINE, ) +CANONICAL_DEFINITION_PATTERNS = { + "X": re.compile(r"^- `X` — \S"), + "F": re.compile(r"^- `F(?: ⊆ X)?` — \S"), + "f": re.compile(r"^- `f(?:\s*:[^`]*)?` — \S"), + "d": re.compile(r"^- `d` — \S"), + "C": re.compile(r"^- `C` — \S"), + "B": re.compile(r"^- `B` — \S"), + "S": re.compile(r"^- `S` — \S"), +} def die(msg: str) -> None: @@ -70,10 +95,19 @@ def section_lines(text: str, heading: str) -> list[str]: def section_has_content(lines: list[str]) -> bool: - """Require visible non-whitespace content, ignoring HTML comments.""" + """Require record-specific visible content, not stock template prompts.""" content = "\n".join(lines) content = re.sub(r"", "", content, flags=re.DOTALL) - return bool(content.strip()) + for raw in content.splitlines(): + line = raw.strip() + if not line: + continue + if line in TEMPLATE_PLACEHOLDER_LINES: + continue + if EMPTY_LABEL_RE.match(line): + continue + return True + return False def normalized_status_category(raw: str) -> str: @@ -130,7 +164,9 @@ def normalized_status_category(raw: str) -> str: for heading in sorted(REQUIRED_V2): if not section_has_content(section_lines(text, heading)): - die(f"{path.relative_to(ROOT)} has empty mandatory section {heading}") + die( + f"{path.relative_to(ROOT)} has empty/template-only mandatory section {heading}" + ) contract = section_lines(text, "## Optimization problem contract") for field in REQUIRED_CONTRACT_FIELDS: @@ -210,5 +246,18 @@ def normalized_status_category(raw: str) -> str: problem_contract = ROOT / "OPTIMIZATION-PROBLEM.md" if not problem_contract.is_file(): die("OPTIMIZATION-PROBLEM.md is missing") +problem_text = problem_contract.read_text(encoding="utf-8") +problem_lines = problem_text.splitlines() +if not problem_lines or problem_lines[0] != "# Optimization Problem Contract": + die("OPTIMIZATION-PROBLEM.md has missing/invalid title") +if "## Canonical contract" not in problem_lines: + die("OPTIMIZATION-PROBLEM.md is missing ## Canonical contract") +canonical = section_lines(problem_text, "## Canonical contract") +canonical_text = "\n".join(canonical) +if "P = (X, F, f, d, C, B, S)" not in canonical_text: + die("OPTIMIZATION-PROBLEM.md is missing canonical P = (X, F, f, d, C, B, S) formula") +for field, pattern in CANONICAL_DEFINITION_PATTERNS.items(): + if not any(pattern.match(line) for line in canonical): + die(f"OPTIMIZATION-PROBLEM.md is missing canonical definition for {field}") print(f"CATALOG_INTEGRITY_OK records={len(records)} frozen_v1={len(FROZEN_V1)}") From c2b00d13416b2c591a3e11ec04d6f9a8b6301634 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 13:42:50 +0930 Subject: [PATCH 008/229] Harden optimization edge-case contracts --- ...UDGET-001-performance-regression-budgets.md | 14 +++++++------- ...PT-FAN-001-shared-materialization-fanout.md | 18 ++++++++++-------- ...NE-001-bound-driven-search-space-pruning.md | 16 ++++++++-------- ...T-REDUCE-001-early-working-set-reduction.md | 16 ++++++++-------- ...-SEARCH-001-budget-aware-adaptive-search.md | 18 +++++++++--------- 5 files changed, 42 insertions(+), 40 deletions(-) diff --git a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md index 3519d67..31be67d 100644 --- a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md +++ b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md @@ -19,17 +19,17 @@ Small performance regressions accumulate because performance is measured occasio - F: gate configurations based on a sufficiently characterized environment and workload, with statistically justified tolerance and no weakening of functional correctness or workload realism - f: target-measured regression-detection quality together with CI noise/false-alarm rate and measurement overhead - d: minimize missed material regressions and flaky/false failures under the target's predeclared multi-objective ordering -- C: the performance gate must not incentivize weakening tests, assertions, evidence, semantic coverage, or representative workload inputs +- C: the performance gate must not incentivize weakening tests, assertions, evidence, semantic coverage, or representative workload inputs; the gate itself must continue to detect known regressions and accept known-good controls within the declared false-positive/false-negative envelope - B: target-specific calibration budget specifying repetitions, environment samples, and allowable CI/runtime measurement cost - S: stop calibration when the declared sample budget is exhausted or the baseline/noise estimate is stable enough to justify the predeclared warning and hard thresholds ## Preserved contract -A performance gate may not incentivize weakening functional tests, correctness, evidence or workload realism. +A performance gate may not incentivize weakening functional tests, correctness, evidence or workload realism. The gate is valid only while its fixture, environment characterization, detection sensitivity and measurement overhead remain inside their declared contract. ## Optimization -Turn a stable, reproducible performance expectation into a regression gate. Compare distributions or robust summaries where noise matters; separate machine/environment drift from code regression; keep cold/warm claims distinct. +Turn a stable, reproducible performance expectation into a regression gate. Compare distributions or robust summaries where noise matters; separate machine/environment drift from code regression; keep cold/warm claims distinct. Keep known-fast and known-regressed control fixtures (or equivalent calibration cases) so the gate can periodically prove it still distinguishes acceptable from materially regressed behavior. ## Before / after evidence @@ -41,16 +41,16 @@ Turn a stable, reproducible performance expectation into a regression gate. Comp ## Validation -Calibrate variance before setting the threshold. Self-test the gate with known fast/slow fixtures and preserve raw samples where practical. +Calibrate variance before setting the threshold. Self-test the gate with known-good and known-regressed fixtures and preserve raw samples where practical. Re-run these controls after runner/toolchain changes and periodically enough to detect stale fixtures or sensitivity drift. Measure false positives, false negatives and gate overhead against predeclared acceptance limits. ## Target-repo adaptation -Never copy another project's milliseconds, bundle sizes or thresholds. Establish the target's own baseline and noise envelope. +Never copy another project's milliseconds, bundle sizes or thresholds. Establish the target's own baseline and noise envelope, define control fixtures, and declare acceptable false-positive/false-negative rates plus a maximum measurement-overhead budget. ## Failure modes -Flaky gates from uncontrolled runners, benchmark gaming, stale fixtures, hardware drift and thresholds so loose they provide no protection. +Flaky gates from uncontrolled runners, benchmark gaming, stale fixtures, hardware drift, thresholds so loose they miss real regressions, thresholds so tight they block good changes, and measurement overhead large enough to damage CI usability or distort the workload under test. ## Rollback trigger -Temporarily disable only when the measurement environment is proven invalid; fix/recalibrate the benchmark rather than deleting the budget because code regressed. +Disable or demote the gate to non-blocking and recalibrate whenever its measurement environment is invalid, its fixture is stale/nonrepresentative, known regressions are no longer detected, known-good controls fail above the declared false-positive limit, observed false negatives exceed the declared limit, or measurement overhead exceeds the predeclared budget. Do **not** disable merely because product code legitimately regressed; in that case keep the valid gate and fix or explicitly accept the regression through the target's normal review process. diff --git a/optimizations/OPT-FAN-001-shared-materialization-fanout.md b/optimizations/OPT-FAN-001-shared-materialization-fanout.md index b9c82e3..ef69024 100644 --- a/optimizations/OPT-FAN-001-shared-materialization-fanout.md +++ b/optimizations/OPT-FAN-001-shared-materialization-fanout.md @@ -14,22 +14,24 @@ The same deterministic transformation is repeated independently for each consume ## Optimization problem contract -- X: target-supported materialization boundaries, representation formats/versions, persistence policies, and raw-versus-materialized retention policies -- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, and trust requirement +- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, and complete materialization-key definitions +- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, and materialization-equivalence requirement - f: measured transformation CPU, replay CPU, fan-out latency, and storage/I/O overhead under the target's declared objective ordering - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: consumers receive the declared representation semantics exactly; verification/security metadata may be removed only under an explicit contract change +- C: consumers receive the declared representation semantics exactly; reuse is allowed only when the materialization key proves the persisted artifact was produced from the same effective source and transformation identity; verification/security metadata may be removed only under an explicit contract change - B: target-specific fan-out/replay benchmark budget declared before tuning; no portable subscriber count, replay size, or retention duration is supplied here - S: stop when the declared budget is exhausted or a validated materialization policy materially improves the target objective without violating C ## Preserved contract -Consumers must receive the same declared representation semantics. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. +Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. ## Optimization Perform an expensive deterministic transform once near production, persist or retain the reusable representation, and fan out/replay those bytes/objects rather than reconstructing them per consumer. +Persist a materialization key beside the artifact. The key must cover, as applicable, source object/content identity or immutable source version, transformation/encoder implementation identity, encoder configuration and dictionaries, schema/format version, feature flags, trust/security policy, and any other effective input that can affect the materialized result. Before reuse, require both: (1) exact agreement with the current materialization key, and (2) artifact integrity/format validity. A key mismatch or unverifiable artifact is a cache miss and requires regeneration. Do not use format validation alone as evidence that an artifact corresponds to current inputs. + ## Before / after evidence - Environment: No controlled target-repository benchmark has been run for this OPT record. @@ -40,16 +42,16 @@ Perform an expensive deterministic transform once near production, persist or re ## Validation -Compare shared materialization against per-consumer reference output, including version changes, corruption, restart/replay and mixed consumer capabilities. +Compare shared materialization against per-consumer reference output, including corruption, restart/replay and mixed consumer capabilities. Independently mutate each key component—source content/version, transform implementation, encoder options/dictionary, schema/format version, feature flags and trust policy—and prove that each output-affecting change invalidates reuse. Also test unchanged-key reuse, tampered artifacts with matching metadata, and migration/version-boundary cases. ## Target-repo adaptation -Choose representation versioning, invalidation, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. +Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, and specify representation versioning, invalidation, integrity checking, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. ## Failure modes -Materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. +Incomplete keys can serve stale representations after source or transform changes; metadata can match while artifact bytes are corrupted; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. ## Rollback trigger -Disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. +Disable reuse immediately if any materialization-key hit or integrity check can return output that differs from a fresh transform for the same current effective inputs. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index f36c134..21046ee 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -19,9 +19,9 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F - f: a scalar real-valued target objective `f : F → R` evaluated on feasible candidates only - d: exactly one of scalar `minimize` or scalar `maximize`; vector, Pareto, lexicographic, or other partial-order objectives are outside this record unless a separately specified and validated frontier-bound mechanism is introduced -- C: the returned incumbent satisfies the original feasibility/semantic contract, and every pruning decision is justified by a separately defined sound scalar region-bound function `b` +- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, and budget exhaustion without a feasible incumbent produces an explicit unknown/no-incumbent outcome rather than a feasibility or optimality claim - B: a finite, predeclared target-specific cap on evaluations, wall time, compute, or equivalent resource consumption; exact-mode search may prove optimality before this cap but may not run without a finite cap -- S: stop immediately when optimality is proven or the frontier is exhausted; otherwise stop when B is exhausted and return the best validated incumbent plus the remaining valid bound/optimality gap without claiming exact completion +- S: stop immediately when optimality is proven or the frontier is exhausted; otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality For each unexplored region `R`, define a bound `b(R)` separately from `f`: @@ -32,11 +32,11 @@ An independently proven infeasible region may also be pruned. A heuristic estima ## Preserved contract -A region may be discarded only when its scalar bound proves it cannot improve the incumbent under the declared scalar objective and constraints. Heuristic guesses are not proof-based pruning, and exhausting B without an optimality proof does not permit an exactness claim. +A region may be discarded only when its scalar bound proves it cannot improve the incumbent under the declared scalar objective and constraints. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, and exhausting B without a feasible incumbent does not permit an infeasibility claim. ## Optimization -Maintain an incumbent, partition the search space, compute a cheap sound `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific scalar bound relation proves the region cannot improve the incumbent. +Maintain an incumbent when one exists, partition the search space, compute a cheap sound `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific scalar bound relation proves the region cannot improve the incumbent. Before the first incumbent exists, sound bounds may prioritize regions or prove individual regions infeasible, but incumbent-based objective pruning is unavailable. A relaxed solution is evidence for a bound, not automatically a feasible final answer. @@ -50,16 +50,16 @@ A relaxed solution is evidence for a bound, not automatically a feasible final a ## Validation -For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness independently by checking the direction-specific inequality against exhaustive feasible values inside each test region. Test pruning separately from search ordering. Verify that budget exhaustion returns an anytime result without an exactness claim, and record the remaining valid optimality gap/bound whenever exact completion was not proven. +For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness independently by checking the direction-specific inequality against exhaustive feasible values inside each test region. Test pruning separately from search ordering. Include fixtures where the first feasible candidate is found late and where B expires before any feasible candidate exists; verify that the latter returns `no-incumbent / feasibility-unknown`, reports only independently valid global-bound information, and makes no infeasibility, optimality, or incumbent-based gap claim. Verify that budget exhaustion with an incumbent returns an anytime result without an exactness claim. ## Target-repo adaptation -The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. +The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. ## Failure modes -Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; an unbounded exact-search policy can consume resources indefinitely. +Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; an unbounded exact-search policy can consume resources indefinitely; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. ## Rollback trigger -Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar bound relation, is applied to an unsupported objective ordering, or whose bound cost exceeds the work it eliminates. Abort exact-mode claims whenever B is exhausted before optimality is proven. +Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar bound relation, is applied to an unsupported objective ordering, or whose bound cost exceeds the work it eliminates. Abort exact-mode claims whenever B is exhausted before optimality is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. diff --git a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md index cf428dd..be1758f 100644 --- a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md +++ b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md @@ -15,20 +15,20 @@ An expensive operation is applied to a large population even though only a small ## Optimization problem contract - X: semantically legal placements and implementations of filtering, culling, limiting, candidate selection, or other working-set reductions in the target pipeline -- F: placements that preserve every candidate and ordering/tie/join semantic required by the final result and satisfy target resource constraints +- F: placements that preserve every candidate and every contractually observable behavior required by the reference pipeline—including output, ordering/tie/join semantics, errors/exceptions, writes, mutations, auditing/telemetry, and other side effects—or that move only across stages proven pure with respect to those effects - f: measured end-to-end pipeline cost and cardinality presented to the expensive stage - d: minimize under the target's predeclared objective ordering -- C: the reordered/reduced pipeline must be semantically equivalent to the reference pipeline for all declared output, ordering, top-k, tie, null, and join semantics +- C: the reordered/reduced pipeline is semantically equivalent to the reference for all declared outputs **and observable effects**; an effectful stage may be bypassed for discarded candidates only when those effects/errors are explicitly proven irrelevant by the target contract - B: target-specific benchmark budget over representative and adversarial selectivity distributions; no portable selectivity threshold is supplied here - S: stop when the declared budget is exhausted or a validated early-reduction placement materially lowers total cost without violating C ## Preserved contract -Moving a reduction earlier is valid only if it is semantically equivalent to the original later reduction, including ordering/top-k/tie and join semantics where relevant. +Moving a reduction earlier is valid only across a pure stage or when the earlier placement preserves the full observable contract of the reference pipeline. That contract includes final values plus ordering/top-k/tie/join semantics, exceptions/error checks, writes/mutations, audit events, metrics or other externally visible side effects where they are significant. ## Optimization -Push selective operations toward the input boundary: filter before join, cull before render, select candidate roots before expensive DSP, prune impossible simulations before full evaluation. Prefer indexed/cheap predicates to expensive composition over the full population. +Push selective operations toward the input boundary only when doing so is semantics-preserving: filter before a pure expensive join/transform, cull before pure rendering work, select candidate roots before pure DSP calculation, or eliminate simulations proven unable to affect any required result/effect. If the expensive stage is effectful, either keep the effectful portion on every candidate that would have reached it in the reference path, split the stage into a pure expensive computation and a required effect layer, or prove those skipped effects/errors are outside the declared contract. Do not optimize away observable behavior merely because the final data rows match. ## Before / after evidence @@ -40,16 +40,16 @@ Push selective operations toward the input boundary: filter before join, cull be ## Validation -Differential-test reordered pipelines against the reference, with emphasis on ties, null/missing values, boundary ordering and rare candidates. +Differential-test reordered pipelines against the reference, with emphasis on ties, null/missing values, boundary ordering and rare candidates. Also compare observable side effects and error behavior: writes/mutations, audit/log/metric events, callbacks, exception/error surfaces and their relevant ordering/counts. Include a deliberately effectful fixture to prove the optimization is rejected or preserves the effects, and a pure-stage fixture where early reduction is admissible. ## Target-repo adaptation -Measure selectivity and reduction cost. A cheap filter with low selectivity may simply add another pass. +Measure selectivity and reduction cost. Classify the expensive stage as pure or effectful before reordering; inventory contractually significant side effects/errors and define how each is preserved. A cheap filter with low selectivity may simply add another pass. ## Failure modes -Illegal predicate reordering, changed top-k semantics, underestimated filtering cost, loss of vectorization and duplicated scans. +Illegal predicate reordering, changed top-k semantics, underestimated filtering cost, loss of vectorization, duplicated scans, skipped writes/audit events/mutations, changed exceptions or validation failures, and reordered side effects can all make an apparently equivalent final result semantically wrong. ## Rollback trigger -Revert if outputs differ or total measured cost does not fall on representative workloads. +Immediately revert if any output, ordering, error, or contractually significant side effect differs from the reference path. Also revert if total measured cost does not fall on representative workloads. diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index 0ce6926..cf28162 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -20,19 +20,19 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra - F: candidates in X that satisfy all hard resource, platform, semantic, and correctness constraints before objective ranking - f: the target-measured objective or objective vector for each feasible candidate, including declared noise/statistical treatment - d: the target's predeclared minimize, maximize, lexicographic, or Pareto ordering -- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for already reserved/in-flight trials -- B: an explicit target-specific maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit and failure/cancellation charging policy are fixed before dispatch begins -- S: stop proposing/dispatching when no additional trial can be reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger +- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources +- B: an explicit target-specific maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit, conservative per-trial reservation rule, and failure/cancellation charging policy are fixed before dispatch begins +- S: stop proposing/dispatching when no additional trial can be conservatively reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger ## Preserved contract -Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard dispatch bound: pending work counts according to the predeclared accounting policy rather than being ignored until completion. +Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard dispatch bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B. ## Optimization Use observations to adapt future evaluations: surrogate/acquisition search for expensive black-box objectives, conditional spaces where parameters only exist under certain choices, progressive domain contraction where justified, and explicit stopping/evaluation budgets. For asynchronous workers, reserve pending regions or otherwise diversify proposals so workers do not redundantly evaluate the same neighborhood. -Before dispatching an asynchronous trial, atomically reserve that trial in the ledger and debit the applicable unit from B (evaluation count, money, compute quota, or the target's declared equivalent). If the reservation would exceed B, do not dispatch. A reserved trial remains budget-accounted while pending. The target must predeclare whether failed/cancelled trials consume the reservation permanently, partially, or are refunded; that rule is applied deterministically and recorded in the ledger. Completion converts the reservation into a completed trial without charging the same budget twice. +Before dispatching an asynchronous trial, atomically reserve a conservative amount of the applicable budget and record the pending trial in the ledger. If `consumed + reserved + proposed_reservation > B`, do not dispatch. Completion converts the actually consumed amount into permanent budget consumption and releases only any **demonstrably unconsumed** remainder of the reservation; it must not double-charge the same usage. A failed or cancelled trial never erases resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute/time budgets, release only the measured or otherwise provable unused portion of the reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. Every reservation, consumption adjustment, release, failure, and cancellation is recorded in the ledger. Parallelism has an information cost: very wide batches receive less feedback between suggestions and can degenerate toward non-adaptive/random search. @@ -46,16 +46,16 @@ Parallelism has an information cost: very wide batches receive less feedback bet ## Validation -Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. For asynchronous search, test the budget boundary with multiple workers contending for the last remaining reservation (for example, 99 of 100 evaluation slots already consumed/reserved) and prove that at most one additional trial can be dispatched. Inject failures and cancellations and verify the declared charge/refund policy without double-debit or budget overshoot. +Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. For asynchronous search, test the budget boundary with multiple workers contending for the last remaining reservation and prove no dispatch can make `consumed + reserved` exceed B. Inject early failures, late failures, partial consumption, and cancellation after measurable work; verify that only demonstrably unconsumed reservation is released, evaluation-count slots are not resurrected after dispatch, and repeated failures cannot create extra budget capacity. ## Target-repo adaptation -Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, atomic reservation mechanism, and failure/cancellation charging policy for the target before enabling asynchronous dispatch. +Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, atomic reservation mechanism, metering source, and failure/cancellation charging policy before enabling asynchronous dispatch. ## Failure modes -Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; ambiguous refund rules can make the ledger disagree with actual resource consumption. +Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism. ## Rollback trigger -Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, or any concurrency test shows dispatch can exceed the declared budget after pending reservations are counted. +Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, or any accounting/concurrency test shows that dispatch, failure, cancellation, or reservation release can cause actual consumption plus outstanding reservations to exceed B. From e01d9a1a7ed13f4481b85a14181af33e93646b9e Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 15:12:33 +0930 Subject: [PATCH 009/229] Harden async caps and catalog indexing --- ...01-concurrent-duplicate-work-coalescing.md | 18 ++++++------ ...T-FAN-001-shared-materialization-fanout.md | 22 +++++++++------ ...SEARCH-001-budget-aware-adaptive-search.md | 20 +++++++------ scripts/check_catalog.py | 28 +++++++++++-------- templates/OPTIMIZATION-RECORD.md | 7 +++-- 5 files changed, 57 insertions(+), 38 deletions(-) mode change 100644 => 100755 scripts/check_catalog.py diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index 70855f0..71e66d2 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -18,19 +18,21 @@ Many callers request the same expensive computation concurrently before any call - F: policies that coalesce only semantically equivalent requests and preserve authorization, timeout, cancellation, result, and error semantics for every joined caller - f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization and waiter-memory overhead - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives a result or error valid for its original request semantics; non-equivalent requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller +- C: every joined caller receives a result or error valid for its original request semantics; non-equivalent requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; once a shared generation enters cancellation/closure it is no longer joinable by new callers - B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied by this record - S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C ## Preserved contract -Coalescing may merge only requests that are semantically equivalent for the shared operation. Each caller retains independent cancellation and timeout semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters, and authorization/result/error semantics must remain explicit. +Coalescing may merge only requests that are semantically equivalent for the same **joinable generation** of the shared operation. Each caller retains independent cancellation and timeout semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once the last waiter leaves and cancellation is initiated, that generation is closed to new joiners before upstream cancellation proceeds asynchronously. ## Optimization -Create an in-flight registry entry for the canonical request key. The first caller starts the shared upstream operation, but **does not own its lifetime**. Every equivalent caller registers as an independent waiter on that shared operation. +Create an in-flight registry entry for the canonical request key. The first caller starts the shared upstream operation, but **does not own its lifetime**. Every equivalent caller registers as an independent waiter on the currently joinable generation. -Cancellation and timeout are per waiter: when one waiter leaves, remove only that waiter. Cancel the upstream operation only when no live waiters remain, or when a separately documented target policy proves that early cancellation is safe. On success or failure, deliver the same shared terminal result/error to all waiters that are still registered, then remove the in-flight entry deterministically. Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new operation after the failed entry is removed. +Cancellation and timeout are per waiter: when one waiter leaves, remove only that waiter. If live waiters remain, keep the shared generation joinable. If the last waiter leaves and the policy calls for upstream cancellation, atomically mark the registry entry **closing/non-joinable** (or remove it from the joinable map) before sending the asynchronous cancellation request upstream. A new caller arriving after that transition must create a fresh generation rather than attach to work already being canceled. The closing generation may remain internally tracked until its terminal completion for cleanup/accounting, but it is not eligible for coalescing. + +On success or failure, deliver the same shared terminal result/error to all waiters still registered to that generation, then remove/retire the entry deterministically. Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. This differs from caching: the reusable result does not exist yet. @@ -44,16 +46,16 @@ This differs from caching: the reusable result does not exist yet. ## Validation -Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; test waiter-specific deadlines; verify shared failure delivery and retry accounting; prove only one upstream evaluation occurs for a coalesced key while all surviving callers terminate correctly. +Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; race a new caller against the last-waiter cancellation transition and prove it never joins the closing generation; test waiter-specific deadlines; verify shared failure delivery and retry accounting; prove only one upstream evaluation occurs per joinable generation while all surviving callers terminate correctly. ## Target-repo adaptation -Define key canonicalization, maximum waiter count, per-waiter cancellation/deadline handling, the exact condition for canceling upstream work, and whether failures are shared as terminal or retried under one explicit shared retry policy. +Define key canonicalization, maximum waiter count, per-waiter cancellation/deadline handling, the exact condition for canceling upstream work, the atomic closing/non-joinable transition, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent work; coupling shared lifetime to the first caller can terminate valid waiters; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; unbounded waiter lists amplify memory; ambiguous retry/error policy can cause correlated or duplicated work. +Over-broad keys merge non-equivalent work; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled generation joinable can attach new callers to doomed work; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; unbounded waiter lists amplify memory; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes any caller's cancellation/result/error semantics, permits one caller to cancel work required by another, leaks orphaned shared operations, increases tail latency materially, or creates unacceptable failure amplification. +Disable if coalescing changes any caller's cancellation/result/error semantics, permits one caller to cancel work required by another, allows a new caller to join a closing/canceled generation, leaks orphaned shared operations, increases tail latency materially, or creates unacceptable failure amplification. diff --git a/optimizations/OPT-FAN-001-shared-materialization-fanout.md b/optimizations/OPT-FAN-001-shared-materialization-fanout.md index ef69024..ab3f103 100644 --- a/optimizations/OPT-FAN-001-shared-materialization-fanout.md +++ b/optimizations/OPT-FAN-001-shared-materialization-fanout.md @@ -14,23 +14,27 @@ The same deterministic transformation is repeated independently for each consume ## Optimization problem contract -- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, and complete materialization-key definitions -- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, and materialization-equivalence requirement +- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, and crash-consistent publication schemes +- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, and publication-atomicity requirement - f: measured transformation CPU, replay CPU, fan-out latency, and storage/I/O overhead under the target's declared objective ordering - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: consumers receive the declared representation semantics exactly; reuse is allowed only when the materialization key proves the persisted artifact was produced from the same effective source and transformation identity; verification/security metadata may be removed only under an explicit contract change +- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state cryptographically or structurally binds the artifact bytes to the complete effective source/transform identity; verification/security metadata may be removed only under an explicit contract change - B: target-specific fan-out/replay benchmark budget declared before tuning; no portable subscriber count, replay size, or retention duration is supplied here - S: stop when the declared budget is exhausted or a validated materialization policy materially improves the target objective without violating C ## Preserved contract -Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. +Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive crashes and interrupted publication. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. ## Optimization Perform an expensive deterministic transform once near production, persist or retain the reusable representation, and fan out/replay those bytes/objects rather than reconstructing them per consumer. -Persist a materialization key beside the artifact. The key must cover, as applicable, source object/content identity or immutable source version, transformation/encoder implementation identity, encoder configuration and dictionaries, schema/format version, feature flags, trust/security policy, and any other effective input that can affect the materialized result. Before reuse, require both: (1) exact agreement with the current materialization key, and (2) artifact integrity/format validity. A key mismatch or unverifiable artifact is a cache miss and requires regeneration. Do not use format validation alone as evidence that an artifact corresponds to current inputs. +Define a materialization key that covers, as applicable, source object/content identity or immutable source version, transformation/encoder implementation identity, encoder configuration and dictionaries, schema/format version, feature flags, trust/security policy, and any other effective input that can affect the materialized result. + +Publish the artifact and its identity as **one committed state**. Acceptable designs include content-addressed storage where the artifact digest is itself part of the committed key, an atomically replaced manifest that contains both the full materialization key and the artifact digest/location, or another crash-consistent transaction that makes old state or new state visible but never a mixed pair. Do not update artifact bytes and their key independently in a way that can expose a new artifact with stale metadata or stale bytes with a new key after a crash. + +Before reuse, require: (1) exact agreement with the current effective-input materialization key, (2) a committed manifest/content-address relation that binds that key to the artifact identity, and (3) artifact integrity/format validity. A key mismatch, missing/incomplete publication marker, digest mismatch, or unverifiable artifact is a cache miss and requires regeneration. Do not use format validation alone as evidence that an artifact corresponds to current inputs. ## Before / after evidence @@ -44,14 +48,16 @@ Persist a materialization key beside the artifact. The key must cover, as applic Compare shared materialization against per-consumer reference output, including corruption, restart/replay and mixed consumer capabilities. Independently mutate each key component—source content/version, transform implementation, encoder options/dictionary, schema/format version, feature flags and trust policy—and prove that each output-affecting change invalidates reuse. Also test unchanged-key reuse, tampered artifacts with matching metadata, and migration/version-boundary cases. +Inject crashes/interruption at every publication boundary: after artifact write but before manifest commit, after provisional metadata write, during atomic replacement, and immediately after commit. After restart, prove that readers see either the previous valid committed materialization or the new valid committed materialization, never a mixed key/artifact state. Verify digest/key mismatch is rejected even when the artifact is otherwise parseable. + ## Target-repo adaptation -Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, and specify representation versioning, invalidation, integrity checking, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. +Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, and specify representation versioning, invalidation, integrity checking, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. ## Failure modes -Incomplete keys can serve stale representations after source or transform changes; metadata can match while artifact bytes are corrupted; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. +Incomplete keys can serve stale representations after source or transform changes; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. ## Rollback trigger -Disable reuse immediately if any materialization-key hit or integrity check can return output that differs from a fresh transform for the same current effective inputs. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. +Disable reuse immediately if any materialization-key hit, publication-recovery path, or integrity check can return output that differs from a fresh transform for the same current effective inputs, or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index cf28162..f452f4c 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -20,19 +20,21 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra - F: candidates in X that satisfy all hard resource, platform, semantic, and correctness constraints before objective ranking - f: the target-measured objective or objective vector for each feasible candidate, including declared noise/statistical treatment - d: the target's predeclared minimize, maximize, lexicographic, or Pareto ordering -- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources -- B: an explicit target-specific maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit, conservative per-trial reservation rule, and failure/cancellation charging policy are fixed before dispatch begins -- S: stop proposing/dispatching when no additional trial can be conservatively reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger +- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources, and every per-trial reservation must be an enforceable upper bound rather than an estimate +- B: an explicit target-specific hard maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit, enforceable per-trial cap mechanism, conservative reservation rule, and failure/cancellation charging policy are fixed before dispatch begins +- S: stop proposing/dispatching when no additional trial can be safely reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger ## Preserved contract -Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard dispatch bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B. +Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B, and no individual trial may consume beyond its reserved cap. ## Optimization Use observations to adapt future evaluations: surrogate/acquisition search for expensive black-box objectives, conditional spaces where parameters only exist under certain choices, progressive domain contraction where justified, and explicit stopping/evaluation budgets. For asynchronous workers, reserve pending regions or otherwise diversify proposals so workers do not redundantly evaluate the same neighborhood. -Before dispatching an asynchronous trial, atomically reserve a conservative amount of the applicable budget and record the pending trial in the ledger. If `consumed + reserved + proposed_reservation > B`, do not dispatch. Completion converts the actually consumed amount into permanent budget consumption and releases only any **demonstrably unconsumed** remainder of the reservation; it must not double-charge the same usage. A failed or cancelled trial never erases resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute/time budgets, release only the measured or otherwise provable unused portion of the reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. Every reservation, consumption adjustment, release, failure, and cancellation is recorded in the ledger. +Before dispatching an asynchronous trial, atomically reserve a conservative amount of the applicable budget and record the pending trial in the ledger. If `consumed + reserved + proposed_reservation > B`, do not dispatch. The reservation must be an **enforceable upper limit** for that trial, not merely an estimate: use a per-trial quota, wall-time deadline with forced cancellation/termination, provider spending cap, cgroup/job resource limit, evaluation-slot ownership, or another mechanism that prevents actual trial consumption from exceeding the reservation. If the target cannot enforce such a cap for a resource dimension, that dimension cannot be advertised as a hard maximum B; instead define a different enforceable budget or explicitly classify the quantity as observational rather than bounded. + +Completion converts the actually consumed amount into permanent budget consumption and releases only any **demonstrably unconsumed** remainder of the reservation; it must not double-charge the same usage. A failed or cancelled trial never erases resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute/time budgets, release only the measured or otherwise provable unused portion of the enforceable reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. Every reservation, cap enforcement action, consumption adjustment, release, failure, cancellation, and forced termination is recorded in the ledger. Parallelism has an information cost: very wide batches receive less feedback between suggestions and can degenerate toward non-adaptive/random search. @@ -46,16 +48,16 @@ Parallelism has an information cost: very wide batches receive less feedback bet ## Validation -Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. For asynchronous search, test the budget boundary with multiple workers contending for the last remaining reservation and prove no dispatch can make `consumed + reserved` exceed B. Inject early failures, late failures, partial consumption, and cancellation after measurable work; verify that only demonstrably unconsumed reservation is released, evaluation-count slots are not resurrected after dispatch, and repeated failures cannot create extra budget capacity. +Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. For asynchronous search, test the budget boundary with multiple workers contending for the last remaining reservation and prove no dispatch can make `consumed + reserved` exceed B. Deliberately run trials that attempt to exceed their per-trial money/compute/time reservation and prove the quota/deadline/termination mechanism prevents the overrun. Inject early failures, late failures, partial consumption, and cancellation after measurable work; verify that only demonstrably unconsumed reservation is released, evaluation-count slots are not resurrected after dispatch, and repeated failures cannot create extra budget capacity. ## Target-repo adaptation -Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, atomic reservation mechanism, metering source, and failure/cancellation charging policy before enabling asynchronous dispatch. +Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, **enforcement mechanism for that reservation**, atomic reservation mechanism, metering source, and failure/cancellation charging policy before enabling asynchronous dispatch. ## Failure modes -Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism. +Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; an unenforced reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism. ## Rollback trigger -Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, or any accounting/concurrency test shows that dispatch, failure, cancellation, or reservation release can cause actual consumption plus outstanding reservations to exceed B. +Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, any trial can consume beyond its enforceable reservation, or any accounting/concurrency test shows that dispatch, failure, cancellation, forced termination, or reservation release can cause actual consumption plus outstanding reservations to exceed B. diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py old mode 100644 new mode 100755 index 2ae1d42..6492f24 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -186,13 +186,12 @@ def normalized_status_category(raw: str) -> str: record_paths = {str(path.relative_to(ROOT)): record_id for record_id, path in records.items()} -# README is the complete human-facing record index. CATALOG may use either -# links or plain/backticked IDs, but any optimization-record link in either -# document must use the target record's stable ID as its label. +# Validate every optimization-record link wherever it appears. README index +# completeness/uniqueness is checked separately from actual catalog table rows, +# so contextual prose links are allowed and do not count as duplicate index rows. for doc_name in ("README.md", "CATALOG.md"): text = (ROOT / doc_name).read_text(encoding="utf-8") links = LINK_RE.findall(text) - linked_ids: list[str] = [] for label, rel in links: target = ROOT / rel if not target.is_file(): @@ -205,19 +204,26 @@ def normalized_status_category(raw: str) -> str: f"record link label mismatch in {doc_name}: '{label}' points to " f"{target_id} ({rel})" ) - linked_ids.append(target_id) if doc_name == "README.md": - counts = Counter(linked_ids) - duplicates = sorted(record_id for record_id, count in counts.items() if count != 1) - if duplicates: - die(f"README.md must index each record exactly once; bad counts for: {', '.join(duplicates)}") + rows = README_ROW_RE.findall(text) + row_ids = [row_id for row_id, _rel, _status in rows] + counts = Counter(row_ids) + bad_counts = sorted(record_id for record_id, count in counts.items() if count != 1) + if bad_counts: + die( + "README.md catalog table must index each record exactly once; " + f"bad row counts for: {', '.join(bad_counts)}" + ) missing_readme = sorted(records.keys() - counts.keys()) if missing_readme: - die(f"README.md is missing record(s): {', '.join(missing_readme)}") + die(f"README.md catalog table is missing record(s): {', '.join(missing_readme)}") + unknown_rows = sorted(counts.keys() - records.keys()) + if unknown_rows: + die(f"README.md catalog table references unknown record(s): {', '.join(unknown_rows)}") row_statuses: dict[str, str] = {} - for row_id, rel, raw_status in README_ROW_RE.findall(text): + for row_id, rel, raw_status in rows: if record_paths.get(rel) != row_id: die(f"README.md row identity mismatch for {row_id}: {rel}") if row_id in row_statuses: diff --git a/templates/OPTIMIZATION-RECORD.md b/templates/OPTIMIZATION-RECORD.md index 9596d31..419b5c6 100644 --- a/templates/OPTIMIZATION-RECORD.md +++ b/templates/OPTIMIZATION-RECORD.md @@ -28,12 +28,15 @@ Define the target using `OPTIMIZATION-PROBLEM.md`. Keep these seven canonical fi - B: - S: -Then record useful classification detail: +Then record every required problem-classification dimension from `OPTIMIZATION-PROBLEM.md`: - Variables: continuous / integer / categorical / conditional / mixed -- Objective behavior: deterministic / noisy / stochastic - Search scope: local / global +- Objective behavior: deterministic / noisy / stochastic - Information: gradient / derivative-free / black-box +- Evaluation cost: cheap / moderate / expensive +- Constraints: bounds / equality / inequality / semantic / resource +- Parallelism: sequential / synchronous batch / asynchronous - Exactness: exact / approximation permitted under explicit error contract ## Preserved contract From 9acebb25380820c73908fb1b901546e38c2977b0 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 15:24:23 +0930 Subject: [PATCH 010/229] Harden atomic state and approximation contracts --- ...PROX-001-contract-bounded-approximation.md | 24 +++++++++++-------- ...01-concurrent-duplicate-work-coalescing.md | 22 +++++++++-------- ...1-signature-bound-incremental-execution.md | 22 +++++++++-------- ...SEARCH-001-budget-aware-adaptive-search.md | 18 +++++++------- 4 files changed, 48 insertions(+), 38 deletions(-) diff --git a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md index 0cacb3e..b1a7a2a 100644 --- a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md +++ b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md @@ -15,22 +15,24 @@ Exact processing has unbounded or unacceptable cost even though the product/scie ## Optimization problem contract -- X: target-supported approximation policies, quality/resource ceilings, sampling/culling/LOD policies, update frequencies, and exact-mode fallback choices -- F: policies whose declared error/degradation metric remains within the target's explicit envelope and whose resource/semantic constraints are satisfied +- X: target-supported approximation policies, quality/resource ceilings, sampling/culling/LOD policies, update frequencies, state-reset rules, evaluation horizons, and exact-mode fallback choices +- F: policies whose declared error/degradation metric remains within the target's explicit envelope over the declared state/composition horizon and whose resource/semantic constraints are satisfied - f: target-measured resource or latency cost, optionally paired with the declared quality/error metric - d: minimize resource/latency cost subject to feasibility in F, or use the target's predeclared multi-objective ordering when quality is ranked rather than hard-bounded -- C: approximation is permitted only by an explicit contract; exact callers are not silently weakened, and an exact reference path or exact fixture remains available where practical -- B: target-specific benchmark/quality-evaluation budget over predeclared ordinary, boundary, and adversarial fixtures -- S: stop when the evaluation budget is exhausted or a validated policy meets the target resource objective while remaining inside the declared quality envelope +- C: approximation is permitted only by an explicit contract; exact callers are not silently weakened; the error norm, aggregation rule, sequence/composition horizon, and reset boundaries are declared before evaluation; an exact reference path or exact fixture remains available where practical +- B: target-specific benchmark/quality-evaluation budget over predeclared ordinary, boundary, adversarial, repeated-application, and long-horizon fixtures +- S: stop when the evaluation budget is exhausted or a validated policy meets the target resource objective while remaining inside the declared quality envelope over the entire declared horizon ## Preserved contract -Approximation is admissible only when the contract explicitly permits it. A previously exact API cannot be silently weakened and still be called correctness-preserving. +Approximation is admissible only when the contract explicitly permits it. A previously exact API cannot be silently weakened and still be called correctness-preserving. For stateful or repeatedly composed approximations, the contract applies over an explicitly declared horizon—not merely to each isolated step—so bounded per-step error is insufficient if drift can accumulate beyond the allowed envelope. ## Optimization Introduce a resource ceiling and degrade only along a declared dimension: sample/cull, lower level of detail, approximate search, bounded stale data, or reduced update frequency. Make the error surface measurable and reversible. +For stateful streaming, simulation, DSP, iterative numerical work, or any repeatedly applied approximation, define the error model before benchmarking: the norm/metric (for example absolute, relative, L2, perceptual, state-distance, or domain-specific), how error composes or is aggregated through time, the maximum sequence length or physical/time horizon over which the envelope must hold, and any reset/checkpoint/re-synchronization boundaries that legitimately restart the horizon. If the system can run longer than the validated horizon without reset, either extend validation to that operational horizon or define a separate long-run drift bound; do not infer long-run safety from one-step ε alone. + ## Before / after evidence - Environment: No controlled target-repository benchmark has been run for this OPT record. @@ -41,16 +43,18 @@ Introduce a resource ceiling and degrade only along a declared dimension: sample ## Validation -Measure error/degradation and resource savings together across ordinary, boundary and adversarial workloads. Keep an exact reference for differential evaluation where practical. +Measure error/degradation and resource savings together across ordinary, boundary and adversarial workloads. Keep an exact reference for differential evaluation where practical. Declare and test the error norm/metric, aggregation rule, sequence/composition horizon, and reset boundaries explicitly. + +For stateful/repeated use, run differential trajectories against the exact path across short, nominal, maximum-supported, and adversarially long sequences. Include biased-error fixtures where each individual step remains within the local ε but errors accumulate in the same direction; verify the cumulative/state error still respects the declared horizon envelope. Test reset/checkpoint boundaries before, at, and after the limit; verify resets actually restore the assumptions used by the next horizon. Where stochastic approximation is used, evaluate both expected and tail/worst-case drift according to the declared statistical contract rather than only average one-step error. ## Target-repo adaptation -Define `ε`, quality metric, workload distribution, escape hatch and exact-mode availability locally. +Define `ε`, the exact quality/error norm, aggregation rule, workload distribution, maximum state/composition horizon, reset/checkpoint semantics, long-run drift policy, escape hatch and exact-mode availability locally. If the target has no finite operational horizon, establish a justified asymptotic/stability bound or periodic re-synchronization rule instead of copying a finite benchmark horizon from another system. ## Failure modes -Unmeasured quality loss, biased sampling, hidden rare-case failures, cumulative error and callers incorrectly assuming exact semantics. +Unmeasured quality loss, biased sampling, hidden rare-case failures, cumulative drift that is invisible to one-step checks, reset boundaries that fail to restore reference assumptions, state-dependent amplification, unstable feedback loops, and callers incorrectly assuming exact semantics. ## Rollback trigger -Disable when error exceeds the declared envelope, reference comparisons drift, or resource savings are not material. +Disable when error exceeds the declared envelope at any point within the supported horizon, cumulative/state drift exceeds the declared bound even though per-step ε holds, reset/checkpoint validation fails, reference comparisons drift beyond contract, or resource savings are not material. diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index 71e66d2..6f0ae2a 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,25 +14,25 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- X: target-supported request-key canonicalizations, shared-operation lifetime policies, waiter limits, per-waiter cancellation policies, and retry/error-sharing policies -- F: policies that coalesce only semantically equivalent requests and preserve authorization, timeout, cancellation, result, and error semantics for every joined caller +- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime policies, waiter limits, per-waiter cancellation policies, and retry/error-sharing policies +- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, and error semantics for every joined caller, and never admit new waiters to a closing or terminal generation - f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization and waiter-memory overhead - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives a result or error valid for its original request semantics; non-equivalent requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; once a shared generation enters cancellation/closure it is no longer joinable by new callers +- C: every joined caller receives a result or error valid for its original request semantics and authorization scope; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; closing/terminal generations are not joinable - B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied by this record - S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C ## Preserved contract -Coalescing may merge only requests that are semantically equivalent for the same **joinable generation** of the shared operation. Each caller retains independent cancellation and timeout semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once the last waiter leaves and cancellation is initiated, that generation is closed to new joiners before upstream cancellation proceeds asynchronously. +Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation and timeout semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. ## Optimization -Create an in-flight registry entry for the canonical request key. The first caller starts the shared upstream operation, but **does not own its lifetime**. Every equivalent caller registers as an independent waiter on the currently joinable generation. +Create an in-flight registry entry for a canonical equivalence key. The key must include every request attribute required to establish safe sharing, including authorization-relevant tenant/principal/visibility scope unless the target instead proves that the upstream result is globally shareable and independently authorizes each delivered result. The first caller starts the shared upstream operation, but **does not own its lifetime**. Every equivalent caller registers as an independent waiter on the currently joinable generation. Cancellation and timeout are per waiter: when one waiter leaves, remove only that waiter. If live waiters remain, keep the shared generation joinable. If the last waiter leaves and the policy calls for upstream cancellation, atomically mark the registry entry **closing/non-joinable** (or remove it from the joinable map) before sending the asynchronous cancellation request upstream. A new caller arriving after that transition must create a fresh generation rather than attach to work already being canceled. The closing generation may remain internally tracked until its terminal completion for cleanup/accounting, but it is not eligible for coalescing. -On success or failure, deliver the same shared terminal result/error to all waiters still registered to that generation, then remove/retire the entry deterministically. Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. +On success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the terminal waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Then snapshot the waiters still registered to that terminal generation, deliver the same shared terminal result/error to that snapshot, and retire/clean up the generation deterministically. Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. This differs from caching: the reusable result does not exist yet. @@ -46,16 +46,18 @@ This differs from caching: the reusable result does not exist yet. ## Validation -Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; race a new caller against the last-waiter cancellation transition and prove it never joins the closing generation; test waiter-specific deadlines; verify shared failure delivery and retry accounting; prove only one upstream evaluation occurs per joinable generation while all surviving callers terminate correctly. +Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; race a new caller against the last-waiter cancellation transition and prove it never joins the closing generation; race a new caller against success/failure completion and prove the terminal generation is made non-joinable before waiter snapshot/notification; test waiter-specific deadlines; verify shared failure delivery and retry accounting; prove only one upstream evaluation occurs per joinable generation while all surviving callers terminate correctly. + +Add authorization-boundary fixtures: issue syntactically identical requests under different tenants, principals, roles, ACL/visibility scopes, or other authorization context. Prove they either map to different equivalence keys **or** that the shared upstream result is explicitly safe to reuse and each caller is independently authorized before delivery. Verify that a result produced under one authorization scope can never leak to another merely because the resource parameters match. ## Target-repo adaptation -Define key canonicalization, maximum waiter count, per-waiter cancellation/deadline handling, the exact condition for canceling upstream work, the atomic closing/non-joinable transition, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. +Define key canonicalization, the authorization/visibility context that participates in equivalence, maximum waiter count, per-waiter cancellation/deadline handling, the exact condition for canceling upstream work, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent work; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled generation joinable can attach new callers to doomed work; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; unbounded waiter lists amplify memory; ambiguous retry/error policy can cause correlated or duplicated work. +Over-broad keys merge non-equivalent or authorization-distinct work; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; snapshotting waiters before terminal closure can strand a late joiner; omitting authorization scope can leak results across principals/tenants; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; unbounded waiter lists amplify memory; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes any caller's cancellation/result/error semantics, permits one caller to cancel work required by another, allows a new caller to join a closing/canceled generation, leaks orphaned shared operations, increases tail latency materially, or creates unacceptable failure amplification. +Disable if coalescing changes any caller's authorization/cancellation/result/error semantics, merges authorization-distinct requests without independent delivery authorization, permits one caller to cancel work required by another, allows a new caller to join a closing or terminal generation, strands a late joiner during terminal notification, leaks orphaned shared operations, increases tail latency materially, or creates unacceptable failure amplification. diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index 6c14555..7cc99e3 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -14,21 +14,23 @@ Expensive work is rerun even though every input capable of affecting its result ## Optimization problem contract -- X: target-supported signature definitions, persistence scopes, invalidation granularities, and output-validity policies -- F: configurations whose signature covers every output-affecting input, whose reuse validates required outputs, and whose failed executions never commit new reusable state -- f: measured repeated-work cost including stage runtime plus signature/metadata/output-validation I/O overhead +- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity policies, and crash-consistent state-publication mechanisms +- F: configurations whose signature covers every output-affecting input, whose reuse validates required outputs, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted executions never publish reusable partial state +- f: measured repeated-work cost including stage runtime plus signature/metadata/output-validation/publication I/O overhead - d: minimize -- C: every reused output is semantically equivalent to a fresh execution for the same effective inputs, with the same failure and output-validity semantics +- C: every reused output is semantically equivalent to a fresh execution for the same effective inputs, with the same failure and output-validity semantics; reuse metadata cannot mix fields from different generations - B: target-specific benchmark/evaluation budget declared before tuning; no portable value is supplied by this record - S: stop when the declared budget is exhausted or a validated configuration meets the predeclared improvement threshold without violating C ## Preserved contract -Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, and an unchanged input signature alone is insufficient when an existing output can be corrupted, overwritten, or otherwise invalidated externally. +Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, and interrupted publication must not expose a signature paired with output identities from another generation. ## Optimization -Compute a deterministic signature over the effective inputs and compare it with successfully persisted prior state. Reuse is allowed only when that signature still matches **and** every required output satisfies a declared validity predicate. Depending on the target, that predicate may be a content digest/version manifest, a trusted immutable/protected artifact identity, or another reproducible integrity check strong enough to detect external mutation. Mere file presence is not sufficient unless the target explicitly guarantees that reused outputs are immutable and protected from modification. Execute when the input signature differs, any required output is missing, or any output-validity check fails. Persist the new signature and output-validity metadata only after successful execution. +Compute a deterministic signature over the effective inputs and compare it with successfully persisted prior state. Reuse is allowed only when that signature still matches **and** every required output satisfies a declared validity predicate. Depending on the target, that predicate may be a content digest/version manifest, a trusted immutable/protected artifact identity, or another reproducible integrity check strong enough to detect external mutation. Mere file presence is not sufficient unless the target explicitly guarantees that reused outputs are immutable and protected from modification. Execute when the input signature differs, any required output is missing, or any output-validity check fails. + +Publish incremental state as one crash-consistent **generation** that binds the input signature to the complete output identity/validity metadata. Do not persist the signature and output metadata as independently authoritative updates. Use an atomic rename/swap of a complete manifest, a transactional store, a content-addressed generation pointer, or another mechanism where readers observe either the previous complete generation or the new complete generation—never a mixture. Only publish the new generation after every output has been produced and validated successfully; an interrupted or failed publication leaves the previous committed generation authoritative and the partial generation non-reusable. Reuse filesystem/configuration metadata lazily only while its own validity predicate still holds. @@ -46,16 +48,16 @@ Wonderbuild demonstrates the mechanism and benchmark shapes, but its historical ## Validation -Test unchanged inputs with valid outputs, changed inputs, missing outputs, failed runs, corrupted persistent state, externally overwritten/corrupted outputs, and stale output-version metadata against a forced-fresh reference path. A mutated output must force reconstruction unless the target's immutable/protected-output contract proves such mutation impossible. +Test unchanged inputs with valid outputs, changed inputs, missing outputs, failed runs, corrupted persistent state, externally overwritten/corrupted outputs, and stale output-version metadata against a forced-fresh reference path. A mutated output must force reconstruction unless the target's immutable/protected-output contract proves such mutation impossible. Exercise interruption/crash injection at every publication boundary: before outputs complete, after outputs complete but before manifest publication, during temporary-manifest write, immediately before/after the atomic generation switch, and during cleanup. After each interruption, prove readers observe only a self-consistent old or new generation and can never pair signature A with output identities/metadata from generation B. ## Target-repo adaptation -Re-profile signature and output-validation cost, hash/version choice, metadata granularity and persistence format. Include environment/toolchain inputs when they affect output. Explicitly choose whether outputs are integrity-checked on reuse or are stored behind an enforceable immutable/protected boundary. +Re-profile signature and output-validation cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Explicitly choose whether outputs are integrity-checked on reuse or are stored behind an enforceable immutable/protected boundary, and define the crash-consistency guarantee for committing the signature plus output identities. ## Failure modes -Incomplete signatures create stale reuse; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. +Incomplete signatures create stale reuse; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. ## Rollback trigger -Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference, if external output mutation can bypass the declared validity predicate, or if signature/integrity maintenance costs more than the avoided work. +Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference, if external output mutation can bypass the declared validity predicate, if crash/interruption testing can expose mixed-generation state, or if signature/integrity/publication maintenance costs more than the avoided work. diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index f452f4c..dca3b97 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -20,21 +20,21 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra - F: candidates in X that satisfy all hard resource, platform, semantic, and correctness constraints before objective ranking - f: the target-measured objective or objective vector for each feasible candidate, including declared noise/statistical treatment - d: the target's predeclared minimize, maximize, lexicographic, or Pareto ordering -- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources, and every per-trial reservation must be an enforceable upper bound rather than an estimate -- B: an explicit target-specific hard maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit, enforceable per-trial cap mechanism, conservative reservation rule, and failure/cancellation charging policy are fixed before dispatch begins +- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources, every per-trial reservation must be an enforceable upper bound rather than an estimate, and dispatch/completion accounting must be linearizable under concurrency +- B: an explicit target-specific hard maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit, enforceable per-trial cap mechanism, conservative reservation rule, atomic accounting boundary, and failure/cancellation charging policy are fixed before dispatch begins - S: stop proposing/dispatching when no additional trial can be safely reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger ## Preserved contract -Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B, and no individual trial may consume beyond its reserved cap. +Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B, no individual trial may consume beyond its reserved cap, and concurrent dispatch/completion transitions must not transiently expose phantom free capacity. ## Optimization Use observations to adapt future evaluations: surrogate/acquisition search for expensive black-box objectives, conditional spaces where parameters only exist under certain choices, progressive domain contraction where justified, and explicit stopping/evaluation budgets. For asynchronous workers, reserve pending regions or otherwise diversify proposals so workers do not redundantly evaluate the same neighborhood. -Before dispatching an asynchronous trial, atomically reserve a conservative amount of the applicable budget and record the pending trial in the ledger. If `consumed + reserved + proposed_reservation > B`, do not dispatch. The reservation must be an **enforceable upper limit** for that trial, not merely an estimate: use a per-trial quota, wall-time deadline with forced cancellation/termination, provider spending cap, cgroup/job resource limit, evaluation-slot ownership, or another mechanism that prevents actual trial consumption from exceeding the reservation. If the target cannot enforce such a cap for a resource dimension, that dimension cannot be advertised as a hard maximum B; instead define a different enforceable budget or explicitly classify the quantity as observational rather than bounded. +Before dispatching an asynchronous trial, enter one atomic/serializable accounting boundary, reserve a conservative amount of the applicable budget, and record the pending trial in the ledger. If `consumed + reserved + proposed_reservation > B`, do not dispatch. The reservation must be an **enforceable upper limit** for that trial, not merely an estimate: use a per-trial quota, wall-time deadline with forced cancellation/termination, provider spending cap, cgroup/job resource limit, evaluation-slot ownership, or another mechanism that prevents actual trial consumption from exceeding the reservation. If the target cannot enforce such a cap for a resource dimension, that dimension cannot be advertised as a hard maximum B; instead define a different enforceable budget or explicitly classify the quantity as observational rather than bounded. -Completion converts the actually consumed amount into permanent budget consumption and releases only any **demonstrably unconsumed** remainder of the reservation; it must not double-charge the same usage. A failed or cancelled trial never erases resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute/time budgets, release only the measured or otherwise provable unused portion of the enforceable reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. Every reservation, cap enforcement action, consumption adjustment, release, failure, cancellation, and forced termination is recorded in the ledger. +Completion, failure, cancellation, and forced termination use the **same atomic accounting boundary** as dispatch reservation. For one terminal transition, atomically: (1) read the trial's reservation, (2) meter/record the amount actually consumed, (3) move that consumed amount into permanent `consumed`, (4) release only the demonstrably unconsumed remainder from `reserved`, and (5) mark the trial terminal. No dispatcher may observe released reservation capacity before the corresponding consumed charge is committed, and concurrent terminal updates must not lose increments. Completion must not double-charge the same usage. A failed or cancelled trial never erases resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute/time budgets, release only the measured or otherwise provable unused portion of the enforceable reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. Every reservation, cap enforcement action, consumption adjustment, release, failure, cancellation, forced termination, and terminal accounting transaction is recorded in the ledger. Parallelism has an information cost: very wide batches receive less feedback between suggestions and can degenerate toward non-adaptive/random search. @@ -50,14 +50,16 @@ Parallelism has an information cost: very wide batches receive less feedback bet Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. For asynchronous search, test the budget boundary with multiple workers contending for the last remaining reservation and prove no dispatch can make `consumed + reserved` exceed B. Deliberately run trials that attempt to exceed their per-trial money/compute/time reservation and prove the quota/deadline/termination mechanism prevents the overrun. Inject early failures, late failures, partial consumption, and cancellation after measurable work; verify that only demonstrably unconsumed reservation is released, evaluation-count slots are not resurrected after dispatch, and repeated failures cannot create extra budget capacity. +Race multiple trial completions/cancellations against one another and against workers attempting the final dispatch slot. Verify the accounting transaction is linearizable: no consumed increment is lost, no reservation is released before its corresponding consumption is charged, and a dispatcher never observes capacity that would make the post-transaction invariant `consumed + reserved <= B` false. + ## Target-repo adaptation -Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, **enforcement mechanism for that reservation**, atomic reservation mechanism, metering source, and failure/cancellation charging policy before enabling asynchronous dispatch. +Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, **enforcement mechanism for that reservation**, one atomic/serializable accounting mechanism shared by reservation and terminal conversion, metering source, and failure/cancellation charging policy before enabling asynchronous dispatch. ## Failure modes -Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; an unenforced reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism. +Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism. ## Rollback trigger -Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, any trial can consume beyond its enforceable reservation, or any accounting/concurrency test shows that dispatch, failure, cancellation, forced termination, or reservation release can cause actual consumption plus outstanding reservations to exceed B. +Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, any trial can consume beyond its enforceable reservation, or any accounting/concurrency test shows that dispatch, completion, failure, cancellation, forced termination, or reservation release can cause actual consumption plus outstanding reservations to exceed B or expose transient free capacity before consumption is committed. From a35bafc9cb9c3e3ff5a1b2cc381bd9f51bf4d819 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 15:52:57 +0930 Subject: [PATCH 011/229] Harden ownership, mutation, and fencing contracts --- ...01-concurrent-duplicate-work-coalescing.md | 22 +++++++------ ...NT-001-partitioned-coordination-domains.md | 22 ++++++++----- ...1-signature-bound-incremental-execution.md | 31 ++++++++++++------- ...SEARCH-001-budget-aware-adaptive-search.md | 16 +++++++--- ...T-SET-001-density-adaptive-compact-sets.md | 24 +++++++------- 5 files changed, 71 insertions(+), 44 deletions(-) diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index 6f0ae2a..3be3f5c 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,17 +14,19 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime policies, waiter limits, per-waiter cancellation policies, and retry/error-sharing policies -- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, and error semantics for every joined caller, and never admit new waiters to a closing or terminal generation -- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization and waiter-memory overhead +- X: target-supported request-key canonicalizations, authorization/equivalence scopes, result-ownership policies, shared-operation lifetime policies, waiter limits, per-waiter cancellation policies, and retry/error-sharing policies +- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result-ownership, and error semantics for every joined caller, and never admit new waiters to a closing or terminal generation +- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, cloning/copy-on-delivery cost where required, and waiter-memory overhead - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives a result or error valid for its original request semantics and authorization scope; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; closing/terminal generations are not joinable +- C: every joined caller receives a result or error valid for its original request semantics, authorization scope, and ownership contract; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; closing/terminal generations are not joinable; mutable caller-owned results are never observably aliased across callers unless the target contract explicitly declares shared mutation semantics - B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied by this record - S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C ## Preserved contract -Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation and timeout semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. +Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout, and result-ownership semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. + +If the upstream result is immutable or explicitly share-safe, all waiters may observe the same value. If callers normally receive mutable/caller-owned objects, the coalescer must preserve isolation by cloning/materializing an independent result per waiter, using copy-on-write with equivalent isolation, or another mechanism that makes one caller's mutation unobservable to other callers. Returning the same mutable object to independent callers is a contract change unless shared mutation is explicitly part of the target API. ## Optimization @@ -32,7 +34,7 @@ Create an in-flight registry entry for a canonical equivalence key. The key must Cancellation and timeout are per waiter: when one waiter leaves, remove only that waiter. If live waiters remain, keep the shared generation joinable. If the last waiter leaves and the policy calls for upstream cancellation, atomically mark the registry entry **closing/non-joinable** (or remove it from the joinable map) before sending the asynchronous cancellation request upstream. A new caller arriving after that transition must create a fresh generation rather than attach to work already being canceled. The closing generation may remain internally tracked until its terminal completion for cleanup/accounting, but it is not eligible for coalescing. -On success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the terminal waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Then snapshot the waiters still registered to that terminal generation, deliver the same shared terminal result/error to that snapshot, and retire/clean up the generation deterministically. Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. +On success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the terminal waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Then snapshot the waiters still registered to that terminal generation. For each waiter, independently authorize delivery and apply the declared result-ownership policy: deliver the same object only when it is immutable/share-safe, otherwise clone/materialize an isolated caller-owned result (or equivalent copy-on-write view) before delivery. Deliver the shared terminal error under the declared error semantics, then retire/clean up the generation deterministically. Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, ownership semantics, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. This differs from caching: the reusable result does not exist yet. @@ -50,14 +52,16 @@ Stress simultaneous identical and non-identical keys; inject upstream failures/t Add authorization-boundary fixtures: issue syntactically identical requests under different tenants, principals, roles, ACL/visibility scopes, or other authorization context. Prove they either map to different equivalence keys **or** that the shared upstream result is explicitly safe to reuse and each caller is independently authorized before delivery. Verify that a result produced under one authorization scope can never leak to another merely because the resource parameters match. +Add ownership-isolation fixtures for mutable results: coalesce multiple callers, mutate one caller's returned object after delivery, and prove every other caller's result remains unchanged and reference-equivalent to an independent call. Test nested/container mutation, retained references, and any copy-on-write transition. If the target declares immutable/share-safe results, attempt mutation or alias observation and verify the immutability/share-safety guarantee rather than assuming it. + ## Target-repo adaptation -Define key canonicalization, the authorization/visibility context that participates in equivalence, maximum waiter count, per-waiter cancellation/deadline handling, the exact condition for canceling upstream work, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. +Define key canonicalization, the authorization/visibility context that participates in equivalence, result ownership/mutability rules, cloning or copy-on-write strategy where needed, maximum waiter count, per-waiter cancellation/deadline handling, the exact condition for canceling upstream work, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent or authorization-distinct work; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; snapshotting waiters before terminal closure can strand a late joiner; omitting authorization scope can leak results across principals/tenants; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; unbounded waiter lists amplify memory; ambiguous retry/error policy can cause correlated or duplicated work. +Over-broad keys merge non-equivalent or authorization-distinct work; returning one mutable object to independent callers can create cross-caller corruption through aliasing; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; snapshotting waiters before terminal closure can strand a late joiner; omitting authorization scope can leak results across principals/tenants; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; unbounded waiter lists amplify memory; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes any caller's authorization/cancellation/result/error semantics, merges authorization-distinct requests without independent delivery authorization, permits one caller to cancel work required by another, allows a new caller to join a closing or terminal generation, strands a late joiner during terminal notification, leaks orphaned shared operations, increases tail latency materially, or creates unacceptable failure amplification. +Disable if coalescing changes any caller's authorization/cancellation/result/error/ownership semantics, exposes mutable result aliasing between callers, merges authorization-distinct requests without independent delivery authorization, permits one caller to cancel work required by another, allows a new caller to join a closing or terminal generation, strands a late joiner during terminal notification, leaks orphaned shared operations, increases tail latency materially, or creates unacceptable failure amplification. diff --git a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md index 37a1386..d5b7ea7 100644 --- a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md +++ b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md @@ -15,22 +15,26 @@ Independent workers serialize on one globally coordinated resource even though t ## Optimization problem contract -- X: target-supported shard/domain counts, namespace splits, worker-to-domain mappings, and merge/aggregation policies -- F: configurations that preserve the target's required uniqueness, ownership, visibility, failure-domain, and ordering guarantees +- X: target-supported shard/domain counts, namespace splits, worker-to-domain mappings, exclusive-ownership/handoff mechanisms, fencing-epoch policies, and merge/aggregation policies +- F: configurations that preserve the target's required uniqueness, ownership, visibility, failure-domain, and ordering guarantees under steady state, reassignment, restart, delayed-old-owner recovery, and split-brain conditions - f: measured coordination contention, tail latency, and coordination overhead under the declared workload - d: minimize under the target's predeclared objective ordering -- C: partitioning must not silently weaken any global invariant; any intentional shift from global to per-domain ordering is a separately declared contract change -- B: target-specific contention/scale benchmark budget declared before tuning; no portable shard count or bit split is supplied here +- C: partitioning must not silently weaken any global invariant; only the currently fenced/authorized owner of a domain may mutate domain-scoped state, stale owners must be rejected after reassignment, and any intentional shift from global to per-domain ordering is a separately declared contract change +- B: target-specific contention/scale/failover benchmark budget declared before tuning; no portable shard count or bit split is supplied here - S: stop when the budget is exhausted or a validated partitioning materially reduces the target bottleneck without violating C ## Preserved contract -Partitioning must not silently weaken uniqueness, ownership, visibility or ordering guarantees. If ordering becomes per-domain rather than global, that is a contract change and must be explicit. +Partitioning must not silently weaken uniqueness, ownership, visibility or ordering guarantees. Reassignment must preserve exclusive authority: once ownership moves, an old worker that remains alive, resumes after a pause, or recovers from a partition must be unable to allocate IDs, process queue ranges, commit writes, or otherwise act as the current owner. If ordering becomes per-domain rather than global, that is a contract change and must be explicit. ## Optimization Factor a global coordination space into independent domains. Encode domain identity into keys/IDs or route work so each domain can advance mostly independently. Prefer a small explicit merge/aggregation boundary to a permanently hot global lock/counter/poller. +For any domain whose ownership can move, pair routing/assignment with an **exclusive handoff and fencing mechanism**. A typical design uses a durable lease/ownership record containing a monotonically increasing epoch (generation/fencing token). A worker may act for a domain only while holding the current valid lease/epoch, and every mutating downstream action must carry or be checked against that epoch so an older owner is rejected even if it is still running. Reassignment must advance the epoch before the replacement begins authoritative work; the old epoch can never become valid again merely because the old process resumes. Where a lease can expire, expiration alone is insufficient unless the storage/queue/allocator accepting writes also enforces the fencing token. + +If two workers temporarily believe they own the same domain, the durable fencing boundary decides which epoch is authoritative. Recovery may retry idempotent work under the new epoch, but it must not accept stale-owner mutations that could duplicate IDs, process the same queue range twice, or overwrite newer state. + ## Before / after evidence - Environment: No controlled target-repository benchmark has been run for this OPT record. @@ -43,14 +47,16 @@ Factor a global coordination space into independent domains. Encode domain ident Check global invariants across all domains, collision/duplicate behavior, rebalance/restart behavior and target-scale contention profiles. +Add ownership-race fixtures. Start owner A for a domain, pause/delay it without terminating it, reassign the domain to owner B with a strictly newer fencing epoch, then resume A and prove every A mutation is rejected while B remains authoritative. Repeat with network partitions, lease expiry, process suspension, delayed messages, reordered retries, and split-brain recovery. For ID allocation, prove no duplicate local/global IDs can be emitted or committed across epochs. For queues, prove stale consumers cannot acknowledge/process the reassigned range authoritatively. For stores/counters, verify stale writes are rejected at the mutation boundary, not merely by the router. Exercise repeated reassignments A→B→C and recovery of both older owners. + ## Target-repo adaptation -Shard counts and bit splits are workload-specific. Measure skew, cache locality, failure domains and merge costs. +Shard counts and bit splits are workload-specific. Measure skew, cache locality, failure domains and merge costs. Define the durable ownership record, lease lifetime if any, monotonically increasing fencing epoch, which downstream operations must validate it, handoff ordering, retry/idempotency behavior, and recovery semantics before allowing dynamic reassignment. ## Failure modes -Hot shards merely move the bottleneck; domain proliferation raises memory/management overhead; rebalancing may violate identity stability; global ordering requirements may make the pattern inadmissible. +Hot shards merely move the bottleneck; domain proliferation raises memory/management overhead; stale owners without fencing can duplicate IDs/work or corrupt state during rebalance; lease expiry without downstream fencing can create split-brain authority; epoch reuse/wraparound or non-durable handoff can resurrect old ownership; rebalancing may violate identity stability; global ordering requirements may make the pattern inadmissible. ## Rollback trigger -Revert if partitioning does not reduce measured contention or if any cross-domain invariant fails. +Revert if partitioning does not reduce measured contention, if any cross-domain invariant fails, if a stale owner can mutate state after reassignment, if split-brain tests admit two authoritative epochs, or if fencing/handoff overhead outweighs the coordination benefit. diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index 7cc99e3..732f49f 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -14,23 +14,28 @@ Expensive work is rerun even though every input capable of affecting its result ## Optimization problem contract -- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity policies, and crash-consistent state-publication mechanisms -- F: configurations whose signature covers every output-affecting input, whose reuse validates required outputs, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted executions never publish reusable partial state -- f: measured repeated-work cost including stage runtime plus signature/metadata/output-validation/publication I/O overhead +- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity policies, input-snapshot/revalidation policies, and crash-consistent state-publication mechanisms +- F: configurations whose signature covers every output-affecting input, whose execution observes one valid effective-input snapshot or revalidates the complete effective-input identity before commit, whose reuse validates required outputs, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted/raced executions never publish reusable partial or mismatched state +- f: measured repeated-work cost including stage runtime plus signature/revalidation/metadata/output-validation/publication I/O overhead - d: minimize -- C: every reused output is semantically equivalent to a fresh execution for the same effective inputs, with the same failure and output-validity semantics; reuse metadata cannot mix fields from different generations +- C: every reused output is semantically equivalent to a fresh execution for the exact effective-input identity recorded in its committed generation, with the same failure and output-validity semantics; reuse metadata cannot mix fields from different generations; concurrent input mutation cannot cause a generation to publish outputs under a stale signature - B: target-specific benchmark/evaluation budget declared before tuning; no portable value is supplied by this record - S: stop when the declared budget is exhausted or a validated configuration meets the predeclared improvement threshold without violating C ## Preserved contract -Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, and interrupted publication must not expose a signature paired with output identities from another generation. +Reused output must be semantically equivalent to a fresh execution for the exact effective inputs identified by the committed generation. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and input mutation during execution must not let output produced from state B be committed under signature A. ## Optimization -Compute a deterministic signature over the effective inputs and compare it with successfully persisted prior state. Reuse is allowed only when that signature still matches **and** every required output satisfies a declared validity predicate. Depending on the target, that predicate may be a content digest/version manifest, a trusted immutable/protected artifact identity, or another reproducible integrity check strong enough to detect external mutation. Mere file presence is not sufficient unless the target explicitly guarantees that reused outputs are immutable and protected from modification. Execute when the input signature differs, any required output is missing, or any output-validity check fails. +Establish the effective-input identity before execution using one of two admissible strategies: -Publish incremental state as one crash-consistent **generation** that binds the input signature to the complete output identity/validity metadata. Do not persist the signature and output metadata as independently authoritative updates. Use an atomic rename/swap of a complete manifest, a transactional store, a content-addressed generation pointer, or another mechanism where readers observe either the previous complete generation or the new complete generation—never a mixture. Only publish the new generation after every output has been produced and validated successfully; an interrupted or failed publication leaves the previous committed generation authoritative and the partial generation non-reusable. +1. **Immutable snapshot:** execute strictly against a snapshot/version whose identity is the signature recorded for the generation; or +2. **Precommit revalidation:** compute the complete effective-input signature before execution, run the stage, then recompute/reauthenticate the complete effective-input signature immediately before publication. If it differs, discard or quarantine the produced outputs and retry from the new identity rather than publishing them under the stale signature. + +Reuse is allowed only when the committed input signature still matches **and** every required output satisfies a declared validity predicate. Depending on the target, that predicate may be a content digest/version manifest, a trusted immutable/protected artifact identity, or another reproducible integrity check strong enough to detect external mutation. Mere file presence is not sufficient unless the target explicitly guarantees that reused outputs are immutable and protected from modification. Execute when the input signature differs, any required output is missing, or any output-validity check fails. + +Publish incremental state as one crash-consistent **generation** that binds the verified input identity to the complete output identity/validity metadata. Do not persist the signature and output metadata as independently authoritative updates. Use an atomic rename/swap of a complete manifest, a transactional store, a content-addressed generation pointer, or another mechanism where readers observe either the previous complete generation or the new complete generation—never a mixture. Only publish after every output has been produced and validated successfully **and** the immutable-snapshot or precommit-revalidation rule proves the recorded input identity still matches the inputs used to produce those outputs. An interrupted, failed, or input-raced publication leaves the previous committed generation authoritative and the partial generation non-reusable. Reuse filesystem/configuration metadata lazily only while its own validity predicate still holds. @@ -48,16 +53,20 @@ Wonderbuild demonstrates the mechanism and benchmark shapes, but its historical ## Validation -Test unchanged inputs with valid outputs, changed inputs, missing outputs, failed runs, corrupted persistent state, externally overwritten/corrupted outputs, and stale output-version metadata against a forced-fresh reference path. A mutated output must force reconstruction unless the target's immutable/protected-output contract proves such mutation impossible. Exercise interruption/crash injection at every publication boundary: before outputs complete, after outputs complete but before manifest publication, during temporary-manifest write, immediately before/after the atomic generation switch, and during cleanup. After each interruption, prove readers observe only a self-consistent old or new generation and can never pair signature A with output identities/metadata from generation B. +Test unchanged inputs with valid outputs, changed inputs, missing outputs, failed runs, corrupted persistent state, externally overwritten/corrupted outputs, stale output-version metadata, and concurrent input mutation against a forced-fresh reference path. A mutated output must force reconstruction unless the target's immutable/protected-output contract proves such mutation impossible. + +For mutable inputs, deliberately change one or more effective inputs while the stage is executing, including changes immediately before commit and changes that revert to the original value. Prove that either execution was bound to an immutable snapshot or the precommit signature comparison detects the change and prevents publication. Verify no generation can bind signature A to output produced from B or from a mixed A/B observation. + +Exercise interruption/crash injection at every publication boundary: before outputs complete, after outputs complete but before manifest publication, during temporary-manifest write, immediately before/after the atomic generation switch, and during cleanup. After each interruption, prove readers observe only a self-consistent old or new generation and can never pair signature A with output identities/metadata from generation B. ## Target-repo adaptation -Re-profile signature and output-validation cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Explicitly choose whether outputs are integrity-checked on reuse or are stored behind an enforceable immutable/protected boundary, and define the crash-consistency guarantee for committing the signature plus output identities. +Re-profile signature and output-validation cost, input-snapshot or revalidation cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Explicitly choose whether outputs are integrity-checked on reuse or are stored behind an enforceable immutable/protected boundary, define how execution is tied to an immutable input snapshot or how complete input identity is revalidated before commit, and define the crash-consistency guarantee for committing the signature plus output identities. ## Failure modes -Incomplete signatures create stale reuse; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. +Incomplete signatures create stale reuse; mutable inputs can change during execution and produce outputs that do not correspond to the pre-run signature; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. ## Rollback trigger -Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference, if external output mutation can bypass the declared validity predicate, if crash/interruption testing can expose mixed-generation state, or if signature/integrity/publication maintenance costs more than the avoided work. +Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference, if concurrent input mutation can publish outputs under a stale identity, if external output mutation can bypass the declared validity predicate, if crash/interruption testing can expose mixed-generation state, or if signature/revalidation/integrity/publication maintenance costs more than the avoided work. diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index dca3b97..d0a4621 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -20,14 +20,16 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra - F: candidates in X that satisfy all hard resource, platform, semantic, and correctness constraints before objective ranking - f: the target-measured objective or objective vector for each feasible candidate, including declared noise/statistical treatment - d: the target's predeclared minimize, maximize, lexicographic, or Pareto ordering -- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources, every per-trial reservation must be an enforceable upper bound rather than an estimate, and dispatch/completion accounting must be linearizable under concurrency +- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources, every per-trial reservation must be an enforceable upper bound rather than an estimate, dispatch/completion accounting must be linearizable under concurrency, and targets that require deterministic search outcomes must use deterministic observation assimilation rather than completion-order updates - B: an explicit target-specific hard maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit, enforceable per-trial cap mechanism, conservative reservation rule, atomic accounting boundary, and failure/cancellation charging policy are fixed before dispatch begins -- S: stop proposing/dispatching when no additional trial can be safely reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger +- S: stop proposing/dispatching when no additional trial can be safely reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger and apply stopping decisions only to the declared deterministic assimilation frontier when determinism is required ## Preserved contract Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B, no individual trial may consume beyond its reserved cap, and concurrent dispatch/completion transitions must not transiently expose phantom free capacity. +If the target requires reproducible search traces or identical selected configurations across runs, asynchronous completion order is not allowed to change the optimizer's logical observation sequence. In that mode, results may finish in any wall-clock order, but they are assimilated into the optimizer only in a deterministic order such as monotonically increasing trial ID or explicit deterministic batches. If completion-order assimilation is intentionally used, the resulting nondeterminism must be declared as a contract change rather than hidden behind a fixed seed. + ## Optimization Use observations to adapt future evaluations: surrogate/acquisition search for expensive black-box objectives, conditional spaces where parameters only exist under certain choices, progressive domain contraction where justified, and explicit stopping/evaluation budgets. For asynchronous workers, reserve pending regions or otherwise diversify proposals so workers do not redundantly evaluate the same neighborhood. @@ -36,6 +38,8 @@ Before dispatching an asynchronous trial, enter one atomic/serializable accounti Completion, failure, cancellation, and forced termination use the **same atomic accounting boundary** as dispatch reservation. For one terminal transition, atomically: (1) read the trial's reservation, (2) meter/record the amount actually consumed, (3) move that consumed amount into permanent `consumed`, (4) release only the demonstrably unconsumed remainder from `reserved`, and (5) mark the trial terminal. No dispatcher may observe released reservation capacity before the corresponding consumed charge is committed, and concurrent terminal updates must not lose increments. Completion must not double-charge the same usage. A failed or cancelled trial never erases resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute/time budgets, release only the measured or otherwise provable unused portion of the enforceable reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. Every reservation, cap enforcement action, consumption adjustment, release, failure, cancellation, forced termination, and terminal accounting transaction is recorded in the ledger. +When deterministic search behavior is required, assign each proposal a stable trial ID at reservation/dispatch time and separate **physical completion** from **logical assimilation**. Buffer terminal results until the next deterministic trial-ID/batch frontier is complete, then update the surrogate/acquisition/stopping state in that fixed order. Failed/cancelled trials contribute their predeclared deterministic terminal observation/status at the same logical position. Later proposals may depend only on observations already admitted through that deterministic frontier. Alternative deterministic batching schemes are admissible if their ordering rule is fixed before execution and replayable from the ledger. + Parallelism has an information cost: very wide batches receive less feedback between suggestions and can degenerate toward non-adaptive/random search. ## Before / after evidence @@ -52,14 +56,16 @@ Keep a deterministic search seed where practical, preserve the full trial ledger Race multiple trial completions/cancellations against one another and against workers attempting the final dispatch slot. Verify the accounting transaction is linearizable: no consumed increment is lost, no reservation is released before its corresponding consumption is charged, and a dispatcher never observes capacity that would make the post-transaction invariant `consumed + reserved <= B` false. +For targets that require deterministic optimization, run the same fixed-seed search repeatedly while deliberately perturbing worker latency/completion order. Verify that the persisted logical observation sequence, surrogate updates, proposals, stopping decision, and selected result are identical. Compare sequential execution with deterministic asynchronous/batched execution where the chosen scheme claims equivalence. Replay the ledger from scratch and prove it reconstructs the same optimizer state and final selection. If the target permits nondeterministic completion-order assimilation, record that explicitly and do not claim deterministic replay equivalence. + ## Target-repo adaptation -Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, **enforcement mechanism for that reservation**, one atomic/serializable accounting mechanism shared by reservation and terminal conversion, metering source, and failure/cancellation charging policy before enabling asynchronous dispatch. +Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, **enforcement mechanism for that reservation**, one atomic/serializable accounting mechanism shared by reservation and terminal conversion, metering source, failure/cancellation charging policy, and deterministic observation-assimilation rule when required before enabling asynchronous dispatch. ## Failure modes -Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism. +Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism; completion-order assimilation can make fixed-seed asynchronous searches produce different traces, stopping points, and selected configurations. ## Rollback trigger -Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, any trial can consume beyond its enforceable reservation, or any accounting/concurrency test shows that dispatch, completion, failure, cancellation, forced termination, or reservation release can cause actual consumption plus outstanding reservations to exceed B or expose transient free capacity before consumption is committed. +Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, any trial can consume beyond its enforceable reservation, any accounting/concurrency test shows that dispatch, completion, failure, cancellation, forced termination, or reservation release can cause actual consumption plus outstanding reservations to exceed B or expose transient free capacity before consumption is committed, or a target that requires determinism cannot reproduce the same logical observation sequence and final selection under perturbed asynchronous completion order. diff --git a/optimizations/OPT-SET-001-density-adaptive-compact-sets.md b/optimizations/OPT-SET-001-density-adaptive-compact-sets.md index a5d1788..d854b63 100644 --- a/optimizations/OPT-SET-001-density-adaptive-compact-sets.md +++ b/optimizations/OPT-SET-001-density-adaptive-compact-sets.md @@ -15,21 +15,21 @@ A single representation performs poorly across regions with very different densi ## Optimization problem contract -- X: target-supported partition widths, sparse/dense container choices, switching thresholds, and serialization layouts -- F: representations that preserve exact membership and set-operation semantics and satisfy target memory/serialization compatibility constraints -- f: measured memory footprint plus target-relevant set-operation and serialization latency +- X: target-supported partition widths, sparse/dense container choices, switching thresholds/hysteresis policies, mutation/conversion policies, and serialization layouts +- F: representations that preserve exact membership and set-operation semantics across all supported mutations and representation transitions and satisfy target memory/serialization compatibility constraints +- f: measured memory footprint plus target-relevant set-operation, mutation/conversion, and serialization latency - d: minimize under the target's predeclared scalar, lexicographic, or Pareto ordering -- C: membership, union, intersection, difference, and persistence round trips match the canonical reference set exactly -- B: target-specific benchmark budget over declared sparse, dense, mixed, and transition-boundary datasets; no portable trial count is supplied here +- C: membership, insertion, deletion, union, intersection, difference, representation transitions, and persistence round trips match the canonical reference set exactly after every mutation +- B: target-specific benchmark budget over declared sparse, dense, mixed, transition-boundary, and mutation-sequence datasets; no portable trial count is supplied here - S: stop when the declared budget is exhausted or a validated representation meets the target objective without violating C ## Preserved contract -Membership and set operations must match the reference set exactly. +Membership and set operations must match the reference set exactly before, during, and after conversion between sparse and dense representations. A threshold crossing is an internal representation change only; it must not drop, duplicate, reorder semantically significant iteration, or corrupt members. ## Optimization -Partition the identifier space and choose a representation per partition according to local density. Keep sparse regions compact while using bitmap-like containers where dense boolean algebra is advantageous. Prefer representations that can be serialized without expanding to a larger intermediate form. +Partition the identifier space and choose a representation per partition according to local density. Keep sparse regions compact while using bitmap-like containers where dense boolean algebra is advantageous. For mutable sets, conversions triggered by insertions/deletions must be deterministic and exact. Consider hysteresis or other anti-thrashing policy when repeated near-threshold mutation would otherwise cause conversion churn, but do not change set semantics to avoid conversions. Prefer representations that can be serialized without expanding to a larger intermediate form. ## Evidence boundary @@ -45,16 +45,18 @@ Jazco reports strong production-scale graph results, but OPT treats the numbers ## Validation -Differential-test membership, union, intersection, difference and persistence against a simple canonical set implementation over sparse, dense and transition-boundary fixtures. +Differential-test membership, insertion, deletion, union, intersection, difference and persistence against a simple canonical set implementation over sparse, dense and transition-boundary fixtures. + +Add mutation-sequence tests that repeatedly cross every sparse↔dense switching threshold in **both directions**. Construct sequences that insert just past the promotion boundary, delete back below the demotion boundary, and repeat for many cycles; include randomized/adversarial churn near the boundary. After every mutation and every representation transition, verify exact membership/cardinality against the reference set, then re-run union/intersection/difference checks and a serialization round trip. Test duplicate insertions, deletion of absent elements, empty/full-ish containers, threshold off-by-one cases, and restart/deserialization followed by further transitions. If hysteresis is used, verify its exact promotion/demotion rules while preserving the same set contents. ## Target-repo adaptation -Benchmark partition sizes and switching thresholds on the real identifier distribution and CPU/cache hierarchy. +Benchmark partition sizes, switching thresholds, hysteresis/conversion policy, and serialization format on the real identifier distribution, mutation pattern, and CPU/cache hierarchy. Immutable/read-mostly and mutation-heavy workloads may justify different policies. ## Failure modes -Conversion churn near thresholds, pathological distributions, serialization incompatibility and hidden temporary allocations can erase the benefit. +Conversion bugs can drop or duplicate members; repeated near-threshold mutation can cause conversion thrash; asymmetric promotion/demotion logic can strand a container in the wrong representation; pathological distributions, serialization incompatibility and hidden temporary allocations can erase the benefit. ## Rollback trigger -Revert when target data does not show a memory/latency win or exact set differential tests fail. +Revert when target data does not show a memory/latency win, any static or mutation-sequence differential test fails, any transition loses/duplicates members, persistence round trips diverge, or conversion churn materially worsens the target workload. From 19cc6d1e8ec1ce926bcbde92874504334246546d Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 16:28:59 +0930 Subject: [PATCH 012/229] Enforce classification and edge-case contracts --- ...PROX-001-contract-bounded-approximation.md | 8 +++ ...DGET-001-performance-regression-budgets.md | 8 +++ ...01-concurrent-duplicate-work-coalescing.md | 46 +++++++++---- ...NT-001-partitioned-coordination-domains.md | 30 +++++---- ...T-CRIT-001-critical-path-prioritization.md | 28 +++++--- ...T-FAN-001-shared-materialization-fanout.md | 28 +++++--- ...1-signature-bound-incremental-execution.md | 39 ++++++----- ...E-001-bound-driven-search-space-pruning.md | 35 +++++++--- ...-REDUCE-001-early-working-set-reduction.md | 8 +++ ...SEARCH-001-budget-aware-adaptive-search.md | 26 +++++--- ...T-SET-001-density-adaptive-compact-sets.md | 38 +++++++---- scripts/check_catalog.py | 64 ++++++++++++++----- 12 files changed, 252 insertions(+), 106 deletions(-) mode change 100755 => 100644 scripts/check_catalog.py diff --git a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md index b1a7a2a..7ff2206 100644 --- a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md +++ b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md @@ -22,6 +22,14 @@ Exact processing has unbounded or unacceptable cost even though the product/scie - C: approximation is permitted only by an explicit contract; exact callers are not silently weakened; the error norm, aggregation rule, sequence/composition horizon, and reset boundaries are declared before evaluation; an exact reference path or exact fixture remains available where practical - B: target-specific benchmark/quality-evaluation budget over predeclared ordinary, boundary, adversarial, repeated-application, and long-horizon fixtures - S: stop when the evaluation budget is exhausted or a validated policy meets the target resource objective while remaining inside the declared quality envelope over the entire declared horizon +- Variables: continuous / integer / categorical / conditional / mixed, depending on approximation policy +- Search scope: local or global, explicitly declared for the target +- Objective behavior: deterministic, noisy, or stochastic depending on the quality/resource metric +- Information: derivative-free / black-box by default +- Evaluation cost: moderate to expensive when exact references or long-horizon trajectories are required +- Constraints: explicit error envelope, semantic/API, resource, horizon/reset, and exact-fallback constraints +- Parallelism: sequential, synchronous batch, or asynchronous according to target evaluation; stateful validation must preserve trajectory semantics +- Exactness: approximation explicitly permitted only inside the declared measurable envelope ## Preserved contract diff --git a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md index 31be67d..417c94f 100644 --- a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md +++ b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md @@ -22,6 +22,14 @@ Small performance regressions accumulate because performance is measured occasio - C: the performance gate must not incentivize weakening tests, assertions, evidence, semantic coverage, or representative workload inputs; the gate itself must continue to detect known regressions and accept known-good controls within the declared false-positive/false-negative envelope - B: target-specific calibration budget specifying repetitions, environment samples, and allowable CI/runtime measurement cost - S: stop calibration when the declared sample budget is exhausted or the baseline/noise estimate is stable enough to justify the predeclared warning and hard thresholds +- Variables: continuous / integer / categorical / mixed metric, statistic, fixture, and threshold choices +- Search scope: local gate/calibration tuning +- Objective behavior: noisy / stochastic measurement distributions +- Information: derivative-free statistical observations +- Evaluation cost: moderate to expensive depending on repetitions and fixture scale +- Constraints: functional correctness, representative workload, statistical tolerance, runner/environment characterization, false-positive/false-negative, and CI-overhead constraints +- Parallelism: sequential or synchronous-batch calibration; parallel sampling only when runner interference is characterized +- Exactness: no semantic approximation; statistical tolerance/noise handling is explicit ## Preserved contract diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index 3be3f5c..f18d487 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,27 +14,41 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- X: target-supported request-key canonicalizations, authorization/equivalence scopes, result-ownership policies, shared-operation lifetime policies, waiter limits, per-waiter cancellation policies, and retry/error-sharing policies -- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result-ownership, and error semantics for every joined caller, and never admit new waiters to a closing or terminal generation -- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, cloning/copy-on-delivery cost where required, and waiter-memory overhead +- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime policies, waiter limits, overflow/backpressure policies, per-waiter cancellation policies, retry/error-sharing policies, and result-ownership policies +- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, and error semantics for every joined caller, bound waiter memory, and never admit new waiters to a closing or terminal generation +- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, waiter-memory, overflow/backpressure, and result-copy overhead - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives a result or error valid for its original request semantics, authorization scope, and ownership contract; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; closing/terminal generations are not joinable; mutable caller-owned results are never observably aliased across callers unless the target contract explicitly declares shared mutation semantics -- B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied by this record +- C: every joined caller receives a result or error valid for its original request semantics, authorization scope, and ownership contract; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; closing/terminal generations are not joinable; waiter overflow has an explicit bounded behavior +- B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied here - S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C +- Variables: categorical / integer / mixed +- Search scope: local policy tuning within one coalescing boundary +- Objective behavior: noisy under concurrent load; semantic equivalence remains deterministic +- Information: derivative-free / black-box performance measurements +- Evaluation cost: moderate to expensive concurrent-load testing +- Constraints: semantic equivalence, authorization, ownership, waiter-memory, cancellation, timeout, and resource constraints +- Parallelism: asynchronous / concurrent +- Exactness: exact request/result semantics; no approximation is introduced ## Preserved contract -Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout, and result-ownership semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. - -If the upstream result is immutable or explicitly share-safe, all waiters may observe the same value. If callers normally receive mutable/caller-owned objects, the coalescer must preserve isolation by cloning/materializing an independent result per waiter, using copy-on-write with equivalent isolation, or another mechanism that makes one caller's mutation unobservable to other callers. Returning the same mutable object to independent callers is a contract change unless shared mutation is explicitly part of the target API. +Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout, result-ownership, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. A configured waiter bound must never be exceeded silently. ## Optimization -Create an in-flight registry entry for a canonical equivalence key. The key must include every request attribute required to establish safe sharing, including authorization-relevant tenant/principal/visibility scope unless the target instead proves that the upstream result is globally shareable and independently authorizes each delivered result. The first caller starts the shared upstream operation, but **does not own its lifetime**. Every equivalent caller registers as an independent waiter on the currently joinable generation. +Create an in-flight registry entry for a canonical equivalence key. The key must include every request attribute required to establish safe sharing, including authorization-relevant tenant/principal/visibility scope unless the target instead proves that the upstream result is globally shareable and independently authorizes each delivered result. + +Atomically create the joinable generation **with the initiating caller already registered as its first waiter before invoking, scheduling, or otherwise allowing the upstream operation to run**. This prevents an immediately/synchronously completing operation from reaching terminal state with an empty waiter set. Only after the first waiter is durably part of the generation may the upstream work begin. + +Equivalent later callers may register as independent waiters only while the generation is joinable and the configured waiter capacity remains. Waiter admission is atomic with capacity accounting. When the final waiter slot is already occupied, apply one explicit target policy rather than silently exceeding the bound: reject/return a documented overload or retryable-backpressure result, block/queue the caller behind a separately bounded admission mechanism, or use another bounded policy with explicit timeout/cancellation semantics. Starting an unconstrained parallel generation for the same equivalence key is not the default overflow behavior because it recreates the duplicate upstream load this pattern is intended to prevent. If a target deliberately permits overflow generations, that concurrency bound and duplicate-work tradeoff must be part of C/B and validated separately. Cancellation and timeout are per waiter: when one waiter leaves, remove only that waiter. If live waiters remain, keep the shared generation joinable. If the last waiter leaves and the policy calls for upstream cancellation, atomically mark the registry entry **closing/non-joinable** (or remove it from the joinable map) before sending the asynchronous cancellation request upstream. A new caller arriving after that transition must create a fresh generation rather than attach to work already being canceled. The closing generation may remain internally tracked until its terminal completion for cleanup/accounting, but it is not eligible for coalescing. -On success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the terminal waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Then snapshot the waiters still registered to that terminal generation. For each waiter, independently authorize delivery and apply the declared result-ownership policy: deliver the same object only when it is immutable/share-safe, otherwise clone/materialize an isolated caller-owned result (or equivalent copy-on-write view) before delivery. Deliver the shared terminal error under the declared error semantics, then retire/clean up the generation deterministically. Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, ownership semantics, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. +On success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the terminal waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Then snapshot the waiters still registered to that terminal generation. + +Define result ownership explicitly. If the terminal value is immutable/share-safe under the target API, the same immutable value may be delivered to all authorized waiters. If callers normally receive mutable or caller-owned results, create an independent defensive clone/copy/copy-on-write handle for each waiter before delivery so one caller cannot observably mutate another caller's result. Deliver the shared terminal error (or per-caller wrapped equivalent where the API requires ownership/context) to the terminal waiter snapshot, then retire/clean up the generation deterministically. + +Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. This differs from caching: the reusable result does not exist yet. @@ -50,18 +64,22 @@ This differs from caching: the reusable result does not exist yet. Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; race a new caller against the last-waiter cancellation transition and prove it never joins the closing generation; race a new caller against success/failure completion and prove the terminal generation is made non-joinable before waiter snapshot/notification; test waiter-specific deadlines; verify shared failure delivery and retry accounting; prove only one upstream evaluation occurs per joinable generation while all surviving callers terminate correctly. +Add an **immediate synchronous-completion** fixture where the upstream operation can finish inline at launch. Prove the initiating caller was already registered before launch and always receives the terminal result/error. + Add authorization-boundary fixtures: issue syntactically identical requests under different tenants, principals, roles, ACL/visibility scopes, or other authorization context. Prove they either map to different equivalence keys **or** that the shared upstream result is explicitly safe to reuse and each caller is independently authorized before delivery. Verify that a result produced under one authorization scope can never leak to another merely because the resource parameters match. -Add ownership-isolation fixtures for mutable results: coalesce multiple callers, mutate one caller's returned object after delivery, and prove every other caller's result remains unchanged and reference-equivalent to an independent call. Test nested/container mutation, retained references, and any copy-on-write transition. If the target declares immutable/share-safe results, attempt mutation or alias observation and verify the immutability/share-safety guarantee rather than assuming it. +Add ownership-isolation fixtures for mutable results: deliver one coalesced computation to multiple callers, mutate one caller's returned object, and prove every other caller's result remains unchanged. If the API declares the shared value immutable, attempt prohibited mutation through all exposed aliases and verify the immutability/share-safety contract. + +Add waiter-overflow races: fill the waiter list to one slot below the maximum, launch multiple equivalent callers concurrently for the final slot, and prove admission is linearizable, capacity is never exceeded, non-admitted callers receive exactly the documented backpressure/overflow behavior, and cancellation/timeouts of queued or rejected callers remain correct. ## Target-repo adaptation -Define key canonicalization, the authorization/visibility context that participates in equivalence, result ownership/mutability rules, cloning or copy-on-write strategy where needed, maximum waiter count, per-waiter cancellation/deadline handling, the exact condition for canceling upstream work, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. +Define key canonicalization, the authorization/visibility context that participates in equivalence, maximum waiter count, bounded overflow/backpressure semantics, result ownership/share-safety policy, per-waiter cancellation/deadline handling, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent or authorization-distinct work; returning one mutable object to independent callers can create cross-caller corruption through aliasing; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; snapshotting waiters before terminal closure can strand a late joiner; omitting authorization scope can leak results across principals/tenants; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; unbounded waiter lists amplify memory; ambiguous retry/error policy can cause correlated or duplicated work. +Over-broad keys merge non-equivalent or authorization-distinct work; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; snapshotting waiters before terminal closure can strand a late joiner; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes any caller's authorization/cancellation/result/error/ownership semantics, exposes mutable result aliasing between callers, merges authorization-distinct requests without independent delivery authorization, permits one caller to cancel work required by another, allows a new caller to join a closing or terminal generation, strands a late joiner during terminal notification, leaks orphaned shared operations, increases tail latency materially, or creates unacceptable failure amplification. +Disable if coalescing changes any caller's authorization/cancellation/result/ownership/error semantics, merges authorization-distinct requests without independent delivery authorization, permits one caller to cancel work required by another, strands the initiating caller on immediate completion, allows a new caller to join a closing or terminal generation, exceeds the configured waiter bound, violates documented overflow/backpressure behavior, permits mutable-result aliasing across callers, leaks orphaned shared operations, increases tail latency materially, or creates unacceptable failure amplification. diff --git a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md index d5b7ea7..6ae6b2e 100644 --- a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md +++ b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md @@ -15,25 +15,33 @@ Independent workers serialize on one globally coordinated resource even though t ## Optimization problem contract -- X: target-supported shard/domain counts, namespace splits, worker-to-domain mappings, exclusive-ownership/handoff mechanisms, fencing-epoch policies, and merge/aggregation policies -- F: configurations that preserve the target's required uniqueness, ownership, visibility, failure-domain, and ordering guarantees under steady state, reassignment, restart, delayed-old-owner recovery, and split-brain conditions +- X: target-supported shard/domain counts, namespace splits, worker-to-domain mappings, merge/aggregation policies, ownership-lease policies, and fencing-epoch schemes +- F: configurations that preserve the target's required uniqueness, exclusive ownership, visibility, failure-domain, and ordering guarantees through assignment, rebalance, restart, and split-brain recovery - f: measured coordination contention, tail latency, and coordination overhead under the declared workload - d: minimize under the target's predeclared objective ordering -- C: partitioning must not silently weaken any global invariant; only the currently fenced/authorized owner of a domain may mutate domain-scoped state, stale owners must be rejected after reassignment, and any intentional shift from global to per-domain ordering is a separately declared contract change +- C: partitioning must not silently weaken any global invariant; any intentional shift from global to per-domain ordering is a separately declared contract change; mutable domain ownership transitions require one active fenced owner for each epoch - B: target-specific contention/scale/failover benchmark budget declared before tuning; no portable shard count or bit split is supplied here - S: stop when the budget is exhausted or a validated partitioning materially reduces the target bottleneck without violating C +- Variables: integer / categorical / mixed +- Search scope: local architecture/partition-policy tuning +- Objective behavior: noisy under concurrent load; ownership/uniqueness invariants are deterministic +- Information: derivative-free / black-box performance measurements +- Evaluation cost: moderate to expensive at target scale and during failover testing +- Constraints: uniqueness, ownership, ordering, visibility, failure-domain, lease/fencing, and resource constraints +- Parallelism: concurrent / asynchronous by construction +- Exactness: exact ownership/uniqueness semantics; no approximation is introduced ## Preserved contract -Partitioning must not silently weaken uniqueness, ownership, visibility or ordering guarantees. Reassignment must preserve exclusive authority: once ownership moves, an old worker that remains alive, resumes after a pause, or recovers from a partition must be unable to allocate IDs, process queue ranges, commit writes, or otherwise act as the current owner. If ordering becomes per-domain rather than global, that is a contract change and must be explicit. +Partitioning must not silently weaken uniqueness, ownership, visibility or ordering guarantees. If ordering becomes per-domain rather than global, that is a contract change and must be explicit. When a domain can be reassigned, only the current fenced owner may mutate that domain; a delayed, partitioned, resumed, or split-brain previous owner must be rejected even if it still believes its old lease is valid. ## Optimization Factor a global coordination space into independent domains. Encode domain identity into keys/IDs or route work so each domain can advance mostly independently. Prefer a small explicit merge/aggregation boundary to a permanently hot global lock/counter/poller. -For any domain whose ownership can move, pair routing/assignment with an **exclusive handoff and fencing mechanism**. A typical design uses a durable lease/ownership record containing a monotonically increasing epoch (generation/fencing token). A worker may act for a domain only while holding the current valid lease/epoch, and every mutating downstream action must carry or be checked against that epoch so an older owner is rejected even if it is still running. Reassignment must advance the epoch before the replacement begins authoritative work; the old epoch can never become valid again merely because the old process resumes. Where a lease can expire, expiration alone is insufficient unless the storage/queue/allocator accepting writes also enforces the fencing token. +For dynamic assignment/rebalance, use an **exclusive handoff with fencing**. A durable coordinator grants ownership together with a monotonically increasing epoch/token. Every state-changing operation that depends on domain ownership carries that epoch, and the authoritative storage/queue/allocation boundary rejects operations from epochs older than the current one. A lease alone is insufficient if an old process can resume after expiry; the fencing token must make stale writes/actions impossible at the mutation boundary. Do not activate the replacement owner until the new epoch is durably authoritative. -If two workers temporarily believe they own the same domain, the durable fencing boundary decides which epoch is authoritative. Recovery may retry idempotent work under the new epoch, but it must not accept stale-owner mutations that could duplicate IDs, process the same queue range twice, or overwrite newer state. +Where local IDs/counters are used, combine the stable domain identity with the fenced ownership epoch or another target-specific mechanism strong enough to prevent duplicate allocation across reassignment. If IDs must remain stable across ownership epochs, separate the stable domain namespace from the fencing metadata while still rejecting stale mutations. ## Before / after evidence @@ -45,18 +53,18 @@ If two workers temporarily believe they own the same domain, the durable fencing ## Validation -Check global invariants across all domains, collision/duplicate behavior, rebalance/restart behavior and target-scale contention profiles. +Check global invariants across all domains, collision/duplicate behavior, restart behavior and target-scale contention profiles. -Add ownership-race fixtures. Start owner A for a domain, pause/delay it without terminating it, reassign the domain to owner B with a strictly newer fencing epoch, then resume A and prove every A mutation is rejected while B remains authoritative. Repeat with network partitions, lease expiry, process suspension, delayed messages, reordered retries, and split-brain recovery. For ID allocation, prove no duplicate local/global IDs can be emitted or committed across epochs. For queues, prove stale consumers cannot acknowledge/process the reassigned range authoritatively. For stores/counters, verify stale writes are rejected at the mutation boundary, not merely by the router. Exercise repeated reassignments A→B→C and recovery of both older owners. +Exercise **rebalance/recovery races**: pause an owner, expire/revoke it, assign a higher fencing epoch to a replacement, then resume the old owner and prove every stale mutation/allocation/queue claim is rejected. Inject network partition and split-brain conditions where both old and new processes run simultaneously. Verify only the highest authoritative epoch can mutate state, no duplicate IDs/work claims are produced, handoff is crash-recoverable, and ownership remains unique through coordinator/storage restarts. Include delayed messages from old epochs arriving after the new owner has already committed work. ## Target-repo adaptation -Shard counts and bit splits are workload-specific. Measure skew, cache locality, failure domains and merge costs. Define the durable ownership record, lease lifetime if any, monotonically increasing fencing epoch, which downstream operations must validate it, handoff ordering, retry/idempotency behavior, and recovery semantics before allowing dynamic reassignment. +Shard counts and bit splits are workload-specific. Measure skew, cache locality, failure domains and merge costs. Define the durable ownership source, lease timeout if used, monotonically increasing fencing epoch/token, authoritative mutation boundary that validates epochs, handoff sequence, and restart/recovery semantics before enabling dynamic reassignment. ## Failure modes -Hot shards merely move the bottleneck; domain proliferation raises memory/management overhead; stale owners without fencing can duplicate IDs/work or corrupt state during rebalance; lease expiry without downstream fencing can create split-brain authority; epoch reuse/wraparound or non-durable handoff can resurrect old ownership; rebalancing may violate identity stability; global ordering requirements may make the pattern inadmissible. +Hot shards merely move the bottleneck; domain proliferation raises memory/management overhead; rebalancing without fencing can allow stale and replacement owners to act concurrently; lease-only ownership can fail when an old process resumes; delayed old-epoch messages can duplicate allocations or queue work; identity stability may be violated; global ordering requirements may make the pattern inadmissible. ## Rollback trigger -Revert if partitioning does not reduce measured contention, if any cross-domain invariant fails, if a stale owner can mutate state after reassignment, if split-brain tests admit two authoritative epochs, or if fencing/handoff overhead outweighs the coordination benefit. +Revert if partitioning does not reduce measured contention, if any cross-domain invariant fails, or if failover/rebalance testing shows a stale owner or old-epoch message can mutate state after a replacement owner becomes authoritative. diff --git a/optimizations/OPT-CRIT-001-critical-path-prioritization.md b/optimizations/OPT-CRIT-001-critical-path-prioritization.md index 022629a..a8e44db 100644 --- a/optimizations/OPT-CRIT-001-critical-path-prioritization.md +++ b/optimizations/OPT-CRIT-001-critical-path-prioritization.md @@ -15,22 +15,32 @@ Non-critical work competes with the dependency chain that determines user-visibl ## Optimization problem contract -- X: target-supported task-priority, prefetch/precompute, lazy/deferred-work, and speculation policies -- F: policies that preserve all semantic deadlines, avoid externally visible speculative side effects before commitment, and satisfy starvation/resource constraints +- X: target-supported task-priority, prefetch/precompute, lazy/deferred-work, speculation, speculative-input identity, and commitment/revalidation policies +- F: policies that preserve all semantic deadlines, avoid externally visible speculative side effects before commitment, commit speculative results only for matching effective inputs, and satisfy starvation/resource constraints - f: measured end-to-end latency of the declared critical dependency path, including resource pressure introduced by speculation/deferment - d: minimize -- C: critical outputs and semantic deadlines are preserved; speculative work is safely discardable; deferred work completes before it becomes semantically required -- B: target-specific trace/benchmark budget covering cold/warm, hit/miss, and wrong-speculation cases; no portable prediction horizon is supplied here +- C: critical outputs and semantic deadlines are preserved; speculative work is safely discardable; any speculative result is bound to the complete effective-input identity and revalidated at commitment/delivery; deferred work completes before it becomes semantically required +- B: target-specific trace/benchmark budget covering cold/warm, hit/miss, wrong-speculation, stale-speculation, and change/revert cases; no portable prediction horizon is supplied here - S: stop when the declared budget is exhausted or a validated policy materially reduces critical-path latency without violating C +- Variables: categorical / conditional / mixed priority, deferment, prefetch, and speculation policies +- Search scope: local critical-path policy tuning +- Objective behavior: noisy under realistic workload timing; semantic identity/deadline checks are deterministic +- Information: derivative-free / black-box latency measurements +- Evaluation cost: moderate to expensive end-to-end tracing/benchmarking +- Constraints: semantic deadlines, starvation, side effects, input identity/freshness, memory/CPU/I/O, and target resource constraints +- Parallelism: asynchronous / concurrent execution is common +- Exactness: exact target semantics; speculative work may be discarded but not committed stale ## Preserved contract -Deferred work must still complete before its semantic deadline. Speculative work must be discardable and must not create externally visible side effects before commitment. +Deferred work must still complete before its semantic deadline. Speculative work must be discardable and must not create externally visible side effects before commitment. A speculative result may be committed/delivered only if it still corresponds to the complete current effective-input identity required by the non-speculative reference path. ## Optimization Execute critical dependencies first; prefetch/precompute likely-soon work only when probability and spare resources justify it; lazily defer non-critical work; avoid work with no demonstrated demand. +Bind every speculative/precomputed result to a complete effective-input identity or immutable snapshot. Prefer speculation against an immutable version/snapshot. Otherwise, immediately before commitment/delivery, recompute or reauthenticate the complete effective-input identity and require it to match the identity under which the speculative result was produced. Presence alone is never a freshness proof. If any relevant input changed while speculation was in flight—even if it later changed back A→B→A unless the target can prove one stable A snapshot was consumed—discard the speculative result and execute/recompute from the current reference identity. Commitment is the semantic boundary: no stale speculative result may become externally visible merely because the speculation itself had no side effects. + ## Before / after evidence - Environment: No controlled target-repository benchmark has been run for this OPT record. @@ -43,14 +53,16 @@ Execute critical dependencies first; prefetch/precompute likely-soon work only w Trace the true dependency path and measure end-to-end latency, not only individual task duration. Test cold/warm, cache-hit/miss and wrong-speculation cases. Explicitly test semantic deadlines, starvation, cancellation, and that speculative work cannot expose side effects before commitment. +Add stale-speculation fixtures: start speculation from input identity A, mutate the effective inputs to B before demand/commitment, and verify A is discarded. Include A→B→A change-and-revert races, delayed speculative completion, version rollback, and concurrent config/schema changes. Compare every committed speculative result against the non-speculative reference path for the exact committed identity, and prove commitment/delivery performs the declared revalidation or uses an immutable snapshot strong enough to make revalidation unnecessary. + ## Target-repo adaptation -Criticality and prediction horizons are workload-specific. Re-profile after topology or user-flow changes. +Criticality and prediction horizons are workload-specific. Re-profile after topology or user-flow changes. Define the complete effective-input identity for each speculative result, choose immutable snapshots or commit-time revalidation, and specify exactly when a stale speculative result is discarded. ## Failure modes -Speculation steals resources from critical work, lazy work causes later latency cliffs, priorities become stale, deferred tasks starve, semantic deadlines are missed, or speculative side effects escape before commitment. +Speculation steals resources from critical work, lazy work causes later latency cliffs, priorities become stale, deferred tasks starve, semantic deadlines are missed, speculative side effects escape before commitment, or stale speculative output is committed after its effective inputs changed. ## Rollback trigger -Immediately disable/revert the policy on any violation of C, including a required task missing its semantic deadline or speculative work exposing an externally visible side effect before commitment. Also disable it if critical-path latency or resource pressure worsens materially. +Immediately disable/revert the policy on any violation of C, including a required task missing its semantic deadline, speculative work exposing an externally visible side effect before commitment, or a speculative result being committed/delivered without matching the current effective-input identity. Also disable it if critical-path latency or resource pressure worsens materially. diff --git a/optimizations/OPT-FAN-001-shared-materialization-fanout.md b/optimizations/OPT-FAN-001-shared-materialization-fanout.md index ab3f103..57147d4 100644 --- a/optimizations/OPT-FAN-001-shared-materialization-fanout.md +++ b/optimizations/OPT-FAN-001-shared-materialization-fanout.md @@ -14,17 +14,25 @@ The same deterministic transformation is repeated independently for each consume ## Optimization problem contract -- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, and crash-consistent publication schemes -- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, and publication-atomicity requirement +- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, immutable-source snapshot/revalidation policies, and crash-consistent publication schemes +- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, source-snapshot consistency, and publication-atomicity requirement - f: measured transformation CPU, replay CPU, fan-out latency, and storage/I/O overhead under the target's declared objective ordering - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state cryptographically or structurally binds the artifact bytes to the complete effective source/transform identity; verification/security metadata may be removed only under an explicit contract change +- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state binds the artifact bytes to one coherent effective source/transform identity; verification/security metadata may be removed only under an explicit contract change - B: target-specific fan-out/replay benchmark budget declared before tuning; no portable subscriber count, replay size, or retention duration is supplied here - S: stop when the declared budget is exhausted or a validated materialization policy materially improves the target objective without violating C +- Variables: categorical / integer / mixed +- Search scope: local materialization-boundary / representation-policy tuning +- Objective behavior: noisy for performance; transformation identity/equivalence is deterministic +- Information: derivative-free / black-box performance measurements +- Evaluation cost: moderate to expensive depending on transform/replay size +- Constraints: semantic equivalence, source-snapshot consistency, integrity, versioning, trust/security, storage, and crash-consistency constraints +- Parallelism: concurrent fan-out/replay; publication must remain race-safe +- Exactness: exact representation semantics; no approximation is introduced ## Preserved contract -Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive crashes and interrupted publication. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. +Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive source mutation, crashes, and interrupted publication. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. ## Optimization @@ -32,9 +40,11 @@ Perform an expensive deterministic transform once near production, persist or re Define a materialization key that covers, as applicable, source object/content identity or immutable source version, transformation/encoder implementation identity, encoder configuration and dictionaries, schema/format version, feature flags, trust/security policy, and any other effective input that can affect the materialized result. +Bind the transform to one coherent source identity. Prefer reading from an immutable source snapshot/version captured together with the materialization key. If the target cannot provide an immutable snapshot, recompute/re-authenticate the **complete** effective-input materialization key immediately before commit and require it to equal the key used to start the transform. Any source/config/transform identity change during execution invalidates the candidate materialization; discard/retry it rather than publishing bytes produced from mixed or newer state under an older key. Change-and-revert (A→B→A) is still a mutation event unless the target can prove the transform observed one stable A snapshot throughout. + Publish the artifact and its identity as **one committed state**. Acceptable designs include content-addressed storage where the artifact digest is itself part of the committed key, an atomically replaced manifest that contains both the full materialization key and the artifact digest/location, or another crash-consistent transaction that makes old state or new state visible but never a mixed pair. Do not update artifact bytes and their key independently in a way that can expose a new artifact with stale metadata or stale bytes with a new key after a crash. -Before reuse, require: (1) exact agreement with the current effective-input materialization key, (2) a committed manifest/content-address relation that binds that key to the artifact identity, and (3) artifact integrity/format validity. A key mismatch, missing/incomplete publication marker, digest mismatch, or unverifiable artifact is a cache miss and requires regeneration. Do not use format validation alone as evidence that an artifact corresponds to current inputs. +Before reuse, require: (1) exact agreement with the current effective-input materialization key, (2) a committed manifest/content-address relation that binds that key to the artifact identity, and (3) artifact integrity/format validity. A key mismatch, missing/incomplete publication marker, digest mismatch, unverifiable artifact, or failed commit-time source revalidation is a cache miss and requires regeneration. Do not use format validation alone as evidence that an artifact corresponds to current inputs. ## Before / after evidence @@ -48,16 +58,18 @@ Before reuse, require: (1) exact agreement with the current effective-input mate Compare shared materialization against per-consumer reference output, including corruption, restart/replay and mixed consumer capabilities. Independently mutate each key component—source content/version, transform implementation, encoder options/dictionary, schema/format version, feature flags and trust policy—and prove that each output-affecting change invalidates reuse. Also test unchanged-key reuse, tampered artifacts with matching metadata, and migration/version-boundary cases. +Exercise **concurrent source mutation**. Start a transform from source identity A, mutate the source/effective transform inputs to B while work is running, and test A→B→A change-and-revert sequences. For snapshot-based targets, prove the transform reads only the immutable A snapshot. For revalidation-based targets, prove the final complete-key comparison rejects/discards any candidate whose effective inputs changed while the transform was executing. Compare accepted materializations with a fresh transform from the exact committed source identity. + Inject crashes/interruption at every publication boundary: after artifact write but before manifest commit, after provisional metadata write, during atomic replacement, and immediately after commit. After restart, prove that readers see either the previous valid committed materialization or the new valid committed materialization, never a mixed key/artifact state. Verify digest/key mismatch is rejected even when the artifact is otherwise parseable. ## Target-repo adaptation -Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, and specify representation versioning, invalidation, integrity checking, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. +Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, specify whether mutable source inputs are consumed from immutable snapshots or protected by complete commit-time key revalidation, and specify representation versioning, invalidation, integrity checking, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. ## Failure modes -Incomplete keys can serve stale representations after source or transform changes; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. +Incomplete keys can serve stale representations after source or transform changes; mutable sources can change during transformation and produce mixed-state output under a stale key; change-and-revert races can fool naive identity checks; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. ## Rollback trigger -Disable reuse immediately if any materialization-key hit, publication-recovery path, or integrity check can return output that differs from a fresh transform for the same current effective inputs, or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. +Disable reuse immediately if any materialization-key hit, source-mutation race, publication-recovery path, or integrity check can return output that differs from a fresh transform for the same exact committed effective inputs, or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index 732f49f..1b10370 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -14,28 +14,33 @@ Expensive work is rerun even though every input capable of affecting its result ## Optimization problem contract -- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity policies, input-snapshot/revalidation policies, and crash-consistent state-publication mechanisms -- F: configurations whose signature covers every output-affecting input, whose execution observes one valid effective-input snapshot or revalidates the complete effective-input identity before commit, whose reuse validates required outputs, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted/raced executions never publish reusable partial or mismatched state -- f: measured repeated-work cost including stage runtime plus signature/revalidation/metadata/output-validation/publication I/O overhead +- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity policies, immutable-input snapshot/revalidation policies, and crash-consistent state-publication mechanisms +- F: configurations whose signature covers every output-affecting input, whose execution consumes one immutable effective-input snapshot or revalidates the complete effective-input identity before commit, whose reuse validates required outputs, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted executions never publish reusable partial state +- f: measured repeated-work cost including stage runtime plus signature/snapshot/metadata/output-validation/publication I/O overhead - d: minimize -- C: every reused output is semantically equivalent to a fresh execution for the exact effective-input identity recorded in its committed generation, with the same failure and output-validity semantics; reuse metadata cannot mix fields from different generations; concurrent input mutation cannot cause a generation to publish outputs under a stale signature +- C: every reused output is semantically equivalent to a fresh execution for the same effective inputs, with the same failure and output-validity semantics; reuse metadata cannot mix fields from different generations; a committed generation cannot bind a pre-execution signature to output produced from changed or mixed inputs - B: target-specific benchmark/evaluation budget declared before tuning; no portable value is supplied by this record - S: stop when the declared budget is exhausted or a validated configuration meets the predeclared improvement threshold without violating C +- Variables: categorical / mixed policy choices for signatures, snapshots, validation, granularity, and publication +- Search scope: local to one incremental stage or pipeline boundary +- Objective behavior: noisy for performance; correctness identity checks are deterministic +- Information: derivative-free / black-box performance measurements +- Evaluation cost: moderate to expensive depending on stage runtime and validation cost +- Constraints: semantic equivalence, integrity, crash consistency, snapshot consistency, and resource constraints +- Parallelism: sequential or pipeline-specific; publication/revalidation must remain race-safe under concurrent producers/consumers +- Exactness: exact reuse semantics; no approximation is introduced ## Preserved contract -Reused output must be semantically equivalent to a fresh execution for the exact effective inputs identified by the committed generation. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and input mutation during execution must not let output produced from state B be committed under signature A. +Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and mutable inputs must not change underneath execution without invalidating the candidate generation. ## Optimization -Establish the effective-input identity before execution using one of two admissible strategies: +Compute a deterministic signature over the effective inputs and compare it with successfully persisted prior state. Reuse is allowed only when that signature still matches **and** every required output satisfies a declared validity predicate. Depending on the target, that predicate may be a content digest/version manifest, a trusted immutable/protected artifact identity, or another reproducible integrity check strong enough to detect external mutation. Mere file presence is not sufficient unless the target explicitly guarantees that reused outputs are immutable and protected from modification. Execute when the input signature differs, any required output is missing, or any output-validity check fails. -1. **Immutable snapshot:** execute strictly against a snapshot/version whose identity is the signature recorded for the generation; or -2. **Precommit revalidation:** compute the complete effective-input signature before execution, run the stage, then recompute/reauthenticate the complete effective-input signature immediately before publication. If it differs, discard or quarantine the produced outputs and retry from the new identity rather than publishing them under the stale signature. +Bind execution to one coherent effective-input identity. Prefer executing against an immutable snapshot/version of every mutable effective input. If the target cannot provide such a snapshot, recompute the **complete** effective-input signature immediately before publication and require it to equal the signature used to start the candidate generation. If any effective input changed—even if it later changes back—discard the candidate generation and retry from a fresh identity; do not publish output produced from a moving or mixed input state under the old signature. -Reuse is allowed only when the committed input signature still matches **and** every required output satisfies a declared validity predicate. Depending on the target, that predicate may be a content digest/version manifest, a trusted immutable/protected artifact identity, or another reproducible integrity check strong enough to detect external mutation. Mere file presence is not sufficient unless the target explicitly guarantees that reused outputs are immutable and protected from modification. Execute when the input signature differs, any required output is missing, or any output-validity check fails. - -Publish incremental state as one crash-consistent **generation** that binds the verified input identity to the complete output identity/validity metadata. Do not persist the signature and output metadata as independently authoritative updates. Use an atomic rename/swap of a complete manifest, a transactional store, a content-addressed generation pointer, or another mechanism where readers observe either the previous complete generation or the new complete generation—never a mixture. Only publish after every output has been produced and validated successfully **and** the immutable-snapshot or precommit-revalidation rule proves the recorded input identity still matches the inputs used to produce those outputs. An interrupted, failed, or input-raced publication leaves the previous committed generation authoritative and the partial generation non-reusable. +Publish incremental state as one crash-consistent **generation** that binds the validated input signature to the complete output identity/validity metadata. Do not persist the signature and output metadata as independently authoritative updates. Use an atomic rename/swap of a complete manifest, a transactional store, a content-addressed generation pointer, or another mechanism where readers observe either the previous complete generation or the new complete generation—never a mixture. Only publish the new generation after every output has been produced and validated successfully **and** the input snapshot/signature has passed the final commit-time identity check; an interrupted, failed, or input-raced publication leaves the previous committed generation authoritative and the candidate generation non-reusable. Reuse filesystem/configuration metadata lazily only while its own validity predicate still holds. @@ -53,20 +58,20 @@ Wonderbuild demonstrates the mechanism and benchmark shapes, but its historical ## Validation -Test unchanged inputs with valid outputs, changed inputs, missing outputs, failed runs, corrupted persistent state, externally overwritten/corrupted outputs, stale output-version metadata, and concurrent input mutation against a forced-fresh reference path. A mutated output must force reconstruction unless the target's immutable/protected-output contract proves such mutation impossible. +Test unchanged inputs with valid outputs, changed inputs, missing outputs, failed runs, corrupted persistent state, externally overwritten/corrupted outputs, and stale output-version metadata against a forced-fresh reference path. A mutated output must force reconstruction unless the target's immutable/protected-output contract proves such mutation impossible. -For mutable inputs, deliberately change one or more effective inputs while the stage is executing, including changes immediately before commit and changes that revert to the original value. Prove that either execution was bound to an immutable snapshot or the precommit signature comparison detects the change and prevents publication. Verify no generation can bind signature A to output produced from B or from a mixed A/B observation. +Exercise **concurrent input mutation**. Start execution from identity A, mutate one or more effective inputs during execution to B (including mixed-state multi-file/config changes), and also test A→B→A change-and-revert sequences. For snapshot-based targets, prove execution reads only the immutable A snapshot. For revalidation-based targets, prove the commit-time complete signature detects any change and discards/retries the candidate instead of publishing it. Compare every accepted generation with a forced-fresh execution over the exact committed input identity. -Exercise interruption/crash injection at every publication boundary: before outputs complete, after outputs complete but before manifest publication, during temporary-manifest write, immediately before/after the atomic generation switch, and during cleanup. After each interruption, prove readers observe only a self-consistent old or new generation and can never pair signature A with output identities/metadata from generation B. +Exercise interruption/crash injection at every publication boundary: before outputs complete, after outputs complete but before final input revalidation, after revalidation but before manifest publication, during temporary-manifest write, immediately before/after the atomic generation switch, and during cleanup. After each interruption, prove readers observe only a self-consistent old or new generation and can never pair signature A with output identities/metadata from generation B. ## Target-repo adaptation -Re-profile signature and output-validation cost, input-snapshot or revalidation cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Explicitly choose whether outputs are integrity-checked on reuse or are stored behind an enforceable immutable/protected boundary, define how execution is tied to an immutable input snapshot or how complete input identity is revalidated before commit, and define the crash-consistency guarantee for committing the signature plus output identities. +Re-profile signature and output-validation cost, snapshot/revalidation cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Explicitly choose whether mutable inputs are consumed from immutable snapshots or protected by complete commit-time signature revalidation, whether outputs are integrity-checked on reuse or stored behind an enforceable immutable/protected boundary, and define the crash-consistency guarantee for committing the signature plus output identities. ## Failure modes -Incomplete signatures create stale reuse; mutable inputs can change during execution and produce outputs that do not correspond to the pre-run signature; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. +Incomplete signatures create stale reuse; input mutation during execution can bind an old signature to new/mixed output; change-and-revert races can evade presence-only checks; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. ## Rollback trigger -Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference, if concurrent input mutation can publish outputs under a stale identity, if external output mutation can bypass the declared validity predicate, if crash/interruption testing can expose mixed-generation state, or if signature/revalidation/integrity/publication maintenance costs more than the avoided work. +Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference, if mutable-input races can publish a generation not tied to one coherent effective-input identity, if external output mutation can bypass the declared validity predicate, if crash/interruption testing can expose mixed-generation state, or if signature/snapshot/integrity/publication maintenance costs more than the avoided work. diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index 21046ee..a9cfbe3 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -19,24 +19,39 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F - f: a scalar real-valued target objective `f : F → R` evaluated on feasible candidates only - d: exactly one of scalar `minimize` or scalar `maximize`; vector, Pareto, lexicographic, or other partial-order objectives are outside this record unless a separately specified and validated frontier-bound mechanism is introduced -- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, and budget exhaustion without a feasible incumbent produces an explicit unknown/no-incumbent outcome rather than a feasibility or optimality claim +- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, the target's observable tie semantics are preserved, and budget exhaustion without a feasible incumbent produces an explicit unknown/no-incumbent outcome rather than a feasibility or optimality claim - B: a finite, predeclared target-specific cap on evaluations, wall time, compute, or equivalent resource consumption; exact-mode search may prove optimality before this cap but may not run without a finite cap -- S: stop immediately when optimality is proven or the frontier is exhausted; otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality +- S: stop immediately when the required optimality/tie contract is proven or the frontier is exhausted; otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality +- Variables: integer / categorical / discrete / mixed +- Search scope: global over the declared candidate space +- Objective behavior: deterministic unless uncertainty/noise is incorporated into a separately sound bound model +- Information: derivative-free; bound/relaxation information is target-specific +- Evaluation cost: moderate to expensive when exhaustive evaluation is infeasible +- Constraints: feasibility, semantic correctness, scalar-bound soundness, tie semantics, and finite-resource constraints +- Parallelism: sequential or parallel only with synchronized incumbent/frontier/bound semantics +- Exactness: exact only when the declared optimality and observable-tie contract is proven; otherwise anytime/incomplete result semantics apply For each unexplored region `R`, define a bound `b(R)` separately from `f`: -- minimizing: `b(R) ≤ inf { f(x) | x ∈ F ∩ R }`; prune `R` only when `b(R) ≥ f(x_incumbent)`; -- maximizing: `b(R) ≥ sup { f(x) | x ∈ F ∩ R }`; prune `R` only when `b(R) ≤ f(x_incumbent)`. +- minimizing: `b(R) ≤ inf { f(x) | x ∈ F ∩ R }`; +- maximizing: `b(R) ≥ sup { f(x) | x ∈ F ∩ R }`. + +If the target contract accepts **any one scalar optimum** and equal-objective candidates are not observably distinct, minimization may prune `R` when `b(R) >= f(x_incumbent)` and maximization may prune when `b(R) <= f(x_incumbent)`. + +If equal-objective candidates remain observable—for example the target requires a deterministic tie winner, a secondary total ordering, or enumeration of all optimal candidates—equality is not enough to discard a region under the scalar bound alone. In that case either: + +- use strict objective pruning while unresolved ties remain (`b(R) > f(x_incumbent)` for minimization; `b(R) < f(x_incumbent)` for maximization), and continue exploring equality-bound regions as required by C; or +- define a separately sound bound over the **complete declared tie ordering/frontier** and validate that stronger bound independently. An independently proven infeasible region may also be pruned. A heuristic estimate that does not satisfy the declared bound relation is search-ordering evidence at most, not a pruning proof. This record does not authorize scalar bounds to prune vector/Pareto or partially ordered objectives. ## Preserved contract -A region may be discarded only when its scalar bound proves it cannot improve the incumbent under the declared scalar objective and constraints. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, and exhausting B without a feasible incumbent does not permit an infeasibility claim. +A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, and exhausting B without a feasible incumbent does not permit an infeasibility claim. ## Optimization -Maintain an incumbent when one exists, partition the search space, compute a cheap sound `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific scalar bound relation proves the region cannot improve the incumbent. Before the first incumbent exists, sound bounds may prioritize regions or prove individual regions infeasible, but incumbent-based objective pruning is unavailable. +Maintain an incumbent when one exists, partition the search space, compute a cheap sound `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific bound plus the target's tie semantics prove the region cannot affect the required answer. Before the first incumbent exists, sound bounds may prioritize regions or prove individual regions infeasible, but incumbent-based objective pruning is unavailable. A relaxed solution is evidence for a bound, not automatically a feasible final answer. @@ -52,14 +67,16 @@ A relaxed solution is evidence for a bound, not automatically a feasible final a For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness independently by checking the direction-specific inequality against exhaustive feasible values inside each test region. Test pruning separately from search ordering. Include fixtures where the first feasible candidate is found late and where B expires before any feasible candidate exists; verify that the latter returns `no-incumbent / feasibility-unknown`, reports only independently valid global-bound information, and makes no infeasibility, optimality, or incumbent-based gap claim. Verify that budget exhaustion with an incumbent returns an anytime result without an exactness claim. +Add **equal-objective tie fixtures**. For an any-one-optimum contract, prove equality pruning cannot alter any observable result. For deterministic tie-winner contracts, construct regions containing equal-objective candidates with better/worse tie ranks and prove equality-bound regions are retained until the declared tie winner is established. For all-optima contracts, prove every equal-objective optimum is enumerated. If using a stronger total-order bound, validate its soundness independently against exhaustive fixtures. + ## Target-repo adaptation -The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. +The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. ## Failure modes -Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; an unbounded exact-search policy can consume resources indefinitely; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. +Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; an unbounded exact-search policy can consume resources indefinitely; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. ## Rollback trigger -Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar bound relation, is applied to an unsupported objective ordering, or whose bound cost exceeds the work it eliminates. Abort exact-mode claims whenever B is exhausted before optimality is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. +Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort exact-mode claims whenever B is exhausted before the full objective/tie contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. diff --git a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md index be1758f..867bb4e 100644 --- a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md +++ b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md @@ -21,6 +21,14 @@ An expensive operation is applied to a large population even though only a small - C: the reordered/reduced pipeline is semantically equivalent to the reference for all declared outputs **and observable effects**; an effectful stage may be bypassed for discarded candidates only when those effects/errors are explicitly proven irrelevant by the target contract - B: target-specific benchmark budget over representative and adversarial selectivity distributions; no portable selectivity threshold is supplied here - S: stop when the declared budget is exhausted or a validated early-reduction placement materially lowers total cost without violating C +- Variables: categorical / conditional / mixed placement and predicate choices +- Search scope: local pipeline-reordering / working-set-reduction decisions +- Objective behavior: noisy for performance; semantic equivalence is deterministic +- Information: derivative-free / black-box performance measurements +- Evaluation cost: moderate to expensive depending on downstream stage cost and workload size +- Constraints: output, ordering/tie/join, side-effect/error, purity, and resource constraints +- Parallelism: sequential pipeline semantics with target-specific parallel execution only where equivalence remains valid +- Exactness: exact observable semantics; no approximation is introduced ## Preserved contract diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index d0a4621..381fe7c 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -20,15 +20,21 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra - F: candidates in X that satisfy all hard resource, platform, semantic, and correctness constraints before objective ranking - f: the target-measured objective or objective vector for each feasible candidate, including declared noise/statistical treatment - d: the target's predeclared minimize, maximize, lexicographic, or Pareto ordering -- C: search may choose where to evaluate but may not weaken correctness, determinism, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources, every per-trial reservation must be an enforceable upper bound rather than an estimate, dispatch/completion accounting must be linearizable under concurrency, and targets that require deterministic search outcomes must use deterministic observation assimilation rather than completion-order updates +- C: search may choose where to evaluate but may not weaken correctness, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources; every per-trial reservation must be an enforceable upper bound rather than an estimate; dispatch/completion accounting must be linearizable; and targets that require deterministic search outcomes must use deterministic observation assimilation independent of wall-clock completion order - B: an explicit target-specific hard maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit, enforceable per-trial cap mechanism, conservative reservation rule, atomic accounting boundary, and failure/cancellation charging policy are fixed before dispatch begins -- S: stop proposing/dispatching when no additional trial can be safely reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger and apply stopping decisions only to the declared deterministic assimilation frontier when determinism is required +- S: stop proposing/dispatching when no additional trial can be safely reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger and apply stopping decisions to the declared deterministic assimilation order when determinism is required +- Variables: continuous / integer / categorical / conditional / mixed +- Search scope: local or global, explicitly declared for the target +- Objective behavior: deterministic, noisy, or stochastic as declared by the target; noise treatment must be explicit +- Information: derivative-free / black-box by default; gradient information may be used only when the selected target mechanism supports it +- Evaluation cost: typically expensive +- Constraints: bounds, semantic correctness, resource, platform, and target-specific equality/inequality constraints +- Parallelism: sequential / synchronous batch / asynchronous, explicitly declared +- Exactness: target evaluations must satisfy C exactly; the search itself need not prove a global optimum unless the target contract requires it ## Preserved contract -Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B, no individual trial may consume beyond its reserved cap, and concurrent dispatch/completion transitions must not transiently expose phantom free capacity. - -If the target requires reproducible search traces or identical selected configurations across runs, asynchronous completion order is not allowed to change the optimizer's logical observation sequence. In that mode, results may finish in any wall-clock order, but they are assimilated into the optimizer only in a deterministic order such as monotonically increasing trial ID or explicit deterministic batches. If completion-order assimilation is intentionally used, the resulting nondeterminism must be declared as a contract change rather than hidden behind a fixed seed. +Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B, no individual trial may consume beyond its reserved cap, and concurrent dispatch/completion transitions must not transiently expose phantom free capacity. If the target requires deterministic selected configurations or trial traces, proposal updates and stopping decisions must not depend on nondeterministic completion order. ## Optimization @@ -38,7 +44,7 @@ Before dispatching an asynchronous trial, enter one atomic/serializable accounti Completion, failure, cancellation, and forced termination use the **same atomic accounting boundary** as dispatch reservation. For one terminal transition, atomically: (1) read the trial's reservation, (2) meter/record the amount actually consumed, (3) move that consumed amount into permanent `consumed`, (4) release only the demonstrably unconsumed remainder from `reserved`, and (5) mark the trial terminal. No dispatcher may observe released reservation capacity before the corresponding consumed charge is committed, and concurrent terminal updates must not lose increments. Completion must not double-charge the same usage. A failed or cancelled trial never erases resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute/time budgets, release only the measured or otherwise provable unused portion of the enforceable reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. Every reservation, cap enforcement action, consumption adjustment, release, failure, cancellation, forced termination, and terminal accounting transaction is recorded in the ledger. -When deterministic search behavior is required, assign each proposal a stable trial ID at reservation/dispatch time and separate **physical completion** from **logical assimilation**. Buffer terminal results until the next deterministic trial-ID/batch frontier is complete, then update the surrogate/acquisition/stopping state in that fixed order. Failed/cancelled trials contribute their predeclared deterministic terminal observation/status at the same logical position. Later proposals may depend only on observations already admitted through that deterministic frontier. Alternative deterministic batching schemes are admissible if their ordering rule is fixed before execution and replayable from the ledger. +For targets that require deterministic search behavior, assign a deterministic trial ID/order at proposal time and **buffer asynchronous completions for assimilation in that declared order** (or use explicit deterministic batches/barriers). Surrogate/model updates, acquisition decisions, domain contraction, portfolio-selection state, and stopping criteria must consume observations according to this deterministic order rather than wall-clock completion order. A fixed random seed alone is not sufficient. If a target chooses completion-order assimilation for throughput, declare the resulting nondeterminism as an explicit contract change rather than claiming deterministic replay. Parallelism has an information cost: very wide batches receive less feedback between suggestions and can degenerate toward non-adaptive/random search. @@ -56,16 +62,16 @@ Keep a deterministic search seed where practical, preserve the full trial ledger Race multiple trial completions/cancellations against one another and against workers attempting the final dispatch slot. Verify the accounting transaction is linearizable: no consumed increment is lost, no reservation is released before its corresponding consumption is charged, and a dispatcher never observes capacity that would make the post-transaction invariant `consumed + reserved <= B` false. -For targets that require deterministic optimization, run the same fixed-seed search repeatedly while deliberately perturbing worker latency/completion order. Verify that the persisted logical observation sequence, surrogate updates, proposals, stopping decision, and selected result are identical. Compare sequential execution with deterministic asynchronous/batched execution where the chosen scheme claims equivalence. Replay the ledger from scratch and prove it reconstructs the same optimizer state and final selection. If the target permits nondeterministic completion-order assimilation, record that explicitly and do not claim deterministic replay equivalence. +For deterministic targets, run the same seeded trial set with deliberately permuted worker speeds/completion orders. Verify observation assimilation follows the declared trial-ID/batch order, the surrogate/search state replays identically, and the selected candidate plus stopping reason match the deterministic reference. Where sequential/parallel equivalence is part of C, compare an asynchronous execution with its deterministic sequential or batch-assimilation replay. If deterministic equivalence is intentionally not required, verify the record/target explicitly labels that nondeterminism instead. ## Target-repo adaptation -Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, **enforcement mechanism for that reservation**, one atomic/serializable accounting mechanism shared by reservation and terminal conversion, metering source, failure/cancellation charging policy, and deterministic observation-assimilation rule when required before enabling asynchronous dispatch. +Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, **enforcement mechanism for that reservation**, one atomic/serializable accounting mechanism shared by reservation and terminal conversion, metering source, failure/cancellation charging policy, and deterministic observation-assimilation policy (when required) before enabling asynchronous dispatch. ## Failure modes -Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism; completion-order assimilation can make fixed-seed asynchronous searches produce different traces, stopping points, and selected configurations. +Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism; wall-clock completion-order assimilation can make supposedly deterministic search traces, proposals, and stopping decisions irreproducible. ## Rollback trigger -Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, any trial can consume beyond its enforceable reservation, any accounting/concurrency test shows that dispatch, completion, failure, cancellation, forced termination, or reservation release can cause actual consumption plus outstanding reservations to exceed B or expose transient free capacity before consumption is committed, or a target that requires determinism cannot reproduce the same logical observation sequence and final selection under perturbed asynchronous completion order. +Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, any trial can consume beyond its enforceable reservation, any accounting/concurrency test shows that dispatch/terminal transitions can violate B, or any target that requires deterministic search fails replay under permuted asynchronous completion orders. diff --git a/optimizations/OPT-SET-001-density-adaptive-compact-sets.md b/optimizations/OPT-SET-001-density-adaptive-compact-sets.md index d854b63..8650414 100644 --- a/optimizations/OPT-SET-001-density-adaptive-compact-sets.md +++ b/optimizations/OPT-SET-001-density-adaptive-compact-sets.md @@ -15,21 +15,33 @@ A single representation performs poorly across regions with very different densi ## Optimization problem contract -- X: target-supported partition widths, sparse/dense container choices, switching thresholds/hysteresis policies, mutation/conversion policies, and serialization layouts -- F: representations that preserve exact membership and set-operation semantics across all supported mutations and representation transitions and satisfy target memory/serialization compatibility constraints -- f: measured memory footprint plus target-relevant set-operation, mutation/conversion, and serialization latency +- X: target-supported partition widths, sparse/dense container choices, switching thresholds, serialization layouts/versions, and migration policies +- F: representations that preserve exact membership and set-operation semantics, preserve contractually significant iteration ordering when one exists, and satisfy target memory/serialization compatibility constraints +- f: measured memory footprint plus target-relevant set-operation and serialization latency - d: minimize under the target's predeclared scalar, lexicographic, or Pareto ordering -- C: membership, insertion, deletion, union, intersection, difference, representation transitions, and persistence round trips match the canonical reference set exactly after every mutation -- B: target-specific benchmark budget over declared sparse, dense, mixed, transition-boundary, and mutation-sequence datasets; no portable trial count is supplied here +- C: membership, union, intersection, difference, iteration behavior where observable, mutation semantics, and persistence/compatibility round trips match the canonical reference contract exactly +- B: target-specific benchmark budget over declared sparse, dense, mixed, transition-boundary, mutation, and compatibility datasets; no portable trial count is supplied here - S: stop when the declared budget is exhausted or a validated representation meets the target objective without violating C +- Variables: integer / categorical / mixed +- Search scope: local representation/threshold/layout tuning +- Objective behavior: noisy for performance; set semantics are deterministic +- Information: derivative-free performance measurements +- Evaluation cost: cheap to moderate per fixture; may become expensive at production scale +- Constraints: exact set semantics, ordering where observable, memory, serialization, version compatibility, and migration constraints +- Parallelism: sequential for representation transitions unless the target separately defines safe concurrent mutation semantics +- Exactness: exact set semantics; no approximation is introduced ## Preserved contract -Membership and set operations must match the reference set exactly before, during, and after conversion between sparse and dense representations. A threshold crossing is an internal representation change only; it must not drop, duplicate, reorder semantically significant iteration, or corrupt members. +Membership and set operations must match the reference set exactly. If iteration order is part of the target API/serialization contract, representation changes must preserve that order exactly. Persisted or exchanged sets must remain readable/writable according to the target's explicit version-compatibility policy; otherwise the layout change requires an explicit migration/version boundary rather than being treated as transparent. ## Optimization -Partition the identifier space and choose a representation per partition according to local density. Keep sparse regions compact while using bitmap-like containers where dense boolean algebra is advantageous. For mutable sets, conversions triggered by insertions/deletions must be deterministic and exact. Consider hysteresis or other anti-thrashing policy when repeated near-threshold mutation would otherwise cause conversion churn, but do not change set semantics to avoid conversions. Prefer representations that can be serialized without expanding to a larger intermediate form. +Partition the identifier space and choose a representation per partition according to local density. Keep sparse regions compact while using bitmap-like containers where dense boolean algebra is advantageous. Prefer representations that can be serialized without expanding to a larger intermediate form. + +For mutable sets, representation switching is part of the state machine: insertion/deletion may cross thresholds in either direction. Conversions must be semantics-preserving and idempotent with respect to the canonical set state, including any observable iteration order and persistence metadata. + +For persisted/mixed-version use, define a versioned serialization contract. Either retain backward/forward compatibility for the required reader/writer matrix or provide an explicit migration/version bump before emitting an incompatible layout. Do not infer external compatibility from a same-version self-round-trip. ## Evidence boundary @@ -45,18 +57,20 @@ Jazco reports strong production-scale graph results, but OPT treats the numbers ## Validation -Differential-test membership, insertion, deletion, union, intersection, difference and persistence against a simple canonical set implementation over sparse, dense and transition-boundary fixtures. +Differential-test membership, union, intersection, difference, iteration (when observable), and persistence against a simple canonical set implementation over sparse, dense and transition-boundary fixtures. + +Exercise **mutable transition sequences**: insert/delete elements so each switching threshold is crossed repeatedly sparse→dense and dense→sparse, including oscillation directly around thresholds. After every mutation and conversion, compare membership, cardinality, union/intersection/difference, and—when part of C—the exact iteration sequence with the canonical reference. Serialize and reload after every transition, then repeat the same comparisons. -Add mutation-sequence tests that repeatedly cross every sparse↔dense switching threshold in **both directions**. Construct sequences that insert just past the promotion boundary, delete back below the demotion boundary, and repeat for many cycles; include randomized/adversarial churn near the boundary. After every mutation and every representation transition, verify exact membership/cardinality against the reference set, then re-run union/intersection/difference checks and a serialization round trip. Test duplicate insertions, deletion of absent elements, empty/full-ish containers, threshold off-by-one cases, and restart/deserialization followed by further transitions. If hysteresis is used, verify its exact promotion/demotion rules while preserving the same set contents. +Exercise **serialization compatibility** independently from same-version round trips. Keep golden fixtures from every required older format/version and prove the new reader accepts them without semantic loss. Where backward writing or forward reading is required, exercise those reader/writer combinations explicitly. If compatibility is intentionally broken, require a versioned migration that converts old persisted state before the new layout becomes authoritative, and verify old consumers cannot silently misinterpret new bytes. ## Target-repo adaptation -Benchmark partition sizes, switching thresholds, hysteresis/conversion policy, and serialization format on the real identifier distribution, mutation pattern, and CPU/cache hierarchy. Immutable/read-mostly and mutation-heavy workloads may justify different policies. +Benchmark partition sizes and switching thresholds on the real identifier distribution and CPU/cache hierarchy. Determine whether iteration order is observable. Define the serialization-version matrix, migration policy, and any hysteresis needed to avoid conversion churn around thresholds. ## Failure modes -Conversion bugs can drop or duplicate members; repeated near-threshold mutation can cause conversion thrash; asymmetric promotion/demotion logic can strand a container in the wrong representation; pathological distributions, serialization incompatibility and hidden temporary allocations can erase the benefit. +Conversion defects can drop/duplicate members; repeated threshold crossing can cause churn; sparse↔dense transitions can reorder iteration; new layouts can make old persisted state unreadable or emit bytes older consumers reject; pathological distributions, serialization incompatibility and hidden temporary allocations can erase the benefit. ## Rollback trigger -Revert when target data does not show a memory/latency win, any static or mutation-sequence differential test fails, any transition loses/duplicates members, persistence round trips diverge, or conversion churn materially worsens the target workload. +Revert when target data does not show a memory/latency win, any mutable-transition differential test fails, observable iteration order changes, or any required persistence/version-compatibility fixture fails. diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py old mode 100755 new mode 100644 index 6492f24..2187a0d --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -29,6 +29,16 @@ "## Rollback trigger", } REQUIRED_CONTRACT_FIELDS = ("X", "F", "f", "d", "C", "B", "S") +REQUIRED_CLASSIFICATION_FIELDS = ( + "Variables", + "Search scope", + "Objective behavior", + "Information", + "Evaluation cost", + "Constraints", + "Parallelism", + "Exactness", +) ALLOWED_V2_STATUS_CATEGORIES = { "Verified", "Verified, environment-specific", @@ -116,6 +126,20 @@ def normalized_status_category(raw: str) -> str: return plain.split(";", 1)[0].strip() +def require_prefixed_fields(path: Path, lines: list[str], fields: tuple[str, ...], section: str) -> None: + """Require exactly one non-empty '- Field:' row for every declared field.""" + for field in fields: + prefix = f"- {field}:" + matches = [line for line in lines if line.startswith(prefix)] + if len(matches) != 1: + die( + f"{path.relative_to(ROOT)} must contain exactly one field " + f"'{prefix}' in {section}" + ) + if not matches[0][len(prefix) :].strip(): + die(f"{path.relative_to(ROOT)} has empty field {field} in {section}") + + records: dict[str, Path] = {} status_categories: dict[str, str] = {} for path in sorted(OPT_DIR.glob("OPT-*.md")): @@ -169,16 +193,18 @@ def normalized_status_category(raw: str) -> str: ) contract = section_lines(text, "## Optimization problem contract") - for field in REQUIRED_CONTRACT_FIELDS: - prefix = f"- {field}:" - matches = [line for line in contract if line.startswith(prefix)] - if len(matches) != 1: - die( - f"{path.relative_to(ROOT)} must contain exactly one contract field " - f"'{prefix}' in ## Optimization problem contract" - ) - if not matches[0][len(prefix) :].strip(): - die(f"{path.relative_to(ROOT)} has empty contract field {field}") + require_prefixed_fields( + path, + contract, + REQUIRED_CONTRACT_FIELDS, + "## Optimization problem contract", + ) + require_prefixed_fields( + path, + contract, + REQUIRED_CLASSIFICATION_FIELDS, + "## Optimization problem contract", + ) missing_frozen = sorted(FROZEN_V1 - records.keys()) if missing_frozen: @@ -206,34 +232,38 @@ def normalized_status_category(raw: str) -> str: ) if doc_name == "README.md": - rows = README_ROW_RE.findall(text) + catalog_lines = section_lines(text, "## Catalog") + if not catalog_lines: + die("README.md is missing a non-empty ## Catalog section") + catalog_section = "\n".join(catalog_lines) + rows = README_ROW_RE.findall(catalog_section) row_ids = [row_id for row_id, _rel, _status in rows] counts = Counter(row_ids) bad_counts = sorted(record_id for record_id, count in counts.items() if count != 1) if bad_counts: die( - "README.md catalog table must index each record exactly once; " + "README.md ## Catalog table must index each record exactly once; " f"bad row counts for: {', '.join(bad_counts)}" ) missing_readme = sorted(records.keys() - counts.keys()) if missing_readme: - die(f"README.md catalog table is missing record(s): {', '.join(missing_readme)}") + die(f"README.md ## Catalog table is missing record(s): {', '.join(missing_readme)}") unknown_rows = sorted(counts.keys() - records.keys()) if unknown_rows: - die(f"README.md catalog table references unknown record(s): {', '.join(unknown_rows)}") + die(f"README.md ## Catalog table references unknown record(s): {', '.join(unknown_rows)}") row_statuses: dict[str, str] = {} for row_id, rel, raw_status in rows: if record_paths.get(rel) != row_id: - die(f"README.md row identity mismatch for {row_id}: {rel}") + die(f"README.md ## Catalog row identity mismatch for {row_id}: {rel}") if row_id in row_statuses: - die(f"README.md has duplicate status row for {row_id}") + die(f"README.md ## Catalog has duplicate status row for {row_id}") row_statuses[row_id] = normalized_status_category(raw_status) for record_id, expected_status in status_categories.items(): observed_status = row_statuses.get(record_id) if observed_status is None: - die(f"README.md has no catalog status cell for post-v1 record {record_id}") + die(f"README.md ## Catalog has no status cell for post-v1 record {record_id}") if observed_status != expected_status: die( f"README.md status mismatch for {record_id}: " From df387d7809f69c7220579e55480ab5896b9eb724 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 16:59:55 +0930 Subject: [PATCH 013/229] Harden mutation, allocator, and parallel-budget invariants --- ...NT-001-partitioned-coordination-domains.md | 22 +++++++------ ...1-signature-bound-incremental-execution.md | 32 +++++++++---------- ...E-001-bound-driven-search-space-pruning.md | 24 ++++++++------ scripts/check_catalog.py | 29 +++++++++++++++-- 4 files changed, 70 insertions(+), 37 deletions(-) mode change 100644 => 100755 scripts/check_catalog.py diff --git a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md index 6ae6b2e..3bb52e1 100644 --- a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md +++ b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md @@ -15,11 +15,11 @@ Independent workers serialize on one globally coordinated resource even though t ## Optimization problem contract -- X: target-supported shard/domain counts, namespace splits, worker-to-domain mappings, merge/aggregation policies, ownership-lease policies, and fencing-epoch schemes -- F: configurations that preserve the target's required uniqueness, exclusive ownership, visibility, failure-domain, and ordering guarantees through assignment, rebalance, restart, and split-brain recovery +- X: target-supported shard/domain counts, namespace splits, worker-to-domain mappings, merge/aggregation policies, ownership-lease policies, fencing-epoch schemes, and restart-safe allocator-state policies +- F: configurations that preserve the target's required uniqueness, exclusive ownership, visibility, failure-domain, and ordering guarantees through assignment, rebalance, same-owner restart, crash recovery, and split-brain recovery - f: measured coordination contention, tail latency, and coordination overhead under the declared workload - d: minimize under the target's predeclared objective ordering -- C: partitioning must not silently weaken any global invariant; any intentional shift from global to per-domain ordering is a separately declared contract change; mutable domain ownership transitions require one active fenced owner for each epoch +- C: partitioning must not silently weaken any global invariant; any intentional shift from global to per-domain ordering is a separately declared contract change; mutable domain ownership transitions require one active fenced owner for each epoch; allocator restart must not reuse IDs/ranges already issued before the crash - B: target-specific contention/scale/failover benchmark budget declared before tuning; no portable shard count or bit split is supplied here - S: stop when the budget is exhausted or a validated partitioning materially reduces the target bottleneck without violating C - Variables: integer / categorical / mixed @@ -27,13 +27,13 @@ Independent workers serialize on one globally coordinated resource even though t - Objective behavior: noisy under concurrent load; ownership/uniqueness invariants are deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive at target scale and during failover testing -- Constraints: uniqueness, ownership, ordering, visibility, failure-domain, lease/fencing, and resource constraints +- Constraints: uniqueness, ownership, ordering, visibility, failure-domain, lease/fencing, durable allocator-state, and resource constraints - Parallelism: concurrent / asynchronous by construction - Exactness: exact ownership/uniqueness semantics; no approximation is introduced ## Preserved contract -Partitioning must not silently weaken uniqueness, ownership, visibility or ordering guarantees. If ordering becomes per-domain rather than global, that is a contract change and must be explicit. When a domain can be reassigned, only the current fenced owner may mutate that domain; a delayed, partitioned, resumed, or split-brain previous owner must be rejected even if it still believes its old lease is valid. +Partitioning must not silently weaken uniqueness, ownership, visibility or ordering guarantees. If ordering becomes per-domain rather than global, that is a contract change and must be explicit. When a domain can be reassigned, only the current fenced owner may mutate that domain; a delayed, partitioned, resumed, or split-brain previous owner must be rejected even if it still believes its old lease is valid. A same-owner process restart is also part of the ownership contract: restarting an allocator must not reset process-local state in a way that can reissue an ID or range already made externally visible. ## Optimization @@ -41,7 +41,9 @@ Factor a global coordination space into independent domains. Encode domain ident For dynamic assignment/rebalance, use an **exclusive handoff with fencing**. A durable coordinator grants ownership together with a monotonically increasing epoch/token. Every state-changing operation that depends on domain ownership carries that epoch, and the authoritative storage/queue/allocation boundary rejects operations from epochs older than the current one. A lease alone is insufficient if an old process can resume after expiry; the fencing token must make stale writes/actions impossible at the mutation boundary. Do not activate the replacement owner until the new epoch is durably authoritative. -Where local IDs/counters are used, combine the stable domain identity with the fenced ownership epoch or another target-specific mechanism strong enough to prevent duplicate allocation across reassignment. If IDs must remain stable across ownership epochs, separate the stable domain namespace from the fencing metadata while still rejecting stale mutations. +Where local IDs/counters are used, combine the stable domain identity with a restart-safe allocation policy. Acceptable designs include: (1) a durable high-water mark advanced atomically **before** an ID/range becomes externally usable, (2) durable allocation of non-overlapping ranges/blocks so a restart resumes from a fresh unissued block and may safely burn any uncertain tail, or (3) a new durable allocator-incarnation epoch on every allocator process restart, with that incarnation identity participating in the emitted ID namespace. Merely combining the domain with the current ownership/fencing epoch is insufficient when the same owner can restart without receiving a new epoch, because a reset process-local counter can reproduce old IDs. + +If IDs must remain stable across allocator restarts and ownership epochs, use a durable monotonic counter/high-water mark or durable non-overlapping range allocator; do **not** reset an ephemeral counter. Persist/reserve advancement before returning the corresponding ID to the caller, or otherwise use a transaction whose crash semantics can prove that recovery never reissues an already-visible value. When commit status is uncertain after a crash, prefer skipping/burning an uncertain range over risking reuse. ## Before / after evidence @@ -55,16 +57,18 @@ Where local IDs/counters are used, combine the stable domain identity with the f Check global invariants across all domains, collision/duplicate behavior, restart behavior and target-scale contention profiles. +Exercise **same-owner allocator restarts** independently of reassignment. Issue IDs/ranges, crash the allocator before and after each persistence/reservation boundary, restart it under the same domain ownership/fencing epoch, and prove it never reissues an externally visible ID/range. Test crashes after durable reservation but before delivery, after delivery but before acknowledgement bookkeeping, and with uncertain commit status. For high-water counters, verify monotonic durable recovery. For block/range allocation, verify recovered allocators never enter a previously issued block and that burning an uncertain tail preserves uniqueness. For allocator-incarnation schemes, verify every restart advances the durable incarnation before any allocation is served. + Exercise **rebalance/recovery races**: pause an owner, expire/revoke it, assign a higher fencing epoch to a replacement, then resume the old owner and prove every stale mutation/allocation/queue claim is rejected. Inject network partition and split-brain conditions where both old and new processes run simultaneously. Verify only the highest authoritative epoch can mutate state, no duplicate IDs/work claims are produced, handoff is crash-recoverable, and ownership remains unique through coordinator/storage restarts. Include delayed messages from old epochs arriving after the new owner has already committed work. ## Target-repo adaptation -Shard counts and bit splits are workload-specific. Measure skew, cache locality, failure domains and merge costs. Define the durable ownership source, lease timeout if used, monotonically increasing fencing epoch/token, authoritative mutation boundary that validates epochs, handoff sequence, and restart/recovery semantics before enabling dynamic reassignment. +Shard counts and bit splits are workload-specific. Measure skew, cache locality, failure domains and merge costs. Define the durable ownership source, lease timeout if used, monotonically increasing fencing epoch/token, authoritative mutation boundary that validates epochs, handoff sequence, and restart/recovery semantics before enabling dynamic reassignment. For allocators, separately define the same-owner restart policy: durable high-water mark, durable non-overlapping range reservation, or durable per-incarnation epoch. Specify exactly which state is made durable before an ID/range can escape and how ambiguous crash outcomes are recovered without reuse. ## Failure modes -Hot shards merely move the bottleneck; domain proliferation raises memory/management overhead; rebalancing without fencing can allow stale and replacement owners to act concurrently; lease-only ownership can fail when an old process resumes; delayed old-epoch messages can duplicate allocations or queue work; identity stability may be violated; global ordering requirements may make the pattern inadmissible. +Hot shards merely move the bottleneck; domain proliferation raises memory/management overhead; rebalancing without fencing can allow stale and replacement owners to act concurrently; lease-only ownership can fail when an old process resumes; delayed old-epoch messages can duplicate allocations or queue work; a same-owner restart can reset an ephemeral local counter and reissue prior IDs even without any fencing race; persisting allocation state after delivery can create crash windows that reuse visible values; identity stability may be violated; global ordering requirements may make the pattern inadmissible. ## Rollback trigger -Revert if partitioning does not reduce measured contention, if any cross-domain invariant fails, or if failover/rebalance testing shows a stale owner or old-epoch message can mutate state after a replacement owner becomes authoritative. +Revert if partitioning does not reduce measured contention, if any cross-domain invariant fails, if failover/rebalance testing shows a stale owner or old-epoch message can mutate state after a replacement owner becomes authoritative, or if same-owner crash/restart testing can reissue any externally visible ID/range or otherwise lose durable allocator progress. diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index 1b10370..92df402 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -14,33 +14,33 @@ Expensive work is rerun even though every input capable of affecting its result ## Optimization problem contract -- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity policies, immutable-input snapshot/revalidation policies, and crash-consistent state-publication mechanisms -- F: configurations whose signature covers every output-affecting input, whose execution consumes one immutable effective-input snapshot or revalidates the complete effective-input identity before commit, whose reuse validates required outputs, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted executions never publish reusable partial state -- f: measured repeated-work cost including stage runtime plus signature/snapshot/metadata/output-validation/publication I/O overhead +- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity policies, immutable-input snapshot/mutation-control policies, monotonic mutation epochs, and crash-consistent state-publication mechanisms +- F: configurations whose signature covers every output-affecting input, whose execution consumes one immutable effective-input snapshot or is protected by a mutation lock/monotonic mutation witness that detects every intervening change, whose reuse validates required outputs, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted/raced executions never publish reusable partial state +- f: measured repeated-work cost including stage runtime plus signature/snapshot/mutation-tracking/metadata/output-validation/publication I/O overhead - d: minimize -- C: every reused output is semantically equivalent to a fresh execution for the same effective inputs, with the same failure and output-validity semantics; reuse metadata cannot mix fields from different generations; a committed generation cannot bind a pre-execution signature to output produced from changed or mixed inputs +- C: every reused output is semantically equivalent to a fresh execution for the same effective inputs, with the same failure and output-validity semantics; reuse metadata cannot mix fields from different generations; a committed generation cannot bind a pre-execution signature to output produced from changed or mixed inputs; A→B→A mutations during execution are detected rather than erased by endpoint equality - B: target-specific benchmark/evaluation budget declared before tuning; no portable value is supplied by this record - S: stop when the declared budget is exhausted or a validated configuration meets the predeclared improvement threshold without violating C -- Variables: categorical / mixed policy choices for signatures, snapshots, validation, granularity, and publication +- Variables: categorical / mixed policy choices for signatures, snapshots, validation, granularity, mutation control, and publication - Search scope: local to one incremental stage or pipeline boundary -- Objective behavior: noisy for performance; correctness identity checks are deterministic +- Objective behavior: noisy for performance; correctness identity/mutation checks are deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive depending on stage runtime and validation cost -- Constraints: semantic equivalence, integrity, crash consistency, snapshot consistency, and resource constraints -- Parallelism: sequential or pipeline-specific; publication/revalidation must remain race-safe under concurrent producers/consumers +- Constraints: semantic equivalence, integrity, crash consistency, snapshot/mutation consistency, and resource constraints +- Parallelism: sequential or pipeline-specific; mutation tracking and publication must remain race-safe under concurrent producers/consumers - Exactness: exact reuse semantics; no approximation is introduced ## Preserved contract -Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and mutable inputs must not change underneath execution without invalidating the candidate generation. +Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and mutable inputs must not change underneath execution without invalidating the candidate generation. Endpoint equality is not enough: if an input changes and later returns to its original bytes, the intervening mutation must still be observable to the publication decision. ## Optimization Compute a deterministic signature over the effective inputs and compare it with successfully persisted prior state. Reuse is allowed only when that signature still matches **and** every required output satisfies a declared validity predicate. Depending on the target, that predicate may be a content digest/version manifest, a trusted immutable/protected artifact identity, or another reproducible integrity check strong enough to detect external mutation. Mere file presence is not sufficient unless the target explicitly guarantees that reused outputs are immutable and protected from modification. Execute when the input signature differs, any required output is missing, or any output-validity check fails. -Bind execution to one coherent effective-input identity. Prefer executing against an immutable snapshot/version of every mutable effective input. If the target cannot provide such a snapshot, recompute the **complete** effective-input signature immediately before publication and require it to equal the signature used to start the candidate generation. If any effective input changed—even if it later changes back—discard the candidate generation and retry from a fresh identity; do not publish output produced from a moving or mixed input state under the old signature. +Bind execution to one coherent effective-input identity. The preferred design is an **immutable snapshot/version** of every mutable effective input. If a snapshot is unavailable, use a mechanism that records *intervening mutation*, not merely endpoint content equality: for example, hold a read/mutation lock for the full execution-to-publication interval, or capture a monotonically increasing version/epoch for every mutable input and require the exact same epoch vector at publication. Every mutation must advance its epoch durably/atomically with the mutation, including a change that later restores the original bytes. For multiple inputs, capture the snapshot/epoch vector coherently under the target's transaction/locking rules so a mixed vector cannot be mistaken for one state. A content signature recomputed at publication may supplement this check, but **must not be the sole fallback** because A→B→A can make endpoint signatures equal. Any lock violation, epoch change, incoherent snapshot, or untrackable mutable input discards the candidate generation and requires retry from a fresh identity. -Publish incremental state as one crash-consistent **generation** that binds the validated input signature to the complete output identity/validity metadata. Do not persist the signature and output metadata as independently authoritative updates. Use an atomic rename/swap of a complete manifest, a transactional store, a content-addressed generation pointer, or another mechanism where readers observe either the previous complete generation or the new complete generation—never a mixture. Only publish the new generation after every output has been produced and validated successfully **and** the input snapshot/signature has passed the final commit-time identity check; an interrupted, failed, or input-raced publication leaves the previous committed generation authoritative and the candidate generation non-reusable. +Publish incremental state as one crash-consistent **generation** that binds the validated input signature and immutable snapshot/epoch identity to the complete output identity/validity metadata. Do not persist the signature and output metadata as independently authoritative updates. Use an atomic rename/swap of a complete manifest, a transactional store, a content-addressed generation pointer, or another mechanism where readers observe either the previous complete generation or the new complete generation—never a mixture. Only publish the new generation after every output has been produced and validated successfully **and** the snapshot/mutation witness has passed the final commit-time check; an interrupted, failed, or input-raced publication leaves the previous committed generation authoritative and the candidate generation non-reusable. Reuse filesystem/configuration metadata lazily only while its own validity predicate still holds. @@ -60,18 +60,18 @@ Wonderbuild demonstrates the mechanism and benchmark shapes, but its historical Test unchanged inputs with valid outputs, changed inputs, missing outputs, failed runs, corrupted persistent state, externally overwritten/corrupted outputs, and stale output-version metadata against a forced-fresh reference path. A mutated output must force reconstruction unless the target's immutable/protected-output contract proves such mutation impossible. -Exercise **concurrent input mutation**. Start execution from identity A, mutate one or more effective inputs during execution to B (including mixed-state multi-file/config changes), and also test A→B→A change-and-revert sequences. For snapshot-based targets, prove execution reads only the immutable A snapshot. For revalidation-based targets, prove the commit-time complete signature detects any change and discards/retries the candidate instead of publishing it. Compare every accepted generation with a forced-fresh execution over the exact committed input identity. +Exercise **concurrent input mutation**, including explicit A→B→A races. Start execution from identity A, mutate one or more effective inputs during execution to B (including mixed-state multi-file/config changes), then restore the original bytes before publication. For snapshot-based targets, prove execution reads only the immutable A snapshot. For lock-based targets, prove the mutation cannot interleave with the protected execution/publication interval. For epoch/version-based targets, prove every mutation increments the monotonic witness and that the final epoch vector differs even when the final content signature returns to A. Reject/discard the candidate on any mutation witness change. Compare every accepted generation with a forced-fresh execution over the exact committed input identity. -Exercise interruption/crash injection at every publication boundary: before outputs complete, after outputs complete but before final input revalidation, after revalidation but before manifest publication, during temporary-manifest write, immediately before/after the atomic generation switch, and during cleanup. After each interruption, prove readers observe only a self-consistent old or new generation and can never pair signature A with output identities/metadata from generation B. +Exercise interruption/crash injection at every publication boundary: before outputs complete, after outputs complete but before final snapshot/epoch validation, after validation but before manifest publication, during temporary-manifest write, immediately before/after the atomic generation switch, and during cleanup. After each interruption, prove readers observe only a self-consistent old or new generation and can never pair signature A with output identities/metadata from generation B. ## Target-repo adaptation -Re-profile signature and output-validation cost, snapshot/revalidation cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Explicitly choose whether mutable inputs are consumed from immutable snapshots or protected by complete commit-time signature revalidation, whether outputs are integrity-checked on reuse or stored behind an enforceable immutable/protected boundary, and define the crash-consistency guarantee for committing the signature plus output identities. +Re-profile signature and output-validation cost, immutable-snapshot or mutation-lock/epoch cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Explicitly choose whether mutable inputs are consumed from immutable snapshots, protected by locks, or guarded by monotonic mutation epochs; define how every mutation advances the witness and how a coherent multi-input witness is captured. Do not advertise commit-time content rehashing alone as sufficient mutation detection. Also define whether outputs are integrity-checked on reuse or stored behind an enforceable immutable/protected boundary and the crash-consistency guarantee for committing the signature plus output identities. ## Failure modes -Incomplete signatures create stale reuse; input mutation during execution can bind an old signature to new/mixed output; change-and-revert races can evade presence-only checks; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. +Incomplete signatures create stale reuse; input mutation during execution can bind an old signature to new/mixed output; A→B→A races can defeat endpoint signature comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent per-input epoch reads can represent no real source state; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. ## Rollback trigger -Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference, if mutable-input races can publish a generation not tied to one coherent effective-input identity, if external output mutation can bypass the declared validity predicate, if crash/interruption testing can expose mixed-generation state, or if signature/snapshot/integrity/publication maintenance costs more than the avoided work. +Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference, if an A→B→A or other mutable-input race can publish a generation without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input identity, if external output mutation can bypass the declared validity predicate, if crash/interruption testing can expose mixed-generation state, or if signature/snapshot/mutation-tracking/integrity/publication maintenance costs more than the avoided work. diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index a9cfbe3..76697fa 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -19,17 +19,17 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F - f: a scalar real-valued target objective `f : F → R` evaluated on feasible candidates only - d: exactly one of scalar `minimize` or scalar `maximize`; vector, Pareto, lexicographic, or other partial-order objectives are outside this record unless a separately specified and validated frontier-bound mechanism is introduced -- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, the target's observable tie semantics are preserved, and budget exhaustion without a feasible incumbent produces an explicit unknown/no-incumbent outcome rather than a feasibility or optimality claim -- B: a finite, predeclared target-specific cap on evaluations, wall time, compute, or equivalent resource consumption; exact-mode search may prove optimality before this cap but may not run without a finite cap +- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, the target's observable tie semantics are preserved, and parallel dispatch cannot oversubscribe the declared hard budget +- B: a finite, predeclared target-specific **enforceable** cap on evaluations, wall time, compute, or equivalent resource consumption; parallel dispatch requires linearizable reservations before candidate/bound work starts, and every wall-time/compute reservation requires a per-operation quota/deadline/termination mechanism strong enough to prevent overrun; a resource that cannot be hard-capped must be labeled observational/best-effort rather than advertised as hard B - S: stop immediately when the required optimality/tie contract is proven or the frontier is exhausted; otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality - Variables: integer / categorical / discrete / mixed - Search scope: global over the declared candidate space - Objective behavior: deterministic unless uncertainty/noise is incorporated into a separately sound bound model - Information: derivative-free; bound/relaxation information is target-specific - Evaluation cost: moderate to expensive when exhaustive evaluation is infeasible -- Constraints: feasibility, semantic correctness, scalar-bound soundness, tie semantics, and finite-resource constraints -- Parallelism: sequential or parallel only with synchronized incumbent/frontier/bound semantics -- Exactness: exact only when the declared optimality and observable-tie contract is proven; otherwise anytime/incomplete result semantics apply +- Constraints: feasibility, semantic correctness, scalar-bound soundness, tie semantics, and enforceable finite-resource constraints +- Parallelism: sequential, or parallel only with synchronized incumbent/frontier/bound state **and linearizable budget reservation/completion accounting plus enforceable per-operation resource caps** +- Exactness: exact only when the declared optimality and observable-tie contract is proven within B; otherwise anytime/incomplete result semantics apply For each unexplored region `R`, define a bound `b(R)` separately from `f`: @@ -47,12 +47,16 @@ An independently proven infeasible region may also be pruned. A heuristic estima ## Preserved contract -A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, and exhausting B without a feasible incumbent does not permit an infeasibility claim. +A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, exhausting B without a feasible incumbent does not permit an infeasibility claim, and parallel execution must preserve the same hard resource ceiling as sequential execution rather than oversubscribing work in flight. ## Optimization Maintain an incumbent when one exists, partition the search space, compute a cheap sound `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific bound plus the target's tie semantics prove the region cannot affect the required answer. Before the first incumbent exists, sound bounds may prioritize regions or prove individual regions infeasible, but incumbent-based objective pruning is unavailable. +For **parallel** search, treat both candidate evaluation and nontrivial bound/relaxation evaluation as budget-consuming operations. Before dispatch, atomically reserve the operation's declared evaluation slot or conservative wall-time/compute quota from one shared budget ledger. If `consumed + reserved + proposed_reservation > B`, do not dispatch. Completion/failure/cancellation converts the reservation to consumed usage and releases only demonstrably unconsumed capacity under the same linearizable accounting boundary, so workers racing for the final slot cannot oversubscribe it. + +Evaluation-count budgets consume/reserve a slot before launch. For wall-time/compute budgets, each launched operation must have an enforceable per-operation upper bound—for example a deadline with forced termination, cgroup/job quota, provider/runtime cap, or equivalent mechanism. If the target cannot prevent one bound/candidate evaluation from running past the nominal reservation, wall-time/compute is **not** a hard B and must be documented as observational/best-effort instead of being used to justify finite-cap correctness claims. + A relaxed solution is evidence for a bound, not automatically a feasible final answer. ## Before / after evidence @@ -67,16 +71,18 @@ A relaxed solution is evidence for a bound, not automatically a feasible final a For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness independently by checking the direction-specific inequality against exhaustive feasible values inside each test region. Test pruning separately from search ordering. Include fixtures where the first feasible candidate is found late and where B expires before any feasible candidate exists; verify that the latter returns `no-incumbent / feasibility-unknown`, reports only independently valid global-bound information, and makes no infeasibility, optimality, or incumbent-based gap claim. Verify that budget exhaustion with an incumbent returns an anytime result without an exactness claim. +Add **parallel budget-boundary fixtures**. Race multiple workers against one remaining evaluation slot and prove only one reservation succeeds. Race bound evaluations and candidate evaluations against the same final capacity and prove both charge the declared ledger. For wall-time/compute budgets, deliberately run an operation that attempts to exceed its reservation and prove the quota/deadline/termination mechanism stops it within the enforceable cap. Race completion/cancellation with new dispatch and verify the accounting transition is linearizable—released capacity is not visible before corresponding consumption is committed, no increments are lost, and `consumed + reserved <= B` always holds for hard-budget dimensions. + Add **equal-objective tie fixtures**. For an any-one-optimum contract, prove equality pruning cannot alter any observable result. For deterministic tie-winner contracts, construct regions containing equal-objective candidates with better/worse tie ranks and prove equality-bound regions are retained until the declared tie winner is established. For all-optima contracts, prove every equal-objective optimum is enumerated. If using a stronger total-order bound, validate its soundness independently against exhaustive fixtures. ## Target-repo adaptation -The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. +The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. For parallel implementations, define one linearizable reservation/completion ledger shared by candidate and bound work, the accounting unit, per-operation reservation amount, metering source, and the enforcement mechanism for wall-time/compute quotas. Downgrade any unenforceable resource limit to best-effort/observational rather than calling it hard B. ## Failure modes -Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; an unbounded exact-search policy can consume resources indefinitely; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. +Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; parallel workers without linearizable reservations can oversubscribe the last evaluation/resource slot; an uncapped candidate/bound evaluation can exceed a nominal wall-time/compute cap before stopping logic observes it; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. ## Rollback trigger -Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort exact-mode claims whenever B is exhausted before the full objective/tie contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. +Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel mode if workers can dispatch without first reserving budget, if concurrent accounting can oversubscribe B, or if any operation can exceed a resource reservation that is claimed as a hard cap. Abort exact-mode claims whenever B is exhausted before the full objective/tie contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py old mode 100644 new mode 100755 index 2187a0d..c1da443 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -39,6 +39,16 @@ "Parallelism", "Exactness", ) +CLASSIFICATION_TEMPLATE_VALUES = { + "Variables": "continuous / integer / categorical / conditional / mixed", + "Search scope": "local / global", + "Objective behavior": "deterministic / noisy / stochastic", + "Information": "gradient / derivative-free / black-box", + "Evaluation cost": "cheap / moderate / expensive", + "Constraints": "bounds / equality / inequality / semantic / resource", + "Parallelism": "sequential / synchronous batch / asynchronous", + "Exactness": "exact / approximation permitted under explicit error contract", +} ALLOWED_V2_STATUS_CATEGORIES = { "Verified", "Verified, environment-specific", @@ -126,8 +136,14 @@ def normalized_status_category(raw: str) -> str: return plain.split(";", 1)[0].strip() -def require_prefixed_fields(path: Path, lines: list[str], fields: tuple[str, ...], section: str) -> None: - """Require exactly one non-empty '- Field:' row for every declared field.""" +def require_prefixed_fields( + path: Path, + lines: list[str], + fields: tuple[str, ...], + section: str, + rejected_values: dict[str, str] | None = None, +) -> None: + """Require exactly one selected non-empty '- Field:' row for every declared field.""" for field in fields: prefix = f"- {field}:" matches = [line for line in lines if line.startswith(prefix)] @@ -136,8 +152,14 @@ def require_prefixed_fields(path: Path, lines: list[str], fields: tuple[str, ... f"{path.relative_to(ROOT)} must contain exactly one field " f"'{prefix}' in {section}" ) - if not matches[0][len(prefix) :].strip(): + value = matches[0][len(prefix) :].strip() + if not value: die(f"{path.relative_to(ROOT)} has empty field {field} in {section}") + if rejected_values is not None and value == rejected_values.get(field): + die( + f"{path.relative_to(ROOT)} has unselected template placeholder " + f"for {field} in {section}: '{value}'" + ) records: dict[str, Path] = {} @@ -204,6 +226,7 @@ def require_prefixed_fields(path: Path, lines: list[str], fields: tuple[str, ... contract, REQUIRED_CLASSIFICATION_FIELDS, "## Optimization problem contract", + rejected_values=CLASSIFICATION_TEMPLATE_VALUES, ) missing_frozen = sorted(FROZEN_V1 - records.keys()) From b8c50688d9807146c3b1f90eb38b38a3c4f187ec Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 17:01:50 +0930 Subject: [PATCH 014/229] Select explicit adaptive-search classification --- optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index 381fe7c..e091a45 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -23,7 +23,7 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra - C: search may choose where to evaluate but may not weaken correctness, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources; every per-trial reservation must be an enforceable upper bound rather than an estimate; dispatch/completion accounting must be linearizable; and targets that require deterministic search outcomes must use deterministic observation assimilation independent of wall-clock completion order - B: an explicit target-specific hard maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit, enforceable per-trial cap mechanism, conservative reservation rule, atomic accounting boundary, and failure/cancellation charging policy are fixed before dispatch begins - S: stop proposing/dispatching when no additional trial can be safely reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger and apply stopping decisions to the declared deterministic assimilation order when determinism is required -- Variables: continuous / integer / categorical / conditional / mixed +- Variables: mixed search spaces; may include continuous, integer, categorical, and conditional dimensions as explicitly declared by the target - Search scope: local or global, explicitly declared for the target - Objective behavior: deterministic, noisy, or stochastic as declared by the target; noise treatment must be explicit - Information: derivative-free / black-box by default; gradient information may be used only when the selected target mechanism supports it From dc31113f875d11c16108418ba8b279cba9f25997 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 17:18:47 +0930 Subject: [PATCH 015/229] Harden mutation, delivery, and catalog invariants --- ...01-concurrent-duplicate-work-coalescing.md | 30 ++++++++------- ...T-CRIT-001-critical-path-prioritization.md | 20 +++++----- ...T-FAN-001-shared-materialization-fanout.md | 22 +++++------ ...1-signature-bound-incremental-execution.md | 26 +++++++------ scripts/check_catalog.py | 37 +++++++++++++++++++ 5 files changed, 90 insertions(+), 45 deletions(-) diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index f18d487..be6765f 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,11 +14,11 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime policies, waiter limits, overflow/backpressure policies, per-waiter cancellation policies, retry/error-sharing policies, and result-ownership policies -- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, and error semantics for every joined caller, bound waiter memory, and never admit new waiters to a closing or terminal generation -- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, waiter-memory, overflow/backpressure, and result-copy overhead +- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime policies, waiter limits, overflow/backpressure policies, per-waiter cancellation/deadline/terminal-claim policies, retry/error-sharing policies, and result-ownership policies +- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, and error semantics for every joined caller, linearize cancellation/deadline against terminal delivery for each waiter, bound waiter memory, and never admit new waiters to a closing or terminal generation +- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, waiter-memory, atomic terminal-claim, overflow/backpressure, and result-copy overhead - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives a result or error valid for its original request semantics, authorization scope, and ownership contract; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; closing/terminal generations are not joinable; waiter overflow has an explicit bounded behavior +- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, and deadline; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; closing/terminal generations are not joinable; waiter overflow has an explicit bounded behavior - B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied here - S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C - Variables: categorical / integer / mixed @@ -26,13 +26,13 @@ Many callers request the same expensive computation concurrently before any call - Objective behavior: noisy under concurrent load; semantic equivalence remains deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive concurrent-load testing -- Constraints: semantic equivalence, authorization, ownership, waiter-memory, cancellation, timeout, and resource constraints +- Constraints: semantic equivalence, authorization, ownership, waiter-memory, cancellation, deadline, terminal-claim, timeout, and resource constraints - Parallelism: asynchronous / concurrent - Exactness: exact request/result semantics; no approximation is introduced ## Preserved contract -Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout, result-ownership, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. A configured waiter bound must never be exceeded silently. +Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout/deadline, result-ownership, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. A configured waiter bound must never be exceeded silently. Each waiter reaches exactly one linearized terminal state; a waiter that has already cancelled or timed out cannot later receive the shared value/error. ## Optimization @@ -42,11 +42,13 @@ Atomically create the joinable generation **with the initiating caller already r Equivalent later callers may register as independent waiters only while the generation is joinable and the configured waiter capacity remains. Waiter admission is atomic with capacity accounting. When the final waiter slot is already occupied, apply one explicit target policy rather than silently exceeding the bound: reject/return a documented overload or retryable-backpressure result, block/queue the caller behind a separately bounded admission mechanism, or use another bounded policy with explicit timeout/cancellation semantics. Starting an unconstrained parallel generation for the same equivalence key is not the default overflow behavior because it recreates the duplicate upstream load this pattern is intended to prevent. If a target deliberately permits overflow generations, that concurrency bound and duplicate-work tradeoff must be part of C/B and validated separately. -Cancellation and timeout are per waiter: when one waiter leaves, remove only that waiter. If live waiters remain, keep the shared generation joinable. If the last waiter leaves and the policy calls for upstream cancellation, atomically mark the registry entry **closing/non-joinable** (or remove it from the joinable map) before sending the asynchronous cancellation request upstream. A new caller arriving after that transition must create a fresh generation rather than attach to work already being canceled. The closing generation may remain internally tracked until its terminal completion for cleanup/accounting, but it is not eligible for coalescing. +Represent each admitted waiter with an atomic terminal state, initially `pending`. Cancellation attempts atomically claim `pending -> cancelled`; timeout/deadline handling atomically claims `pending -> timed-out`. A terminal result/error notifier may claim `pending -> delivered-success` or `pending -> delivered-error` only if the waiter's declared deadline has not expired at the claim point. If the deadline is already expired, the notifier must instead leave/transition that waiter to the target's timed-out state and must not deliver the shared terminal value/error. For explicit cancellation racing completion, whichever atomic transition claims `pending` first wins; the losing transition is a no-op for that waiter. These claim semantics are part of the public request contract and must not depend on scheduler timing after the claim. -On success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the terminal waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Then snapshot the waiters still registered to that terminal generation. +Cancellation and timeout are otherwise per waiter: when one waiter leaves through a winning cancellation/timeout claim, remove only that waiter from the live-waiter accounting. If live waiters remain, keep the shared generation joinable. If the last live waiter leaves and the policy calls for upstream cancellation, atomically mark the registry entry **closing/non-joinable** (or remove it from the joinable map) before sending the asynchronous cancellation request upstream. A new caller arriving after that transition must create a fresh generation rather than attach to work already being canceled. The closing generation may remain internally tracked until its terminal completion for cleanup/accounting, but it is not eligible for coalescing. -Define result ownership explicitly. If the terminal value is immutable/share-safe under the target API, the same immutable value may be delivered to all authorized waiters. If callers normally receive mutable or caller-owned results, create an independent defensive clone/copy/copy-on-write handle for each waiter before delivery so one caller cannot observably mutate another caller's result. Deliver the shared terminal error (or per-caller wrapped equivalent where the API requires ownership/context) to the terminal waiter snapshot, then retire/clean up the generation deterministically. +On upstream success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the candidate waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Snapshot the waiter records, but do not treat membership in that snapshot as entitlement to delivery: for each waiter, perform the atomic per-waiter terminal claim described above immediately before delivery. A waiter whose cancellation/timeout claim already won is skipped. + +Define result ownership explicitly. If a terminal-success claim wins and the terminal value is immutable/share-safe under the target API, the same immutable value may be delivered to all authorized success-claimed waiters. If callers normally receive mutable or caller-owned results, create an independent defensive clone/copy/copy-on-write handle for each waiter after its successful terminal claim and before delivery so one caller cannot observably mutate another caller's result. Deliver a terminal failure only to waiters whose `delivered-error` claim wins (or a per-caller wrapped equivalent where the API requires ownership/context), then retire/clean up the generation deterministically. Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. @@ -64,7 +66,9 @@ This differs from caching: the reusable result does not exist yet. Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; race a new caller against the last-waiter cancellation transition and prove it never joins the closing generation; race a new caller against success/failure completion and prove the terminal generation is made non-joinable before waiter snapshot/notification; test waiter-specific deadlines; verify shared failure delivery and retry accounting; prove only one upstream evaluation occurs per joinable generation while all surviving callers terminate correctly. -Add an **immediate synchronous-completion** fixture where the upstream operation can finish inline at launch. Prove the initiating caller was already registered before launch and always receives the terminal result/error. +Add **terminal-delivery races** for both upstream success and upstream failure. Pause after the terminal waiter snapshot, then race explicit cancellation and deadline expiry against each waiter's delivery claim. Prove exactly one `pending -> terminal` transition wins, cancelled/timed-out waiters never receive a later value/error, completion that legitimately claims before cancellation preserves the declared completion result, and an already-expired deadline cannot be bypassed merely because the timeout worker has not run yet. Repeat under high concurrency and verify no waiter observes two terminal outcomes. + +Add an **immediate synchronous-completion** fixture where the upstream operation can finish inline at launch. Prove the initiating caller was already registered before launch and always receives the terminal result/error unless its own cancellation/deadline claim wins under the same rules. Add authorization-boundary fixtures: issue syntactically identical requests under different tenants, principals, roles, ACL/visibility scopes, or other authorization context. Prove they either map to different equivalence keys **or** that the shared upstream result is explicitly safe to reuse and each caller is independently authorized before delivery. Verify that a result produced under one authorization scope can never leak to another merely because the resource parameters match. @@ -74,12 +78,12 @@ Add waiter-overflow races: fill the waiter list to one slot below the maximum, l ## Target-repo adaptation -Define key canonicalization, the authorization/visibility context that participates in equivalence, maximum waiter count, bounded overflow/backpressure semantics, result ownership/share-safety policy, per-waiter cancellation/deadline handling, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. +Define key canonicalization, the authorization/visibility context that participates in equivalence, maximum waiter count, bounded overflow/backpressure semantics, result ownership/share-safety policy, per-waiter atomic terminal-state representation, cancellation/deadline winning semantics, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent or authorization-distinct work; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; snapshotting waiters before terminal closure can strand a late joiner; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. +Over-broad keys merge non-equivalent or authorization-distinct work; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; non-linearized cancellation/deadline versus delivery can produce late values/errors or double terminal outcomes; treating terminal snapshot membership as delivery entitlement can notify a waiter after it has timed out; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes any caller's authorization/cancellation/result/ownership/error semantics, merges authorization-distinct requests without independent delivery authorization, permits one caller to cancel work required by another, strands the initiating caller on immediate completion, allows a new caller to join a closing or terminal generation, exceeds the configured waiter bound, violates documented overflow/backpressure behavior, permits mutable-result aliasing across callers, leaks orphaned shared operations, increases tail latency materially, or creates unacceptable failure amplification. +Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/error semantics; if a cancelled/timed-out waiter can receive a later terminal value/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if the waiter bound or documented overflow behavior is violated; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. diff --git a/optimizations/OPT-CRIT-001-critical-path-prioritization.md b/optimizations/OPT-CRIT-001-critical-path-prioritization.md index a8e44db..1f1727e 100644 --- a/optimizations/OPT-CRIT-001-critical-path-prioritization.md +++ b/optimizations/OPT-CRIT-001-critical-path-prioritization.md @@ -15,11 +15,11 @@ Non-critical work competes with the dependency chain that determines user-visibl ## Optimization problem contract -- X: target-supported task-priority, prefetch/precompute, lazy/deferred-work, speculation, speculative-input identity, and commitment/revalidation policies -- F: policies that preserve all semantic deadlines, avoid externally visible speculative side effects before commitment, commit speculative results only for matching effective inputs, and satisfy starvation/resource constraints +- X: target-supported task-priority, prefetch/precompute, lazy/deferred-work, speculation, speculative-input identity, mutation-control, and commitment policies +- F: policies that preserve all semantic deadlines, avoid externally visible speculative side effects before commitment, commit speculative results only from one stable effective-input generation, and satisfy starvation/resource constraints - f: measured end-to-end latency of the declared critical dependency path, including resource pressure introduced by speculation/deferment - d: minimize -- C: critical outputs and semantic deadlines are preserved; speculative work is safely discardable; any speculative result is bound to the complete effective-input identity and revalidated at commitment/delivery; deferred work completes before it becomes semantically required +- C: critical outputs and semantic deadlines are preserved; speculative work is safely discardable; any speculative result is bound to a complete immutable snapshot or full-duration mutation witness so intervening A→B→A changes cannot be erased before commitment; deferred work completes before it becomes semantically required - B: target-specific trace/benchmark budget covering cold/warm, hit/miss, wrong-speculation, stale-speculation, and change/revert cases; no portable prediction horizon is supplied here - S: stop when the declared budget is exhausted or a validated policy materially reduces critical-path latency without violating C - Variables: categorical / conditional / mixed priority, deferment, prefetch, and speculation policies @@ -27,19 +27,19 @@ Non-critical work competes with the dependency chain that determines user-visibl - Objective behavior: noisy under realistic workload timing; semantic identity/deadline checks are deterministic - Information: derivative-free / black-box latency measurements - Evaluation cost: moderate to expensive end-to-end tracing/benchmarking -- Constraints: semantic deadlines, starvation, side effects, input identity/freshness, memory/CPU/I/O, and target resource constraints +- Constraints: semantic deadlines, starvation, side effects, input identity/mutation freshness, memory/CPU/I/O, and target resource constraints - Parallelism: asynchronous / concurrent execution is common - Exactness: exact target semantics; speculative work may be discarded but not committed stale ## Preserved contract -Deferred work must still complete before its semantic deadline. Speculative work must be discardable and must not create externally visible side effects before commitment. A speculative result may be committed/delivered only if it still corresponds to the complete current effective-input identity required by the non-speculative reference path. +Deferred work must still complete before its semantic deadline. Speculative work must be discardable and must not create externally visible side effects before commitment. A speculative result may be committed/delivered only if it was produced from one coherent effective-input generation equivalent to the non-speculative reference path; endpoint equality after an intervening mutation is not sufficient. ## Optimization Execute critical dependencies first; prefetch/precompute likely-soon work only when probability and spare resources justify it; lazily defer non-critical work; avoid work with no demonstrated demand. -Bind every speculative/precomputed result to a complete effective-input identity or immutable snapshot. Prefer speculation against an immutable version/snapshot. Otherwise, immediately before commitment/delivery, recompute or reauthenticate the complete effective-input identity and require it to match the identity under which the speculative result was produced. Presence alone is never a freshness proof. If any relevant input changed while speculation was in flight—even if it later changed back A→B→A unless the target can prove one stable A snapshot was consumed—discard the speculative result and execute/recompute from the current reference identity. Commitment is the semantic boundary: no stale speculative result may become externally visible merely because the speculation itself had no side effects. +Bind every speculative/precomputed result to a complete effective-input identity for the **full speculation-to-commit interval**. Prefer speculation against an immutable snapshot/version. If snapshots are unavailable, use a full-duration mutation/read lock or capture a monotonically increasing, non-reusable version/epoch for every mutable effective input and require the same coherent epoch vector at commitment/delivery. Every relevant mutation must advance its witness, including A→B→A changes that restore original bytes. A commit-time hash/identity comparison may supplement the mutation witness but must not be the sole freshness proof. Any lock violation, epoch change, incoherent witness, or untrackable mutable input discards the speculative result and forces execution/recomputation from the current reference identity. Commitment is the semantic boundary: no stale speculative result may become externally visible merely because the speculation itself had no side effects. ## Before / after evidence @@ -53,16 +53,16 @@ Bind every speculative/precomputed result to a complete effective-input identity Trace the true dependency path and measure end-to-end latency, not only individual task duration. Test cold/warm, cache-hit/miss and wrong-speculation cases. Explicitly test semantic deadlines, starvation, cancellation, and that speculative work cannot expose side effects before commitment. -Add stale-speculation fixtures: start speculation from input identity A, mutate the effective inputs to B before demand/commitment, and verify A is discarded. Include A→B→A change-and-revert races, delayed speculative completion, version rollback, and concurrent config/schema changes. Compare every committed speculative result against the non-speculative reference path for the exact committed identity, and prove commitment/delivery performs the declared revalidation or uses an immutable snapshot strong enough to make revalidation unnecessary. +Add stale-speculation fixtures with explicit **A→B→A** races. Start speculation from identity A, mutate the effective inputs to B while speculation reads/runs, then restore original bytes before demand/commitment. For snapshot-based targets, prove speculation consumed only immutable A. For lock-based targets, prove the mutation cannot interleave. For epoch/version-based targets, prove every mutation increments the monotonic witness and that the final witness exposes the intervening change even though endpoint content equals A. Also test delayed speculative completion, version rollback, and concurrent config/schema changes. Compare every committed speculative result against the non-speculative reference path for the exact committed identity. ## Target-repo adaptation -Criticality and prediction horizons are workload-specific. Re-profile after topology or user-flow changes. Define the complete effective-input identity for each speculative result, choose immutable snapshots or commit-time revalidation, and specify exactly when a stale speculative result is discarded. +Criticality and prediction horizons are workload-specific. Re-profile after topology or user-flow changes. Define the complete effective-input identity for each speculative result and choose immutable snapshots, full-duration mutation locks, or monotonic epochs that record every intervening change. Specify exactly when a stale speculative result is discarded. Do not rely on commit-time endpoint revalidation alone to detect change-and-revert races. ## Failure modes -Speculation steals resources from critical work, lazy work causes later latency cliffs, priorities become stale, deferred tasks starve, semantic deadlines are missed, speculative side effects escape before commitment, or stale speculative output is committed after its effective inputs changed. +Speculation steals resources from critical work, lazy work causes later latency cliffs, priorities become stale, deferred tasks starve, semantic deadlines are missed, speculative side effects escape before commitment, A→B→A mutations can fool endpoint-only freshness checks, non-monotonic/reused epochs can erase intervening changes, or stale speculative output is committed after its effective inputs changed. ## Rollback trigger -Immediately disable/revert the policy on any violation of C, including a required task missing its semantic deadline, speculative work exposing an externally visible side effect before commitment, or a speculative result being committed/delivered without matching the current effective-input identity. Also disable it if critical-path latency or resource pressure worsens materially. +Immediately disable/revert the policy on any violation of C, including a required task missing its semantic deadline, speculative work exposing an externally visible side effect before commitment, or a speculative result being committed/delivered without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input generation. Also disable it if critical-path latency or resource pressure worsens materially. diff --git a/optimizations/OPT-FAN-001-shared-materialization-fanout.md b/optimizations/OPT-FAN-001-shared-materialization-fanout.md index 57147d4..c1ef8f2 100644 --- a/optimizations/OPT-FAN-001-shared-materialization-fanout.md +++ b/optimizations/OPT-FAN-001-shared-materialization-fanout.md @@ -14,11 +14,11 @@ The same deterministic transformation is repeated independently for each consume ## Optimization problem contract -- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, immutable-source snapshot/revalidation policies, and crash-consistent publication schemes -- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, source-snapshot consistency, and publication-atomicity requirement +- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, immutable-source snapshot/mutation-control policies, monotonic source/config epochs, and crash-consistent publication schemes +- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, source-snapshot/mutation-consistency, and publication-atomicity requirement - f: measured transformation CPU, replay CPU, fan-out latency, and storage/I/O overhead under the target's declared objective ordering - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state binds the artifact bytes to one coherent effective source/transform identity; verification/security metadata may be removed only under an explicit contract change +- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state binds the artifact bytes to one coherent effective source/transform identity and no intervening mutable-input change can be erased by endpoint equality; verification/security metadata may be removed only under an explicit contract change - B: target-specific fan-out/replay benchmark budget declared before tuning; no portable subscriber count, replay size, or retention duration is supplied here - S: stop when the declared budget is exhausted or a validated materialization policy materially improves the target objective without violating C - Variables: categorical / integer / mixed @@ -26,13 +26,13 @@ The same deterministic transformation is repeated independently for each consume - Objective behavior: noisy for performance; transformation identity/equivalence is deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive depending on transform/replay size -- Constraints: semantic equivalence, source-snapshot consistency, integrity, versioning, trust/security, storage, and crash-consistency constraints +- Constraints: semantic equivalence, source-snapshot/mutation consistency, integrity, versioning, trust/security, storage, and crash-consistency constraints - Parallelism: concurrent fan-out/replay; publication must remain race-safe - Exactness: exact representation semantics; no approximation is introduced ## Preserved contract -Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive source mutation, crashes, and interrupted publication. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. +Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive source mutation, change-and-revert races, crashes, and interrupted publication. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. ## Optimization @@ -40,11 +40,11 @@ Perform an expensive deterministic transform once near production, persist or re Define a materialization key that covers, as applicable, source object/content identity or immutable source version, transformation/encoder implementation identity, encoder configuration and dictionaries, schema/format version, feature flags, trust/security policy, and any other effective input that can affect the materialized result. -Bind the transform to one coherent source identity. Prefer reading from an immutable source snapshot/version captured together with the materialization key. If the target cannot provide an immutable snapshot, recompute/re-authenticate the **complete** effective-input materialization key immediately before commit and require it to equal the key used to start the transform. Any source/config/transform identity change during execution invalidates the candidate materialization; discard/retry it rather than publishing bytes produced from mixed or newer state under an older key. Change-and-revert (A→B→A) is still a mutation event unless the target can prove the transform observed one stable A snapshot throughout. +Bind the transform to one coherent source identity for the **entire transform-to-commit interval**. Prefer reading every mutable effective input from an immutable snapshot/version captured together with the materialization key. If immutable snapshots are unavailable, use a mechanism that records intervening mutation rather than comparing endpoint content alone: hold an appropriate mutation/read lock for the full interval, or capture a monotonically increasing, non-reusable version/epoch for each mutable source/config/transform input and require the same coherent epoch vector at commit. Every mutation must advance its witness durably/atomically with the mutation, including A→B→A changes that restore the original bytes. A final content/key recomputation may supplement this witness but must not be the sole protection. Any lock violation, epoch change, incoherent multi-input witness, or untrackable mutable input invalidates the candidate materialization; discard/retry it rather than publishing mixed-state bytes. Publish the artifact and its identity as **one committed state**. Acceptable designs include content-addressed storage where the artifact digest is itself part of the committed key, an atomically replaced manifest that contains both the full materialization key and the artifact digest/location, or another crash-consistent transaction that makes old state or new state visible but never a mixed pair. Do not update artifact bytes and their key independently in a way that can expose a new artifact with stale metadata or stale bytes with a new key after a crash. -Before reuse, require: (1) exact agreement with the current effective-input materialization key, (2) a committed manifest/content-address relation that binds that key to the artifact identity, and (3) artifact integrity/format validity. A key mismatch, missing/incomplete publication marker, digest mismatch, unverifiable artifact, or failed commit-time source revalidation is a cache miss and requires regeneration. Do not use format validation alone as evidence that an artifact corresponds to current inputs. +Before reuse, require: (1) exact agreement with the current effective-input materialization key and immutable snapshot/monotonic mutation identity, (2) a committed manifest/content-address relation that binds that identity to the artifact identity, and (3) artifact integrity/format validity. A key mismatch, mutation-epoch mismatch, missing/incomplete publication marker, digest mismatch, or unverifiable artifact is a cache miss and requires regeneration. Do not use format validation or endpoint key equality alone as evidence that an artifact corresponds to one stable input generation. ## Before / after evidence @@ -58,18 +58,18 @@ Before reuse, require: (1) exact agreement with the current effective-input mate Compare shared materialization against per-consumer reference output, including corruption, restart/replay and mixed consumer capabilities. Independently mutate each key component—source content/version, transform implementation, encoder options/dictionary, schema/format version, feature flags and trust policy—and prove that each output-affecting change invalidates reuse. Also test unchanged-key reuse, tampered artifacts with matching metadata, and migration/version-boundary cases. -Exercise **concurrent source mutation**. Start a transform from source identity A, mutate the source/effective transform inputs to B while work is running, and test A→B→A change-and-revert sequences. For snapshot-based targets, prove the transform reads only the immutable A snapshot. For revalidation-based targets, prove the final complete-key comparison rejects/discards any candidate whose effective inputs changed while the transform was executing. Compare accepted materializations with a fresh transform from the exact committed source identity. +Exercise **concurrent source mutation**, including explicit A→B→A races. Start a transform from source identity A, mutate one or more source/config/transform inputs to B while the transform is running, then restore the original bytes before commit. For snapshot-based targets, prove the transform reads only the immutable A snapshot. For lock-based targets, prove mutation cannot interleave with the protected transform/publication interval. For epoch/version-based targets, prove every mutation advances the monotonic witness and the final witness differs even when the final content/key returns to A. Reject/discard the candidate on any witness change and compare every accepted materialization with a fresh transform from the exact committed source identity. Inject crashes/interruption at every publication boundary: after artifact write but before manifest commit, after provisional metadata write, during atomic replacement, and immediately after commit. After restart, prove that readers see either the previous valid committed materialization or the new valid committed materialization, never a mixed key/artifact state. Verify digest/key mismatch is rejected even when the artifact is otherwise parseable. ## Target-repo adaptation -Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, specify whether mutable source inputs are consumed from immutable snapshots or protected by complete commit-time key revalidation, and specify representation versioning, invalidation, integrity checking, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. +Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, specify whether mutable inputs are consumed from immutable snapshots, protected by full-duration locks, or guarded by monotonic mutation epochs, and define how a coherent multi-input witness is captured. Specify representation versioning, invalidation, integrity checking, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. Do not advertise commit-time endpoint rehashing alone as sufficient mutation detection. ## Failure modes -Incomplete keys can serve stale representations after source or transform changes; mutable sources can change during transformation and produce mixed-state output under a stale key; change-and-revert races can fool naive identity checks; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. +Incomplete keys can serve stale representations after source or transform changes; mutable sources can change during transformation and produce mixed-state output under a stale key; A→B→A races can defeat endpoint key comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent epoch vectors can describe no real source state; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. ## Rollback trigger -Disable reuse immediately if any materialization-key hit, source-mutation race, publication-recovery path, or integrity check can return output that differs from a fresh transform for the same exact committed effective inputs, or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. +Disable reuse immediately if any materialization-key hit, source-mutation race, publication-recovery path, or integrity check can return output that differs from a fresh transform for the same exact committed effective inputs; if an A→B→A race can evade the snapshot/lock/monotonic mutation witness; or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index 92df402..c0e3815 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -14,30 +14,32 @@ Expensive work is rerun even though every input capable of affecting its result ## Optimization problem contract -- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity policies, immutable-input snapshot/mutation-control policies, monotonic mutation epochs, and crash-consistent state-publication mechanisms -- F: configurations whose signature covers every output-affecting input, whose execution consumes one immutable effective-input snapshot or is protected by a mutation lock/monotonic mutation witness that detects every intervening change, whose reuse validates required outputs, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted/raced executions never publish reusable partial state -- f: measured repeated-work cost including stage runtime plus signature/snapshot/mutation-tracking/metadata/output-validation/publication I/O overhead +- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity/consumption policies, immutable-output handles, immutable-input snapshot/mutation-control policies, monotonic mutation epochs, and crash-consistent state-publication mechanisms +- F: configurations whose signature covers every output-affecting input, whose execution consumes one immutable effective-input snapshot or is protected by a mutation lock/monotonic mutation witness that detects every intervening change, whose reuse validates required outputs and binds downstream consumption to the exact validated output versions, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted/raced executions never publish reusable partial state +- f: measured repeated-work cost including stage runtime plus signature/snapshot/mutation-tracking/metadata/output-validation/output-pinning/publication I/O overhead - d: minimize -- C: every reused output is semantically equivalent to a fresh execution for the same effective inputs, with the same failure and output-validity semantics; reuse metadata cannot mix fields from different generations; a committed generation cannot bind a pre-execution signature to output produced from changed or mixed inputs; A→B→A mutations during execution are detected rather than erased by endpoint equality +- C: every reused output consumed downstream is the same immutable/versioned output instance whose validity predicate passed, and is semantically equivalent to a fresh execution for the same effective inputs with the same failure semantics; reuse metadata cannot mix fields from different generations; a committed generation cannot bind a pre-execution signature to output produced from changed or mixed inputs; A→B→A mutations during execution are detected rather than erased by endpoint equality - B: target-specific benchmark/evaluation budget declared before tuning; no portable value is supplied by this record - S: stop when the declared budget is exhausted or a validated configuration meets the predeclared improvement threshold without violating C -- Variables: categorical / mixed policy choices for signatures, snapshots, validation, granularity, mutation control, and publication +- Variables: categorical / mixed policy choices for signatures, snapshots, validation, output pinning, granularity, mutation control, and publication - Search scope: local to one incremental stage or pipeline boundary - Objective behavior: noisy for performance; correctness identity/mutation checks are deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive depending on stage runtime and validation cost -- Constraints: semantic equivalence, integrity, crash consistency, snapshot/mutation consistency, and resource constraints -- Parallelism: sequential or pipeline-specific; mutation tracking and publication must remain race-safe under concurrent producers/consumers +- Constraints: semantic equivalence, input/output integrity, crash consistency, snapshot/mutation consistency, validated-output consumption, and resource constraints +- Parallelism: sequential or pipeline-specific; mutation tracking, output pinning, and publication must remain race-safe under concurrent producers/consumers - Exactness: exact reuse semantics; no approximation is introduced ## Preserved contract -Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and mutable inputs must not change underneath execution without invalidating the candidate generation. Endpoint equality is not enough: if an input changes and later returns to its original bytes, the intervening mutation must still be observable to the publication decision. +Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and mutable inputs must not change underneath execution without invalidating the candidate generation. Endpoint equality is not enough: if an input changes and later returns to its original bytes, the intervening mutation must still be observable to the publication decision. Likewise, validating a mutable output path is not enough unless downstream consumption is pinned to that exact validated version. ## Optimization Compute a deterministic signature over the effective inputs and compare it with successfully persisted prior state. Reuse is allowed only when that signature still matches **and** every required output satisfies a declared validity predicate. Depending on the target, that predicate may be a content digest/version manifest, a trusted immutable/protected artifact identity, or another reproducible integrity check strong enough to detect external mutation. Mere file presence is not sufficient unless the target explicitly guarantees that reused outputs are immutable and protected from modification. Execute when the input signature differs, any required output is missing, or any output-validity check fails. +Treat output validation and output consumption as one identity-bound operation. A successful validity check must yield or pin the exact immutable/versioned output handle that downstream consumers will read: for example a content-addressed object, immutable artifact/version ID, snapshot handle, open file descriptor tied to a protected inode/version where the platform guarantees the needed semantics, or another target-specific stable handle. Do **not** validate bytes at a mutable pathname/object name and then later reopen that name for consumption, because another writer may replace it between validation and read. If the storage system cannot provide an immutable/versioned handle, hold an appropriate lock from validation through the downstream read/consumption, or copy the validated bytes into an immutable snapshot and consume that snapshot. Every consumer on a reuse hit must be bound to the validated handle/version, not merely to the same logical path. + Bind execution to one coherent effective-input identity. The preferred design is an **immutable snapshot/version** of every mutable effective input. If a snapshot is unavailable, use a mechanism that records *intervening mutation*, not merely endpoint content equality: for example, hold a read/mutation lock for the full execution-to-publication interval, or capture a monotonically increasing version/epoch for every mutable input and require the exact same epoch vector at publication. Every mutation must advance its epoch durably/atomically with the mutation, including a change that later restores the original bytes. For multiple inputs, capture the snapshot/epoch vector coherently under the target's transaction/locking rules so a mixed vector cannot be mistaken for one state. A content signature recomputed at publication may supplement this check, but **must not be the sole fallback** because A→B→A can make endpoint signatures equal. Any lock violation, epoch change, incoherent snapshot, or untrackable mutable input discards the candidate generation and requires retry from a fresh identity. Publish incremental state as one crash-consistent **generation** that binds the validated input signature and immutable snapshot/epoch identity to the complete output identity/validity metadata. Do not persist the signature and output metadata as independently authoritative updates. Use an atomic rename/swap of a complete manifest, a transactional store, a content-addressed generation pointer, or another mechanism where readers observe either the previous complete generation or the new complete generation—never a mixture. Only publish the new generation after every output has been produced and validated successfully **and** the snapshot/mutation witness has passed the final commit-time check; an interrupted, failed, or input-raced publication leaves the previous committed generation authoritative and the candidate generation non-reusable. @@ -60,18 +62,20 @@ Wonderbuild demonstrates the mechanism and benchmark shapes, but its historical Test unchanged inputs with valid outputs, changed inputs, missing outputs, failed runs, corrupted persistent state, externally overwritten/corrupted outputs, and stale output-version metadata against a forced-fresh reference path. A mutated output must force reconstruction unless the target's immutable/protected-output contract proves such mutation impossible. +Exercise an **output validation-to-consumption race**. Arrange a reuse hit for output A, validate A successfully, then have another writer replace the mutable path/object with B before the consumer reads. Prove the consumer still reads the pinned immutable/versioned A that was validated, or prove the lock prevents replacement until consumption completes. Repeat with delete/recreate, atomic rename, symlink/object-pointer replacement, version rollback, and multiple required outputs where one is swapped after validation. A test that merely corrupts output before validation is insufficient; the mutation must occur after the validity predicate succeeds and before/downstream consumption. + Exercise **concurrent input mutation**, including explicit A→B→A races. Start execution from identity A, mutate one or more effective inputs during execution to B (including mixed-state multi-file/config changes), then restore the original bytes before publication. For snapshot-based targets, prove execution reads only the immutable A snapshot. For lock-based targets, prove the mutation cannot interleave with the protected execution/publication interval. For epoch/version-based targets, prove every mutation increments the monotonic witness and that the final epoch vector differs even when the final content signature returns to A. Reject/discard the candidate on any mutation witness change. Compare every accepted generation with a forced-fresh execution over the exact committed input identity. Exercise interruption/crash injection at every publication boundary: before outputs complete, after outputs complete but before final snapshot/epoch validation, after validation but before manifest publication, during temporary-manifest write, immediately before/after the atomic generation switch, and during cleanup. After each interruption, prove readers observe only a self-consistent old or new generation and can never pair signature A with output identities/metadata from generation B. ## Target-repo adaptation -Re-profile signature and output-validation cost, immutable-snapshot or mutation-lock/epoch cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Explicitly choose whether mutable inputs are consumed from immutable snapshots, protected by locks, or guarded by monotonic mutation epochs; define how every mutation advances the witness and how a coherent multi-input witness is captured. Do not advertise commit-time content rehashing alone as sufficient mutation detection. Also define whether outputs are integrity-checked on reuse or stored behind an enforceable immutable/protected boundary and the crash-consistency guarantee for committing the signature plus output identities. +Re-profile signature and output-validation cost, immutable-output handle/pinning cost, immutable-input snapshot or mutation-lock/epoch cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Define how a successful output-validity check returns/pins the exact version consumed downstream; if mutable storage is unavoidable, define the lock scope or immutable-copy boundary. Explicitly choose whether mutable inputs are consumed from immutable snapshots, protected by locks, or guarded by monotonic mutation epochs; define how every mutation advances the witness and how a coherent multi-input witness is captured. Do not advertise commit-time content rehashing alone as sufficient mutation detection. Also define the crash-consistency guarantee for committing the signature plus output identities. ## Failure modes -Incomplete signatures create stale reuse; input mutation during execution can bind an old signature to new/mixed output; A→B→A races can defeat endpoint signature comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent per-input epoch reads can represent no real source state; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. +Incomplete signatures create stale reuse; input mutation during execution can bind an old signature to new/mixed output; A→B→A races can defeat endpoint signature comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent per-input epoch reads can represent no real source state; validating a mutable output name and reopening it later can consume different unvalidated bytes; output-version handles that are not actually immutable/pinned can create time-of-check/time-of-use reuse bugs; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. ## Rollback trigger -Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference, if an A→B→A or other mutable-input race can publish a generation without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input identity, if external output mutation can bypass the declared validity predicate, if crash/interruption testing can expose mixed-generation state, or if signature/snapshot/mutation-tracking/integrity/publication maintenance costs more than the avoided work. +Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference; if a consumer can read bytes/objects different from the exact output version whose validity predicate passed; if an A→B→A or other mutable-input race can publish a generation without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input identity; if external output mutation can bypass the declared validity/consumption binding; if crash/interruption testing can expose mixed-generation state; or if signature/snapshot/mutation-tracking/output-pinning/integrity/publication maintenance costs more than the avoided work. diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index c1da443..7ca3d06 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -84,6 +84,10 @@ r"\s*\|[^|]*\|\s*([^|]+?)\s*\|", re.MULTILINE, ) +CATALOG_DECISION_ROW_RE = re.compile( + r"^\|[^|\n]*\|\s*\[(OPT-[A-Z]+-\d{3})\]\((optimizations/[^)#]+\.md)\)\s*\|", + re.MULTILINE, +) CANONICAL_DEFINITION_PATTERNS = { "X": re.compile(r"^- `X` — \S"), "F": re.compile(r"^- `F(?: ⊆ X)?` — \S"), @@ -302,6 +306,39 @@ def require_prefixed_fields( if record_id not in catalog_ids: die(f"{record_id} ({path.name}) is not mentioned in CATALOG.md") +# The quick decision table is a distinct advertised decision surface. Mentions +# in later descriptive sections must not be allowed to mask a missing table row. +decision_lines = section_lines(catalog, "## Quick decision table") +if not decision_lines: + die("CATALOG.md is missing a non-empty ## Quick decision table section") +decision_section = "\n".join(decision_lines) +decision_rows = CATALOG_DECISION_ROW_RE.findall(decision_section) +decision_ids = [record_id for record_id, _rel in decision_rows] +decision_counts = Counter(decision_ids) +bad_decision_counts = sorted( + record_id for record_id, count in decision_counts.items() if count != 1 +) +if bad_decision_counts: + die( + "CATALOG.md ## Quick decision table must index each record exactly once; " + f"bad row counts for: {', '.join(bad_decision_counts)}" + ) +missing_decision = sorted(records.keys() - decision_counts.keys()) +if missing_decision: + die( + "CATALOG.md ## Quick decision table is missing record(s): " + f"{', '.join(missing_decision)}" + ) +unknown_decision = sorted(decision_counts.keys() - records.keys()) +if unknown_decision: + die( + "CATALOG.md ## Quick decision table references unknown record(s): " + f"{', '.join(unknown_decision)}" + ) +for row_id, rel in decision_rows: + if record_paths.get(rel) != row_id: + die(f"CATALOG.md ## Quick decision table row identity mismatch for {row_id}: {rel}") + problem_contract = ROOT / "OPTIMIZATION-PROBLEM.md" if not problem_contract.is_file(): die("OPTIMIZATION-PROBLEM.md is missing") From 3d3efff2fdb980295bb50bd2f195fdea6337b8fb Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 17:52:46 +0930 Subject: [PATCH 016/229] Harden delivery preparation and rendered catalog checks --- ...01-concurrent-duplicate-work-coalescing.md | 30 +++-- ...T-FAN-001-shared-materialization-fanout.md | 24 ++-- ...SEARCH-001-budget-aware-adaptive-search.md | 16 +-- scripts/check_catalog.py | 112 ++++++++++++++++-- 4 files changed, 142 insertions(+), 40 deletions(-) diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index be6765f..5d955bc 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,11 +14,11 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime policies, waiter limits, overflow/backpressure policies, per-waiter cancellation/deadline/terminal-claim policies, retry/error-sharing policies, and result-ownership policies -- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, and error semantics for every joined caller, linearize cancellation/deadline against terminal delivery for each waiter, bound waiter memory, and never admit new waiters to a closing or terminal generation -- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, waiter-memory, atomic terminal-claim, overflow/backpressure, and result-copy overhead +- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime policies, waiter limits, overflow/backpressure policies, per-waiter cancellation/deadline/terminal-claim policies, result-preparation/clone-failure policies, retry/error-sharing policies, and result-ownership policies +- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, preparation-failure, and error semantics for every joined caller, linearize cancellation/deadline against terminal delivery for each waiter, bound waiter memory, and never admit new waiters to a closing or terminal generation +- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, waiter-memory, result preparation/cloning, atomic terminal-claim, overflow/backpressure, and result-copy overhead - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, and deadline; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; closing/terminal generations are not joinable; waiter overflow has an explicit bounded behavior +- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, deadline, and result-preparation outcome; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; closing/terminal generations are not joinable; waiter overflow has an explicit bounded behavior - B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied here - S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C - Variables: categorical / integer / mixed @@ -26,13 +26,13 @@ Many callers request the same expensive computation concurrently before any call - Objective behavior: noisy under concurrent load; semantic equivalence remains deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive concurrent-load testing -- Constraints: semantic equivalence, authorization, ownership, waiter-memory, cancellation, deadline, terminal-claim, timeout, and resource constraints +- Constraints: semantic equivalence, authorization, ownership, result-preparation failure, waiter-memory, cancellation, deadline, terminal-claim, timeout, and resource constraints - Parallelism: asynchronous / concurrent - Exactness: exact request/result semantics; no approximation is introduced ## Preserved contract -Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout/deadline, result-ownership, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. A configured waiter bound must never be exceeded silently. Each waiter reaches exactly one linearized terminal state; a waiter that has already cancelled or timed out cannot later receive the shared value/error. +Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout/deadline, result-ownership, preparation-failure, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. A configured waiter bound must never be exceeded silently. Each waiter reaches exactly one linearized terminal state; a waiter that has already cancelled or timed out cannot later receive the shared value/error, and a mutable-result clone failure cannot leave a waiter marked successful without a deliverable value. ## Optimization @@ -42,15 +42,17 @@ Atomically create the joinable generation **with the initiating caller already r Equivalent later callers may register as independent waiters only while the generation is joinable and the configured waiter capacity remains. Waiter admission is atomic with capacity accounting. When the final waiter slot is already occupied, apply one explicit target policy rather than silently exceeding the bound: reject/return a documented overload or retryable-backpressure result, block/queue the caller behind a separately bounded admission mechanism, or use another bounded policy with explicit timeout/cancellation semantics. Starting an unconstrained parallel generation for the same equivalence key is not the default overflow behavior because it recreates the duplicate upstream load this pattern is intended to prevent. If a target deliberately permits overflow generations, that concurrency bound and duplicate-work tradeoff must be part of C/B and validated separately. -Represent each admitted waiter with an atomic terminal state, initially `pending`. Cancellation attempts atomically claim `pending -> cancelled`; timeout/deadline handling atomically claims `pending -> timed-out`. A terminal result/error notifier may claim `pending -> delivered-success` or `pending -> delivered-error` only if the waiter's declared deadline has not expired at the claim point. If the deadline is already expired, the notifier must instead leave/transition that waiter to the target's timed-out state and must not deliver the shared terminal value/error. For explicit cancellation racing completion, whichever atomic transition claims `pending` first wins; the losing transition is a no-op for that waiter. These claim semantics are part of the public request contract and must not depend on scheduler timing after the claim. +Represent each admitted waiter with an atomic terminal state, initially `pending`. Cancellation attempts atomically claim `pending -> cancelled`; timeout/deadline handling atomically claims `pending -> timed-out`. A terminal notifier may claim a delivery outcome only if the waiter's declared deadline has not expired at the claim point. If the deadline is already expired, the notifier must instead leave/transition that waiter to the target's timed-out state and must not deliver the shared terminal value/error. For explicit cancellation racing completion, whichever atomic transition claims `pending` first wins; the losing transition is a no-op for that waiter. These claim semantics are part of the public request contract and must not depend on scheduler timing after the claim. Cancellation and timeout are otherwise per waiter: when one waiter leaves through a winning cancellation/timeout claim, remove only that waiter from the live-waiter accounting. If live waiters remain, keep the shared generation joinable. If the last live waiter leaves and the policy calls for upstream cancellation, atomically mark the registry entry **closing/non-joinable** (or remove it from the joinable map) before sending the asynchronous cancellation request upstream. A new caller arriving after that transition must create a fresh generation rather than attach to work already being canceled. The closing generation may remain internally tracked until its terminal completion for cleanup/accounting, but it is not eligible for coalescing. -On upstream success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the candidate waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Snapshot the waiter records, but do not treat membership in that snapshot as entitlement to delivery: for each waiter, perform the atomic per-waiter terminal claim described above immediately before delivery. A waiter whose cancellation/timeout claim already won is skipped. +On upstream success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the candidate waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Snapshot the waiter records, but do not treat membership in that snapshot as entitlement to delivery: each waiter still competes through its atomic terminal state. -Define result ownership explicitly. If a terminal-success claim wins and the terminal value is immutable/share-safe under the target API, the same immutable value may be delivered to all authorized success-claimed waiters. If callers normally receive mutable or caller-owned results, create an independent defensive clone/copy/copy-on-write handle for each waiter after its successful terminal claim and before delivery so one caller cannot observably mutate another caller's result. Deliver a terminal failure only to waiters whose `delivered-error` claim wins (or a per-caller wrapped equivalent where the API requires ownership/context), then retire/clean up the generation deterministically. +For an upstream **failure**, no mutable success value must be prepared. Immediately before delivering the shared failure (or a per-caller wrapped equivalent where the API requires ownership/context), attempt the waiter's `pending -> delivered-error` claim subject to the same deadline rule. A waiter whose cancellation/timeout claim already won is skipped. -Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. +For an upstream **success**, define result ownership explicitly. If the terminal value is immutable/share-safe under the target API, no per-waiter clone is needed; immediately before delivery, attempt `pending -> delivered-success` subject to the same deadline/cancellation race, and deliver only if that claim wins. If callers normally receive mutable or caller-owned results, **prepare the independent defensive clone/copy/copy-on-write handle while the waiter is still `pending`, before claiming successful delivery**. Preparation is not entitlement to delivery: cancellation or timeout may win while preparation is in progress, in which case discard/release the prepared value and deliver nothing. If preparation succeeds, attempt `pending -> delivered-success`; deliver the prepared value only when that claim wins, otherwise discard it. If preparation fails because of allocation, serialization, quota, or another declared preparation error, attempt `pending -> delivered-error` with that preparation failure (again subject to deadline/cancellation). If cancellation/timeout already won, discard the preparation error for delivery purposes. This ensures no waiter is permanently marked successful before a deliverable isolated value exists and every waiter still observes exactly one terminal outcome. + +Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, result-preparation behavior, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. Retire/clean up the generation deterministically after all candidate waiters have either reached their terminal claim or been safely removed under the target cleanup policy. This differs from caching: the reusable result does not exist yet. @@ -68,6 +70,8 @@ Stress simultaneous identical and non-identical keys; inject upstream failures/t Add **terminal-delivery races** for both upstream success and upstream failure. Pause after the terminal waiter snapshot, then race explicit cancellation and deadline expiry against each waiter's delivery claim. Prove exactly one `pending -> terminal` transition wins, cancelled/timed-out waiters never receive a later value/error, completion that legitimately claims before cancellation preserves the declared completion result, and an already-expired deadline cannot be bypassed merely because the timeout worker has not run yet. Repeat under high concurrency and verify no waiter observes two terminal outcomes. +Add **clone/preparation-failure fixtures** for mutable results. Force allocation, serialization, copy-on-write setup, or quota failure while preparing a per-waiter value. Verify a waiter is still `pending` until preparation succeeds; successful preparation followed by a winning cancellation/timeout causes the prepared value to be discarded; failed preparation can atomically resolve to exactly one `delivered-error` only if cancellation/timeout has not already won; and no clone failure can leave a waiter in `delivered-success` without an actual value. Race clone success/failure against cancellation and deadline expiry repeatedly under load. + Add an **immediate synchronous-completion** fixture where the upstream operation can finish inline at launch. Prove the initiating caller was already registered before launch and always receives the terminal result/error unless its own cancellation/deadline claim wins under the same rules. Add authorization-boundary fixtures: issue syntactically identical requests under different tenants, principals, roles, ACL/visibility scopes, or other authorization context. Prove they either map to different equivalence keys **or** that the shared upstream result is explicitly safe to reuse and each caller is independently authorized before delivery. Verify that a result produced under one authorization scope can never leak to another merely because the resource parameters match. @@ -78,12 +82,12 @@ Add waiter-overflow races: fill the waiter list to one slot below the maximum, l ## Target-repo adaptation -Define key canonicalization, the authorization/visibility context that participates in equivalence, maximum waiter count, bounded overflow/backpressure semantics, result ownership/share-safety policy, per-waiter atomic terminal-state representation, cancellation/deadline winning semantics, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. +Define key canonicalization, the authorization/visibility context that participates in equivalence, maximum waiter count, bounded overflow/backpressure semantics, result ownership/share-safety policy, how mutable per-waiter results are prepared and how preparation failures surface, the per-waiter atomic terminal-state representation, cancellation/deadline winning semantics, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent or authorization-distinct work; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; non-linearized cancellation/deadline versus delivery can produce late values/errors or double terminal outcomes; treating terminal snapshot membership as delivery entitlement can notify a waiter after it has timed out; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. +Over-broad keys merge non-equivalent or authorization-distinct work; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; non-linearized cancellation/deadline versus delivery can produce late values/errors or double terminal outcomes; claiming success before a mutable per-waiter value is successfully prepared can strand a waiter with no deliverable result; clone/preparation failure can race cancellation and create inconsistent outcomes if not claimed atomically; treating terminal snapshot membership as delivery entitlement can notify a waiter after it has timed out; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/error semantics; if a cancelled/timed-out waiter can receive a later terminal value/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if the waiter bound or documented overflow behavior is violated; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. +Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/preparation-error semantics; if a waiter can enter `delivered-success` before an isolated deliverable result exists; if clone/preparation failure can produce no terminal outcome or a second terminal outcome; if a cancelled/timed-out waiter can receive a later terminal value/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if the waiter bound or documented overflow behavior is violated; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. diff --git a/optimizations/OPT-FAN-001-shared-materialization-fanout.md b/optimizations/OPT-FAN-001-shared-materialization-fanout.md index c1ef8f2..9d59a6e 100644 --- a/optimizations/OPT-FAN-001-shared-materialization-fanout.md +++ b/optimizations/OPT-FAN-001-shared-materialization-fanout.md @@ -14,11 +14,11 @@ The same deterministic transformation is repeated independently for each consume ## Optimization problem contract -- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, immutable-source snapshot/mutation-control policies, monotonic source/config epochs, and crash-consistent publication schemes -- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, source-snapshot/mutation-consistency, and publication-atomicity requirement +- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, immutable-source snapshot/mutation-control policies, monotonic source/config epochs, artifact-version pinning/consumption policies, and crash-consistent publication schemes +- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, source-snapshot/mutation-consistency, validation-to-consumption identity, and publication-atomicity requirement - f: measured transformation CPU, replay CPU, fan-out latency, and storage/I/O overhead under the target's declared objective ordering - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state binds the artifact bytes to one coherent effective source/transform identity and no intervening mutable-input change can be erased by endpoint equality; verification/security metadata may be removed only under an explicit contract change +- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state binds the artifact bytes to one coherent effective source/transform identity, no intervening mutable-input change can be erased by endpoint equality, and every consumer reads the **same immutable/versioned artifact instance that was validated** rather than re-resolving a mutable alias after validation; verification/security metadata may be removed only under an explicit contract change - B: target-specific fan-out/replay benchmark budget declared before tuning; no portable subscriber count, replay size, or retention duration is supplied here - S: stop when the declared budget is exhausted or a validated materialization policy materially improves the target objective without violating C - Variables: categorical / integer / mixed @@ -26,13 +26,13 @@ The same deterministic transformation is repeated independently for each consume - Objective behavior: noisy for performance; transformation identity/equivalence is deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive depending on transform/replay size -- Constraints: semantic equivalence, source-snapshot/mutation consistency, integrity, versioning, trust/security, storage, and crash-consistency constraints -- Parallelism: concurrent fan-out/replay; publication must remain race-safe +- Constraints: semantic equivalence, source-snapshot/mutation consistency, artifact-version pinning, integrity, versioning, trust/security, storage, and crash-consistency constraints +- Parallelism: concurrent fan-out/replay; publication and consumption pinning must remain race-safe - Exactness: exact representation semantics; no approximation is introduced ## Preserved contract -Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive source mutation, change-and-revert races, crashes, and interrupted publication. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. +Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive source mutation, change-and-revert races, crashes, interrupted publication, and concurrent replacement of mutable aliases. Validation is meaningful only if the consumer subsequently reads the exact artifact version/handle that passed validation. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. ## Optimization @@ -44,7 +44,9 @@ Bind the transform to one coherent source identity for the **entire transform-to Publish the artifact and its identity as **one committed state**. Acceptable designs include content-addressed storage where the artifact digest is itself part of the committed key, an atomically replaced manifest that contains both the full materialization key and the artifact digest/location, or another crash-consistent transaction that makes old state or new state visible but never a mixed pair. Do not update artifact bytes and their key independently in a way that can expose a new artifact with stale metadata or stale bytes with a new key after a crash. -Before reuse, require: (1) exact agreement with the current effective-input materialization key and immutable snapshot/monotonic mutation identity, (2) a committed manifest/content-address relation that binds that identity to the artifact identity, and (3) artifact integrity/format validity. A key mismatch, mutation-epoch mismatch, missing/incomplete publication marker, digest mismatch, or unverifiable artifact is a cache miss and requires regeneration. Do not use format validation or endpoint key equality alone as evidence that an artifact corresponds to one stable input generation. +Before reuse, require: (1) exact agreement with the current effective-input materialization key and immutable snapshot/monotonic mutation identity, (2) a committed manifest/content-address relation that binds that identity to the artifact identity, and (3) artifact integrity/format validity. **Validation must return or retain an immutable/versioned artifact handle/snapshot that uniquely identifies the validated bytes. Every fan-out/replay consumer must read through that same pinned handle/version.** Do not validate mutable pathname/object-name A and later re-resolve that alias for consumption. If the target cannot provide immutable/versioned handles, hold an appropriate read/replacement lock from the integrity check through the complete consumer read, or first create an immutable snapshot and validate/consume that snapshot. A key mismatch, mutation-epoch mismatch, missing/incomplete publication marker, digest mismatch, unverifiable artifact, or inability to bind consumption to the validated bytes is a cache miss/fail-closed condition and requires regeneration or safe fallback. Do not use format validation, endpoint key equality, or a mutable location name alone as evidence that the bytes consumed are the bytes validated. + +For multiple consumers, each may hold its own reference to the same immutable validated artifact version, or the system may retain one immutable snapshot for the fan-out/replay lifetime. Reclamation/retention must not invalidate a pinned consumer handle before that consumer completes. A mutable alias may advance to a newer committed materialization for later callers without changing the version already pinned by an in-flight consumer. ## Before / after evidence @@ -60,16 +62,18 @@ Compare shared materialization against per-consumer reference output, including Exercise **concurrent source mutation**, including explicit A→B→A races. Start a transform from source identity A, mutate one or more source/config/transform inputs to B while the transform is running, then restore the original bytes before commit. For snapshot-based targets, prove the transform reads only the immutable A snapshot. For lock-based targets, prove mutation cannot interleave with the protected transform/publication interval. For epoch/version-based targets, prove every mutation advances the monotonic witness and the final witness differs even when the final content/key returns to A. Reject/discard the candidate on any witness change and compare every accepted materialization with a fresh transform from the exact committed source identity. +Add a **post-validation replacement race**. Validate committed artifact A, pause before a consumer reads it, replace the mutable alias/path/object name with a different valid artifact B, then resume consumption. Prove a pinned immutable/versioned handle still yields exactly A (or fails closed if A was invalidated by the target's retention contract), never unvalidated B. Repeat with fan-out consumers at staggered start times, concurrent manifest advancement, replay after alias replacement, reclamation pressure, and mutable object-store/version aliases. For lock-based targets, prove replacement cannot occur until the protected consumer read completes. For snapshot-based targets, prove validation and consumption address the same snapshot digest/version. + Inject crashes/interruption at every publication boundary: after artifact write but before manifest commit, after provisional metadata write, during atomic replacement, and immediately after commit. After restart, prove that readers see either the previous valid committed materialization or the new valid committed materialization, never a mixed key/artifact state. Verify digest/key mismatch is rejected even when the artifact is otherwise parseable. ## Target-repo adaptation -Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, specify whether mutable inputs are consumed from immutable snapshots, protected by full-duration locks, or guarded by monotonic mutation epochs, and define how a coherent multi-input witness is captured. Specify representation versioning, invalidation, integrity checking, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. Do not advertise commit-time endpoint rehashing alone as sufficient mutation detection. +Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, specify whether mutable inputs are consumed from immutable snapshots, protected by full-duration locks, or guarded by monotonic mutation epochs, and define how a coherent multi-input witness is captured. Specify representation versioning, invalidation, integrity checking, **the immutable/versioned artifact handle or lock/snapshot that binds validation through consumption**, retention/reclamation semantics for pinned consumers, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. Do not advertise commit-time endpoint rehashing or mutable-path validation alone as sufficient identity protection. ## Failure modes -Incomplete keys can serve stale representations after source or transform changes; mutable sources can change during transformation and produce mixed-state output under a stale key; A→B→A races can defeat endpoint key comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent epoch vectors can describe no real source state; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. +Incomplete keys can serve stale representations after source or transform changes; mutable sources can change during transformation and produce mixed-state output under a stale key; A→B→A races can defeat endpoint key comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent epoch vectors can describe no real source state; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; a mutable alias can be replaced after validation and before consumption, delivering unvalidated bytes; reclamation can invalidate a pinned artifact prematurely; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. ## Rollback trigger -Disable reuse immediately if any materialization-key hit, source-mutation race, publication-recovery path, or integrity check can return output that differs from a fresh transform for the same exact committed effective inputs; if an A→B→A race can evade the snapshot/lock/monotonic mutation witness; or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. +Disable reuse immediately if any materialization-key hit, source-mutation race, publication-recovery path, integrity check, or validation-to-consumption race can return output that differs from a fresh transform for the same exact committed effective inputs; if an A→B→A race can evade the snapshot/lock/monotonic mutation witness; if a mutable alias replacement can make a consumer read bytes other than the exact artifact version that passed validation; if a pinned artifact can be reclaimed before consumption completes; or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index e091a45..9d3d149 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -20,9 +20,9 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra - F: candidates in X that satisfy all hard resource, platform, semantic, and correctness constraints before objective ranking - f: the target-measured objective or objective vector for each feasible candidate, including declared noise/statistical treatment - d: the target's predeclared minimize, maximize, lexicographic, or Pareto ordering -- C: search may choose where to evaluate but may not weaken correctness, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources; every per-trial reservation must be an enforceable upper bound rather than an estimate; dispatch/completion accounting must be linearizable; and targets that require deterministic search outcomes must use deterministic observation assimilation independent of wall-clock completion order +- C: search may choose where to evaluate but may not weaken correctness, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources; every per-trial reservation must be an enforceable upper bound rather than an estimate; dispatch/completion accounting must be linearizable; and targets that require deterministic search outcomes must use deterministic observation assimilation **and deterministic proposal/dispatch/refill scheduling** independent of wall-clock completion order - B: an explicit target-specific hard maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit, enforceable per-trial cap mechanism, conservative reservation rule, atomic accounting boundary, and failure/cancellation charging policy are fixed before dispatch begins -- S: stop proposing/dispatching when no additional trial can be safely reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger and apply stopping decisions to the declared deterministic assimilation order when determinism is required +- S: stop proposing/dispatching when no additional trial can be safely reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger and apply proposal, dispatch/refill, assimilation, and stopping decisions to the declared deterministic schedule when determinism is required - Variables: mixed search spaces; may include continuous, integer, categorical, and conditional dimensions as explicitly declared by the target - Search scope: local or global, explicitly declared for the target - Objective behavior: deterministic, noisy, or stochastic as declared by the target; noise treatment must be explicit @@ -34,7 +34,7 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra ## Preserved contract -Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B, no individual trial may consume beyond its reserved cap, and concurrent dispatch/completion transitions must not transiently expose phantom free capacity. If the target requires deterministic selected configurations or trial traces, proposal updates and stopping decisions must not depend on nondeterministic completion order. +Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B, no individual trial may consume beyond its reserved cap, and concurrent dispatch/completion transitions must not transiently expose phantom free capacity. If the target requires deterministic selected configurations or trial traces, **both the observation prefix used to create each proposal and the schedule that decides when a new proposal may be generated/dispatched must be deterministic**; worker completion timing may not change the proposal sequence. ## Optimization @@ -44,7 +44,7 @@ Before dispatching an asynchronous trial, enter one atomic/serializable accounti Completion, failure, cancellation, and forced termination use the **same atomic accounting boundary** as dispatch reservation. For one terminal transition, atomically: (1) read the trial's reservation, (2) meter/record the amount actually consumed, (3) move that consumed amount into permanent `consumed`, (4) release only the demonstrably unconsumed remainder from `reserved`, and (5) mark the trial terminal. No dispatcher may observe released reservation capacity before the corresponding consumed charge is committed, and concurrent terminal updates must not lose increments. Completion must not double-charge the same usage. A failed or cancelled trial never erases resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute/time budgets, release only the measured or otherwise provable unused portion of the enforceable reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. Every reservation, cap enforcement action, consumption adjustment, release, failure, cancellation, forced termination, and terminal accounting transaction is recorded in the ledger. -For targets that require deterministic search behavior, assign a deterministic trial ID/order at proposal time and **buffer asynchronous completions for assimilation in that declared order** (or use explicit deterministic batches/barriers). Surrogate/model updates, acquisition decisions, domain contraction, portfolio-selection state, and stopping criteria must consume observations according to this deterministic order rather than wall-clock completion order. A fixed random seed alone is not sufficient. If a target chooses completion-order assimilation for throughput, declare the resulting nondeterminism as an explicit contract change rather than claiming deterministic replay. +For targets that require deterministic search behavior, assign deterministic trial IDs and define a **deterministic proposal frontier**. A new proposal may be generated only from a declared ordered observation prefix that is the same in every replay. Buffer out-of-order completions until that prefix is available. Do **not** immediately refill whichever worker happens to become free if doing so would let wall-clock completion order choose the model state used for the next proposal. Acceptable deterministic designs include fixed deterministic batches/barriers, or an ordered-prefix scheduler where proposal `k+1` is generated only after the exact predeclared prefix needed for that proposal has been assimilated and its dispatch slot/order is determined independently of worker-speed races. Surrogate/model updates, acquisition decisions, domain contraction, portfolio-selection state, proposal generation, dispatch/refill decisions, and stopping criteria must consume the same deterministic state sequence. A fixed random seed plus buffered assimilation alone is not sufficient if worker availability can still change which proposal is generated next. If a target chooses immediate completion-driven refill for throughput, declare the resulting nondeterminism as an explicit contract change rather than claiming deterministic replay. Parallelism has an information cost: very wide batches receive less feedback between suggestions and can degenerate toward non-adaptive/random search. @@ -62,16 +62,16 @@ Keep a deterministic search seed where practical, preserve the full trial ledger Race multiple trial completions/cancellations against one another and against workers attempting the final dispatch slot. Verify the accounting transaction is linearizable: no consumed increment is lost, no reservation is released before its corresponding consumption is charged, and a dispatcher never observes capacity that would make the post-transaction invariant `consumed + reserved <= B` false. -For deterministic targets, run the same seeded trial set with deliberately permuted worker speeds/completion orders. Verify observation assimilation follows the declared trial-ID/batch order, the surrogate/search state replays identically, and the selected candidate plus stopping reason match the deterministic reference. Where sequential/parallel equivalence is part of C, compare an asynchronous execution with its deterministic sequential or batch-assimilation replay. If deterministic equivalence is intentionally not required, verify the record/target explicitly labels that nondeterminism instead. +For deterministic targets, run the same seeded search with deliberately permuted worker speeds and completion orders, including the case where trial 2 finishes before trial 1 and frees a worker first. Verify out-of-order completion **does not permit proposal 3 to be generated from a different observation prefix**. The complete proposal sequence, parameter values, deterministic trial IDs, logical dispatch/refill order, surrogate/search states, selected candidate, and stopping reason must match the deterministic reference. Test both fixed-batch/barrier scheduling and any ordered-prefix scheduler the target claims to support. Where sequential/parallel equivalence is part of C, compare the asynchronous execution with its deterministic sequential or batch replay. If completion-driven refill is intentionally retained, verify the target explicitly labels the search trace nondeterministic instead of claiming replay equivalence. ## Target-repo adaptation -Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, **enforcement mechanism for that reservation**, one atomic/serializable accounting mechanism shared by reservation and terminal conversion, metering source, failure/cancellation charging policy, and deterministic observation-assimilation policy (when required) before enabling asynchronous dispatch. +Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, **enforcement mechanism for that reservation**, one atomic/serializable accounting mechanism shared by reservation and terminal conversion, metering source, failure/cancellation charging policy, deterministic observation-assimilation policy, **deterministic proposal frontier and dispatch/refill schedule** (when required), and the exact condition under which a freed worker may receive new work before enabling asynchronous dispatch. ## Failure modes -Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism; wall-clock completion-order assimilation can make supposedly deterministic search traces, proposals, and stopping decisions irreproducible. +Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism; wall-clock completion-order assimilation can make supposedly deterministic search traces irreproducible; **immediate worker refill can also make proposals nondeterministic even when assimilation itself is buffered**. ## Rollback trigger -Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, any trial can consume beyond its enforceable reservation, any accounting/concurrency test shows that dispatch/terminal transitions can violate B, or any target that requires deterministic search fails replay under permuted asynchronous completion orders. +Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, any trial can consume beyond its enforceable reservation, any accounting/concurrency test shows that dispatch/terminal transitions can violate B, or any target that requires deterministic search produces different proposals, logical dispatch/refill order, model states, selected candidates, or stopping reasons under permuted asynchronous completion orders. diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 7ca3d06..cb05bd5 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -88,6 +88,11 @@ r"^\|[^|\n]*\|\s*\[(OPT-[A-Z]+-\d{3})\]\((optimizations/[^)#]+\.md)\)\s*\|", re.MULTILINE, ) +HEADING_RE = re.compile(r"^#{1,6}(?:\s|$)") +THEMATIC_BREAK_RE = re.compile(r"^(?:-{3,}|\*{3,}|_{3,})$") +FENCE_RE = re.compile(r"^(?:```|~~~)") +LIST_MARKER_ONLY_RE = re.compile(r"^(?:[-+*]|\d+[.)])$") +TABLE_SEPARATOR_CELL_RE = re.compile(r"^:?-{3,}:?$") CANONICAL_DEFINITION_PATTERNS = { "X": re.compile(r"^- `X` — \S"), "F": re.compile(r"^- `F(?: ⊆ X)?` — \S"), @@ -118,10 +123,89 @@ def section_lines(text: str, heading: str) -> list[str]: return lines[start:end] +def strip_html_comments(text: str) -> str: + return re.sub(r"", "", text, flags=re.DOTALL) + + +def visible_nonfenced_lines(lines: list[str]) -> list[str]: + """Return rendered-ish Markdown lines, excluding comments and fenced examples.""" + cleaned = strip_html_comments("\n".join(lines)) + visible: list[str] = [] + fence: str | None = None + for raw in cleaned.splitlines(): + stripped = raw.strip() + if fence is not None: + if stripped.startswith(fence): + fence = None + continue + if stripped.startswith("```"): + fence = "```" + continue + if stripped.startswith("~~~"): + fence = "~~~" + continue + visible.append(raw) + return visible + + +def markdown_table_cells(line: str) -> list[str] | None: + stripped = line.strip() + if not stripped.startswith("|"): + return None + return [cell.strip() for cell in stripped.strip("|").split("|")] + + +def extract_markdown_table( + lines: list[str], expected_headers: tuple[str, ...], context: str +) -> list[str]: + """Extract one visible Markdown table by exact header, excluding comments/fences.""" + visible = visible_nonfenced_lines(lines) + expected = list(expected_headers) + for i, line in enumerate(visible): + cells = markdown_table_cells(line) + if cells != expected: + continue + if i + 1 >= len(visible): + die(f"{context} table has no separator row") + separator = markdown_table_cells(visible[i + 1]) + if ( + separator is None + or len(separator) != len(expected) + or not all(TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in separator) + ): + die(f"{context} table has an invalid separator row") + table = [line, visible[i + 1]] + for row in visible[i + 2 :]: + if markdown_table_cells(row) is None: + break + table.append(row) + return table + die(f"{context} is missing the expected Markdown table") + + +def is_structural_only_line(line: str) -> bool: + """Return true for Markdown scaffolding that does not state record content.""" + if HEADING_RE.match(line): + return True + if THEMATIC_BREAK_RE.fullmatch(line): + return True + if FENCE_RE.match(line): + return True + if LIST_MARKER_ONLY_RE.fullmatch(line): + return True + if line == ">": + return True + cells = markdown_table_cells(line) + if cells is not None and cells and all( + TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in cells + ): + return True + return False + + def section_has_content(lines: list[str]) -> bool: - """Require record-specific visible content, not stock template prompts.""" - content = "\n".join(lines) - content = re.sub(r"", "", content, flags=re.DOTALL) + """Require record-specific visible content, not prompts or Markdown scaffolding.""" + content = strip_html_comments("\n".join(lines)) for raw in content.splitlines(): line = raw.strip() if not line: @@ -130,6 +214,8 @@ def section_has_content(lines: list[str]) -> bool: continue if EMPTY_LABEL_RE.match(line): continue + if is_structural_only_line(line): + continue return True return False @@ -215,7 +301,7 @@ def require_prefixed_fields( for heading in sorted(REQUIRED_V2): if not section_has_content(section_lines(text, heading)): die( - f"{path.relative_to(ROOT)} has empty/template-only mandatory section {heading}" + f"{path.relative_to(ROOT)} has empty/template/structural-only mandatory section {heading}" ) contract = section_lines(text, "## Optimization problem contract") @@ -240,7 +326,7 @@ def require_prefixed_fields( record_paths = {str(path.relative_to(ROOT)): record_id for record_id, path in records.items()} # Validate every optimization-record link wherever it appears. README index -# completeness/uniqueness is checked separately from actual catalog table rows, +# completeness/uniqueness is checked separately from its rendered catalog table, # so contextual prose links are allowed and do not count as duplicate index rows. for doc_name in ("README.md", "CATALOG.md"): text = (ROOT / doc_name).read_text(encoding="utf-8") @@ -262,8 +348,12 @@ def require_prefixed_fields( catalog_lines = section_lines(text, "## Catalog") if not catalog_lines: die("README.md is missing a non-empty ## Catalog section") - catalog_section = "\n".join(catalog_lines) - rows = README_ROW_RE.findall(catalog_section) + catalog_table = extract_markdown_table( + catalog_lines, + ("ID", "Optimization", "Status", "Core idea"), + "README.md ## Catalog", + ) + rows = README_ROW_RE.findall("\n".join(catalog_table)) row_ids = [row_id for row_id, _rel, _status in rows] counts = Counter(row_ids) bad_counts = sorted(record_id for record_id, count in counts.items() if count != 1) @@ -311,8 +401,12 @@ def require_prefixed_fields( decision_lines = section_lines(catalog, "## Quick decision table") if not decision_lines: die("CATALOG.md is missing a non-empty ## Quick decision table section") -decision_section = "\n".join(decision_lines) -decision_rows = CATALOG_DECISION_ROW_RE.findall(decision_section) +decision_table = extract_markdown_table( + decision_lines, + ("Bottleneck / problem shape", "First record to inspect", "Core idea"), + "CATALOG.md ## Quick decision table", +) +decision_rows = CATALOG_DECISION_ROW_RE.findall("\n".join(decision_table)) decision_ids = [record_id for record_id, _rel in decision_rows] decision_counts = Counter(decision_ids) bad_decision_counts = sorted( From 1804ba2894dca025af1714885d53c2e406c7bcac Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 18:16:15 +0930 Subject: [PATCH 017/229] Align approximation rollback with declared envelope --- ...PPROX-001-contract-bounded-approximation.md | 18 +++++++++++------- 1 file changed, 11 insertions(+), 7 deletions(-) diff --git a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md index 7ff2206..8c9567b 100644 --- a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md +++ b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md @@ -19,7 +19,7 @@ Exact processing has unbounded or unacceptable cost even though the product/scie - F: policies whose declared error/degradation metric remains within the target's explicit envelope over the declared state/composition horizon and whose resource/semantic constraints are satisfied - f: target-measured resource or latency cost, optionally paired with the declared quality/error metric - d: minimize resource/latency cost subject to feasibility in F, or use the target's predeclared multi-objective ordering when quality is ranked rather than hard-bounded -- C: approximation is permitted only by an explicit contract; exact callers are not silently weakened; the error norm, aggregation rule, sequence/composition horizon, and reset boundaries are declared before evaluation; an exact reference path or exact fixture remains available where practical +- C: approximation is permitted only by an explicit contract; exact callers are not silently weakened; the error norm, aggregation rule, sequence/composition horizon, reset boundaries, and whether the envelope is hard worst-case or statistical/confidence/tail-based are declared before evaluation; an exact reference path or exact fixture remains available where practical - B: target-specific benchmark/quality-evaluation budget over predeclared ordinary, boundary, adversarial, repeated-application, and long-horizon fixtures - S: stop when the evaluation budget is exhausted or a validated policy meets the target resource objective while remaining inside the declared quality envelope over the entire declared horizon - Variables: continuous / integer / categorical / conditional / mixed, depending on approximation policy @@ -33,7 +33,7 @@ Exact processing has unbounded or unacceptable cost even though the product/scie ## Preserved contract -Approximation is admissible only when the contract explicitly permits it. A previously exact API cannot be silently weakened and still be called correctness-preserving. For stateful or repeatedly composed approximations, the contract applies over an explicitly declared horizon—not merely to each isolated step—so bounded per-step error is insufficient if drift can accumulate beyond the allowed envelope. +Approximation is admissible only when the contract explicitly permits it. A previously exact API cannot be silently weakened and still be called correctness-preserving. For stateful or repeatedly composed approximations, the contract applies over an explicitly declared horizon—not merely to each isolated step—so bounded per-step error is insufficient if drift can accumulate beyond the allowed envelope. The contract must also state whether compliance is pointwise/worst-case or statistical; a stochastic envelope is judged by its declared aggregation, confidence, exceedance-probability, quantile, or tail criterion rather than by silently substituting a hard per-sample limit. ## Optimization @@ -41,6 +41,8 @@ Introduce a resource ceiling and degrade only along a declared dimension: sample For stateful streaming, simulation, DSP, iterative numerical work, or any repeatedly applied approximation, define the error model before benchmarking: the norm/metric (for example absolute, relative, L2, perceptual, state-distance, or domain-specific), how error composes or is aggregated through time, the maximum sequence length or physical/time horizon over which the envelope must hold, and any reset/checkpoint/re-synchronization boundaries that legitimately restart the horizon. If the system can run longer than the validated horizon without reset, either extend validation to that operational horizon or define a separate long-run drift bound; do not infer long-run safety from one-step ε alone. +For stochastic/noisy approximations, also define the statistical compliance rule before evaluation: the sampling unit and workload distribution, aggregation statistic, confidence level or interval procedure, tolerated exceedance probability, quantile/tail bound, and the sample/evaluation budget used to decide compliance. Do not reinterpret a statistical guarantee as a pointwise worst-case guarantee, and do not weaken a declared hard worst-case envelope into an average-case claim after observing data. + ## Before / after evidence - Environment: No controlled target-repository benchmark has been run for this OPT record. @@ -51,18 +53,20 @@ For stateful streaming, simulation, DSP, iterative numerical work, or any repeat ## Validation -Measure error/degradation and resource savings together across ordinary, boundary and adversarial workloads. Keep an exact reference for differential evaluation where practical. Declare and test the error norm/metric, aggregation rule, sequence/composition horizon, and reset boundaries explicitly. +Measure error/degradation and resource savings together across ordinary, boundary and adversarial workloads. Keep an exact reference for differential evaluation where practical. Declare and test the error norm/metric, aggregation rule, sequence/composition horizon, reset boundaries, and hard-versus-statistical envelope semantics explicitly. + +For stateful/repeated use, run differential trajectories against the exact path across short, nominal, maximum-supported, and adversarially long sequences. Include biased-error fixtures where each individual step remains within the local ε but errors accumulate in the same direction; verify the cumulative/state error still respects the declared horizon envelope. Test reset/checkpoint boundaries before, at, and after the limit; verify resets actually restore the assumptions used by the next horizon. -For stateful/repeated use, run differential trajectories against the exact path across short, nominal, maximum-supported, and adversarially long sequences. Include biased-error fixtures where each individual step remains within the local ε but errors accumulate in the same direction; verify the cumulative/state error still respects the declared horizon envelope. Test reset/checkpoint boundaries before, at, and after the limit; verify resets actually restore the assumptions used by the next horizon. Where stochastic approximation is used, evaluate both expected and tail/worst-case drift according to the declared statistical contract rather than only average one-step error. +Where stochastic approximation is used, evaluate the declared expected, quantile, exceedance-probability, confidence, tail, or worst-case criterion against the exact path as specified by C. Include fixtures where individual samples exceed a nominal pointwise value while the declared statistical envelope remains satisfied, and fixtures where the configured tail/confidence/exceedance criterion truly fails. Verify rollback decisions distinguish those cases rather than triggering on one sample unless the contract explicitly declares a hard single-sample/worst-case bound. ## Target-repo adaptation -Define `ε`, the exact quality/error norm, aggregation rule, workload distribution, maximum state/composition horizon, reset/checkpoint semantics, long-run drift policy, escape hatch and exact-mode availability locally. If the target has no finite operational horizon, establish a justified asymptotic/stability bound or periodic re-synchronization rule instead of copying a finite benchmark horizon from another system. +Define `ε`, the exact quality/error norm, aggregation rule, workload distribution, maximum state/composition horizon, reset/checkpoint semantics, long-run drift policy, escape hatch and exact-mode availability locally. Explicitly classify the quality envelope as hard pointwise/worst-case or statistical, and for statistical contracts specify the confidence/tail/exceedance rule and decision sample budget. If the target has no finite operational horizon, establish a justified asymptotic/stability bound or periodic re-synchronization rule instead of copying a finite benchmark horizon from another system. ## Failure modes -Unmeasured quality loss, biased sampling, hidden rare-case failures, cumulative drift that is invisible to one-step checks, reset boundaries that fail to restore reference assumptions, state-dependent amplification, unstable feedback loops, and callers incorrectly assuming exact semantics. +Unmeasured quality loss, biased sampling, hidden rare-case failures, cumulative drift that is invisible to one-step checks, reset boundaries that fail to restore reference assumptions, state-dependent amplification, unstable feedback loops, misclassifying a statistical envelope as a hard pointwise bound (or vice versa), and callers incorrectly assuming exact semantics. ## Rollback trigger -Disable when error exceeds the declared envelope at any point within the supported horizon, cumulative/state drift exceeds the declared bound even though per-step ε holds, reset/checkpoint validation fails, reference comparisons drift beyond contract, or resource savings are not material. +Evaluate rollback against the **declared envelope semantics**. For a hard pointwise/worst-case contract, disable immediately when any supported-horizon observation exceeds the declared bound. For a stochastic/statistical contract, disable when the predeclared aggregation, confidence, exceedance-probability, quantile, or tail criterion fails under its stated evaluation procedure; an isolated sample beyond a nominal pointwise value is not by itself a contract violation unless the contract says it is. In all cases, disable when cumulative/state drift violates its declared bound, reset/checkpoint validation fails, reference comparisons violate C, a catastrophic semantic/safety constraint is breached, or resource savings are not material. From 3bb0e5a7c41bb981ffc08bf7c42c71c93a0ce13f Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 18:16:53 +0930 Subject: [PATCH 018/229] Count leased regions in pruning frontier --- ...E-001-bound-driven-search-space-pruning.md | 22 +++++++++++-------- 1 file changed, 13 insertions(+), 9 deletions(-) diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index 76697fa..22dd49e 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -19,17 +19,17 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F - f: a scalar real-valued target objective `f : F → R` evaluated on feasible candidates only - d: exactly one of scalar `minimize` or scalar `maximize`; vector, Pareto, lexicographic, or other partial-order objectives are outside this record unless a separately specified and validated frontier-bound mechanism is introduced -- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, the target's observable tie semantics are preserved, and parallel dispatch cannot oversubscribe the declared hard budget +- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, the target's observable tie semantics are preserved, parallel dispatch cannot oversubscribe the declared hard budget, and frontier exhaustion is declared only after all queued **and leased/in-flight** regions are accounted for - B: a finite, predeclared target-specific **enforceable** cap on evaluations, wall time, compute, or equivalent resource consumption; parallel dispatch requires linearizable reservations before candidate/bound work starts, and every wall-time/compute reservation requires a per-operation quota/deadline/termination mechanism strong enough to prevent overrun; a resource that cannot be hard-capped must be labeled observational/best-effort rather than advertised as hard B -- S: stop immediately when the required optimality/tie contract is proven or the frontier is exhausted; otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality +- S: stop immediately when the required optimality/tie contract is proven, or when the **global frontier is exhausted**, meaning there are no queued regions, no leased/in-flight regions still capable of producing candidates/children, and no unpublished child/frontier updates owned by active work. Otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality - Variables: integer / categorical / discrete / mixed - Search scope: global over the declared candidate space - Objective behavior: deterministic unless uncertainty/noise is incorporated into a separately sound bound model - Information: derivative-free; bound/relaxation information is target-specific - Evaluation cost: moderate to expensive when exhaustive evaluation is infeasible -- Constraints: feasibility, semantic correctness, scalar-bound soundness, tie semantics, and enforceable finite-resource constraints -- Parallelism: sequential, or parallel only with synchronized incumbent/frontier/bound state **and linearizable budget reservation/completion accounting plus enforceable per-operation resource caps** -- Exactness: exact only when the declared optimality and observable-tie contract is proven within B; otherwise anytime/incomplete result semantics apply +- Constraints: feasibility, semantic correctness, scalar-bound soundness, tie semantics, global-frontier accounting, and enforceable finite-resource constraints +- Parallelism: sequential, or parallel only with synchronized incumbent/frontier/bound state, **leased/in-flight region accounting**, and linearizable budget reservation/completion accounting plus enforceable per-operation resource caps +- Exactness: exact only when the declared optimality and observable-tie contract is proven within B, including proof that no queued or leased region can still affect the answer; otherwise anytime/incomplete result semantics apply For each unexplored region `R`, define a bound `b(R)` separately from `f`: @@ -47,12 +47,14 @@ An independently proven infeasible region may also be pruned. A heuristic estima ## Preserved contract -A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, exhausting B without a feasible incumbent does not permit an infeasibility claim, and parallel execution must preserve the same hard resource ceiling as sequential execution rather than oversubscribing work in flight. +A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, exhausting B without a feasible incumbent does not permit an infeasibility claim, and parallel execution must preserve the same hard resource ceiling as sequential execution rather than oversubscribing work in flight. A temporarily empty shared queue is **not** frontier exhaustion while any worker owns a leased region that may still produce a candidate, proof obligation, or child region. ## Optimization Maintain an incumbent when one exists, partition the search space, compute a cheap sound `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific bound plus the target's tie semantics prove the region cannot affect the required answer. Before the first incumbent exists, sound bounds may prioritize regions or prove individual regions infeasible, but incumbent-based objective pruning is unavailable. +For **parallel** search, define one global frontier lifecycle. A region remains part of the frontier from enqueue until it is either (a) soundly pruned/closed, or (b) replaced by its child regions through an atomic/linearizable completion transition. Dequeuing for worker ownership therefore changes a region from `queued` to `leased/in-flight`; it does **not** remove that region from the global frontier. A worker that branches a leased region must publish all resulting children and close/release the parent as one frontier-accounting transition, or use another protocol that cannot expose a moment where the queue is empty even though unpublished descendants still exist. Worker failure/cancellation must return or recover the lease so unexplored work is not silently lost. + For **parallel** search, treat both candidate evaluation and nontrivial bound/relaxation evaluation as budget-consuming operations. Before dispatch, atomically reserve the operation's declared evaluation slot or conservative wall-time/compute quota from one shared budget ledger. If `consumed + reserved + proposed_reservation > B`, do not dispatch. Completion/failure/cancellation converts the reservation to consumed usage and releases only demonstrably unconsumed capacity under the same linearizable accounting boundary, so workers racing for the final slot cannot oversubscribe it. Evaluation-count budgets consume/reserve a slot before launch. For wall-time/compute budgets, each launched operation must have an enforceable per-operation upper bound—for example a deadline with forced termination, cgroup/job quota, provider/runtime cap, or equivalent mechanism. If the target cannot prevent one bound/candidate evaluation from running past the nominal reservation, wall-time/compute is **not** a hard B and must be documented as observational/best-effort instead of being used to justify finite-cap correctness claims. @@ -71,18 +73,20 @@ A relaxed solution is evidence for a bound, not automatically a feasible final a For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness independently by checking the direction-specific inequality against exhaustive feasible values inside each test region. Test pruning separately from search ordering. Include fixtures where the first feasible candidate is found late and where B expires before any feasible candidate exists; verify that the latter returns `no-incumbent / feasibility-unknown`, reports only independently valid global-bound information, and makes no infeasibility, optimality, or incumbent-based gap claim. Verify that budget exhaustion with an incumbent returns an anytime result without an exactness claim. +Add **parallel frontier-exhaustion races**. Use a fixture where the last queued region is leased by one worker, making the shared queue empty, then pause that worker before it publishes one or more child regions. Prove the coordinator does not declare exhaustion or exact optimality while that lease remains live. Resume the worker and verify the children become searchable and the final result matches exhaustive/scalar search. Also inject worker failure/cancellation while holding the last lease and verify the region is recovered/requeued or otherwise completed without losing unexplored work. Test simultaneous parent-close/child-publish transitions and prove there is no observation in which both queued and leased frontier counts reach zero before all descendants are durably accounted for. + Add **parallel budget-boundary fixtures**. Race multiple workers against one remaining evaluation slot and prove only one reservation succeeds. Race bound evaluations and candidate evaluations against the same final capacity and prove both charge the declared ledger. For wall-time/compute budgets, deliberately run an operation that attempts to exceed its reservation and prove the quota/deadline/termination mechanism stops it within the enforceable cap. Race completion/cancellation with new dispatch and verify the accounting transition is linearizable—released capacity is not visible before corresponding consumption is committed, no increments are lost, and `consumed + reserved <= B` always holds for hard-budget dimensions. Add **equal-objective tie fixtures**. For an any-one-optimum contract, prove equality pruning cannot alter any observable result. For deterministic tie-winner contracts, construct regions containing equal-objective candidates with better/worse tie ranks and prove equality-bound regions are retained until the declared tie winner is established. For all-optima contracts, prove every equal-objective optimum is enumerated. If using a stronger total-order bound, validate its soundness independently against exhaustive fixtures. ## Target-repo adaptation -The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. For parallel implementations, define one linearizable reservation/completion ledger shared by candidate and bound work, the accounting unit, per-operation reservation amount, metering source, and the enforcement mechanism for wall-time/compute quotas. Downgrade any unenforceable resource limit to best-effort/observational rather than calling it hard B. +The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. For parallel implementations, define the global frontier state machine, lease ownership/recovery rules, parent-close/child-publish atomicity, and the exact exhaustion predicate over queued plus leased/in-flight work. Also define one linearizable reservation/completion ledger shared by candidate and bound work, the accounting unit, per-operation reservation amount, metering source, and the enforcement mechanism for wall-time/compute quotas. Downgrade any unenforceable resource limit to best-effort/observational rather than calling it hard B. ## Failure modes -Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; parallel workers without linearizable reservations can oversubscribe the last evaluation/resource slot; an uncapped candidate/bound evaluation can exceed a nominal wall-time/compute cap before stopping logic observes it; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. +Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; treating queue-empty as frontier-empty can declare exact completion while a leased region still owns unexplored descendants; losing a worker lease can silently drop search regions; non-atomic parent-close/child-publication can create false exhaustion; parallel workers without linearizable reservations can oversubscribe the last evaluation/resource slot; an uncapped candidate/bound evaluation can exceed a nominal wall-time/compute cap before stopping logic observes it; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. ## Rollback trigger -Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel mode if workers can dispatch without first reserving budget, if concurrent accounting can oversubscribe B, or if any operation can exceed a resource reservation that is claimed as a hard cap. Abort exact-mode claims whenever B is exhausted before the full objective/tie contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. +Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel/exact mode if frontier exhaustion can be observed while any leased/in-flight region may still produce work, if parent-close/child-publication or lease recovery can lose unexplored regions, if workers can dispatch without first reserving budget, if concurrent accounting can oversubscribe B, or if any operation can exceed a resource reservation that is claimed as a hard cap. Abort exact-mode claims whenever B is exhausted before the full objective/tie/frontier contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. From f831e38624620fd3b67974b711c2f424d8b726fc Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 18:17:47 +0930 Subject: [PATCH 019/229] Harden rendered catalog schema validation --- scripts/check_catalog.py | 22 ++++++++++++++-------- 1 file changed, 14 insertions(+), 8 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index cb05bd5..304fd26 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -158,7 +158,7 @@ def markdown_table_cells(line: str) -> list[str] | None: def extract_markdown_table( lines: list[str], expected_headers: tuple[str, ...], context: str ) -> list[str]: - """Extract one visible Markdown table by exact header, excluding comments/fences.""" + """Extract one visible Markdown table by exact header and validate every row width.""" visible = visible_nonfenced_lines(lines) expected = list(expected_headers) for i, line in enumerate(visible): @@ -176,8 +176,14 @@ def extract_markdown_table( die(f"{context} table has an invalid separator row") table = [line, visible[i + 1]] for row in visible[i + 2 :]: - if markdown_table_cells(row) is None: + row_cells = markdown_table_cells(row) + if row_cells is None: break + if len(row_cells) != len(expected): + die( + f"{context} table row has {len(row_cells)} column(s); " + f"expected {len(expected)}: {row.strip()}" + ) table.append(row) return table die(f"{context} is missing the expected Markdown table") @@ -204,9 +210,8 @@ def is_structural_only_line(line: str) -> bool: def section_has_content(lines: list[str]) -> bool: - """Require record-specific visible content, not prompts or Markdown scaffolding.""" - content = strip_html_comments("\n".join(lines)) - for raw in content.splitlines(): + """Require record-specific rendered content, not prompts/scaffolding/examples.""" + for raw in visible_nonfenced_lines(lines): line = raw.strip() if not line: continue @@ -233,13 +238,14 @@ def require_prefixed_fields( section: str, rejected_values: dict[str, str] | None = None, ) -> None: - """Require exactly one selected non-empty '- Field:' row for every declared field.""" + """Require one visible selected non-empty '- Field:' row for every field.""" + visible = visible_nonfenced_lines(lines) for field in fields: prefix = f"- {field}:" - matches = [line for line in lines if line.startswith(prefix)] + matches = [line for line in visible if line.startswith(prefix)] if len(matches) != 1: die( - f"{path.relative_to(ROOT)} must contain exactly one field " + f"{path.relative_to(ROOT)} must contain exactly one visible field " f"'{prefix}' in {section}" ) value = matches[0][len(prefix) :].strip() From 512e5a7ff8423adb7fb7f124c941399b81c634fe Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 18:38:22 +0930 Subject: [PATCH 020/229] Harden rendered catalog validation --- scripts/check_catalog.py | 71 ++++++++++++++++++++++------------------ 1 file changed, 39 insertions(+), 32 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 304fd26..c08453e 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -108,21 +108,6 @@ def die(msg: str) -> None: raise SystemExit(f"catalog-integrity: {msg}") -def section_lines(text: str, heading: str) -> list[str]: - """Return lines belonging to one exact level-2 Markdown section.""" - lines = text.splitlines() - try: - start = lines.index(heading) + 1 - except ValueError: - return [] - end = len(lines) - for i in range(start, len(lines)): - if lines[i].startswith("## "): - end = i - break - return lines[start:end] - - def strip_html_comments(text: str) -> str: return re.sub(r"", "", text, flags=re.DOTALL) @@ -148,6 +133,26 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: return visible +def visible_text(text: str) -> str: + """Return visible, non-fenced Markdown text for semantic integrity checks.""" + return "\n".join(visible_nonfenced_lines(text.splitlines())) + + +def section_lines(text: str, heading: str) -> list[str]: + """Return one exact visible level-2 Markdown section.""" + lines = visible_nonfenced_lines(text.splitlines()) + try: + start = lines.index(heading) + 1 + except ValueError: + return [] + end = len(lines) + for i in range(start, len(lines)): + if lines[i].startswith("## "): + end = i + break + return lines[start:end] + + def markdown_table_cells(line: str) -> list[str] | None: stripped = line.strip() if not stripped.startswith("|"): @@ -262,11 +267,11 @@ def require_prefixed_fields( status_categories: dict[str, str] = {} for path in sorted(OPT_DIR.glob("OPT-*.md")): text = path.read_text(encoding="utf-8") - lines = text.splitlines() + lines = visible_nonfenced_lines(text.splitlines()) first = lines[0] if lines else "" match = ID_RE.match(first) if not match: - die(f"bad record heading: {path.relative_to(ROOT)}") + die(f"bad or hidden record heading: {path.relative_to(ROOT)}") record_id = match.group(1) filename_match = FILENAME_ID_RE.match(path.name) @@ -286,7 +291,7 @@ def require_prefixed_fields( status_matches = [STATUS_RE.match(line) for line in lines] statuses = [m.group(1).strip() for m in status_matches if m is not None] if len(statuses) != 1: - die(f"{path.relative_to(ROOT)} must contain exactly one Status line") + die(f"{path.relative_to(ROOT)} must contain exactly one visible Status line") if not statuses[0]: die(f"{path.relative_to(ROOT)} has empty Status") @@ -302,7 +307,7 @@ def require_prefixed_fields( headings = {line for line in lines if line.startswith("## ")} missing = sorted(REQUIRED_V2 - headings) if missing: - die(f"{path.relative_to(ROOT)} missing sections: {', '.join(missing)}") + die(f"{path.relative_to(ROOT)} missing visible sections: {', '.join(missing)}") for heading in sorted(REQUIRED_V2): if not section_has_content(section_lines(text, heading)): @@ -331,16 +336,17 @@ def require_prefixed_fields( record_paths = {str(path.relative_to(ROOT)): record_id for record_id, path in records.items()} -# Validate every optimization-record link wherever it appears. README index +# Validate every visible optimization-record link wherever it appears. README index # completeness/uniqueness is checked separately from its rendered catalog table, # so contextual prose links are allowed and do not count as duplicate index rows. for doc_name in ("README.md", "CATALOG.md"): text = (ROOT / doc_name).read_text(encoding="utf-8") - links = LINK_RE.findall(text) + rendered = visible_text(text) + links = LINK_RE.findall(rendered) for label, rel in links: target = ROOT / rel if not target.is_file(): - die(f"broken record link in {doc_name}: {rel}") + die(f"broken visible record link in {doc_name}: {rel}") target_id = record_paths.get(rel) if target_id is None: die(f"record link in {doc_name} is not a discovered OPT record: {rel}") @@ -353,7 +359,7 @@ def require_prefixed_fields( if doc_name == "README.md": catalog_lines = section_lines(text, "## Catalog") if not catalog_lines: - die("README.md is missing a non-empty ## Catalog section") + die("README.md is missing a non-empty visible ## Catalog section") catalog_table = extract_markdown_table( catalog_lines, ("ID", "Optimization", "Status", "Core idea"), @@ -394,19 +400,20 @@ def require_prefixed_fields( ) catalog = (ROOT / "CATALOG.md").read_text(encoding="utf-8") -catalog_ids = set(OPT_TOKEN_RE.findall(catalog)) +rendered_catalog = visible_text(catalog) +catalog_ids = set(OPT_TOKEN_RE.findall(rendered_catalog)) unknown_catalog_ids = sorted(catalog_ids - records.keys()) if unknown_catalog_ids: - die(f"CATALOG.md references unknown record ID(s): {', '.join(unknown_catalog_ids)}") + die(f"CATALOG.md references unknown visible record ID(s): {', '.join(unknown_catalog_ids)}") for record_id, path in records.items(): if record_id not in catalog_ids: - die(f"{record_id} ({path.name}) is not mentioned in CATALOG.md") + die(f"{record_id} ({path.name}) is not visibly mentioned in CATALOG.md") # The quick decision table is a distinct advertised decision surface. Mentions # in later descriptive sections must not be allowed to mask a missing table row. decision_lines = section_lines(catalog, "## Quick decision table") if not decision_lines: - die("CATALOG.md is missing a non-empty ## Quick decision table section") + die("CATALOG.md is missing a non-empty visible ## Quick decision table section") decision_table = extract_markdown_table( decision_lines, ("Bottleneck / problem shape", "First record to inspect", "Core idea"), @@ -443,17 +450,17 @@ def require_prefixed_fields( if not problem_contract.is_file(): die("OPTIMIZATION-PROBLEM.md is missing") problem_text = problem_contract.read_text(encoding="utf-8") -problem_lines = problem_text.splitlines() +problem_lines = visible_nonfenced_lines(problem_text.splitlines()) if not problem_lines or problem_lines[0] != "# Optimization Problem Contract": - die("OPTIMIZATION-PROBLEM.md has missing/invalid title") + die("OPTIMIZATION-PROBLEM.md has missing/hidden/invalid title") if "## Canonical contract" not in problem_lines: - die("OPTIMIZATION-PROBLEM.md is missing ## Canonical contract") + die("OPTIMIZATION-PROBLEM.md is missing visible ## Canonical contract") canonical = section_lines(problem_text, "## Canonical contract") canonical_text = "\n".join(canonical) if "P = (X, F, f, d, C, B, S)" not in canonical_text: - die("OPTIMIZATION-PROBLEM.md is missing canonical P = (X, F, f, d, C, B, S) formula") + die("OPTIMIZATION-PROBLEM.md is missing visible canonical P = (X, F, f, d, C, B, S) formula") for field, pattern in CANONICAL_DEFINITION_PATTERNS.items(): if not any(pattern.match(line) for line in canonical): - die(f"OPTIMIZATION-PROBLEM.md is missing canonical definition for {field}") + die(f"OPTIMIZATION-PROBLEM.md is missing visible canonical definition for {field}") print(f"CATALOG_INTEGRITY_OK records={len(records)} frozen_v1={len(FROZEN_V1)}") From 5f5f6ec24d336a2498a29d19cab6e818fb397c53 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 18:39:05 +0930 Subject: [PATCH 021/229] Linearize incremental generation publication --- ...1-signature-bound-incremental-execution.md | 34 +++++++++++-------- 1 file changed, 19 insertions(+), 15 deletions(-) diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index c0e3815..d583812 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -14,25 +14,25 @@ Expensive work is rerun even though every input capable of affecting its result ## Optimization problem contract -- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity/consumption policies, immutable-output handles, immutable-input snapshot/mutation-control policies, monotonic mutation epochs, and crash-consistent state-publication mechanisms -- F: configurations whose signature covers every output-affecting input, whose execution consumes one immutable effective-input snapshot or is protected by a mutation lock/monotonic mutation witness that detects every intervening change, whose reuse validates required outputs and binds downstream consumption to the exact validated output versions, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted/raced executions never publish reusable partial state -- f: measured repeated-work cost including stage runtime plus signature/snapshot/mutation-tracking/metadata/output-validation/output-pinning/publication I/O overhead +- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity/consumption policies, immutable-output handles, immutable-input snapshot/mutation-control policies, monotonic mutation epochs, compare-and-publish activation policies, and crash-consistent state-publication mechanisms +- F: configurations whose signature covers every output-affecting input, whose execution consumes one immutable effective-input snapshot or is protected by a mutation lock/monotonic mutation witness that detects every intervening change, whose reuse validates required outputs and binds downstream consumption to the exact validated output versions, whose authoritative generation switch is linearized with the witnessed input state, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted/raced executions never publish reusable partial or stale-current state +- f: measured repeated-work cost including stage runtime plus signature/snapshot/mutation-tracking/metadata/output-validation/output-pinning/compare-and-publish/publication I/O overhead - d: minimize -- C: every reused output consumed downstream is the same immutable/versioned output instance whose validity predicate passed, and is semantically equivalent to a fresh execution for the same effective inputs with the same failure semantics; reuse metadata cannot mix fields from different generations; a committed generation cannot bind a pre-execution signature to output produced from changed or mixed inputs; A→B→A mutations during execution are detected rather than erased by endpoint equality +- C: every reused output consumed downstream is the same immutable/versioned output instance whose validity predicate passed, and is semantically equivalent to a fresh execution for the same effective inputs with the same failure semantics; reuse metadata cannot mix fields from different generations; a committed generation cannot bind a pre-execution signature to output produced from changed or mixed inputs; A→B→A mutations during execution are detected rather than erased by endpoint equality; and no input mutation may linearize between the final accepted input witness and activation of that generation as authoritative for those inputs - B: target-specific benchmark/evaluation budget declared before tuning; no portable value is supplied by this record - S: stop when the declared budget is exhausted or a validated configuration meets the predeclared improvement threshold without violating C -- Variables: categorical / mixed policy choices for signatures, snapshots, validation, output pinning, granularity, mutation control, and publication +- Variables: categorical / mixed policy choices for signatures, snapshots, validation, output pinning, granularity, mutation control, compare-and-publish activation, and publication - Search scope: local to one incremental stage or pipeline boundary - Objective behavior: noisy for performance; correctness identity/mutation checks are deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive depending on stage runtime and validation cost -- Constraints: semantic equivalence, input/output integrity, crash consistency, snapshot/mutation consistency, validated-output consumption, and resource constraints -- Parallelism: sequential or pipeline-specific; mutation tracking, output pinning, and publication must remain race-safe under concurrent producers/consumers +- Constraints: semantic equivalence, input/output integrity, crash consistency, snapshot/mutation consistency, validated-output consumption, publication linearizability, and resource constraints +- Parallelism: sequential or pipeline-specific; mutation tracking, output pinning, activation, and publication must remain race-safe under concurrent producers/consumers - Exactness: exact reuse semantics; no approximation is introduced ## Preserved contract -Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and mutable inputs must not change underneath execution without invalidating the candidate generation. Endpoint equality is not enough: if an input changes and later returns to its original bytes, the intervening mutation must still be observable to the publication decision. Likewise, validating a mutable output path is not enough unless downstream consumption is pinned to that exact validated version. +Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and mutable inputs must not change underneath execution without invalidating the candidate generation. Endpoint equality is not enough: if an input changes and later returns to its original bytes, the intervening mutation must still be observable to the publication decision. Likewise, validating a mutable output path is not enough unless downstream consumption is pinned to that exact validated version. Finally, validating the input witness and later switching the authoritative generation are not two independent steps: generation activation must linearize against the same input state that was validated. ## Optimization @@ -40,9 +40,11 @@ Compute a deterministic signature over the effective inputs and compare it with Treat output validation and output consumption as one identity-bound operation. A successful validity check must yield or pin the exact immutable/versioned output handle that downstream consumers will read: for example a content-addressed object, immutable artifact/version ID, snapshot handle, open file descriptor tied to a protected inode/version where the platform guarantees the needed semantics, or another target-specific stable handle. Do **not** validate bytes at a mutable pathname/object name and then later reopen that name for consumption, because another writer may replace it between validation and read. If the storage system cannot provide an immutable/versioned handle, hold an appropriate lock from validation through the downstream read/consumption, or copy the validated bytes into an immutable snapshot and consume that snapshot. Every consumer on a reuse hit must be bound to the validated handle/version, not merely to the same logical path. -Bind execution to one coherent effective-input identity. The preferred design is an **immutable snapshot/version** of every mutable effective input. If a snapshot is unavailable, use a mechanism that records *intervening mutation*, not merely endpoint content equality: for example, hold a read/mutation lock for the full execution-to-publication interval, or capture a monotonically increasing version/epoch for every mutable input and require the exact same epoch vector at publication. Every mutation must advance its epoch durably/atomically with the mutation, including a change that later restores the original bytes. For multiple inputs, capture the snapshot/epoch vector coherently under the target's transaction/locking rules so a mixed vector cannot be mistaken for one state. A content signature recomputed at publication may supplement this check, but **must not be the sole fallback** because A→B→A can make endpoint signatures equal. Any lock violation, epoch change, incoherent snapshot, or untrackable mutable input discards the candidate generation and requires retry from a fresh identity. +Bind execution to one coherent effective-input identity. The preferred design is an **immutable snapshot/version** of every mutable effective input. If a snapshot is unavailable, use a mechanism that records *intervening mutation*, not merely endpoint content equality: for example, hold a read/mutation lock for the full execution-through-activation interval, or capture a monotonically increasing version/epoch for every mutable input. Every mutation must advance its epoch durably/atomically with the mutation, including a change that later restores the original bytes. For multiple inputs, capture the snapshot/epoch vector coherently under the target's transaction/locking rules so a mixed vector cannot be mistaken for one state. A content signature recomputed at publication may supplement this check, but **must not be the sole fallback** because A→B→A can make endpoint signatures equal. Any lock violation, epoch change, incoherent snapshot, or untrackable mutable input discards the candidate generation and requires retry from a fresh identity. -Publish incremental state as one crash-consistent **generation** that binds the validated input signature and immutable snapshot/epoch identity to the complete output identity/validity metadata. Do not persist the signature and output metadata as independently authoritative updates. Use an atomic rename/swap of a complete manifest, a transactional store, a content-addressed generation pointer, or another mechanism where readers observe either the previous complete generation or the new complete generation—never a mixture. Only publish the new generation after every output has been produced and validated successfully **and** the snapshot/mutation witness has passed the final commit-time check; an interrupted, failed, or input-raced publication leaves the previous committed generation authoritative and the candidate generation non-reusable. +The **authoritative generation switch must be linearized with that witness**. A plain sequence of “read epochs; they match; later replace the current-generation manifest” is insufficient. For lock-based designs, retain the mutation/read lock through the atomic generation-pointer/manifest switch. For epoch/version designs, use one transaction, CAS, compare-and-swap manifest operation, or equivalent serialization boundary that atomically verifies the complete witnessed epoch vector still matches **and** activates the new generation; if the comparison fails, the candidate remains non-current and must be discarded or retried. For immutable-snapshot designs, immutable candidate artifacts may be produced independently, but making one authoritative for the live mutable source still requires an atomic check that the live source identity/version remains the captured snapshot identity at the activation point. There must be no interval in which an input mutation can win after the accepted witness check but before the generation becomes authoritative. + +Publish incremental state as one crash-consistent **generation** that binds the validated input signature and immutable snapshot/epoch identity to the complete output identity/validity metadata. Do not persist the signature and output metadata as independently authoritative updates. Use an atomic rename/swap of a complete manifest, a transactional store, a content-addressed generation pointer, or another mechanism where readers observe either the previous complete generation or the new complete generation—never a mixture. Only activate the new generation after every output has been produced and validated successfully **and** the input witness plus authoritative-generation switch have succeeded in the same linearizable activation boundary; an interrupted, failed, input-raced, or compare-and-publish-failed candidate leaves the previous committed generation authoritative and the candidate generation non-reusable. Reuse filesystem/configuration metadata lazily only while its own validity predicate still holds. @@ -64,18 +66,20 @@ Test unchanged inputs with valid outputs, changed inputs, missing outputs, faile Exercise an **output validation-to-consumption race**. Arrange a reuse hit for output A, validate A successfully, then have another writer replace the mutable path/object with B before the consumer reads. Prove the consumer still reads the pinned immutable/versioned A that was validated, or prove the lock prevents replacement until consumption completes. Repeat with delete/recreate, atomic rename, symlink/object-pointer replacement, version rollback, and multiple required outputs where one is swapped after validation. A test that merely corrupts output before validation is insufficient; the mutation must occur after the validity predicate succeeds and before/downstream consumption. -Exercise **concurrent input mutation**, including explicit A→B→A races. Start execution from identity A, mutate one or more effective inputs during execution to B (including mixed-state multi-file/config changes), then restore the original bytes before publication. For snapshot-based targets, prove execution reads only the immutable A snapshot. For lock-based targets, prove the mutation cannot interleave with the protected execution/publication interval. For epoch/version-based targets, prove every mutation increments the monotonic witness and that the final epoch vector differs even when the final content signature returns to A. Reject/discard the candidate on any mutation witness change. Compare every accepted generation with a forced-fresh execution over the exact committed input identity. +Exercise **concurrent input mutation**, including explicit A→B→A races. Start execution from identity A, mutate one or more effective inputs during execution to B (including mixed-state multi-file/config changes), then restore the original bytes before publication. For snapshot-based targets, prove execution reads only the immutable A snapshot. For lock-based targets, prove the mutation cannot interleave with the protected execution/activation interval. For epoch/version-based targets, prove every mutation increments the monotonic witness and that the final epoch vector differs even when the final content signature returns to A. Reject/discard the candidate on any mutation witness change. Compare every accepted generation with a forced-fresh execution over the exact committed input identity. + +Add the **final-check-to-generation-switch race** explicitly. Pause after the implementation has obtained what would otherwise be its successful final input witness but before the authoritative manifest/generation pointer is switched. Attempt an input mutation in that exact interval. A lock-based implementation must block the mutation until after activation; an epoch/CAS implementation must make either the mutation or the generation switch win one serialization order, and if the mutation wins the compare-and-publish must fail rather than activate stale A as current. Repeat with multi-input epoch vectors and concurrent publishers. No test may accept a design where “check succeeded” and “manifest switched” are separate unprotected events. -Exercise interruption/crash injection at every publication boundary: before outputs complete, after outputs complete but before final snapshot/epoch validation, after validation but before manifest publication, during temporary-manifest write, immediately before/after the atomic generation switch, and during cleanup. After each interruption, prove readers observe only a self-consistent old or new generation and can never pair signature A with output identities/metadata from generation B. +Exercise interruption/crash injection at every publication boundary: before outputs complete, after outputs complete but before final input-witness/activation transaction, during the atomic compare-and-publish or lock-protected generation switch, immediately before/after the switch, and during cleanup. After each interruption, prove readers observe only a self-consistent old or new generation and can never pair signature A with output identities/metadata from generation B or activate a generation whose input witness lost the activation race. ## Target-repo adaptation -Re-profile signature and output-validation cost, immutable-output handle/pinning cost, immutable-input snapshot or mutation-lock/epoch cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Define how a successful output-validity check returns/pins the exact version consumed downstream; if mutable storage is unavoidable, define the lock scope or immutable-copy boundary. Explicitly choose whether mutable inputs are consumed from immutable snapshots, protected by locks, or guarded by monotonic mutation epochs; define how every mutation advances the witness and how a coherent multi-input witness is captured. Do not advertise commit-time content rehashing alone as sufficient mutation detection. Also define the crash-consistency guarantee for committing the signature plus output identities. +Re-profile signature and output-validation cost, immutable-output handle/pinning cost, immutable-input snapshot or mutation-lock/epoch cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Define how a successful output-validity check returns/pins the exact version consumed downstream; if mutable storage is unavoidable, define the lock scope or immutable-copy boundary. Explicitly choose whether mutable inputs are consumed from immutable snapshots, protected by locks, or guarded by monotonic mutation epochs; define how every mutation advances the witness and how a coherent multi-input witness is captured. Define the **linearizable activation primitive**: either the mutation lock remains held through the authoritative generation switch or the system atomically compare-and-publishes against the complete witnessed epoch/version vector. Do not advertise commit-time content rehashing, an epoch read followed by a later manifest swap, or any other check-then-publish sequence as sufficient mutation protection. Also define the crash-consistency guarantee for committing the signature plus output identities. ## Failure modes -Incomplete signatures create stale reuse; input mutation during execution can bind an old signature to new/mixed output; A→B→A races can defeat endpoint signature comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent per-input epoch reads can represent no real source state; validating a mutable output name and reopening it later can consume different unvalidated bytes; output-version handles that are not actually immutable/pinned can create time-of-check/time-of-use reuse bugs; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. +Incomplete signatures create stale reuse; input mutation during execution can bind an old signature to new/mixed output; A→B→A races can defeat endpoint signature comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent per-input epoch reads can represent no real source state; a mutation that wins after the final witness read but before an unprotected manifest switch can make stale state authoritative; validating a mutable output name and reopening it later can consume different unvalidated bytes; output-version handles that are not actually immutable/pinned can create time-of-check/time-of-use reuse bugs; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. ## Rollback trigger -Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference; if a consumer can read bytes/objects different from the exact output version whose validity predicate passed; if an A→B→A or other mutable-input race can publish a generation without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input identity; if external output mutation can bypass the declared validity/consumption binding; if crash/interruption testing can expose mixed-generation state; or if signature/snapshot/mutation-tracking/output-pinning/integrity/publication maintenance costs more than the avoided work. +Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference; if a consumer can read bytes/objects different from the exact output version whose validity predicate passed; if an A→B→A or other mutable-input race can publish a generation without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input identity; if an input mutation can linearize between the accepted final witness and authoritative generation activation; if external output mutation can bypass the declared validity/consumption binding; if crash/interruption testing can expose mixed-generation or stale-current state; or if signature/snapshot/mutation-tracking/output-pinning/integrity/activation/publication maintenance costs more than the avoided work. From 83ff67cb074d37e55cdf6c0f3ce85e1c67bd55e9 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 18:39:49 +0930 Subject: [PATCH 022/229] Linearize coalesced operation launch --- ...01-concurrent-duplicate-work-coalescing.md | 28 +++++++++++-------- 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index 5d955bc..ace234f 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,11 +14,11 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime policies, waiter limits, overflow/backpressure policies, per-waiter cancellation/deadline/terminal-claim policies, result-preparation/clone-failure policies, retry/error-sharing policies, and result-ownership policies -- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, preparation-failure, and error semantics for every joined caller, linearize cancellation/deadline against terminal delivery for each waiter, bound waiter memory, and never admit new waiters to a closing or terminal generation -- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, waiter-memory, result preparation/cloning, atomic terminal-claim, overflow/backpressure, and result-copy overhead +- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime and launch-state policies, waiter limits, overflow/backpressure policies, per-waiter cancellation/deadline/terminal-claim policies, result-preparation/clone-failure policies, retry/error-sharing policies, and result-ownership policies +- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, preparation-failure, launch-cancellation, and error semantics for every joined caller, linearize cancellation/deadline against terminal delivery for each waiter, linearize unstarted-to-running launch against closing/last-waiter cancellation, bound waiter memory, and never admit new waiters to a closing or terminal generation +- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, waiter-memory, launch-state synchronization, result preparation/cloning, atomic terminal-claim, overflow/backpressure, and result-copy overhead - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, deadline, and result-preparation outcome; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; closing/terminal generations are not joinable; waiter overflow has an explicit bounded behavior +- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, deadline, launch state, and result-preparation outcome; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; no upstream operation may start after its generation has already become closing due to loss of all live waiters; closing/terminal generations are not joinable; waiter overflow has an explicit bounded behavior - B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied here - S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C - Variables: categorical / integer / mixed @@ -26,25 +26,27 @@ Many callers request the same expensive computation concurrently before any call - Objective behavior: noisy under concurrent load; semantic equivalence remains deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive concurrent-load testing -- Constraints: semantic equivalence, authorization, ownership, result-preparation failure, waiter-memory, cancellation, deadline, terminal-claim, timeout, and resource constraints +- Constraints: semantic equivalence, authorization, ownership, launch-state linearizability, result-preparation failure, waiter-memory, cancellation, deadline, terminal-claim, timeout, and resource constraints - Parallelism: asynchronous / concurrent - Exactness: exact request/result semantics; no approximation is introduced ## Preserved contract -Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout/deadline, result-ownership, preparation-failure, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. A configured waiter bound must never be exceeded silently. Each waiter reaches exactly one linearized terminal state; a waiter that has already cancelled or timed out cannot later receive the shared value/error, and a mutable-result clone failure cannot leave a waiter marked successful without a deliverable value. +Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout/deadline, result-ownership, preparation-failure, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. A configured waiter bound must never be exceeded silently. Each waiter reaches exactly one linearized terminal state; a waiter that has already cancelled or timed out cannot later receive the shared value/error, and a mutable-result clone failure cannot leave a waiter marked successful without a deliverable value. The shared operation also has a linearized launch lifecycle: a generation closed before launch may never subsequently start ownerless upstream work. ## Optimization Create an in-flight registry entry for a canonical equivalence key. The key must include every request attribute required to establish safe sharing, including authorization-relevant tenant/principal/visibility scope unless the target instead proves that the upstream result is globally shareable and independently authorizes each delivered result. -Atomically create the joinable generation **with the initiating caller already registered as its first waiter before invoking, scheduling, or otherwise allowing the upstream operation to run**. This prevents an immediately/synchronously completing operation from reaching terminal state with an empty waiter set. Only after the first waiter is durably part of the generation may the upstream work begin. +Atomically create the joinable generation **with the initiating caller already registered as its first waiter** and with an explicit launch state such as `unstarted`. Do not invoke, schedule, or otherwise permit upstream work yet. This prevents an immediately/synchronously completing operation from reaching terminal state with an empty waiter set while also giving early cancellation a state it can close before any work exists. + +Linearize upstream launch against the generation's live-waiter and closing state. Under the same registry lock/CAS/transactional boundary used for generation state, permit `unstarted -> running` only while the generation remains joinable and has at least one live `pending` waiter. Install or bind a **sticky upstream cancellation token/handle** as part of that transition, before releasing the serialization boundary. If the last waiter cancels/times out while the generation is still `unstarted`, transition it to `closing/non-joinable` and make any later launch attempt fail; no upstream work is started. If `unstarted -> running` wins first but actual invocation/scheduling occurs immediately afterward, any last-waiter cancellation that races in that interval must set the already-bound sticky cancellation token. The launcher must check/attach that token before or atomically with invocation so a cancellation that has already won cannot be lost merely because the external operation object did not yet exist. There must be no path where the generation is closed with zero live waiters and a creator later launches uncancelled work from stale local state. Equivalent later callers may register as independent waiters only while the generation is joinable and the configured waiter capacity remains. Waiter admission is atomic with capacity accounting. When the final waiter slot is already occupied, apply one explicit target policy rather than silently exceeding the bound: reject/return a documented overload or retryable-backpressure result, block/queue the caller behind a separately bounded admission mechanism, or use another bounded policy with explicit timeout/cancellation semantics. Starting an unconstrained parallel generation for the same equivalence key is not the default overflow behavior because it recreates the duplicate upstream load this pattern is intended to prevent. If a target deliberately permits overflow generations, that concurrency bound and duplicate-work tradeoff must be part of C/B and validated separately. Represent each admitted waiter with an atomic terminal state, initially `pending`. Cancellation attempts atomically claim `pending -> cancelled`; timeout/deadline handling atomically claims `pending -> timed-out`. A terminal notifier may claim a delivery outcome only if the waiter's declared deadline has not expired at the claim point. If the deadline is already expired, the notifier must instead leave/transition that waiter to the target's timed-out state and must not deliver the shared terminal value/error. For explicit cancellation racing completion, whichever atomic transition claims `pending` first wins; the losing transition is a no-op for that waiter. These claim semantics are part of the public request contract and must not depend on scheduler timing after the claim. -Cancellation and timeout are otherwise per waiter: when one waiter leaves through a winning cancellation/timeout claim, remove only that waiter from the live-waiter accounting. If live waiters remain, keep the shared generation joinable. If the last live waiter leaves and the policy calls for upstream cancellation, atomically mark the registry entry **closing/non-joinable** (or remove it from the joinable map) before sending the asynchronous cancellation request upstream. A new caller arriving after that transition must create a fresh generation rather than attach to work already being canceled. The closing generation may remain internally tracked until its terminal completion for cleanup/accounting, but it is not eligible for coalescing. +Cancellation and timeout are otherwise per waiter: when one waiter leaves through a winning cancellation/timeout claim, remove only that waiter from the live-waiter accounting. If live waiters remain, keep the shared generation joinable. If the **last** live waiter leaves, atomically make the generation closing/non-joinable. If it is still `unstarted`, this closure permanently prevents the launch transition. If it is already `running`, set/trigger the sticky upstream cancellation token according to the declared policy. A new caller arriving after the closing transition must create a fresh generation rather than attach to work being prevented/canceled. The closing generation may remain internally tracked until its launch-prevention or terminal cleanup is complete, but it is not eligible for coalescing. On upstream success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the candidate waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Snapshot the waiter records, but do not treat membership in that snapshot as entitlement to delivery: each waiter still competes through its atomic terminal state. @@ -68,11 +70,13 @@ This differs from caching: the reusable result does not exist yet. Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; race a new caller against the last-waiter cancellation transition and prove it never joins the closing generation; race a new caller against success/failure completion and prove the terminal generation is made non-joinable before waiter snapshot/notification; test waiter-specific deadlines; verify shared failure delivery and retry accounting; prove only one upstream evaluation occurs per joinable generation while all surviving callers terminate correctly. +Add an explicit **registration-to-launch cancellation race**. Pause after the generation and initiating waiter have been registered but before `unstarted -> running`. Cancel or time out that initiating waiter as the last live waiter, then release the launcher. Prove the generation becomes closing/non-joinable and upstream work is never started. In the opposite interleaving, let `unstarted -> running` win but pause before the external invocation exists; then cancel the last waiter and prove the sticky token is already set/observable so the subsequent invocation is suppressed or immediately canceled according to policy. Repeat under high contention and prove no zero-waiter generation can leak a running/hung upstream operation. + Add **terminal-delivery races** for both upstream success and upstream failure. Pause after the terminal waiter snapshot, then race explicit cancellation and deadline expiry against each waiter's delivery claim. Prove exactly one `pending -> terminal` transition wins, cancelled/timed-out waiters never receive a later value/error, completion that legitimately claims before cancellation preserves the declared completion result, and an already-expired deadline cannot be bypassed merely because the timeout worker has not run yet. Repeat under high concurrency and verify no waiter observes two terminal outcomes. Add **clone/preparation-failure fixtures** for mutable results. Force allocation, serialization, copy-on-write setup, or quota failure while preparing a per-waiter value. Verify a waiter is still `pending` until preparation succeeds; successful preparation followed by a winning cancellation/timeout causes the prepared value to be discarded; failed preparation can atomically resolve to exactly one `delivered-error` only if cancellation/timeout has not already won; and no clone failure can leave a waiter in `delivered-success` without an actual value. Race clone success/failure against cancellation and deadline expiry repeatedly under load. -Add an **immediate synchronous-completion** fixture where the upstream operation can finish inline at launch. Prove the initiating caller was already registered before launch and always receives the terminal result/error unless its own cancellation/deadline claim wins under the same rules. +Add an **immediate synchronous-completion** fixture where the upstream operation can finish inline once launch actually begins. Prove the initiating caller was already registered before launch and always receives the terminal result/error unless its own cancellation/deadline claim wins under the same rules. Add authorization-boundary fixtures: issue syntactically identical requests under different tenants, principals, roles, ACL/visibility scopes, or other authorization context. Prove they either map to different equivalence keys **or** that the shared upstream result is explicitly safe to reuse and each caller is independently authorized before delivery. Verify that a result produced under one authorization scope can never leak to another merely because the resource parameters match. @@ -82,12 +86,12 @@ Add waiter-overflow races: fill the waiter list to one slot below the maximum, l ## Target-repo adaptation -Define key canonicalization, the authorization/visibility context that participates in equivalence, maximum waiter count, bounded overflow/backpressure semantics, result ownership/share-safety policy, how mutable per-waiter results are prepared and how preparation failures surface, the per-waiter atomic terminal-state representation, cancellation/deadline winning semantics, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. +Define key canonicalization, the authorization/visibility context that participates in equivalence, maximum waiter count, bounded overflow/backpressure semantics, result ownership/share-safety policy, how mutable per-waiter results are prepared and how preparation failures surface, the per-waiter atomic terminal-state representation, cancellation/deadline winning semantics, the generation launch states and serialization primitive for `unstarted -> running` versus `closing`, the sticky cancellation-token/handle semantics used before an external operation object exists, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent or authorization-distinct work; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; non-linearized cancellation/deadline versus delivery can produce late values/errors or double terminal outcomes; claiming success before a mutable per-waiter value is successfully prepared can strand a waiter with no deliverable result; clone/preparation failure can race cancellation and create inconsistent outcomes if not claimed atomically; treating terminal snapshot membership as delivery entitlement can notify a waiter after it has timed out; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. +Over-broad keys merge non-equivalent or authorization-distinct work; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; registering first but launching from stale creator state after the last waiter already closed an unstarted generation can leak ownerless work; cancellation issued before an external operation exists can be lost without a sticky token or atomic launch state; non-linearized cancellation/deadline versus delivery can produce late values/errors or double terminal outcomes; claiming success before a mutable per-waiter value is successfully prepared can strand a waiter with no deliverable result; clone/preparation failure can race cancellation and create inconsistent outcomes if not claimed atomically; treating terminal snapshot membership as delivery entitlement can notify a waiter after it has timed out; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/preparation-error semantics; if a waiter can enter `delivered-success` before an isolated deliverable result exists; if clone/preparation failure can produce no terminal outcome or a second terminal outcome; if a cancelled/timed-out waiter can receive a later terminal value/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if the waiter bound or documented overflow behavior is violated; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. +Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/preparation-error semantics; if an upstream operation can start after its generation has become closing with no live waiters; if a pre-launch cancellation can be lost because no cancellation token/operation object existed yet; if a waiter can enter `delivered-success` before an isolated deliverable result exists; if clone/preparation failure can produce no terminal outcome or a second terminal outcome; if a cancelled/timed-out waiter can receive a later terminal value/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if the waiter bound or documented overflow behavior is violated; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. From 747f9a1ac79277d7e9d97487dfd7d73041720883 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 18:40:26 +0930 Subject: [PATCH 023/229] Separate approximation tuning and certification --- ...PROX-001-contract-bounded-approximation.md | 28 +++++++++++-------- 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md index 8c9567b..cddcc1d 100644 --- a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md +++ b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md @@ -15,25 +15,25 @@ Exact processing has unbounded or unacceptable cost even though the product/scie ## Optimization problem contract -- X: target-supported approximation policies, quality/resource ceilings, sampling/culling/LOD policies, update frequencies, state-reset rules, evaluation horizons, and exact-mode fallback choices -- F: policies whose declared error/degradation metric remains within the target's explicit envelope over the declared state/composition horizon and whose resource/semantic constraints are satisfied +- X: target-supported approximation policies, quality/resource ceilings, sampling/culling/LOD policies, update frequencies, state-reset rules, evaluation horizons, exact-mode fallback choices, and stochastic tuning/certification procedures +- F: policies whose declared error/degradation metric remains within the target's explicit envelope over the declared state/composition horizon and whose resource/semantic constraints are satisfied; when a stochastic policy is selected from multiple candidates, feasibility certification is based on independent held-out conformance data or a predeclared selection-aware simultaneous-confidence/multiple-testing procedure rather than naive reuse of the tuning samples - f: target-measured resource or latency cost, optionally paired with the declared quality/error metric - d: minimize resource/latency cost subject to feasibility in F, or use the target's predeclared multi-objective ordering when quality is ranked rather than hard-bounded -- C: approximation is permitted only by an explicit contract; exact callers are not silently weakened; the error norm, aggregation rule, sequence/composition horizon, reset boundaries, and whether the envelope is hard worst-case or statistical/confidence/tail-based are declared before evaluation; an exact reference path or exact fixture remains available where practical -- B: target-specific benchmark/quality-evaluation budget over predeclared ordinary, boundary, adversarial, repeated-application, and long-horizon fixtures -- S: stop when the evaluation budget is exhausted or a validated policy meets the target resource objective while remaining inside the declared quality envelope over the entire declared horizon +- C: approximation is permitted only by an explicit contract; exact callers are not silently weakened; the error norm, aggregation rule, sequence/composition horizon, reset boundaries, whether the envelope is hard worst-case or statistical/confidence/tail-based, and the stochastic tuning-versus-certification procedure are declared before evaluation; an exact reference path or exact fixture remains available where practical +- B: target-specific benchmark/quality-evaluation budget over predeclared ordinary, boundary, adversarial, repeated-application, and long-horizon fixtures; for stochastic policy search, tuning/selection evaluations and independent certification evaluations (or the budget used by the predeclared simultaneous-confidence procedure) are accounted separately +- S: stop when the evaluation budget is exhausted or a selected policy meets the target resource objective and passes the declared conformance certification while remaining inside the quality envelope over the entire declared horizon - Variables: continuous / integer / categorical / conditional / mixed, depending on approximation policy - Search scope: local or global, explicitly declared for the target - Objective behavior: deterministic, noisy, or stochastic depending on the quality/resource metric - Information: derivative-free / black-box by default -- Evaluation cost: moderate to expensive when exact references or long-horizon trajectories are required -- Constraints: explicit error envelope, semantic/API, resource, horizon/reset, and exact-fallback constraints +- Evaluation cost: moderate to expensive when exact references, held-out certification, or long-horizon trajectories are required +- Constraints: explicit error envelope, semantic/API, resource, horizon/reset, selection-aware statistical certification, and exact-fallback constraints - Parallelism: sequential, synchronous batch, or asynchronous according to target evaluation; stateful validation must preserve trajectory semantics - Exactness: approximation explicitly permitted only inside the declared measurable envelope ## Preserved contract -Approximation is admissible only when the contract explicitly permits it. A previously exact API cannot be silently weakened and still be called correctness-preserving. For stateful or repeatedly composed approximations, the contract applies over an explicitly declared horizon—not merely to each isolated step—so bounded per-step error is insufficient if drift can accumulate beyond the allowed envelope. The contract must also state whether compliance is pointwise/worst-case or statistical; a stochastic envelope is judged by its declared aggregation, confidence, exceedance-probability, quantile, or tail criterion rather than by silently substituting a hard per-sample limit. +Approximation is admissible only when the contract explicitly permits it. A previously exact API cannot be silently weakened and still be called correctness-preserving. For stateful or repeatedly composed approximations, the contract applies over an explicitly declared horizon—not merely to each isolated step—so bounded per-step error is insufficient if drift can accumulate beyond the allowed envelope. The contract must also state whether compliance is pointwise/worst-case or statistical; a stochastic envelope is judged by its declared aggregation, confidence, exceedance-probability, quantile, or tail criterion rather than by silently substituting a hard per-sample limit. When multiple stochastic policies are tuned or screened, choosing the apparent winner changes the sampling distribution: the data used to optimize/select a policy cannot be treated as independent nominal-confidence certification evidence unless the declared procedure explicitly accounts for that selection. ## Optimization @@ -43,6 +43,8 @@ For stateful streaming, simulation, DSP, iterative numerical work, or any repeat For stochastic/noisy approximations, also define the statistical compliance rule before evaluation: the sampling unit and workload distribution, aggregation statistic, confidence level or interval procedure, tolerated exceedance probability, quantile/tail bound, and the sample/evaluation budget used to decide compliance. Do not reinterpret a statistical guarantee as a pointwise worst-case guarantee, and do not weaken a declared hard worst-case envelope into an average-case claim after observing data. +If more than one stochastic approximation policy is tuned, compared, adaptively searched, thresholded, or screened using sampled error data, **separate selection from certification**. The default pattern is to use one predeclared tuning/selection set (or stream) to choose the candidate and then evaluate that frozen candidate on an independent held-out conformance set drawn from the declared operational distribution. If independent holdout is impractical, use a predeclared selection-aware method that preserves the advertised guarantee across the entire candidate-selection procedure—for example simultaneous confidence bounds, family-wise/multiple-testing correction, valid selective-inference/e-process machinery, or another target-justified method. A nominal per-policy confidence interval computed on the same samples used to select the best-looking policy is not certification. Record exactly which evaluations influenced policy selection and which evaluations supported the final compliance claim. + ## Before / after evidence - Environment: No controlled target-repository benchmark has been run for this OPT record. @@ -53,20 +55,22 @@ For stochastic/noisy approximations, also define the statistical compliance rule ## Validation -Measure error/degradation and resource savings together across ordinary, boundary and adversarial workloads. Keep an exact reference for differential evaluation where practical. Declare and test the error norm/metric, aggregation rule, sequence/composition horizon, reset boundaries, and hard-versus-statistical envelope semantics explicitly. +Measure error/degradation and resource savings together across ordinary, boundary and adversarial workloads. Keep an exact reference for differential evaluation where practical. Declare and test the error norm/metric, aggregation rule, sequence/composition horizon, reset boundaries, hard-versus-statistical envelope semantics, and tuning-versus-certification procedure explicitly. For stateful/repeated use, run differential trajectories against the exact path across short, nominal, maximum-supported, and adversarially long sequences. Include biased-error fixtures where each individual step remains within the local ε but errors accumulate in the same direction; verify the cumulative/state error still respects the declared horizon envelope. Test reset/checkpoint boundaries before, at, and after the limit; verify resets actually restore the assumptions used by the next horizon. Where stochastic approximation is used, evaluate the declared expected, quantile, exceedance-probability, confidence, tail, or worst-case criterion against the exact path as specified by C. Include fixtures where individual samples exceed a nominal pointwise value while the declared statistical envelope remains satisfied, and fixtures where the configured tail/confidence/exceedance criterion truly fails. Verify rollback decisions distinguish those cases rather than triggering on one sample unless the contract explicitly declares a hard single-sample/worst-case bound. +Add **selection-bias fixtures** whenever multiple stochastic policies are considered. Generate several candidate policies whose apparent sampled errors vary by chance, select the best-looking candidate using the declared tuning procedure, and prove that the final compliance decision uses either fresh held-out samples unavailable to selection or the declared simultaneous/selection-aware inference procedure. Verify the tuning samples alone cannot certify the selected winner at nominal per-policy confidence. Include repeated/adaptive candidate selection, early stopping, and candidate-count changes; confirm the advertised confidence/tail/exceedance guarantee remains valid under the complete selection procedure. Persist an audit trail labeling each evaluation as tuning/selection, certification, or both only when the declared selection-aware method formally permits dual use. + ## Target-repo adaptation -Define `ε`, the exact quality/error norm, aggregation rule, workload distribution, maximum state/composition horizon, reset/checkpoint semantics, long-run drift policy, escape hatch and exact-mode availability locally. Explicitly classify the quality envelope as hard pointwise/worst-case or statistical, and for statistical contracts specify the confidence/tail/exceedance rule and decision sample budget. If the target has no finite operational horizon, establish a justified asymptotic/stability bound or periodic re-synchronization rule instead of copying a finite benchmark horizon from another system. +Define `ε`, the exact quality/error norm, aggregation rule, workload distribution, maximum state/composition horizon, reset/checkpoint semantics, long-run drift policy, escape hatch and exact-mode availability locally. Explicitly classify the quality envelope as hard pointwise/worst-case or statistical, and for statistical contracts specify the confidence/tail/exceedance rule and decision sample budget. If multiple policies are searched or compared, predeclare the tuning/selection dataset or stream, the independent certification dataset/budget, **or** the exact simultaneous-confidence/multiple-testing/selective-inference method that makes data reuse valid; record which observations affected selection versus certification. If the target has no finite operational horizon, establish a justified asymptotic/stability bound or periodic re-synchronization rule instead of copying a finite benchmark horizon from another system. ## Failure modes -Unmeasured quality loss, biased sampling, hidden rare-case failures, cumulative drift that is invisible to one-step checks, reset boundaries that fail to restore reference assumptions, state-dependent amplification, unstable feedback loops, misclassifying a statistical envelope as a hard pointwise bound (or vice versa), and callers incorrectly assuming exact semantics. +Unmeasured quality loss, biased sampling, hidden rare-case failures, cumulative drift that is invisible to one-step checks, reset boundaries that fail to restore reference assumptions, state-dependent amplification, unstable feedback loops, selecting the best-looking stochastic policy and then certifying it on the same data with naive per-policy confidence, undisclosed adaptive candidate search/early stopping that invalidates nominal error guarantees, misclassifying a statistical envelope as a hard pointwise bound (or vice versa), and callers incorrectly assuming exact semantics. ## Rollback trigger -Evaluate rollback against the **declared envelope semantics**. For a hard pointwise/worst-case contract, disable immediately when any supported-horizon observation exceeds the declared bound. For a stochastic/statistical contract, disable when the predeclared aggregation, confidence, exceedance-probability, quantile, or tail criterion fails under its stated evaluation procedure; an isolated sample beyond a nominal pointwise value is not by itself a contract violation unless the contract says it is. In all cases, disable when cumulative/state drift violates its declared bound, reset/checkpoint validation fails, reference comparisons violate C, a catastrophic semantic/safety constraint is breached, or resource savings are not material. +Evaluate rollback against the **declared envelope semantics and certification procedure**. For a hard pointwise/worst-case contract, disable immediately when any supported-horizon observation exceeds the declared bound. For a stochastic/statistical contract, disable when the predeclared aggregation, confidence, exceedance-probability, quantile, or tail criterion fails under its stated **selection-aware certification** procedure; an isolated sample beyond a nominal pointwise value is not by itself a contract violation unless the contract says it is. Treat a selected policy as uncertified—and disable or fall back—if held-out certification fails, if tuning and certification evidence are mixed contrary to the declared procedure, or if the simultaneous/multiple-testing/selective-inference assumptions required for data reuse are violated. In all cases, disable when cumulative/state drift violates its declared bound, reset/checkpoint validation fails, reference comparisons violate C, a catastrophic semantic/safety constraint is breached, or resource savings are not material. From 711f1fbae3559fa3a81e1c59c42dcca69f76acd5 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:06:33 +0930 Subject: [PATCH 024/229] Harden Markdown fence parsing --- scripts/check_catalog.py | 244 ++++++++++++++++++--------------------- 1 file changed, 112 insertions(+), 132 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index c08453e..b7fbf27 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -73,6 +73,7 @@ "What can make this optimization invalid, slower, less robust or misleading?", "Define the measured or semantic condition that disables/reverts the optimization.", } + LINK_RE = re.compile(r"\[([^\]]+)\]\((optimizations/[^)#]+\.md)\)") ID_RE = re.compile(r"^# (OPT-[A-Z]+-\d{3}) — ") FILENAME_ID_RE = re.compile(r"^(OPT-[A-Z]+-\d{3})-") @@ -90,9 +91,9 @@ ) HEADING_RE = re.compile(r"^#{1,6}(?:\s|$)") THEMATIC_BREAK_RE = re.compile(r"^(?:-{3,}|\*{3,}|_{3,})$") -FENCE_RE = re.compile(r"^(?:```|~~~)") LIST_MARKER_ONLY_RE = re.compile(r"^(?:[-+*]|\d+[.)])$") TABLE_SEPARATOR_CELL_RE = re.compile(r"^:?-{3,}:?$") +FENCE_OPEN_RE = re.compile(r"^(`{3,}|~{3,})(.*)$") CANONICAL_DEFINITION_PATTERNS = { "X": re.compile(r"^- `X` — \S"), "F": re.compile(r"^- `F(?: ⊆ X)?` — \S"), @@ -113,28 +114,33 @@ def strip_html_comments(text: str) -> str: def visible_nonfenced_lines(lines: list[str]) -> list[str]: - """Return rendered-ish Markdown lines, excluding comments and fenced examples.""" + """Return rendered-ish Markdown lines, excluding comments and fenced blocks.""" cleaned = strip_html_comments("\n".join(lines)) visible: list[str] = [] - fence: str | None = None + fence_char: str | None = None + fence_len = 0 + for raw in cleaned.splitlines(): stripped = raw.strip() - if fence is not None: - if stripped.startswith(fence): - fence = None + if fence_char is not None: + close = re.fullmatch(rf"{re.escape(fence_char)}{{{fence_len},}}\s*", stripped) + if close is not None: + fence_char = None + fence_len = 0 continue - if stripped.startswith("```"): - fence = "```" - continue - if stripped.startswith("~~~"): - fence = "~~~" + + opener = FENCE_OPEN_RE.match(stripped) + if opener is not None: + run = opener.group(1) + fence_char = run[0] + fence_len = len(run) continue + visible.append(raw) return visible def visible_text(text: str) -> str: - """Return visible, non-fenced Markdown text for semantic integrity checks.""" return "\n".join(visible_nonfenced_lines(text.splitlines())) @@ -163,12 +169,11 @@ def markdown_table_cells(line: str) -> list[str] | None: def extract_markdown_table( lines: list[str], expected_headers: tuple[str, ...], context: str ) -> list[str]: - """Extract one visible Markdown table by exact header and validate every row width.""" + """Extract one visible table by exact header and validate every row width.""" visible = visible_nonfenced_lines(lines) expected = list(expected_headers) for i, line in enumerate(visible): - cells = markdown_table_cells(line) - if cells != expected: + if markdown_table_cells(line) != expected: continue if i + 1 >= len(visible): die(f"{context} table has no separator row") @@ -181,12 +186,12 @@ def extract_markdown_table( die(f"{context} table has an invalid separator row") table = [line, visible[i + 1]] for row in visible[i + 2 :]: - row_cells = markdown_table_cells(row) - if row_cells is None: + cells = markdown_table_cells(row) + if cells is None: break - if len(row_cells) != len(expected): + if len(cells) != len(expected): die( - f"{context} table row has {len(row_cells)} column(s); " + f"{context} table row has {len(cells)} column(s); " f"expected {len(expected)}: {row.strip()}" ) table.append(row) @@ -195,43 +200,29 @@ def extract_markdown_table( def is_structural_only_line(line: str) -> bool: - """Return true for Markdown scaffolding that does not state record content.""" - if HEADING_RE.match(line): - return True - if THEMATIC_BREAK_RE.fullmatch(line): - return True - if FENCE_RE.match(line): - return True - if LIST_MARKER_ONLY_RE.fullmatch(line): + if HEADING_RE.match(line) or THEMATIC_BREAK_RE.fullmatch(line): return True - if line == ">": + if LIST_MARKER_ONLY_RE.fullmatch(line) or line == ">": return True cells = markdown_table_cells(line) - if cells is not None and cells and all( - TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in cells - ): - return True - return False + return bool( + cells + and all(TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in cells) + ) def section_has_content(lines: list[str]) -> bool: - """Require record-specific rendered content, not prompts/scaffolding/examples.""" for raw in visible_nonfenced_lines(lines): line = raw.strip() - if not line: - continue - if line in TEMPLATE_PLACEHOLDER_LINES: + if not line or line in TEMPLATE_PLACEHOLDER_LINES: continue - if EMPTY_LABEL_RE.match(line): - continue - if is_structural_only_line(line): + if EMPTY_LABEL_RE.match(line) or is_structural_only_line(line): continue return True return False def normalized_status_category(raw: str) -> str: - """Normalize light Markdown emphasis, then return the category before ';'.""" plain = re.sub(r"[*_`]", "", raw).strip() return plain.split(";", 1)[0].strip() @@ -243,7 +234,6 @@ def require_prefixed_fields( section: str, rejected_values: dict[str, str] | None = None, ) -> None: - """Require one visible selected non-empty '- Field:' row for every field.""" visible = visible_nonfenced_lines(lines) for field in fields: prefix = f"- {field}:" @@ -277,58 +267,54 @@ def require_prefixed_fields( filename_match = FILENAME_ID_RE.match(path.name) if not filename_match: die(f"record filename does not begin with an OPT ID: {path.relative_to(ROOT)}") - filename_id = filename_match.group(1) - if filename_id != record_id: + if filename_match.group(1) != record_id: die( f"record ID mismatch: {path.relative_to(ROOT)} declares {record_id} " - f"but filename encodes {filename_id}" + f"but filename encodes {filename_match.group(1)}" ) - if record_id in records: die(f"duplicate record id {record_id}: {records[record_id]} and {path}") records[record_id] = path - status_matches = [STATUS_RE.match(line) for line in lines] - statuses = [m.group(1).strip() for m in status_matches if m is not None] + statuses = [m.group(1).strip() for line in lines if (m := STATUS_RE.match(line))] if len(statuses) != 1: die(f"{path.relative_to(ROOT)} must contain exactly one visible Status line") if not statuses[0]: die(f"{path.relative_to(ROOT)} has empty Status") - if record_id not in FROZEN_V1: - status_category = statuses[0].split(";", 1)[0].strip() - if status_category not in ALLOWED_V2_STATUS_CATEGORIES: - die( - f"{path.relative_to(ROOT)} uses undefined status category " - f"'{status_category}'" - ) - status_categories[record_id] = status_category + if record_id in FROZEN_V1: + continue - headings = {line for line in lines if line.startswith("## ")} - missing = sorted(REQUIRED_V2 - headings) - if missing: - die(f"{path.relative_to(ROOT)} missing visible sections: {', '.join(missing)}") + status_category = statuses[0].split(";", 1)[0].strip() + if status_category not in ALLOWED_V2_STATUS_CATEGORIES: + die( + f"{path.relative_to(ROOT)} uses undefined status category " + f"'{status_category}'" + ) + status_categories[record_id] = status_category - for heading in sorted(REQUIRED_V2): - if not section_has_content(section_lines(text, heading)): - die( - f"{path.relative_to(ROOT)} has empty/template/structural-only mandatory section {heading}" - ) + headings = {line for line in lines if line.startswith("## ")} + missing = sorted(REQUIRED_V2 - headings) + if missing: + die(f"{path.relative_to(ROOT)} missing visible sections: {', '.join(missing)}") - contract = section_lines(text, "## Optimization problem contract") - require_prefixed_fields( - path, - contract, - REQUIRED_CONTRACT_FIELDS, - "## Optimization problem contract", - ) - require_prefixed_fields( - path, - contract, - REQUIRED_CLASSIFICATION_FIELDS, - "## Optimization problem contract", - rejected_values=CLASSIFICATION_TEMPLATE_VALUES, - ) + for heading in sorted(REQUIRED_V2): + if not section_has_content(section_lines(text, heading)): + die( + f"{path.relative_to(ROOT)} has empty/template/structural-only mandatory section {heading}" + ) + + contract = section_lines(text, "## Optimization problem contract") + require_prefixed_fields( + path, contract, REQUIRED_CONTRACT_FIELDS, "## Optimization problem contract" + ) + require_prefixed_fields( + path, + contract, + REQUIRED_CLASSIFICATION_FIELDS, + "## Optimization problem contract", + rejected_values=CLASSIFICATION_TEMPLATE_VALUES, + ) missing_frozen = sorted(FROZEN_V1 - records.keys()) if missing_frozen: @@ -336,14 +322,10 @@ def require_prefixed_fields( record_paths = {str(path.relative_to(ROOT)): record_id for record_id, path in records.items()} -# Validate every visible optimization-record link wherever it appears. README index -# completeness/uniqueness is checked separately from its rendered catalog table, -# so contextual prose links are allowed and do not count as duplicate index rows. for doc_name in ("README.md", "CATALOG.md"): text = (ROOT / doc_name).read_text(encoding="utf-8") rendered = visible_text(text) - links = LINK_RE.findall(rendered) - for label, rel in links: + for label, rel in LINK_RE.findall(rendered): target = ROOT / rel if not target.is_file(): die(f"broken visible record link in {doc_name}: {rel}") @@ -356,52 +338,53 @@ def require_prefixed_fields( f"{target_id} ({rel})" ) - if doc_name == "README.md": - catalog_lines = section_lines(text, "## Catalog") - if not catalog_lines: - die("README.md is missing a non-empty visible ## Catalog section") - catalog_table = extract_markdown_table( - catalog_lines, - ("ID", "Optimization", "Status", "Core idea"), - "README.md ## Catalog", + if doc_name != "README.md": + continue + + catalog_lines = section_lines(text, "## Catalog") + if not catalog_lines: + die("README.md is missing a non-empty visible ## Catalog section") + catalog_table = extract_markdown_table( + catalog_lines, + ("ID", "Optimization", "Status", "Core idea"), + "README.md ## Catalog", + ) + rows = README_ROW_RE.findall("\n".join(catalog_table)) + counts = Counter(row_id for row_id, _rel, _status in rows) + bad_counts = sorted(record_id for record_id, count in counts.items() if count != 1) + if bad_counts: + die( + "README.md ## Catalog table must index each record exactly once; " + f"bad row counts for: {', '.join(bad_counts)}" ) - rows = README_ROW_RE.findall("\n".join(catalog_table)) - row_ids = [row_id for row_id, _rel, _status in rows] - counts = Counter(row_ids) - bad_counts = sorted(record_id for record_id, count in counts.items() if count != 1) - if bad_counts: + missing_readme = sorted(records.keys() - counts.keys()) + if missing_readme: + die(f"README.md ## Catalog table is missing record(s): {', '.join(missing_readme)}") + unknown_rows = sorted(counts.keys() - records.keys()) + if unknown_rows: + die(f"README.md ## Catalog table references unknown record(s): {', '.join(unknown_rows)}") + + row_statuses: dict[str, str] = {} + for row_id, rel, raw_status in rows: + if record_paths.get(rel) != row_id: + die(f"README.md ## Catalog row identity mismatch for {row_id}: {rel}") + if row_id in row_statuses: + die(f"README.md ## Catalog has duplicate status row for {row_id}") + row_statuses[row_id] = normalized_status_category(raw_status) + + for record_id, expected_status in status_categories.items(): + observed_status = row_statuses.get(record_id) + if observed_status is None: + die(f"README.md ## Catalog has no status cell for post-v1 record {record_id}") + if observed_status != expected_status: die( - "README.md ## Catalog table must index each record exactly once; " - f"bad row counts for: {', '.join(bad_counts)}" + f"README.md status mismatch for {record_id}: " + f"record='{expected_status}' README='{observed_status}'" ) - missing_readme = sorted(records.keys() - counts.keys()) - if missing_readme: - die(f"README.md ## Catalog table is missing record(s): {', '.join(missing_readme)}") - unknown_rows = sorted(counts.keys() - records.keys()) - if unknown_rows: - die(f"README.md ## Catalog table references unknown record(s): {', '.join(unknown_rows)}") - - row_statuses: dict[str, str] = {} - for row_id, rel, raw_status in rows: - if record_paths.get(rel) != row_id: - die(f"README.md ## Catalog row identity mismatch for {row_id}: {rel}") - if row_id in row_statuses: - die(f"README.md ## Catalog has duplicate status row for {row_id}") - row_statuses[row_id] = normalized_status_category(raw_status) - - for record_id, expected_status in status_categories.items(): - observed_status = row_statuses.get(record_id) - if observed_status is None: - die(f"README.md ## Catalog has no status cell for post-v1 record {record_id}") - if observed_status != expected_status: - die( - f"README.md status mismatch for {record_id}: " - f"record='{expected_status}' README='{observed_status}'" - ) catalog = (ROOT / "CATALOG.md").read_text(encoding="utf-8") -rendered_catalog = visible_text(catalog) -catalog_ids = set(OPT_TOKEN_RE.findall(rendered_catalog)) +visible_catalog = visible_text(catalog) +catalog_ids = set(OPT_TOKEN_RE.findall(visible_catalog)) unknown_catalog_ids = sorted(catalog_ids - records.keys()) if unknown_catalog_ids: die(f"CATALOG.md references unknown visible record ID(s): {', '.join(unknown_catalog_ids)}") @@ -409,8 +392,6 @@ def require_prefixed_fields( if record_id not in catalog_ids: die(f"{record_id} ({path.name}) is not visibly mentioned in CATALOG.md") -# The quick decision table is a distinct advertised decision surface. Mentions -# in later descriptive sections must not be allowed to mask a missing table row. decision_lines = section_lines(catalog, "## Quick decision table") if not decision_lines: die("CATALOG.md is missing a non-empty visible ## Quick decision table section") @@ -420,8 +401,7 @@ def require_prefixed_fields( "CATALOG.md ## Quick decision table", ) decision_rows = CATALOG_DECISION_ROW_RE.findall("\n".join(decision_table)) -decision_ids = [record_id for record_id, _rel in decision_rows] -decision_counts = Counter(decision_ids) +decision_counts = Counter(record_id for record_id, _rel in decision_rows) bad_decision_counts = sorted( record_id for record_id, count in decision_counts.items() if count != 1 ) @@ -450,10 +430,10 @@ def require_prefixed_fields( if not problem_contract.is_file(): die("OPTIMIZATION-PROBLEM.md is missing") problem_text = problem_contract.read_text(encoding="utf-8") -problem_lines = visible_nonfenced_lines(problem_text.splitlines()) -if not problem_lines or problem_lines[0] != "# Optimization Problem Contract": +problem_visible = visible_nonfenced_lines(problem_text.splitlines()) +if not problem_visible or problem_visible[0] != "# Optimization Problem Contract": die("OPTIMIZATION-PROBLEM.md has missing/hidden/invalid title") -if "## Canonical contract" not in problem_lines: +if "## Canonical contract" not in problem_visible: die("OPTIMIZATION-PROBLEM.md is missing visible ## Canonical contract") canonical = section_lines(problem_text, "## Canonical contract") canonical_text = "\n".join(canonical) From 832de5982f88c78ffcafdadbad491a07b4cef90b Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:07:01 +0930 Subject: [PATCH 025/229] Linearize speculative commitment --- ...T-CRIT-001-critical-path-prioritization.md | 20 ++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/optimizations/OPT-CRIT-001-critical-path-prioritization.md b/optimizations/OPT-CRIT-001-critical-path-prioritization.md index 1f1727e..519c98a 100644 --- a/optimizations/OPT-CRIT-001-critical-path-prioritization.md +++ b/optimizations/OPT-CRIT-001-critical-path-prioritization.md @@ -19,7 +19,7 @@ Non-critical work competes with the dependency chain that determines user-visibl - F: policies that preserve all semantic deadlines, avoid externally visible speculative side effects before commitment, commit speculative results only from one stable effective-input generation, and satisfy starvation/resource constraints - f: measured end-to-end latency of the declared critical dependency path, including resource pressure introduced by speculation/deferment - d: minimize -- C: critical outputs and semantic deadlines are preserved; speculative work is safely discardable; any speculative result is bound to a complete immutable snapshot or full-duration mutation witness so intervening A→B→A changes cannot be erased before commitment; deferred work completes before it becomes semantically required +- C: critical outputs and semantic deadlines are preserved; speculative work is safely discardable; any speculative result is bound to a complete immutable snapshot or full-duration mutation witness, and validation of that witness is linearized with commitment so intervening or final-window A→B→A/input changes cannot be erased before visibility; deferred work completes before it becomes semantically required - B: target-specific trace/benchmark budget covering cold/warm, hit/miss, wrong-speculation, stale-speculation, and change/revert cases; no portable prediction horizon is supplied here - S: stop when the declared budget is exhausted or a validated policy materially reduces critical-path latency without violating C - Variables: categorical / conditional / mixed priority, deferment, prefetch, and speculation policies @@ -27,19 +27,23 @@ Non-critical work competes with the dependency chain that determines user-visibl - Objective behavior: noisy under realistic workload timing; semantic identity/deadline checks are deterministic - Information: derivative-free / black-box latency measurements - Evaluation cost: moderate to expensive end-to-end tracing/benchmarking -- Constraints: semantic deadlines, starvation, side effects, input identity/mutation freshness, memory/CPU/I/O, and target resource constraints +- Constraints: semantic deadlines, starvation, side effects, input identity/mutation freshness, commitment linearizability, memory/CPU/I/O, and target resource constraints - Parallelism: asynchronous / concurrent execution is common - Exactness: exact target semantics; speculative work may be discarded but not committed stale ## Preserved contract -Deferred work must still complete before its semantic deadline. Speculative work must be discardable and must not create externally visible side effects before commitment. A speculative result may be committed/delivered only if it was produced from one coherent effective-input generation equivalent to the non-speculative reference path; endpoint equality after an intervening mutation is not sufficient. +Deferred work must still complete before its semantic deadline. Speculative work must be discardable and must not create externally visible side effects before commitment. A speculative result may be committed/delivered only if it was produced from one coherent effective-input generation equivalent to the non-speculative reference path; endpoint equality after an intervening mutation is not sufficient, and a successful freshness check is not sufficient unless the checked identity remains authoritative through the commit that makes the result visible. ## Optimization Execute critical dependencies first; prefetch/precompute likely-soon work only when probability and spare resources justify it; lazily defer non-critical work; avoid work with no demonstrated demand. -Bind every speculative/precomputed result to a complete effective-input identity for the **full speculation-to-commit interval**. Prefer speculation against an immutable snapshot/version. If snapshots are unavailable, use a full-duration mutation/read lock or capture a monotonically increasing, non-reusable version/epoch for every mutable effective input and require the same coherent epoch vector at commitment/delivery. Every relevant mutation must advance its witness, including A→B→A changes that restore original bytes. A commit-time hash/identity comparison may supplement the mutation witness but must not be the sole freshness proof. Any lock violation, epoch change, incoherent witness, or untrackable mutable input discards the speculative result and forces execution/recomputation from the current reference identity. Commitment is the semantic boundary: no stale speculative result may become externally visible merely because the speculation itself had no side effects. +Bind every speculative/precomputed result to a complete effective-input identity for the **full speculation-to-commit interval**. Prefer speculation against an immutable snapshot/version. If snapshots are unavailable, use a full-duration mutation/read lock or capture a monotonically increasing, non-reusable version/epoch for every mutable effective input. Every relevant mutation must advance its witness, including A→B→A changes that restore original bytes. A commit-time hash/identity comparison may supplement the mutation witness but must not be the sole freshness proof. + +Freshness validation and commitment must be **one linearizable operation**. For lock-based targets, hold the mutation/read lock through the exact commit/publication/delivery transition that makes the speculative result externally visible. For epoch/version-based targets, use an atomic compare-and-commit/conditional transaction that verifies the complete coherent epoch vector is still the witnessed vector and, only if that comparison succeeds in the same atomic boundary, publishes the result. A separate `check epochs; later publish` sequence is not sufficient. If the compare-and-commit loses a race, discard the speculative result and execute/recompute from the current reference identity. Any lock violation, epoch change, incoherent witness, or untrackable mutable input likewise forces discard/recompute. + +Commitment is the semantic boundary: no stale speculative result may become externally visible merely because the speculation itself had no side effects. ## Before / after evidence @@ -55,14 +59,16 @@ Trace the true dependency path and measure end-to-end latency, not only individu Add stale-speculation fixtures with explicit **A→B→A** races. Start speculation from identity A, mutate the effective inputs to B while speculation reads/runs, then restore original bytes before demand/commitment. For snapshot-based targets, prove speculation consumed only immutable A. For lock-based targets, prove the mutation cannot interleave. For epoch/version-based targets, prove every mutation increments the monotonic witness and that the final witness exposes the intervening change even though endpoint content equals A. Also test delayed speculative completion, version rollback, and concurrent config/schema changes. Compare every committed speculative result against the non-speculative reference path for the exact committed identity. +Add a **final validation-to-commit race**. Pause immediately after the last ordinary witness comparison but before the result would become visible, then mutate an effective input. For lock-based designs, prove the mutation is blocked until after commitment. For epoch/version designs, prove the atomic compare-and-commit rejects the stale speculative result rather than publishing it. Repeat with A→B→A and multi-input epoch-vector changes. No fixture may pass by doing an ordinary comparison followed by a separate publication step. + ## Target-repo adaptation -Criticality and prediction horizons are workload-specific. Re-profile after topology or user-flow changes. Define the complete effective-input identity for each speculative result and choose immutable snapshots, full-duration mutation locks, or monotonic epochs that record every intervening change. Specify exactly when a stale speculative result is discarded. Do not rely on commit-time endpoint revalidation alone to detect change-and-revert races. +Criticality and prediction horizons are workload-specific. Re-profile after topology or user-flow changes. Define the complete effective-input identity for each speculative result and choose immutable snapshots, full-duration mutation locks, or monotonic epochs that record every intervening change. Define the linearization boundary that couples freshness validation to external commitment: lock-through-commit or atomic compare-and-commit. Specify exactly when a stale speculative result is discarded. Do not rely on commit-time endpoint revalidation alone, or on check-then-publish epoch validation, to establish freshness. ## Failure modes -Speculation steals resources from critical work, lazy work causes later latency cliffs, priorities become stale, deferred tasks starve, semantic deadlines are missed, speculative side effects escape before commitment, A→B→A mutations can fool endpoint-only freshness checks, non-monotonic/reused epochs can erase intervening changes, or stale speculative output is committed after its effective inputs changed. +Speculation steals resources from critical work, lazy work causes later latency cliffs, priorities become stale, deferred tasks starve, semantic deadlines are missed, speculative side effects escape before commitment, A→B→A mutations can fool endpoint-only freshness checks, non-monotonic/reused epochs can erase intervening changes, a check-then-publish window can expose stale speculation after a successful freshness check, or stale speculative output is committed after its effective inputs changed. ## Rollback trigger -Immediately disable/revert the policy on any violation of C, including a required task missing its semantic deadline, speculative work exposing an externally visible side effect before commitment, or a speculative result being committed/delivered without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input generation. Also disable it if critical-path latency or resource pressure worsens materially. +Immediately disable/revert the policy on any violation of C, including a required task missing its semantic deadline, speculative work exposing an externally visible side effect before commitment, a speculative result being committed/delivered without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input generation, or any test showing freshness validation can be separated from commitment so a mutation can win in between. Also disable it if critical-path latency or resource pressure worsens materially. From b23e9822c25757ec4857a81b500c671e11565f01 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:07:22 +0930 Subject: [PATCH 026/229] Validate performance gates out of sample --- ...DGET-001-performance-regression-budgets.md | 26 ++++++++++++------- 1 file changed, 16 insertions(+), 10 deletions(-) diff --git a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md index 417c94f..c36ca85 100644 --- a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md +++ b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md @@ -16,29 +16,31 @@ Small performance regressions accumulate because performance is measured occasio ## Optimization problem contract - X: target-supported metric/fixture/statistic/threshold configurations for a performance-regression gate -- F: gate configurations based on a sufficiently characterized environment and workload, with statistically justified tolerance and no weakening of functional correctness or workload realism +- F: gate configurations based on a sufficiently characterized environment and workload, with statistically justified tolerance, selection-aware validation, and no weakening of functional correctness or workload realism - f: target-measured regression-detection quality together with CI noise/false-alarm rate and measurement overhead - d: minimize missed material regressions and flaky/false failures under the target's predeclared multi-objective ordering -- C: the performance gate must not incentivize weakening tests, assertions, evidence, semantic coverage, or representative workload inputs; the gate itself must continue to detect known regressions and accept known-good controls within the declared false-positive/false-negative envelope -- B: target-specific calibration budget specifying repetitions, environment samples, and allowable CI/runtime measurement cost -- S: stop calibration when the declared sample budget is exhausted or the baseline/noise estimate is stable enough to justify the predeclared warning and hard thresholds +- C: the performance gate must not incentivize weakening tests, assertions, evidence, semantic coverage, or representative workload inputs; once a candidate gate is selected, its claimed false-positive/false-negative performance must be established on independent control executions or under a predeclared selection-aware procedure that accounts for every configuration tried +- B: target-specific calibration and certification budget specifying repetitions, environment samples, held-out/control executions, and allowable CI/runtime measurement cost +- S: stop calibration when the declared sample budget is exhausted or the baseline/noise estimate is stable enough to freeze one candidate gate for independent certification; promote it only if the certification contract passes - Variables: continuous / integer / categorical / mixed metric, statistic, fixture, and threshold choices - Search scope: local gate/calibration tuning - Objective behavior: noisy / stochastic measurement distributions - Information: derivative-free statistical observations - Evaluation cost: moderate to expensive depending on repetitions and fixture scale -- Constraints: functional correctness, representative workload, statistical tolerance, runner/environment characterization, false-positive/false-negative, and CI-overhead constraints +- Constraints: functional correctness, representative workload, statistical tolerance, runner/environment characterization, false-positive/false-negative, selection bias, and CI-overhead constraints - Parallelism: sequential or synchronous-batch calibration; parallel sampling only when runner interference is characterized - Exactness: no semantic approximation; statistical tolerance/noise handling is explicit ## Preserved contract -A performance gate may not incentivize weakening functional tests, correctness, evidence or workload realism. The gate is valid only while its fixture, environment characterization, detection sensitivity and measurement overhead remain inside their declared contract. +A performance gate may not incentivize weakening functional tests, correctness, evidence or workload realism. The gate is valid only while its fixture, environment characterization, detection sensitivity and measurement overhead remain inside their declared contract. Calibration evidence used to choose among competing gates is not automatically valid certification evidence for the selected gate. ## Optimization Turn a stable, reproducible performance expectation into a regression gate. Compare distributions or robust summaries where noise matters; separate machine/environment drift from code regression; keep cold/warm claims distinct. Keep known-fast and known-regressed control fixtures (or equivalent calibration cases) so the gate can periodically prove it still distinguishes acceptable from materially regressed behavior. +When multiple metric/fixture/statistic/threshold configurations are explored, treat that search as model selection. Use calibration/tuning data to choose the candidate, then freeze its complete configuration before certification. The default certification path is an independent held-out set of known-good and known-regressed executions that played no role in choosing the gate. If holding out controls is impractical, use a predeclared nested-resampling, simultaneous-confidence, multiple-testing, or other selection-aware procedure whose error guarantees cover the full configuration search—not nominal per-candidate estimates computed after selecting the best one. + ## Before / after evidence - Environment: No controlled target-repository budget calibration has been run for this OPT record. @@ -49,16 +51,20 @@ Turn a stable, reproducible performance expectation into a regression gate. Comp ## Validation -Calibrate variance before setting the threshold. Self-test the gate with known-good and known-regressed fixtures and preserve raw samples where practical. Re-run these controls after runner/toolchain changes and periodically enough to detect stale fixtures or sensitivity drift. Measure false positives, false negatives and gate overhead against predeclared acceptance limits. +Calibrate variance before setting the threshold. During tuning, compare candidate metric/fixture/statistic/threshold configurations using explicitly designated calibration data and preserve raw samples where practical. Once one gate is selected, **freeze the entire gate configuration before measuring its claimed detection performance**. + +Certify the frozen gate on independent known-good and known-regressed control executions that were not used to select it. Measure false positives, false negatives, and gate overhead against predeclared acceptance limits. If independent controls are unavailable, use a predeclared nested-resampling or selection-aware procedure that accounts for every candidate/configuration examined, and report the resulting adjusted uncertainty/error rates rather than reusing naive in-sample estimates. + +Record which executions were used for calibration/selection versus certification. Re-run independent controls after runner/toolchain changes and periodically enough to detect stale fixtures or sensitivity drift. Add an explicit overfitting fixture where several candidate gates are tuned on one noisy control sample set; prove the gate cannot be promoted merely because one candidate looked best on those same samples. ## Target-repo adaptation -Never copy another project's milliseconds, bundle sizes or thresholds. Establish the target's own baseline and noise envelope, define control fixtures, and declare acceptable false-positive/false-negative rates plus a maximum measurement-overhead budget. +Never copy another project's milliseconds, bundle sizes or thresholds. Establish the target's own baseline and noise envelope, define control fixtures, and declare acceptable false-positive/false-negative rates plus a maximum measurement-overhead budget. Predeclare how calibration/selection is separated from certification: held-out controls by default, or a justified nested/selection-aware alternative. Preserve the candidate-search history needed to audit the claimed certification error rates. ## Failure modes -Flaky gates from uncontrolled runners, benchmark gaming, stale fixtures, hardware drift, thresholds so loose they miss real regressions, thresholds so tight they block good changes, and measurement overhead large enough to damage CI usability or distort the workload under test. +Flaky gates from uncontrolled runners, benchmark gaming, stale fixtures, hardware drift, thresholds so loose they miss real regressions, thresholds so tight they block good changes, selection bias from evaluating a chosen gate on the same controls used to tune it, unreported configuration search that invalidates nominal error rates, and measurement overhead large enough to damage CI usability or distort the workload under test. ## Rollback trigger -Disable or demote the gate to non-blocking and recalibrate whenever its measurement environment is invalid, its fixture is stale/nonrepresentative, known regressions are no longer detected, known-good controls fail above the declared false-positive limit, observed false negatives exceed the declared limit, or measurement overhead exceeds the predeclared budget. Do **not** disable merely because product code legitimately regressed; in that case keep the valid gate and fix or explicitly accept the regression through the target's normal review process. +Disable or demote the gate to non-blocking and recalibrate whenever its measurement environment is invalid, its fixture is stale/nonrepresentative, independent/selection-aware certification no longer meets the declared false-positive/false-negative limits, known regressions are no longer detected, known-good controls fail above the declared false-positive limit, observed false negatives exceed the declared limit, or measurement overhead exceeds the predeclared budget. Do **not** disable merely because product code legitimately regressed; in that case keep the valid gate and fix or explicitly accept the regression through the target's normal review process. From f26204a92c4bbc8070412cdeff593c4620e63f63 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 19:07:49 +0930 Subject: [PATCH 027/229] Fence allocator incarnations --- ...NT-001-partitioned-coordination-domains.md | 28 +++++++++++-------- 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md index 3bb52e1..fbf1dbe 100644 --- a/optimizations/OPT-CONT-001-partitioned-coordination-domains.md +++ b/optimizations/OPT-CONT-001-partitioned-coordination-domains.md @@ -15,11 +15,11 @@ Independent workers serialize on one globally coordinated resource even though t ## Optimization problem contract -- X: target-supported shard/domain counts, namespace splits, worker-to-domain mappings, merge/aggregation policies, ownership-lease policies, fencing-epoch schemes, and restart-safe allocator-state policies -- F: configurations that preserve the target's required uniqueness, exclusive ownership, visibility, failure-domain, and ordering guarantees through assignment, rebalance, same-owner restart, crash recovery, and split-brain recovery +- X: target-supported shard/domain counts, namespace splits, worker-to-domain mappings, merge/aggregation policies, ownership-lease policies, fencing-epoch schemes, allocator-incarnation fencing, and restart-safe allocator-state policies +- F: configurations that preserve the target's required uniqueness, exclusive ownership, visibility, failure-domain, and ordering guarantees through assignment, rebalance, same-owner restart, crash recovery, process replacement, pause/resume, and split-brain recovery - f: measured coordination contention, tail latency, and coordination overhead under the declared workload - d: minimize under the target's predeclared objective ordering -- C: partitioning must not silently weaken any global invariant; any intentional shift from global to per-domain ordering is a separately declared contract change; mutable domain ownership transitions require one active fenced owner for each epoch; allocator restart must not reuse IDs/ranges already issued before the crash +- C: partitioning must not silently weaken any global invariant; any intentional shift from global to per-domain ordering is a separately declared contract change; every mutable ownership/allocator incarnation must be fenced at the authoritative mutation boundary so a superseded process cannot continue acting merely because its emitted IDs remain unique; allocator restart must not reuse IDs/ranges already issued before the crash - B: target-specific contention/scale/failover benchmark budget declared before tuning; no portable shard count or bit split is supplied here - S: stop when the budget is exhausted or a validated partitioning materially reduces the target bottleneck without violating C - Variables: integer / categorical / mixed @@ -27,21 +27,23 @@ Independent workers serialize on one globally coordinated resource even though t - Objective behavior: noisy under concurrent load; ownership/uniqueness invariants are deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive at target scale and during failover testing -- Constraints: uniqueness, ownership, ordering, visibility, failure-domain, lease/fencing, durable allocator-state, and resource constraints +- Constraints: uniqueness, ownership, ordering, visibility, failure-domain, lease/fencing, allocator-incarnation fencing, durable allocator-state, and resource constraints - Parallelism: concurrent / asynchronous by construction - Exactness: exact ownership/uniqueness semantics; no approximation is introduced ## Preserved contract -Partitioning must not silently weaken uniqueness, ownership, visibility or ordering guarantees. If ordering becomes per-domain rather than global, that is a contract change and must be explicit. When a domain can be reassigned, only the current fenced owner may mutate that domain; a delayed, partitioned, resumed, or split-brain previous owner must be rejected even if it still believes its old lease is valid. A same-owner process restart is also part of the ownership contract: restarting an allocator must not reset process-local state in a way that can reissue an ID or range already made externally visible. +Partitioning must not silently weaken uniqueness, ownership, visibility or ordering guarantees. If ordering becomes per-domain rather than global, that is a contract change and must be explicit. When a domain can be reassigned or an owner process can be replaced, only the currently authoritative fenced owner/incarnation may mutate that domain; a delayed, partitioned, resumed, or split-brain previous process must be rejected even if it still believes its old lease is valid. A same-owner process restart is also part of the ownership contract: restarting an allocator must not reset process-local state in a way that can reissue an ID/range already made externally visible, and a replacement process must not coexist as an unfenced second owner with a paused predecessor. ## Optimization Factor a global coordination space into independent domains. Encode domain identity into keys/IDs or route work so each domain can advance mostly independently. Prefer a small explicit merge/aggregation boundary to a permanently hot global lock/counter/poller. -For dynamic assignment/rebalance, use an **exclusive handoff with fencing**. A durable coordinator grants ownership together with a monotonically increasing epoch/token. Every state-changing operation that depends on domain ownership carries that epoch, and the authoritative storage/queue/allocation boundary rejects operations from epochs older than the current one. A lease alone is insufficient if an old process can resume after expiry; the fencing token must make stale writes/actions impossible at the mutation boundary. Do not activate the replacement owner until the new epoch is durably authoritative. +For dynamic assignment/rebalance, use an **exclusive handoff with fencing**. A durable coordinator grants ownership together with a monotonically increasing epoch/token. Every state-changing operation that depends on domain ownership carries that fencing identity, and the authoritative storage/queue/allocation boundary rejects operations from older identities. A lease alone is insufficient if an old process can resume after expiry; the fencing token must make stale writes/actions impossible at the mutation boundary. Do not activate the replacement owner until its new fencing identity is durably authoritative. -Where local IDs/counters are used, combine the stable domain identity with a restart-safe allocation policy. Acceptable designs include: (1) a durable high-water mark advanced atomically **before** an ID/range becomes externally usable, (2) durable allocation of non-overlapping ranges/blocks so a restart resumes from a fresh unissued block and may safely burn any uncertain tail, or (3) a new durable allocator-incarnation epoch on every allocator process restart, with that incarnation identity participating in the emitted ID namespace. Merely combining the domain with the current ownership/fencing epoch is insufficient when the same owner can restart without receiving a new epoch, because a reset process-local counter can reproduce old IDs. +Treat allocator process replacement as an ownership transition unless the target explicitly proves concurrent incarnations are harmless. A new allocator-incarnation token is safe only when it participates in the **authoritative fencing check**, not merely in the emitted ID namespace. Acceptable designs include: (1) advance the domain ownership/fencing epoch for every replacement allocator process, so the predecessor becomes stale automatically; or (2) maintain a separate monotonically increasing allocator-incarnation epoch that every mutation/allocation request carries and the authoritative boundary validates together with the domain ownership epoch. In both designs, replacing a paused/partitioned allocator invalidates the previous incarnation before the replacement may serve work. Merely embedding a fresh incarnation value in IDs prevents collisions but does **not** preserve exclusive ownership if the superseded process can still mutate queues/storage/state. + +Where local IDs/counters are used, combine this fencing rule with a restart-safe allocation policy. Acceptable durable allocation designs include: (1) a high-water mark advanced atomically **before** an ID/range becomes externally usable, or (2) durable allocation of non-overlapping ranges/blocks so a restart resumes from a fresh unissued block and may safely burn any uncertain tail. A per-incarnation namespace may additionally participate in emitted IDs, but for exclusive-owner targets it cannot substitute for fencing the superseded incarnation at the mutation boundary. If IDs must remain stable across allocator restarts and ownership epochs, use a durable monotonic counter/high-water mark or durable non-overlapping range allocator; do **not** reset an ephemeral counter. Persist/reserve advancement before returning the corresponding ID to the caller, or otherwise use a transaction whose crash semantics can prove that recovery never reissues an already-visible value. When commit status is uncertain after a crash, prefer skipping/burning an uncertain range over risking reuse. @@ -57,18 +59,20 @@ If IDs must remain stable across allocator restarts and ownership epochs, use a Check global invariants across all domains, collision/duplicate behavior, restart behavior and target-scale contention profiles. -Exercise **same-owner allocator restarts** independently of reassignment. Issue IDs/ranges, crash the allocator before and after each persistence/reservation boundary, restart it under the same domain ownership/fencing epoch, and prove it never reissues an externally visible ID/range. Test crashes after durable reservation but before delivery, after delivery but before acknowledgement bookkeeping, and with uncertain commit status. For high-water counters, verify monotonic durable recovery. For block/range allocation, verify recovered allocators never enter a previously issued block and that burning an uncertain tail preserves uniqueness. For allocator-incarnation schemes, verify every restart advances the durable incarnation before any allocation is served. +Exercise **same-owner allocator restarts** independently of reassignment. Issue IDs/ranges, crash the allocator before and after each persistence/reservation boundary, restart it, and prove it never reissues an externally visible ID/range. Test crashes after durable reservation but before delivery, after delivery but before acknowledgement bookkeeping, and with uncertain commit status. For high-water counters, verify monotonic durable recovery. For block/range allocation, verify recovered allocators never enter a previously issued block and that burning an uncertain tail preserves uniqueness. + +Exercise **superseded-incarnation races** separately from ID-collision tests. Pause or partition allocator incarnation A without proving it dead, start replacement incarnation B, make B's fencing identity authoritative, then resume A. Prove the authoritative mutation/allocation boundary rejects every operation from A even if A's generated IDs would be collision-free because of a different incarnation namespace. Repeat with delayed A messages, queue acknowledgements, counter updates, and storage mutations. If the target uses a separate allocator-incarnation epoch, prove both ownership epoch and incarnation epoch are validated wherever exclusivity matters. If the target instead deliberately allows concurrent incarnations, state that as a contract change and verify all affected operations are designed for multi-writer semantics. -Exercise **rebalance/recovery races**: pause an owner, expire/revoke it, assign a higher fencing epoch to a replacement, then resume the old owner and prove every stale mutation/allocation/queue claim is rejected. Inject network partition and split-brain conditions where both old and new processes run simultaneously. Verify only the highest authoritative epoch can mutate state, no duplicate IDs/work claims are produced, handoff is crash-recoverable, and ownership remains unique through coordinator/storage restarts. Include delayed messages from old epochs arriving after the new owner has already committed work. +Exercise **rebalance/recovery races**: pause an owner, expire/revoke it, assign a higher fencing identity to a replacement, then resume the old owner and prove every stale mutation/allocation/queue claim is rejected. Inject network partition and split-brain conditions where both old and new processes run simultaneously. Verify only the highest authoritative fencing identity can mutate state, no duplicate IDs/work claims are produced, handoff is crash-recoverable, and ownership remains unique through coordinator/storage restarts. Include delayed messages from old epochs arriving after the new owner has already committed work. ## Target-repo adaptation -Shard counts and bit splits are workload-specific. Measure skew, cache locality, failure domains and merge costs. Define the durable ownership source, lease timeout if used, monotonically increasing fencing epoch/token, authoritative mutation boundary that validates epochs, handoff sequence, and restart/recovery semantics before enabling dynamic reassignment. For allocators, separately define the same-owner restart policy: durable high-water mark, durable non-overlapping range reservation, or durable per-incarnation epoch. Specify exactly which state is made durable before an ID/range can escape and how ambiguous crash outcomes are recovered without reuse. +Shard counts and bit splits are workload-specific. Measure skew, cache locality, failure domains and merge costs. Define the durable ownership source, lease timeout if used, monotonically increasing fencing epoch/token, authoritative mutation boundary that validates fencing identity, handoff sequence, and restart/recovery semantics before enabling dynamic reassignment. For allocators, separately define the same-owner restart policy and the replacement-process fencing policy. Either advance the ownership epoch for every replacement or make allocator-incarnation epochs first-class fencing tokens at the mutation boundary. Do not rely on incarnation namespacing alone unless concurrent incarnations are explicitly admissible. Specify exactly which state is made durable before an ID/range can escape and how ambiguous crash outcomes are recovered without reuse. ## Failure modes -Hot shards merely move the bottleneck; domain proliferation raises memory/management overhead; rebalancing without fencing can allow stale and replacement owners to act concurrently; lease-only ownership can fail when an old process resumes; delayed old-epoch messages can duplicate allocations or queue work; a same-owner restart can reset an ephemeral local counter and reissue prior IDs even without any fencing race; persisting allocation state after delivery can create crash windows that reuse visible values; identity stability may be violated; global ordering requirements may make the pattern inadmissible. +Hot shards merely move the bottleneck; domain proliferation raises memory/management overhead; rebalancing without fencing can allow stale and replacement owners to act concurrently; lease-only ownership can fail when an old process resumes; delayed old-epoch messages can duplicate allocations or queue work; a fresh allocator-incarnation namespace can hide ID collisions while still allowing two owners to mutate the same domain; a same-owner restart can reset an ephemeral local counter and reissue prior IDs even without any fencing race; persisting allocation state after delivery can create crash windows that reuse visible values; identity stability may be violated; global ordering requirements may make the pattern inadmissible. ## Rollback trigger -Revert if partitioning does not reduce measured contention, if any cross-domain invariant fails, if failover/rebalance testing shows a stale owner or old-epoch message can mutate state after a replacement owner becomes authoritative, or if same-owner crash/restart testing can reissue any externally visible ID/range or otherwise lose durable allocator progress. +Revert if partitioning does not reduce measured contention, if any cross-domain invariant fails, if failover/rebalance/replacement testing shows a stale owner or superseded allocator incarnation can mutate state after a replacement becomes authoritative, if incarnation namespacing prevents duplicate IDs but does not fence the old process where exclusive ownership is required, or if same-owner crash/restart testing can reissue any externally visible ID/range or otherwise lose durable allocator progress. From ad61dc8d59d2e25bd79c11fbd11fb5781d3ab5cb Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:09:50 +0930 Subject: [PATCH 028/229] Harden rendered Markdown integrity checks --- scripts/check_catalog.py | 38 ++++++++++++++++++++++++++++---------- 1 file changed, 28 insertions(+), 10 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index b7fbf27..4e286bb 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -93,7 +93,7 @@ THEMATIC_BREAK_RE = re.compile(r"^(?:-{3,}|\*{3,}|_{3,})$") LIST_MARKER_ONLY_RE = re.compile(r"^(?:[-+*]|\d+[.)])$") TABLE_SEPARATOR_CELL_RE = re.compile(r"^:?-{3,}:?$") -FENCE_OPEN_RE = re.compile(r"^(`{3,}|~{3,})(.*)$") +FENCE_OPEN_RE = re.compile(r"^ {0,3}(`{3,}|~{3,})(.*)$") CANONICAL_DEFINITION_PATTERNS = { "X": re.compile(r"^- `X` — \S"), "F": re.compile(r"^- `F(?: ⊆ X)?` — \S"), @@ -110,7 +110,20 @@ def die(msg: str) -> None: def strip_html_comments(text: str) -> str: - return re.sub(r"", "", text, flags=re.DOTALL) + """Strip HTML comments; an unmatched opener hides the remainder through EOF.""" + visible: list[str] = [] + cursor = 0 + while True: + start = text.find("", start + 4) + if end < 0: + break + cursor = end + 3 + return "".join(visible) def visible_nonfenced_lines(lines: list[str]) -> list[str]: @@ -121,17 +134,23 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: fence_len = 0 for raw in cleaned.splitlines(): - stripped = raw.strip() if fence_char is not None: - close = re.fullmatch(rf"{re.escape(fence_char)}{{{fence_len},}}\s*", stripped) + close = re.fullmatch( + rf" {{0,3}}{re.escape(fence_char)}{{{fence_len},}}[ \t]*", raw + ) if close is not None: fence_char = None fence_len = 0 continue - opener = FENCE_OPEN_RE.match(stripped) + opener = FENCE_OPEN_RE.match(raw) if opener is not None: run = opener.group(1) + info = opener.group(2) + # CommonMark forbids backticks inside an opening backtick fence's info string. + if run[0] == "`" and "`" in info: + visible.append(raw) + continue fence_char = run[0] fence_len = len(run) continue @@ -169,7 +188,7 @@ def markdown_table_cells(line: str) -> list[str] | None: def extract_markdown_table( lines: list[str], expected_headers: tuple[str, ...], context: str ) -> list[str]: - """Extract one visible table by exact header and validate every row width.""" + """Extract one visible table and require well-formed, non-empty data cells.""" visible = visible_nonfenced_lines(lines) expected = list(expected_headers) for i, line in enumerate(visible): @@ -194,6 +213,8 @@ def extract_markdown_table( f"{context} table row has {len(cells)} column(s); " f"expected {len(expected)}: {row.strip()}" ) + if any(not cell for cell in cells): + die(f"{context} table row contains an empty required cell: {row.strip()}") table.append(row) return table die(f"{context} is missing the expected Markdown table") @@ -205,10 +226,7 @@ def is_structural_only_line(line: str) -> bool: if LIST_MARKER_ONLY_RE.fullmatch(line) or line == ">": return True cells = markdown_table_cells(line) - return bool( - cells - and all(TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in cells) - ) + return bool(cells and all(TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in cells)) def section_has_content(lines: list[str]) -> bool: From 4849ee887c2699d89fe48173e5101a1c37542506 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:10:12 +0930 Subject: [PATCH 029/229] Restore canonical contract and bounded source guidance --- README.md | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 0da10ec..c1e102a 100644 --- a/README.md +++ b/README.md @@ -42,9 +42,9 @@ The immutable `v1.0.0` release and its five original records are formalized by t The new [`OPTIMIZATION-PROBLEM.md`](OPTIMIZATION-PROBLEM.md) supplies a canonical problem contract for future records: -`P = (X, F, f, direction, C, B, S)` +`P = (X, F, f, d, C, B, S)` -for search space, feasible set, objective, direction, correctness/semantic constraints, evaluation budget and stopping rule. +where `d` is the objective direction/order; the remaining components are search space, feasible set, objective, correctness/semantic constraints, evaluation budget and stopping rule. ## Source material @@ -54,7 +54,9 @@ for search space, feasible set, objective, direction, correctness/semantic const - [`sources/WPO.md`](sources/WPO.md) — critical-path and performance-budget discovery source. - [`sources/MATHEMATICAL-OPTIMIZATION.md`](sources/MATHEMATICAL-OPTIMIZATION.md) — mathematical/combinatorial problem vocabulary and pruning foundations. - [`power_module.md`](power_module.md) — E8/qutrit DSP architecture that motivated **OPT-DSP-001**. -- [`suxen.zip`](suxen.zip) — opaque source archive, still **not promoted as optimization evidence**. +- [`sources/SUXEN.md`](sources/SUXEN.md) — provenance and the required bounded recursive inventory procedure for the opaque `suxen.zip` source candidate. +- [`scripts/inventory_zip.py`](scripts/inventory_zip.py) — bounded recursive ZIP inventory entry point; use the explicit limits documented in `sources/SUXEN.md` rather than generic/unbounded extraction. +- [`suxen.zip`](suxen.zip) — opaque source archive, still **not promoted as optimization evidence** until the bounded inventory identifies reusable mechanisms. ## Integrity gate From 15c10c89e4d226ffa7d6b691651df0c5f42b8132 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:10:44 +0930 Subject: [PATCH 030/229] Linearize materialization validation with commit --- ...T-FAN-001-shared-materialization-fanout.md | 26 +++++++++++-------- 1 file changed, 15 insertions(+), 11 deletions(-) diff --git a/optimizations/OPT-FAN-001-shared-materialization-fanout.md b/optimizations/OPT-FAN-001-shared-materialization-fanout.md index 9d59a6e..1a2bb98 100644 --- a/optimizations/OPT-FAN-001-shared-materialization-fanout.md +++ b/optimizations/OPT-FAN-001-shared-materialization-fanout.md @@ -14,11 +14,11 @@ The same deterministic transformation is repeated independently for each consume ## Optimization problem contract -- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, immutable-source snapshot/mutation-control policies, monotonic source/config epochs, artifact-version pinning/consumption policies, and crash-consistent publication schemes -- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, source-snapshot/mutation-consistency, validation-to-consumption identity, and publication-atomicity requirement +- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, immutable-source snapshot/mutation-control policies, monotonic source/config epochs, source-witness compare-and-publish policies, artifact-version pinning/consumption policies, and crash-consistent publication schemes +- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, source-snapshot/mutation-consistency, **source-validation-to-publication linearizability**, validation-to-consumption identity, and publication-atomicity requirement - f: measured transformation CPU, replay CPU, fan-out latency, and storage/I/O overhead under the target's declared objective ordering - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state binds the artifact bytes to one coherent effective source/transform identity, no intervening mutable-input change can be erased by endpoint equality, and every consumer reads the **same immutable/versioned artifact instance that was validated** rather than re-resolving a mutable alias after validation; verification/security metadata may be removed only under an explicit contract change +- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state binds the artifact bytes to one coherent effective source/transform identity, no intervening mutable-input change can be erased by endpoint equality, the final source witness is linearized with the materialization commit, and every consumer reads the **same immutable/versioned artifact instance that was validated** rather than re-resolving a mutable alias after validation; verification/security metadata may be removed only under an explicit contract change - B: target-specific fan-out/replay benchmark budget declared before tuning; no portable subscriber count, replay size, or retention duration is supplied here - S: stop when the declared budget is exhausted or a validated materialization policy materially improves the target objective without violating C - Variables: categorical / integer / mixed @@ -26,13 +26,13 @@ The same deterministic transformation is repeated independently for each consume - Objective behavior: noisy for performance; transformation identity/equivalence is deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive depending on transform/replay size -- Constraints: semantic equivalence, source-snapshot/mutation consistency, artifact-version pinning, integrity, versioning, trust/security, storage, and crash-consistency constraints -- Parallelism: concurrent fan-out/replay; publication and consumption pinning must remain race-safe +- Constraints: semantic equivalence, source-snapshot/mutation consistency, source-validation/publication linearizability, artifact-version pinning, integrity, versioning, trust/security, storage, and crash-consistency constraints +- Parallelism: concurrent fan-out/replay; source publication, artifact publication, and consumption pinning must remain race-safe - Exactness: exact representation semantics; no approximation is introduced ## Preserved contract -Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive source mutation, change-and-revert races, crashes, interrupted publication, and concurrent replacement of mutable aliases. Validation is meaningful only if the consumer subsequently reads the exact artifact version/handle that passed validation. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. +Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive source mutation, change-and-revert races, final-check-to-commit races, crashes, interrupted publication, and concurrent replacement of mutable aliases. Validation is meaningful only if both publication and downstream consumption remain bound to the exact source/artifact identities that were validated. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. ## Optimization @@ -40,7 +40,9 @@ Perform an expensive deterministic transform once near production, persist or re Define a materialization key that covers, as applicable, source object/content identity or immutable source version, transformation/encoder implementation identity, encoder configuration and dictionaries, schema/format version, feature flags, trust/security policy, and any other effective input that can affect the materialized result. -Bind the transform to one coherent source identity for the **entire transform-to-commit interval**. Prefer reading every mutable effective input from an immutable snapshot/version captured together with the materialization key. If immutable snapshots are unavailable, use a mechanism that records intervening mutation rather than comparing endpoint content alone: hold an appropriate mutation/read lock for the full interval, or capture a monotonically increasing, non-reusable version/epoch for each mutable source/config/transform input and require the same coherent epoch vector at commit. Every mutation must advance its witness durably/atomically with the mutation, including A→B→A changes that restore the original bytes. A final content/key recomputation may supplement this witness but must not be the sole protection. Any lock violation, epoch change, incoherent multi-input witness, or untrackable mutable input invalidates the candidate materialization; discard/retry it rather than publishing mixed-state bytes. +Bind the transform to one coherent source identity for the **entire transform-to-commit interval**. Prefer reading every mutable effective input from an immutable snapshot/version captured together with the materialization key. If immutable snapshots are unavailable, use a mechanism that records intervening mutation rather than comparing endpoint content alone: hold an appropriate mutation/read lock for the full interval, or capture a monotonically increasing, non-reusable version/epoch for each mutable source/config/transform input. Every mutation must advance its witness durably/atomically with the mutation, including A→B→A changes that restore the original bytes. A final content/key recomputation may supplement this witness but must not be the sole protection. Any lock violation, epoch change, incoherent multi-input witness, or untrackable mutable input invalidates the candidate materialization; discard/retry it rather than publishing mixed-state bytes. + +For lock/epoch-based targets, make the **final witness validation and materialization activation one linearizable transition**. A check followed later by manifest publication is insufficient because the source may mutate in that gap. Either hold the mutation/read lock through the authoritative artifact/manifest switch, or use an atomic compare-and-publish/transaction that verifies the complete witnessed epoch vector and publishes the new materialization only if those epochs are still current in the same serialization domain that advances them. If any epoch changed, the activation must fail and the candidate remains non-authoritative. Immutable snapshots satisfy this requirement only when the committed materialization remains explicitly bound to that immutable source version rather than to a mutable alias. Publish the artifact and its identity as **one committed state**. Acceptable designs include content-addressed storage where the artifact digest is itself part of the committed key, an atomically replaced manifest that contains both the full materialization key and the artifact digest/location, or another crash-consistent transaction that makes old state or new state visible but never a mixed pair. Do not update artifact bytes and their key independently in a way that can expose a new artifact with stale metadata or stale bytes with a new key after a crash. @@ -60,7 +62,9 @@ For multiple consumers, each may hold its own reference to the same immutable va Compare shared materialization against per-consumer reference output, including corruption, restart/replay and mixed consumer capabilities. Independently mutate each key component—source content/version, transform implementation, encoder options/dictionary, schema/format version, feature flags and trust policy—and prove that each output-affecting change invalidates reuse. Also test unchanged-key reuse, tampered artifacts with matching metadata, and migration/version-boundary cases. -Exercise **concurrent source mutation**, including explicit A→B→A races. Start a transform from source identity A, mutate one or more source/config/transform inputs to B while the transform is running, then restore the original bytes before commit. For snapshot-based targets, prove the transform reads only the immutable A snapshot. For lock-based targets, prove mutation cannot interleave with the protected transform/publication interval. For epoch/version-based targets, prove every mutation advances the monotonic witness and the final witness differs even when the final content/key returns to A. Reject/discard the candidate on any witness change and compare every accepted materialization with a fresh transform from the exact committed source identity. +Exercise **concurrent source mutation**, including explicit A→B→A races. Start a transform from source identity A, mutate one or more source/config/transform inputs to B while the transform is running, then restore the original bytes before commit. For snapshot-based targets, prove the transform reads only the immutable A snapshot. For lock-based targets, prove mutation cannot interleave with the protected transform/publication interval. For epoch/version-based targets, prove every mutation advances the monotonic witness and the final witness differs even when the final content/key returns to A. Reject/discard the candidate on any mutation witness change and compare every accepted materialization with a fresh transform from the exact committed source identity. + +Add a **final source-validation-to-publication race**. Pause after the last source epoch/vector check but before the new materialization becomes authoritative, mutate one source/config/transform input, then resume publication. For lock-based targets, prove the mutation cannot occur until after the authoritative switch. For epoch/CAS-based targets, prove the compare-and-publish fails because the current epoch vector no longer equals the witnessed vector; no consumer may observe the candidate as committed. Repeat with A→B→A content restoration, multiple inputs, and a concurrent manifest reader. A successful endpoint rehash after the mutation is not sufficient evidence. Add a **post-validation replacement race**. Validate committed artifact A, pause before a consumer reads it, replace the mutable alias/path/object name with a different valid artifact B, then resume consumption. Prove a pinned immutable/versioned handle still yields exactly A (or fails closed if A was invalidated by the target's retention contract), never unvalidated B. Repeat with fan-out consumers at staggered start times, concurrent manifest advancement, replay after alias replacement, reclamation pressure, and mutable object-store/version aliases. For lock-based targets, prove replacement cannot occur until the protected consumer read completes. For snapshot-based targets, prove validation and consumption address the same snapshot digest/version. @@ -68,12 +72,12 @@ Inject crashes/interruption at every publication boundary: after artifact write ## Target-repo adaptation -Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, specify whether mutable inputs are consumed from immutable snapshots, protected by full-duration locks, or guarded by monotonic mutation epochs, and define how a coherent multi-input witness is captured. Specify representation versioning, invalidation, integrity checking, **the immutable/versioned artifact handle or lock/snapshot that binds validation through consumption**, retention/reclamation semantics for pinned consumers, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. Do not advertise commit-time endpoint rehashing or mutable-path validation alone as sufficient identity protection. +Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, specify whether mutable inputs are consumed from immutable snapshots, protected by full-duration locks, or guarded by monotonic mutation epochs, and define how a coherent multi-input witness is captured. For epoch/lock targets, define the serialization domain that makes final witness validation atomic with authoritative materialization publication: lock-through-commit or atomic compare-and-publish against the complete epoch vector. Specify representation versioning, invalidation, integrity checking, **the immutable/versioned artifact handle or lock/snapshot that binds validation through consumption**, retention/reclamation semantics for pinned consumers, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. Do not advertise commit-time endpoint rehashing or mutable-path validation alone as sufficient identity protection. ## Failure modes -Incomplete keys can serve stale representations after source or transform changes; mutable sources can change during transformation and produce mixed-state output under a stale key; A→B→A races can defeat endpoint key comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent epoch vectors can describe no real source state; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; a mutable alias can be replaced after validation and before consumption, delivering unvalidated bytes; reclamation can invalidate a pinned artifact prematurely; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. +Incomplete keys can serve stale representations after source or transform changes; mutable sources can change during transformation and produce mixed-state output under a stale key; A→B→A races can defeat endpoint key comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent epoch vectors can describe no real source state; a check-then-publish gap can authorize A-derived bytes after the source already advanced to B; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; a mutable alias can be replaced after validation and before consumption, delivering unvalidated bytes; reclamation can invalidate a pinned artifact prematurely; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. ## Rollback trigger -Disable reuse immediately if any materialization-key hit, source-mutation race, publication-recovery path, integrity check, or validation-to-consumption race can return output that differs from a fresh transform for the same exact committed effective inputs; if an A→B→A race can evade the snapshot/lock/monotonic mutation witness; if a mutable alias replacement can make a consumer read bytes other than the exact artifact version that passed validation; if a pinned artifact can be reclaimed before consumption completes; or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. +Disable reuse immediately if any materialization-key hit, source-mutation race, final source-validation-to-publication race, publication-recovery path, integrity check, or validation-to-consumption race can return output that differs from a fresh transform for the same exact committed effective inputs; if an A→B→A race can evade the snapshot/lock/monotonic mutation witness; if source epochs can change after a successful final check yet the candidate can still become authoritative; if a mutable alias replacement can make a consumer read bytes other than the exact artifact version that passed validation; if a pinned artifact can be reclaimed before consumption completes; or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. From b66b184ef1d3ffcc1dba39bc5730e501f861adbf Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:11:23 +0930 Subject: [PATCH 031/229] Cover whole-search hard resource budgets --- ...E-001-bound-driven-search-space-pruning.md | 22 ++++++++++--------- 1 file changed, 12 insertions(+), 10 deletions(-) diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index 22dd49e..de2d4bb 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -19,8 +19,8 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F - f: a scalar real-valued target objective `f : F → R` evaluated on feasible candidates only - d: exactly one of scalar `minimize` or scalar `maximize`; vector, Pareto, lexicographic, or other partial-order objectives are outside this record unless a separately specified and validated frontier-bound mechanism is introduced -- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, the target's observable tie semantics are preserved, parallel dispatch cannot oversubscribe the declared hard budget, and frontier exhaustion is declared only after all queued **and leased/in-flight** regions are accounted for -- B: a finite, predeclared target-specific **enforceable** cap on evaluations, wall time, compute, or equivalent resource consumption; parallel dispatch requires linearizable reservations before candidate/bound work starts, and every wall-time/compute reservation requires a per-operation quota/deadline/termination mechanism strong enough to prevent overrun; a resource that cannot be hard-capped must be labeled observational/best-effort rather than advertised as hard B +- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, the target's observable tie semantics are preserved, parallel dispatch cannot oversubscribe the declared hard budget, **every unit of resource consumption covered by a hard wall-time/compute B is accounted for or enclosed by an enforceable whole-search cap**, and frontier exhaustion is declared only after all queued **and leased/in-flight** regions are accounted for +- B: a finite, predeclared target-specific **enforceable** cap. Evaluation-count budgets may count only the declared candidate/bound evaluations. A hard wall-time/compute/resource B must cover the **entire search**, including candidate/bound evaluation, branching, child generation, frontier coordination, serialization, incumbent maintenance, synchronization, cleanup and any other algorithm work that consumes the bounded resource; enforce that with a whole-search deadline/quota or complete metering/reservation. A resource dimension that cannot be capped over the complete search must be labeled observational/best-effort rather than advertised as hard B - S: stop immediately when the required optimality/tie contract is proven, or when the **global frontier is exhausted**, meaning there are no queued regions, no leased/in-flight regions still capable of producing candidates/children, and no unpublished child/frontier updates owned by active work. Otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality - Variables: integer / categorical / discrete / mixed - Search scope: global over the declared candidate space @@ -28,7 +28,7 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - Information: derivative-free; bound/relaxation information is target-specific - Evaluation cost: moderate to expensive when exhaustive evaluation is infeasible - Constraints: feasibility, semantic correctness, scalar-bound soundness, tie semantics, global-frontier accounting, and enforceable finite-resource constraints -- Parallelism: sequential, or parallel only with synchronized incumbent/frontier/bound state, **leased/in-flight region accounting**, and linearizable budget reservation/completion accounting plus enforceable per-operation resource caps +- Parallelism: sequential, or parallel only with synchronized incumbent/frontier/bound state, **leased/in-flight region accounting**, linearizable budget reservation/completion accounting where used, and an enforceable whole-search cap for every wall-time/compute dimension advertised as hard - Exactness: exact only when the declared optimality and observable-tie contract is proven within B, including proof that no queued or leased region can still affect the answer; otherwise anytime/incomplete result semantics apply For each unexplored region `R`, define a bound `b(R)` separately from `f`: @@ -47,7 +47,7 @@ An independently proven infeasible region may also be pruned. A heuristic estima ## Preserved contract -A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, exhausting B without a feasible incumbent does not permit an infeasibility claim, and parallel execution must preserve the same hard resource ceiling as sequential execution rather than oversubscribing work in flight. A temporarily empty shared queue is **not** frontier exhaustion while any worker owns a leased region that may still produce a candidate, proof obligation, or child region. +A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, exhausting B without a feasible incumbent does not permit an infeasibility claim, and parallel execution must preserve the same hard resource ceiling as sequential execution rather than oversubscribing work in flight. A temporarily empty shared queue is **not** frontier exhaustion while any worker owns a leased region that may still produce a candidate, proof obligation, or child region. Likewise, a hard wall-time/compute B applies to the whole search, not just its explicit evaluation calls. ## Optimization @@ -55,9 +55,11 @@ Maintain an incumbent when one exists, partition the search space, compute a che For **parallel** search, define one global frontier lifecycle. A region remains part of the frontier from enqueue until it is either (a) soundly pruned/closed, or (b) replaced by its child regions through an atomic/linearizable completion transition. Dequeuing for worker ownership therefore changes a region from `queued` to `leased/in-flight`; it does **not** remove that region from the global frontier. A worker that branches a leased region must publish all resulting children and close/release the parent as one frontier-accounting transition, or use another protocol that cannot expose a moment where the queue is empty even though unpublished descendants still exist. Worker failure/cancellation must return or recover the lease so unexplored work is not silently lost. -For **parallel** search, treat both candidate evaluation and nontrivial bound/relaxation evaluation as budget-consuming operations. Before dispatch, atomically reserve the operation's declared evaluation slot or conservative wall-time/compute quota from one shared budget ledger. If `consumed + reserved + proposed_reservation > B`, do not dispatch. Completion/failure/cancellation converts the reservation to consumed usage and releases only demonstrably unconsumed capacity under the same linearizable accounting boundary, so workers racing for the final slot cannot oversubscribe it. +For evaluation-count budgets, candidate evaluation and nontrivial bound/relaxation evaluation consume/reserve the declared slots before dispatch. For hard wall-time, compute, memory, provider-cost or equivalent resource budgets, the cap must apply to **all search work**, not merely those evaluations. Acceptable designs include an enforceable whole-search deadline/quota/cgroup/job/provider cap, or complete metering where branching, child construction, queue/frontier operations, serialization, incumbent updates, synchronization and cleanup are all charged under the same global ledger. Per-evaluation reservations may still be used internally, but they do not by themselves prove a whole-search wall-time/compute bound. -Evaluation-count budgets consume/reserve a slot before launch. For wall-time/compute budgets, each launched operation must have an enforceable per-operation upper bound—for example a deadline with forced termination, cgroup/job quota, provider/runtime cap, or equivalent mechanism. If the target cannot prevent one bound/candidate evaluation from running past the nominal reservation, wall-time/compute is **not** a hard B and must be documented as observational/best-effort instead of being used to justify finite-cap correctness claims. +When using reservations, atomically reserve the applicable budget before covered work starts. If `consumed + reserved + proposed_reservation > B`, do not start that work. Completion/failure/cancellation converts the reservation to consumed usage and releases only demonstrably unconsumed capacity under the same linearizable accounting boundary. For a whole-search deadline/quota design, every worker and coordinator path must be subordinate to that cap, including non-evaluation overhead and cleanup required before returning a result. + +If the target meters only candidate/bound evaluations, then only **evaluation count** may be claimed as a hard B from that accounting. Nominal wall-time/compute targets in that design are observational/best-effort and cannot justify finite-cap correctness claims. Similarly, if branching/frontier/serialization overhead can escape an otherwise claimed resource cap, that resource dimension is not hard-bounded. A relaxed solution is evidence for a bound, not automatically a feasible final answer. @@ -75,18 +77,18 @@ For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness i Add **parallel frontier-exhaustion races**. Use a fixture where the last queued region is leased by one worker, making the shared queue empty, then pause that worker before it publishes one or more child regions. Prove the coordinator does not declare exhaustion or exact optimality while that lease remains live. Resume the worker and verify the children become searchable and the final result matches exhaustive/scalar search. Also inject worker failure/cancellation while holding the last lease and verify the region is recovered/requeued or otherwise completed without losing unexplored work. Test simultaneous parent-close/child-publish transitions and prove there is no observation in which both queued and leased frontier counts reach zero before all descendants are durably accounted for. -Add **parallel budget-boundary fixtures**. Race multiple workers against one remaining evaluation slot and prove only one reservation succeeds. Race bound evaluations and candidate evaluations against the same final capacity and prove both charge the declared ledger. For wall-time/compute budgets, deliberately run an operation that attempts to exceed its reservation and prove the quota/deadline/termination mechanism stops it within the enforceable cap. Race completion/cancellation with new dispatch and verify the accounting transition is linearizable—released capacity is not visible before corresponding consumption is committed, no increments are lost, and `consumed + reserved <= B` always holds for hard-budget dimensions. +Add **parallel budget-boundary fixtures**. Race multiple workers against one remaining evaluation slot and prove only one reservation succeeds for an evaluation-count B. Race bound evaluations and candidate evaluations against the same final evaluation capacity and prove both charge the declared ledger. For a hard wall-time/compute/resource B, add fixtures where evaluation itself is cheap but branching, child generation, frontier coordination, serialization or incumbent maintenance deliberately dominates resource use; prove the whole-search quota/deadline stops or charges that overhead before the cap is exceeded. Race completion/cancellation with new work and verify global accounting is linearizable where a ledger is used. For every hard dimension, prove total covered search consumption remains within B—not merely `candidate_eval + bound_eval` consumption. Add **equal-objective tie fixtures**. For an any-one-optimum contract, prove equality pruning cannot alter any observable result. For deterministic tie-winner contracts, construct regions containing equal-objective candidates with better/worse tie ranks and prove equality-bound regions are retained until the declared tie winner is established. For all-optima contracts, prove every equal-objective optimum is enumerated. If using a stronger total-order bound, validate its soundness independently against exhaustive fixtures. ## Target-repo adaptation -The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. For parallel implementations, define the global frontier state machine, lease ownership/recovery rules, parent-close/child-publish atomicity, and the exact exhaustion predicate over queued plus leased/in-flight work. Also define one linearizable reservation/completion ledger shared by candidate and bound work, the accounting unit, per-operation reservation amount, metering source, and the enforcement mechanism for wall-time/compute quotas. Downgrade any unenforceable resource limit to best-effort/observational rather than calling it hard B. +The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. For parallel implementations, define the global frontier state machine, lease ownership/recovery rules, parent-close/child-publish atomicity, and the exact exhaustion predicate over queued plus leased/in-flight work. Also define the budget accounting unit and enforcement boundary. If B is evaluation count, reserve candidate/bound slots linearly. If B is wall time, compute, memory, money or another resource, specify the **whole-search** enforcement mechanism or the complete ledger coverage for evaluations plus branching, frontier work, serialization, incumbent maintenance and coordination. Downgrade any dimension that can escape that enforcement boundary to best-effort/observational rather than calling it hard B. ## Failure modes -Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; treating queue-empty as frontier-empty can declare exact completion while a leased region still owns unexplored descendants; losing a worker lease can silently drop search regions; non-atomic parent-close/child-publication can create false exhaustion; parallel workers without linearizable reservations can oversubscribe the last evaluation/resource slot; an uncapped candidate/bound evaluation can exceed a nominal wall-time/compute cap before stopping logic observes it; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. +Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; treating queue-empty as frontier-empty can declare exact completion while a leased region still owns unexplored descendants; losing a worker lease can silently drop search regions; non-atomic parent-close/child-publication can create false exhaustion; parallel workers without linearizable evaluation reservations can oversubscribe the last evaluation slot; branching/child/frontier/serialization/incumbent overhead can exceed a nominal wall-time/compute B if only evaluations are charged; an unenforced coordinator/cleanup path can outlive a claimed whole-search deadline; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. ## Rollback trigger -Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel/exact mode if frontier exhaustion can be observed while any leased/in-flight region may still produce work, if parent-close/child-publication or lease recovery can lose unexplored regions, if workers can dispatch without first reserving budget, if concurrent accounting can oversubscribe B, or if any operation can exceed a resource reservation that is claimed as a hard cap. Abort exact-mode claims whenever B is exhausted before the full objective/tie/frontier contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. +Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel/exact mode if frontier exhaustion can be observed while any leased/in-flight region may still produce work, if parent-close/child-publication or lease recovery can lose unexplored regions, if workers can oversubscribe an evaluation-count budget, or if any resource consumption path can escape a dimension advertised as a hard whole-search B. Abort exact-mode claims whenever B is exhausted before the full objective/tie/frontier contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. From badf18053481c9fe4d024f65815dc3d66a04c4f9 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:46:58 +0930 Subject: [PATCH 032/229] Harden rendered Markdown catalog parsing --- scripts/check_catalog.py | 157 +++++++++++++++++++++++++++------------ 1 file changed, 110 insertions(+), 47 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 4e286bb..fff3500 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -75,25 +75,22 @@ } LINK_RE = re.compile(r"\[([^\]]+)\]\((optimizations/[^)#]+\.md)\)") +RECORD_LINK_CELL_RE = re.compile( + r"^\[(OPT-[A-Z]+-\d{3})\]\((optimizations/[^)#]+\.md)\)$" +) ID_RE = re.compile(r"^# (OPT-[A-Z]+-\d{3}) — ") FILENAME_ID_RE = re.compile(r"^(OPT-[A-Z]+-\d{3})-") STATUS_RE = re.compile(r"^\*\*Status:\*\*\s*(.*?)\s*$") OPT_TOKEN_RE = re.compile(r"\bOPT-[A-Z]+-\d{3}\b") EMPTY_LABEL_RE = re.compile(r"^-\s+[^:]+:\s*$") -README_ROW_RE = re.compile( - r"^\|\s*\[(OPT-[A-Z]+-\d{3})\]\((optimizations/[^)#]+\.md)\)" - r"\s*\|[^|]*\|\s*([^|]+?)\s*\|", - re.MULTILINE, -) -CATALOG_DECISION_ROW_RE = re.compile( - r"^\|[^|\n]*\|\s*\[(OPT-[A-Z]+-\d{3})\]\((optimizations/[^)#]+\.md)\)\s*\|", - re.MULTILINE, -) HEADING_RE = re.compile(r"^#{1,6}(?:\s|$)") -THEMATIC_BREAK_RE = re.compile(r"^(?:-{3,}|\*{3,}|_{3,})$") +THEMATIC_BREAK_RE = re.compile( + r"^(?:\*(?:[ \t]*\*){2,}|-(?:[ \t]*-){2,}|_(?:[ \t]*_){2,})[ \t]*$" +) LIST_MARKER_ONLY_RE = re.compile(r"^(?:[-+*]|\d+[.)])$") TABLE_SEPARATOR_CELL_RE = re.compile(r"^:?-{3,}:?$") FENCE_OPEN_RE = re.compile(r"^ {0,3}(`{3,}|~{3,})(.*)$") +EMPHASIS_WRAPPERS = ("**", "__", "~~", "*", "_") CANONICAL_DEFINITION_PATTERNS = { "X": re.compile(r"^- `X` — \S"), "F": re.compile(r"^- `F(?: ⊆ X)?` — \S"), @@ -109,31 +106,48 @@ def die(msg: str) -> None: raise SystemExit(f"catalog-integrity: {msg}") -def strip_html_comments(text: str) -> str: - """Strip HTML comments; an unmatched opener hides the remainder through EOF.""" - visible: list[str] = [] +def strip_html_comments_from_visible_line( + raw: str, in_comment: bool +) -> tuple[str, bool]: + """Remove HTML comments from a non-fenced line, carrying unmatched state.""" + out: list[str] = [] cursor = 0 - while True: - start = text.find("") + if end < 0: + return "", True + cursor = end + 3 + in_comment = False + + while cursor < len(raw): + start = raw.find("", start + 4) + out.append(raw[cursor:start]) + end = raw.find("-->", start + 4) if end < 0: + in_comment = True break cursor = end + 3 - return "".join(visible) + + return "".join(out), in_comment def visible_nonfenced_lines(lines: list[str]) -> list[str]: - """Return rendered-ish Markdown lines, excluding comments and fenced blocks.""" - cleaned = strip_html_comments("\n".join(lines)) + """Return rendered-ish Markdown lines, excluding comments and fenced blocks. + + Fence state is determined from the original Markdown line before HTML comments + are removed, so a fence-looking line with trailing comment text cannot become + a valid closer after preprocessing. + """ visible: list[str] = [] fence_char: str | None = None fence_len = 0 + in_comment = False - for raw in cleaned.splitlines(): + for raw in lines: if fence_char is not None: close = re.fullmatch( rf" {{0,3}}{re.escape(fence_char)}{{{fence_len},}}[ \t]*", raw @@ -143,19 +157,29 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: fence_len = 0 continue - opener = FENCE_OPEN_RE.match(raw) - if opener is not None: - run = opener.group(1) - info = opener.group(2) - # CommonMark forbids backticks inside an opening backtick fence's info string. - if run[0] == "`" and "`" in info: - visible.append(raw) + if in_comment: + rendered, in_comment = strip_html_comments_from_visible_line(raw, True) + if in_comment: continue - fence_char = run[0] - fence_len = len(run) - continue + raw_for_parse = rendered + else: + # A fence opener is recognized from the original line. This matters + # because HTML comment syntax in a fence info string is literal text. + opener = FENCE_OPEN_RE.match(raw) + if opener is not None: + run = opener.group(1) + info = opener.group(2) + if run[0] != "`" or "`" not in info: + fence_char = run[0] + fence_len = len(run) + continue + raw_for_parse, in_comment = strip_html_comments_from_visible_line(raw, False) + + if raw_for_parse: + visible.append(raw_for_parse) + elif not in_comment and raw == "": + visible.append("") - visible.append(raw) return visible @@ -187,8 +211,8 @@ def markdown_table_cells(line: str) -> list[str] | None: def extract_markdown_table( lines: list[str], expected_headers: tuple[str, ...], context: str -) -> list[str]: - """Extract one visible table and require well-formed, non-empty data cells.""" +) -> list[list[str]]: + """Extract one visible table and return validated data rows as cell lists.""" visible = visible_nonfenced_lines(lines) expected = list(expected_headers) for i, line in enumerate(visible): @@ -203,7 +227,8 @@ def extract_markdown_table( or not all(TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in separator) ): die(f"{context} table has an invalid separator row") - table = [line, visible[i + 1]] + + rows: list[list[str]] = [] for row in visible[i + 2 :]: cells = markdown_table_cells(row) if cells is None: @@ -215,11 +240,37 @@ def extract_markdown_table( ) if any(not cell for cell in cells): die(f"{context} table row contains an empty required cell: {row.strip()}") - table.append(row) - return table + rows.append(cells) + return rows die(f"{context} is missing the expected Markdown table") +def unwrap_markdown_emphasis(cell: str) -> str: + """Remove balanced outer emphasis wrappers; do not unwrap code spans.""" + value = cell.strip() + changed = True + while changed: + changed = False + for marker in EMPHASIS_WRAPPERS: + if ( + len(value) > 2 * len(marker) + and value.startswith(marker) + and value.endswith(marker) + ): + value = value[len(marker) : -len(marker)].strip() + changed = True + break + return value + + +def parse_record_link_cell(cell: str, context: str) -> tuple[str, str]: + value = unwrap_markdown_emphasis(cell) + match = RECORD_LINK_CELL_RE.fullmatch(value) + if match is None: + die(f"{context} has invalid record-link cell: {cell}") + return match.group(1), match.group(2) + + def is_structural_only_line(line: str) -> bool: if HEADING_RE.match(line) or THEMATIC_BREAK_RE.fullmatch(line): return True @@ -350,7 +401,7 @@ def require_prefixed_fields( target_id = record_paths.get(rel) if target_id is None: die(f"record link in {doc_name} is not a discovered OPT record: {rel}") - if label.strip() != target_id: + if re.sub(r"[*_~]", "", label).strip() != target_id: die( f"record link label mismatch in {doc_name}: '{label}' points to " f"{target_id} ({rel})" @@ -362,13 +413,18 @@ def require_prefixed_fields( catalog_lines = section_lines(text, "## Catalog") if not catalog_lines: die("README.md is missing a non-empty visible ## Catalog section") - catalog_table = extract_markdown_table( + catalog_rows = extract_markdown_table( catalog_lines, ("ID", "Optimization", "Status", "Core idea"), "README.md ## Catalog", ) - rows = README_ROW_RE.findall("\n".join(catalog_table)) - counts = Counter(row_id for row_id, _rel, _status in rows) + + parsed_rows: list[tuple[str, str, str]] = [] + for cells in catalog_rows: + row_id, rel = parse_record_link_cell(cells[0], "README.md ## Catalog") + parsed_rows.append((row_id, rel, cells[2])) + + counts = Counter(row_id for row_id, _rel, _status in parsed_rows) bad_counts = sorted(record_id for record_id, count in counts.items() if count != 1) if bad_counts: die( @@ -383,7 +439,7 @@ def require_prefixed_fields( die(f"README.md ## Catalog table references unknown record(s): {', '.join(unknown_rows)}") row_statuses: dict[str, str] = {} - for row_id, rel, raw_status in rows: + for row_id, rel, raw_status in parsed_rows: if record_paths.get(rel) != row_id: die(f"README.md ## Catalog row identity mismatch for {row_id}: {rel}") if row_id in row_statuses: @@ -413,13 +469,20 @@ def require_prefixed_fields( decision_lines = section_lines(catalog, "## Quick decision table") if not decision_lines: die("CATALOG.md is missing a non-empty visible ## Quick decision table section") -decision_table = extract_markdown_table( +decision_rows = extract_markdown_table( decision_lines, ("Bottleneck / problem shape", "First record to inspect", "Core idea"), "CATALOG.md ## Quick decision table", ) -decision_rows = CATALOG_DECISION_ROW_RE.findall("\n".join(decision_table)) -decision_counts = Counter(record_id for record_id, _rel in decision_rows) + +parsed_decisions: list[tuple[str, str]] = [] +for cells in decision_rows: + row_id, rel = parse_record_link_cell( + cells[1], "CATALOG.md ## Quick decision table" + ) + parsed_decisions.append((row_id, rel)) + +decision_counts = Counter(record_id for record_id, _rel in parsed_decisions) bad_decision_counts = sorted( record_id for record_id, count in decision_counts.items() if count != 1 ) @@ -440,7 +503,7 @@ def require_prefixed_fields( "CATALOG.md ## Quick decision table references unknown record(s): " f"{', '.join(unknown_decision)}" ) -for row_id, rel in decision_rows: +for row_id, rel in parsed_decisions: if record_paths.get(rel) != row_id: die(f"CATALOG.md ## Quick decision table row identity mismatch for {row_id}: {rel}") From d0cba2e1b3e2f49a0f498d84e76838c26025527b Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:47:43 +0930 Subject: [PATCH 033/229] Clarify additive and elapsed search budgets --- ...SEARCH-001-budget-aware-adaptive-search.md | 26 +++++++++++-------- 1 file changed, 15 insertions(+), 11 deletions(-) diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index 9d3d149..dfcb504 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -20,9 +20,9 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra - F: candidates in X that satisfy all hard resource, platform, semantic, and correctness constraints before objective ranking - f: the target-measured objective or objective vector for each feasible candidate, including declared noise/statistical treatment - d: the target's predeclared minimize, maximize, lexicographic, or Pareto ordering -- C: search may choose where to evaluate but may not weaken correctness, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must not exceed B after accounting for consumed and conservatively reserved in-flight resources; every per-trial reservation must be an enforceable upper bound rather than an estimate; dispatch/completion accounting must be linearizable; and targets that require deterministic search outcomes must use deterministic observation assimilation **and deterministic proposal/dispatch/refill scheduling** independent of wall-clock completion order -- B: an explicit target-specific hard maximum evaluation, wall-time, compute, monetary, or equivalent resource budget declared before the search starts; the accounting unit, enforceable per-trial cap mechanism, conservative reservation rule, atomic accounting boundary, and failure/cancellation charging policy are fixed before dispatch begins -- S: stop proposing/dispatching when no additional trial can be safely reserved within B, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger and apply proposal, dispatch/refill, assimilation, and stopping decisions to the declared deterministic schedule when determinism is required +- C: search may choose where to evaluate but may not weaken correctness, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must preserve the declared budget model under concurrency; additive resources such as evaluation count, compute, and spend use linearizable consumed/reserved accounting with enforceable per-trial caps, while elapsed wall-time budgets use one enforceable absolute search deadline shared by every worker; and targets that require deterministic search outcomes must use deterministic observation assimilation **and deterministic proposal/dispatch/refill scheduling** independent of wall-clock completion order +- B: an explicit target-specific hard maximum declared before the search starts together with its **budget semantics**: additive resources (for example evaluation count, compute units, or money) use conservative enforceable reservations from one shared ledger, whereas elapsed wall time uses one absolute monotonic search deadline that bounds the whole concurrent search rather than summing overlapping worker seconds; the accounting unit, enforcement mechanism, atomic boundary, and failure/cancellation charging policy are fixed before dispatch begins +- S: stop proposing/dispatching when no additional work is admissible under B, when the absolute wall-time deadline has arrived, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger and apply proposal, dispatch/refill, assimilation, and stopping decisions to the declared deterministic schedule when determinism is required - Variables: mixed search spaces; may include continuous, integer, categorical, and conditional dimensions as explicitly declared by the target - Search scope: local or global, explicitly declared for the target - Objective behavior: deterministic, noisy, or stochastic as declared by the target; noise treatment must be explicit @@ -34,15 +34,17 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra ## Preserved contract -Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound: actual consumed resources plus all still-reserved in-flight capacity must remain within B, no individual trial may consume beyond its reserved cap, and concurrent dispatch/completion transitions must not transiently expose phantom free capacity. If the target requires deterministic selected configurations or trial traces, **both the observation prefix used to create each proposal and the schedule that decides when a new proposal may be generated/dispatched must be deterministic**; worker completion timing may not change the proposal sequence. +Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound according to its declared semantics. For **additive** resources, actual consumed resources plus all still-reserved in-flight capacity must remain within B, no individual trial may consume beyond its reserved cap, and concurrent dispatch/completion transitions must not transiently expose phantom free capacity. For **elapsed wall time**, all workers share one absolute search deadline; overlapping trials do not consume duplicate elapsed seconds, but no proposal, trial, retry, assimilation step, or cleanup that is part of the bounded search may continue past the enforceable deadline except target-declared bounded termination cleanup. If the target requires deterministic selected configurations or trial traces, **both the observation prefix used to create each proposal and the schedule that decides when a new proposal may be generated/dispatched must be deterministic**; worker completion timing may not change the proposal sequence. ## Optimization Use observations to adapt future evaluations: surrogate/acquisition search for expensive black-box objectives, conditional spaces where parameters only exist under certain choices, progressive domain contraction where justified, and explicit stopping/evaluation budgets. For asynchronous workers, reserve pending regions or otherwise diversify proposals so workers do not redundantly evaluate the same neighborhood. -Before dispatching an asynchronous trial, enter one atomic/serializable accounting boundary, reserve a conservative amount of the applicable budget, and record the pending trial in the ledger. If `consumed + reserved + proposed_reservation > B`, do not dispatch. The reservation must be an **enforceable upper limit** for that trial, not merely an estimate: use a per-trial quota, wall-time deadline with forced cancellation/termination, provider spending cap, cgroup/job resource limit, evaluation-slot ownership, or another mechanism that prevents actual trial consumption from exceeding the reservation. If the target cannot enforce such a cap for a resource dimension, that dimension cannot be advertised as a hard maximum B; instead define a different enforceable budget or explicitly classify the quantity as observational rather than bounded. +First classify each hard budget dimension. **Additive budgets**—for example evaluation slots, billable compute, accelerator-seconds, or monetary spend—use one atomic/serializable accounting ledger. Before dispatching a trial against an additive budget, reserve a conservative amount and record the pending trial. If `consumed + reserved + proposed_reservation > B`, do not dispatch. The reservation must be an **enforceable upper limit** for that trial, not merely an estimate: use an evaluation-slot token, provider spending cap, cgroup/job compute quota, or another mechanism that prevents actual additive consumption from exceeding the reservation. If the target cannot enforce such a cap for an additive resource dimension, that dimension cannot be advertised as a hard maximum B; define a different enforceable budget or classify the quantity as observational. -Completion, failure, cancellation, and forced termination use the **same atomic accounting boundary** as dispatch reservation. For one terminal transition, atomically: (1) read the trial's reservation, (2) meter/record the amount actually consumed, (3) move that consumed amount into permanent `consumed`, (4) release only the demonstrably unconsumed remainder from `reserved`, and (5) mark the trial terminal. No dispatcher may observe released reservation capacity before the corresponding consumed charge is committed, and concurrent terminal updates must not lose increments. Completion must not double-charge the same usage. A failed or cancelled trial never erases resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute/time budgets, release only the measured or otherwise provable unused portion of the enforceable reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. Every reservation, cap enforcement action, consumption adjustment, release, failure, cancellation, forced termination, and terminal accounting transaction is recorded in the ledger. +An **elapsed wall-time budget is different**. At search start, compute one absolute deadline from a monotonic clock and make every worker, trial, retry, proposal, model update, and stopping decision subordinate to that same deadline. Do not add overlapping worker durations into `consumed + reserved`; two trials that run concurrently until the same ten-minute deadline consume at most ten minutes of elapsed search time, not twenty. A trial-specific timeout may be shorter, but never later than the remaining global deadline. Dispatch must stop when insufficient time remains for the target's declared safe launch/termination policy, and the runtime must be able to cancel/terminate in-flight work at the global deadline if wall time is claimed as hard. + +Completion, failure, cancellation, and forced termination for **additive** resources use the same atomic accounting boundary as dispatch reservation. For one terminal transition, atomically: (1) read the trial's reservation, (2) meter/record the amount actually consumed, (3) move that consumed amount into permanent `consumed`, (4) release only the demonstrably unconsumed remainder from `reserved`, and (5) mark the trial terminal. No dispatcher may observe released reservation capacity before the corresponding consumed charge is committed, and concurrent terminal updates must not lose increments. Completion must not double-charge the same usage. A failed or cancelled trial never erases additive resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute budgets, release only the measured or otherwise provable unused portion of the enforceable reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. For elapsed wall time, record start/finish/cancellation times for audit but enforce B via the shared absolute deadline rather than a refundable additive reservation. Every reservation, deadline/cap enforcement action, consumption adjustment, release, failure, cancellation, forced termination, and terminal accounting transaction is recorded in the ledger. For targets that require deterministic search behavior, assign deterministic trial IDs and define a **deterministic proposal frontier**. A new proposal may be generated only from a declared ordered observation prefix that is the same in every replay. Buffer out-of-order completions until that prefix is available. Do **not** immediately refill whichever worker happens to become free if doing so would let wall-clock completion order choose the model state used for the next proposal. Acceptable deterministic designs include fixed deterministic batches/barriers, or an ordered-prefix scheduler where proposal `k+1` is generated only after the exact predeclared prefix needed for that proposal has been assimilated and its dispatch slot/order is determined independently of worker-speed races. Surrogate/model updates, acquisition decisions, domain contraction, portfolio-selection state, proposal generation, dispatch/refill decisions, and stopping criteria must consume the same deterministic state sequence. A fixed random seed plus buffered assimilation alone is not sufficient if worker availability can still change which proposal is generated next. If a target chooses immediate completion-driven refill for throughput, declare the resulting nondeterminism as an explicit contract change rather than claiming deterministic replay. @@ -58,20 +60,22 @@ Parallelism has an information cost: very wide batches receive less feedback bet ## Validation -Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. For asynchronous search, test the budget boundary with multiple workers contending for the last remaining reservation and prove no dispatch can make `consumed + reserved` exceed B. Deliberately run trials that attempt to exceed their per-trial money/compute/time reservation and prove the quota/deadline/termination mechanism prevents the overrun. Inject early failures, late failures, partial consumption, and cancellation after measurable work; verify that only demonstrably unconsumed reservation is released, evaluation-count slots are not resurrected after dispatch, and repeated failures cannot create extra budget capacity. +Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. For asynchronous search with **additive** budgets, test the boundary with multiple workers contending for the last remaining reservation and prove no dispatch can make `consumed + reserved` exceed B. Deliberately run trials that attempt to exceed their per-trial money/compute reservation and prove the quota mechanism prevents the overrun. Inject early failures, late failures, partial consumption, and cancellation after measurable work; verify that only demonstrably unconsumed reservation is released, evaluation-count slots are not resurrected after dispatch, and repeated failures cannot create extra budget capacity. + +For **elapsed wall-time** B, use a controlled monotonic clock and launch multiple workers concurrently under one shared deadline. Verify two trials each permitted to run until the same ten-minute deadline are admissible without requiring twenty minutes of additive reservation. Permute worker count, start order, and completion times; prove the search stops launching work as the deadline approaches, every in-flight worker observes the same deadline, forced termination completes within the declared enforcement bound, and total elapsed search lifetime never exceeds B plus only the explicitly declared bounded termination-cleanup allowance. Ensure no retry or model/proposal step can reset or extend the original deadline. -Race multiple trial completions/cancellations against one another and against workers attempting the final dispatch slot. Verify the accounting transaction is linearizable: no consumed increment is lost, no reservation is released before its corresponding consumption is charged, and a dispatcher never observes capacity that would make the post-transaction invariant `consumed + reserved <= B` false. +Race multiple additive-resource trial completions/cancellations against one another and against workers attempting the final dispatch slot. Verify the accounting transaction is linearizable: no consumed increment is lost, no reservation is released before its corresponding consumption is charged, and a dispatcher never observes capacity that would make the post-transaction invariant `consumed + reserved <= B` false. For deterministic targets, run the same seeded search with deliberately permuted worker speeds and completion orders, including the case where trial 2 finishes before trial 1 and frees a worker first. Verify out-of-order completion **does not permit proposal 3 to be generated from a different observation prefix**. The complete proposal sequence, parameter values, deterministic trial IDs, logical dispatch/refill order, surrogate/search states, selected candidate, and stopping reason must match the deterministic reference. Test both fixed-batch/barrier scheduling and any ordered-prefix scheduler the target claims to support. Where sequential/parallel equivalence is part of C, compare the asynchronous execution with its deterministic sequential or batch replay. If completion-driven refill is intentionally retained, verify the target explicitly labels the search trace nondeterministic instead of claiming replay equivalence. ## Target-repo adaptation -Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define the budget accounting unit, conservative per-trial reservation amount, **enforcement mechanism for that reservation**, one atomic/serializable accounting mechanism shared by reservation and terminal conversion, metering source, failure/cancellation charging policy, deterministic observation-assimilation policy, **deterministic proposal frontier and dispatch/refill schedule** (when required), and the exact condition under which a freed worker may receive new work before enabling asynchronous dispatch. +Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define each budget dimension as either **additive** or **elapsed wall time**. For additive resources, define the accounting unit, conservative per-trial reservation amount, enforcement mechanism, one atomic/serializable reservation/completion ledger, metering source, and failure/cancellation charging policy. For elapsed wall time, define the monotonic absolute search deadline, maximum bounded termination-cleanup interval, worker cancellation/termination mechanism, and the minimum remaining-time rule for new dispatch. Also define the deterministic observation-assimilation policy, **deterministic proposal frontier and dispatch/refill schedule** (when required), and the exact condition under which a freed worker may receive new work before enabling asynchronous dispatch. ## Failure modes -Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; over-conservative reservations can reduce useful parallelism; wall-clock completion-order assimilation can make supposedly deterministic search traces irreproducible; **immediate worker refill can also make proposals nondeterministic even when assimilation itself is buffered**. +Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an additive evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced additive reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; **treating elapsed wall time as an additive per-worker resource can falsely reject valid overlapping trials and serialize the search**; conversely, a nominal wall-time limit without one enforceable shared deadline can let work continue past B; wall-clock completion-order assimilation can make supposedly deterministic search traces irreproducible; **immediate worker refill can also make proposals nondeterministic even when assimilation itself is buffered**. ## Rollback trigger -Stop adaptive search when its overhead exceeds evaluation savings, the budget is exhausted, repeated validation does not confirm the selected improvement, any trial can consume beyond its enforceable reservation, any accounting/concurrency test shows that dispatch/terminal transitions can violate B, or any target that requires deterministic search produces different proposals, logical dispatch/refill order, model states, selected candidates, or stopping reasons under permuted asynchronous completion orders. +Stop adaptive search when its overhead exceeds evaluation savings, the declared budget is exhausted, repeated validation does not confirm the selected improvement, any additive trial can consume beyond its enforceable reservation, additive accounting/concurrency tests can violate B, any hard elapsed-wall-time run can exceed its shared absolute deadline beyond the declared bounded cleanup allowance, a retry/worker can extend or reset that deadline, or any target that requires deterministic search produces different proposals, logical dispatch/refill order, model states, selected candidates, or stopping reasons under permuted asynchronous completion orders. From 7fd4eb0acdbe816c8019fc77ea316d01e8f35db4 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 20:48:30 +0930 Subject: [PATCH 034/229] Bound coalescer admission across all generations --- ...01-concurrent-duplicate-work-coalescing.md | 36 +++++++++++-------- 1 file changed, 21 insertions(+), 15 deletions(-) diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index ace234f..9891edd 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,47 +14,51 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime and launch-state policies, waiter limits, overflow/backpressure policies, per-waiter cancellation/deadline/terminal-claim policies, result-preparation/clone-failure policies, retry/error-sharing policies, and result-ownership policies -- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, preparation-failure, launch-cancellation, and error semantics for every joined caller, linearize cancellation/deadline against terminal delivery for each waiter, linearize unstarted-to-running launch against closing/last-waiter cancellation, bound waiter memory, and never admit new waiters to a closing or terminal generation -- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, waiter-memory, launch-state synchronization, result preparation/cloning, atomic terminal-claim, overflow/backpressure, and result-copy overhead +- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime and launch-state policies, per-generation waiter limits, **global/per-tenant in-flight generation and waiter budgets**, overflow/backpressure policies, per-waiter cancellation/deadline/terminal-claim policies, result-preparation/clone-failure policies, retry/error-sharing policies, and result-ownership policies +- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, preparation-failure, launch-cancellation, and error semantics for every joined caller, linearize cancellation/deadline against terminal delivery for each waiter, linearize unstarted-to-running launch against closing/last-waiter cancellation, **bound total registry/generation/waiter memory across all keys and retained closing generations**, and never admit new waiters to a closing or terminal generation +- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, global/per-tenant admission accounting, generation/waiter memory, launch-state synchronization, result preparation/cloning, atomic terminal-claim, overflow/backpressure, and result-copy overhead - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, deadline, launch state, and result-preparation outcome; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; no upstream operation may start after its generation has already become closing due to loss of all live waiters; closing/terminal generations are not joinable; waiter overflow has an explicit bounded behavior -- B: target-specific concurrent-load test budget declared before tuning; no portable request count or duration is supplied here +- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, deadline, launch state, and result-preparation outcome; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; no upstream operation may start after its generation has already become closing due to loss of all live waiters; closing/terminal generations are not joinable; **global/per-tenant generation and waiter admission limits are never exceeded**, including under high-cardinality keys and retained closing generations; overload has an explicit bounded result +- B: target-specific concurrent-load test budget plus explicit global/per-tenant coalescer admission limits (maximum in-flight generations, waiter records, and any bounded overflow queue); no portable request count, duration, or memory cap is supplied here - S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C - Variables: categorical / integer / mixed - Search scope: local policy tuning within one coalescing boundary - Objective behavior: noisy under concurrent load; semantic equivalence remains deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive concurrent-load testing -- Constraints: semantic equivalence, authorization, ownership, launch-state linearizability, result-preparation failure, waiter-memory, cancellation, deadline, terminal-claim, timeout, and resource constraints +- Constraints: semantic equivalence, authorization, ownership, launch-state linearizability, result-preparation failure, **global/per-tenant registry and waiter memory**, cancellation, deadline, terminal-claim, timeout, and resource constraints - Parallelism: asynchronous / concurrent - Exactness: exact request/result semantics; no approximation is introduced ## Preserved contract -Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout/deadline, result-ownership, preparation-failure, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. A configured waiter bound must never be exceeded silently. Each waiter reaches exactly one linearized terminal state; a waiter that has already cancelled or timed out cannot later receive the shared value/error, and a mutable-result clone failure cannot leave a waiter marked successful without a deliverable value. The shared operation also has a linearized launch lifecycle: a generation closed before launch may never subsequently start ownerless upstream work. +Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout/deadline, result-ownership, preparation-failure, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. **Memory/admission bounds apply across the whole coalescer, not only inside one generation:** high-cardinality keys, fresh generations, and retained closing generations must all consume explicit global/per-tenant generation and waiter capacity until their state is actually retired. No request may bypass those caps merely because it is the first waiter for a new key. Each waiter reaches exactly one linearized terminal state; a waiter that has already cancelled or timed out cannot later receive the shared value/error, and a mutable-result clone failure cannot leave a waiter marked successful without a deliverable value. The shared operation also has a linearized launch lifecycle: a generation closed before launch may never subsequently start ownerless upstream work. ## Optimization Create an in-flight registry entry for a canonical equivalence key. The key must include every request attribute required to establish safe sharing, including authorization-relevant tenant/principal/visibility scope unless the target instead proves that the upstream result is globally shareable and independently authorizes each delivered result. +Before creating a **new generation**, atomically reserve both (a) one generation slot from the applicable global/per-tenant in-flight-generation budget and (b) one waiter slot for the initiating caller. The reservation and registry insertion must be one linearizable admission decision. If either capacity is exhausted, do not allocate a partial generation or untracked waiter; return/block/queue according to the declared bounded overload policy. A distinct equivalence key does not get a free first waiter merely because no matching generation exists yet. + Atomically create the joinable generation **with the initiating caller already registered as its first waiter** and with an explicit launch state such as `unstarted`. Do not invoke, schedule, or otherwise permit upstream work yet. This prevents an immediately/synchronously completing operation from reaching terminal state with an empty waiter set while also giving early cancellation a state it can close before any work exists. Linearize upstream launch against the generation's live-waiter and closing state. Under the same registry lock/CAS/transactional boundary used for generation state, permit `unstarted -> running` only while the generation remains joinable and has at least one live `pending` waiter. Install or bind a **sticky upstream cancellation token/handle** as part of that transition, before releasing the serialization boundary. If the last waiter cancels/times out while the generation is still `unstarted`, transition it to `closing/non-joinable` and make any later launch attempt fail; no upstream work is started. If `unstarted -> running` wins first but actual invocation/scheduling occurs immediately afterward, any last-waiter cancellation that races in that interval must set the already-bound sticky cancellation token. The launcher must check/attach that token before or atomically with invocation so a cancellation that has already won cannot be lost merely because the external operation object did not yet exist. There must be no path where the generation is closed with zero live waiters and a creator later launches uncancelled work from stale local state. -Equivalent later callers may register as independent waiters only while the generation is joinable and the configured waiter capacity remains. Waiter admission is atomic with capacity accounting. When the final waiter slot is already occupied, apply one explicit target policy rather than silently exceeding the bound: reject/return a documented overload or retryable-backpressure result, block/queue the caller behind a separately bounded admission mechanism, or use another bounded policy with explicit timeout/cancellation semantics. Starting an unconstrained parallel generation for the same equivalence key is not the default overflow behavior because it recreates the duplicate upstream load this pattern is intended to prevent. If a target deliberately permits overflow generations, that concurrency bound and duplicate-work tradeoff must be part of C/B and validated separately. +Equivalent later callers may register as independent waiters only while the generation is joinable and **both** the per-generation waiter capacity and applicable global/per-tenant waiter capacity remain. Waiter admission is atomic with both counters. When any required capacity is exhausted, apply one explicit target policy rather than silently exceeding the bound: reject/return a documented overload or retryable-backpressure result, block/queue the caller behind a separately bounded admission mechanism, or use another bounded policy with explicit timeout/cancellation semantics. Starting an unconstrained parallel generation for the same equivalence key is not the default overflow behavior because it recreates the duplicate upstream load this pattern is intended to prevent. If a target deliberately permits overflow generations, those generations still consume the global/per-tenant generation budget, and their concurrency/duplicate-work tradeoff must be part of C/B and validated separately. + +**Closing and terminal-but-not-yet-retired generations continue to consume their generation slot and any still-live waiter/cleanup capacity until deterministic retirement actually releases those resources.** This prevents a churn attack from repeatedly canceling callers, leaving expensive upstream operations closing, and creating unlimited fresh generations that evade the advertised memory bound. Resource release must be atomic with retirement so admission cannot observe capacity before the corresponding registry state is gone. Represent each admitted waiter with an atomic terminal state, initially `pending`. Cancellation attempts atomically claim `pending -> cancelled`; timeout/deadline handling atomically claims `pending -> timed-out`. A terminal notifier may claim a delivery outcome only if the waiter's declared deadline has not expired at the claim point. If the deadline is already expired, the notifier must instead leave/transition that waiter to the target's timed-out state and must not deliver the shared terminal value/error. For explicit cancellation racing completion, whichever atomic transition claims `pending` first wins; the losing transition is a no-op for that waiter. These claim semantics are part of the public request contract and must not depend on scheduler timing after the claim. -Cancellation and timeout are otherwise per waiter: when one waiter leaves through a winning cancellation/timeout claim, remove only that waiter from the live-waiter accounting. If live waiters remain, keep the shared generation joinable. If the **last** live waiter leaves, atomically make the generation closing/non-joinable. If it is still `unstarted`, this closure permanently prevents the launch transition. If it is already `running`, set/trigger the sticky upstream cancellation token according to the declared policy. A new caller arriving after the closing transition must create a fresh generation rather than attach to work being prevented/canceled. The closing generation may remain internally tracked until its launch-prevention or terminal cleanup is complete, but it is not eligible for coalescing. +Cancellation and timeout are otherwise per waiter: when one waiter leaves through a winning cancellation/timeout claim, remove only that waiter from the live-waiter accounting. If live waiters remain, keep the shared generation joinable. If the **last** live waiter leaves, atomically make the generation closing/non-joinable. If it is still `unstarted`, this closure permanently prevents the launch transition. If it is already `running`, set/trigger the sticky upstream cancellation token according to the declared policy. A new caller arriving after the closing transition must create a fresh generation **only if global/per-tenant admission capacity permits it** rather than attach to work being prevented/canceled. The closing generation may remain internally tracked until its launch-prevention or terminal cleanup is complete, and it continues consuming its generation budget during that interval. -On upstream success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the candidate waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation and cannot attach to the completed one. Snapshot the waiter records, but do not treat membership in that snapshot as entitlement to delivery: each waiter still competes through its atomic terminal state. +On upstream success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the candidate waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation only through normal bounded admission and cannot attach to the completed one. Snapshot the waiter records, but do not treat membership in that snapshot as entitlement to delivery: each waiter still competes through its atomic terminal state. For an upstream **failure**, no mutable success value must be prepared. Immediately before delivering the shared failure (or a per-caller wrapped equivalent where the API requires ownership/context), attempt the waiter's `pending -> delivered-error` claim subject to the same deadline rule. A waiter whose cancellation/timeout claim already won is skipped. For an upstream **success**, define result ownership explicitly. If the terminal value is immutable/share-safe under the target API, no per-waiter clone is needed; immediately before delivery, attempt `pending -> delivered-success` subject to the same deadline/cancellation race, and deliver only if that claim wins. If callers normally receive mutable or caller-owned results, **prepare the independent defensive clone/copy/copy-on-write handle while the waiter is still `pending`, before claiming successful delivery**. Preparation is not entitlement to delivery: cancellation or timeout may win while preparation is in progress, in which case discard/release the prepared value and deliver nothing. If preparation succeeds, attempt `pending -> delivered-success`; deliver the prepared value only when that claim wins, otherwise discard it. If preparation fails because of allocation, serialization, quota, or another declared preparation error, attempt `pending -> delivered-error` with that preparation failure (again subject to deadline/cancellation). If cancellation/timeout already won, discard the preparation error for delivery purposes. This ensures no waiter is permanently marked successful before a deliverable isolated value exists and every waiter still observes exactly one terminal outcome. -Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, result-preparation behavior, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired. Retire/clean up the generation deterministically after all candidate waiters have either reached their terminal claim or been safely removed under the target cleanup policy. +Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, result-preparation behavior, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired and must pass normal global/per-tenant admission. Retire/clean up the generation deterministically after all candidate waiters have either reached their terminal claim or been safely removed under the target cleanup policy, then atomically release the associated admission capacity. This differs from caching: the reusable result does not exist yet. @@ -70,6 +74,8 @@ This differs from caching: the reusable result does not exist yet. Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; race a new caller against the last-waiter cancellation transition and prove it never joins the closing generation; race a new caller against success/failure completion and prove the terminal generation is made non-joinable before waiter snapshot/notification; test waiter-specific deadlines; verify shared failure delivery and retry accounting; prove only one upstream evaluation occurs per joinable generation while all surviving callers terminate correctly. +Add a **cross-key/global-admission saturation fixture**. Generate many distinct equivalence keys so every request attempts to create its own generation, and separately churn through keys whose prior generations remain `closing` because upstream cancellation/cleanup is delayed. Fill the global and per-tenant generation/waiter budgets to their limits, race additional first callers and later waiters, and prove admission is linearizable: counts never exceed the configured bounds, retained closing generations continue to occupy capacity until retirement, no first waiter bypasses the global cap, and every non-admitted caller receives exactly the declared overload/backpressure behavior. Repeat with multiple tenants to verify one tenant cannot consume capacity reserved for another when per-tenant isolation is part of C. + Add an explicit **registration-to-launch cancellation race**. Pause after the generation and initiating waiter have been registered but before `unstarted -> running`. Cancel or time out that initiating waiter as the last live waiter, then release the launcher. Prove the generation becomes closing/non-joinable and upstream work is never started. In the opposite interleaving, let `unstarted -> running` win but pause before the external invocation exists; then cancel the last waiter and prove the sticky token is already set/observable so the subsequent invocation is suppressed or immediately canceled according to policy. Repeat under high contention and prove no zero-waiter generation can leak a running/hung upstream operation. Add **terminal-delivery races** for both upstream success and upstream failure. Pause after the terminal waiter snapshot, then race explicit cancellation and deadline expiry against each waiter's delivery claim. Prove exactly one `pending -> terminal` transition wins, cancelled/timed-out waiters never receive a later value/error, completion that legitimately claims before cancellation preserves the declared completion result, and an already-expired deadline cannot be bypassed merely because the timeout worker has not run yet. Repeat under high concurrency and verify no waiter observes two terminal outcomes. @@ -82,16 +88,16 @@ Add authorization-boundary fixtures: issue syntactically identical requests unde Add ownership-isolation fixtures for mutable results: deliver one coalesced computation to multiple callers, mutate one caller's returned object, and prove every other caller's result remains unchanged. If the API declares the shared value immutable, attempt prohibited mutation through all exposed aliases and verify the immutability/share-safety contract. -Add waiter-overflow races: fill the waiter list to one slot below the maximum, launch multiple equivalent callers concurrently for the final slot, and prove admission is linearizable, capacity is never exceeded, non-admitted callers receive exactly the documented backpressure/overflow behavior, and cancellation/timeouts of queued or rejected callers remain correct. +Add per-generation waiter-overflow races: fill one generation's waiter list to one slot below the maximum, launch multiple equivalent callers concurrently for the final slot, and prove admission is linearizable, local and global capacity are never exceeded, non-admitted callers receive exactly the documented backpressure/overflow behavior, and cancellation/timeouts of queued or rejected callers remain correct. ## Target-repo adaptation -Define key canonicalization, the authorization/visibility context that participates in equivalence, maximum waiter count, bounded overflow/backpressure semantics, result ownership/share-safety policy, how mutable per-waiter results are prepared and how preparation failures surface, the per-waiter atomic terminal-state representation, cancellation/deadline winning semantics, the generation launch states and serialization primitive for `unstarted -> running` versus `closing`, the sticky cancellation-token/handle semantics used before an external operation object exists, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup of retired generations, and whether failures are shared as terminal or retried under one explicit shared retry policy. +Define key canonicalization, the authorization/visibility context that participates in equivalence, **global and per-tenant maximum in-flight generation counts, total waiter-record limits, whether closing/terminal cleanup consumes those limits, any separately bounded overflow queue**, per-generation waiter count, bounded overload/backpressure semantics, result ownership/share-safety policy, how mutable per-waiter results are prepared and how preparation failures surface, the per-waiter atomic terminal-state representation, cancellation/deadline winning semantics, the generation launch states and serialization primitive for `unstarted -> running` versus `closing`, the sticky cancellation-token/handle semantics used before an external operation object exists, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup/retirement and capacity release, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent or authorization-distinct work; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; registering first but launching from stale creator state after the last waiter already closed an unstarted generation can leak ownerless work; cancellation issued before an external operation exists can be lost without a sticky token or atomic launch state; non-linearized cancellation/deadline versus delivery can produce late values/errors or double terminal outcomes; claiming success before a mutable per-waiter value is successfully prepared can strand a waiter with no deliverable result; clone/preparation failure can race cancellation and create inconsistent outcomes if not claimed atomically; treating terminal snapshot membership as delivery entitlement can notify a waiter after it has timed out; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. +Over-broad keys merge non-equivalent or authorization-distinct work; **per-generation-only waiter caps can still permit unbounded total memory under high-cardinality keys or churned closing generations**; releasing admission capacity before a closing generation is truly retired can let registry state exceed the advertised bound; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; registering first but launching from stale creator state after the last waiter already closed an unstarted generation can leak ownerless work; cancellation issued before an external operation exists can be lost without a sticky token or atomic launch state; non-linearized cancellation/deadline versus delivery can produce late values/errors or double terminal outcomes; claiming success before a mutable per-waiter value is successfully prepared can strand a waiter with no deliverable result; clone/preparation failure can race cancellation and create inconsistent outcomes if not claimed atomically; treating terminal snapshot membership as delivery entitlement can notify a waiter after it has timed out; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/preparation-error semantics; if an upstream operation can start after its generation has become closing with no live waiters; if a pre-launch cancellation can be lost because no cancellation token/operation object existed yet; if a waiter can enter `delivered-success` before an isolated deliverable result exists; if clone/preparation failure can produce no terminal outcome or a second terminal outcome; if a cancelled/timed-out waiter can receive a later terminal value/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if the waiter bound or documented overflow behavior is violated; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. +Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/preparation-error semantics; if **global/per-tenant in-flight generation, waiter-record, or overflow-queue limits can be exceeded across many keys or retained closing generations**; if admission capacity can be reused before the corresponding generation/waiter state is actually retired; if an upstream operation can start after its generation has become closing with no live waiters; if a pre-launch cancellation can be lost because no cancellation token/operation object existed yet; if a waiter can enter `delivered-success` before an isolated deliverable result exists; if clone/preparation failure can produce no terminal outcome or a second terminal outcome; if a cancelled/timed-out waiter can receive a later terminal value/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. From a184fd7a6f64258f146aaba31cde7716ed1b9881 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 21:18:38 +0930 Subject: [PATCH 035/229] Exclude raw HTML blocks from catalog validation --- scripts/check_catalog.py | 74 ++++++++++++++++++++++++++++++++++++++-- 1 file changed, 72 insertions(+), 2 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index fff3500..3396eb8 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -90,6 +90,17 @@ LIST_MARKER_ONLY_RE = re.compile(r"^(?:[-+*]|\d+[.)])$") TABLE_SEPARATOR_CELL_RE = re.compile(r"^:?-{3,}:?$") FENCE_OPEN_RE = re.compile(r"^ {0,3}(`{3,}|~{3,})(.*)$") +RAW_HTML_TYPE1_OPEN_RE = re.compile( + r"^ {0,3}<(?Pscript|pre|style|textarea)(?:[ \t]|>|$)", re.IGNORECASE +) +RAW_HTML_DECLARATION_OPEN_RE = re.compile(r"^ {0,3}]|$)", + re.IGNORECASE, +) +RAW_HTML_COMPLETE_TAG_RE = re.compile( + r"^ {0,3}]*)?/?>[ \t]*$" +) EMPHASIS_WRAPPERS = ("**", "__", "~~", "*", "_") CANONICAL_DEFINITION_PATTERNS = { "X": re.compile(r"^- `X` — \S"), @@ -135,17 +146,46 @@ def strip_html_comments_from_visible_line( return "".join(out), in_comment +def raw_html_block_start(raw: str) -> tuple[str, str | None] | None: + """Return a raw-HTML block mode for CommonMark-style block starts. + + Markdown inside a raw HTML block is not parsed as Markdown, so it must not + satisfy schema headings or fields. Modes are: tag (until matching close), + token (until literal terminator), and blank (until the first blank line). + """ + type1 = RAW_HTML_TYPE1_OPEN_RE.match(raw) + if type1 is not None: + return "tag", type1.group("tag").lower() + if re.match(r"^ {0,3}<\?", raw): + return "token", "?>" + if re.match(r"^ {0,3}".replace(" ", "") + if RAW_HTML_DECLARATION_OPEN_RE.match(raw): + return "token", ">" + if RAW_HTML_BLOCK_TAG_RE.match(raw) or RAW_HTML_COMPLETE_TAG_RE.match(raw): + return "blank", None + return None + + +def raw_html_tag_closes(raw: str, tag: str) -> bool: + return re.search(rf"", raw, re.IGNORECASE) is not None + + def visible_nonfenced_lines(lines: list[str]) -> list[str]: - """Return rendered-ish Markdown lines, excluding comments and fenced blocks. + """Return rendered-ish Markdown lines, excluding non-Markdown constructs. Fence state is determined from the original Markdown line before HTML comments are removed, so a fence-looking line with trailing comment text cannot become - a valid closer after preprocessing. + a valid closer after preprocessing. Raw HTML blocks are also excluded because + Markdown-looking source inside them is not rendered as Markdown headings, + fields, links, or tables. """ visible: list[str] = [] fence_char: str | None = None fence_len = 0 in_comment = False + html_mode: str | None = None + html_end: str | None = None for raw in lines: if fence_char is not None: @@ -157,6 +197,24 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: fence_len = 0 continue + if html_mode is not None: + if html_mode == "tag": + if html_end is not None and raw_html_tag_closes(raw, html_end): + html_mode = None + html_end = None + continue + if html_mode == "token": + if html_end is not None and html_end in raw: + html_mode = None + html_end = None + continue + if html_mode == "blank": + if raw.strip() == "": + html_mode = None + html_end = None + visible.append("") + continue + if in_comment: rendered, in_comment = strip_html_comments_from_visible_line(raw, True) if in_comment: @@ -173,6 +231,18 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: fence_char = run[0] fence_len = len(run) continue + + html_start = raw_html_block_start(raw) + if html_start is not None: + html_mode, html_end = html_start + if html_mode == "tag" and html_end is not None and raw_html_tag_closes(raw, html_end): + html_mode = None + html_end = None + elif html_mode == "token" and html_end is not None and html_end in raw: + html_mode = None + html_end = None + continue + raw_for_parse, in_comment = strip_html_comments_from_visible_line(raw, False) if raw_for_parse: From 6224e2f5902eb98c050cbdac444ef171c11081cb Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 21:51:45 +0930 Subject: [PATCH 036/229] Require semantic predicate rewrites for early reduction --- ...-REDUCE-001-early-working-set-reduction.md | 24 +++++++++++-------- 1 file changed, 14 insertions(+), 10 deletions(-) diff --git a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md index 867bb4e..4be8dca 100644 --- a/optimizations/OPT-REDUCE-001-early-working-set-reduction.md +++ b/optimizations/OPT-REDUCE-001-early-working-set-reduction.md @@ -14,11 +14,11 @@ An expensive operation is applied to a large population even though only a small ## Optimization problem contract -- X: semantically legal placements and implementations of filtering, culling, limiting, candidate selection, or other working-set reductions in the target pipeline -- F: placements that preserve every candidate and every contractually observable behavior required by the reference pipeline—including output, ordering/tie/join semantics, errors/exceptions, writes, mutations, auditing/telemetry, and other side effects—or that move only across stages proven pure with respect to those effects +- X: semantically legal placements and implementations of filtering, culling, limiting, candidate selection, or other working-set reductions in the target pipeline, including any target-specific predicate rewrite required to move a reduction across a stage +- F: placements for which every reduction predicate is proven to commute with every crossed stage or is replaced by a semantics-preserving pre-stage predicate, while also preserving every candidate and every contractually observable behavior required by the reference pipeline—including output, ordering/tie/join semantics, errors/exceptions, writes, mutations, auditing/telemetry, and other side effects. Purity of a crossed stage proves only that displaced side effects are absent; it does not by itself prove that moving the predicate preserves values or membership - f: measured end-to-end pipeline cost and cardinality presented to the expensive stage - d: minimize under the target's predeclared objective ordering -- C: the reordered/reduced pipeline is semantically equivalent to the reference for all declared outputs **and observable effects**; an effectful stage may be bypassed for discarded candidates only when those effects/errors are explicitly proven irrelevant by the target contract +- C: the reordered/reduced pipeline is semantically equivalent to the reference for all declared outputs **and observable effects**; for every crossed transformation `T` and post-stage predicate `p`, the early form must either use an equivalent predicate `p'` satisfying the target's declared commutation/rewrite law (for example `p(T(x)) = p'(x)` for every relevant `x`) or otherwise prove equivalent candidate membership and downstream semantics; an effectful stage may be bypassed for discarded candidates only when those effects/errors are explicitly proven irrelevant by the target contract - B: target-specific benchmark budget over representative and adversarial selectivity distributions; no portable selectivity threshold is supplied here - S: stop when the declared budget is exhausted or a validated early-reduction placement materially lowers total cost without violating C - Variables: categorical / conditional / mixed placement and predicate choices @@ -26,17 +26,19 @@ An expensive operation is applied to a large population even though only a small - Objective behavior: noisy for performance; semantic equivalence is deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive depending on downstream stage cost and workload size -- Constraints: output, ordering/tie/join, side-effect/error, purity, and resource constraints +- Constraints: predicate-commutation/rewrite equivalence, output, ordering/tie/join, side-effect/error, purity, and resource constraints - Parallelism: sequential pipeline semantics with target-specific parallel execution only where equivalence remains valid - Exactness: exact observable semantics; no approximation is introduced ## Preserved contract -Moving a reduction earlier is valid only across a pure stage or when the earlier placement preserves the full observable contract of the reference pipeline. That contract includes final values plus ordering/top-k/tie/join semantics, exceptions/error checks, writes/mutations, audit events, metrics or other externally visible side effects where they are significant. +Moving a reduction earlier is valid only when the reduction predicate remains semantically equivalent across **every crossed stage**. Purity is necessary only to establish that moving past the stage does not displace observable effects; purity alone is not sufficient to justify the reorder. For a transformation `T` followed by predicate `p`, the early placement must either prove that `p` commutes with `T` or use a correctly rewritten pre-stage predicate `p'` with an invariant such as `p(T(x)) = p'(x)` for every relevant input, together with preservation of any downstream values/order required after `T`. The full observable contract also includes final values, ordering/top-k/tie/join semantics, exceptions/error checks, writes/mutations, audit events, metrics, and other externally visible side effects where they are significant. ## Optimization -Push selective operations toward the input boundary only when doing so is semantics-preserving: filter before a pure expensive join/transform, cull before pure rendering work, select candidate roots before pure DSP calculation, or eliminate simulations proven unable to affect any required result/effect. If the expensive stage is effectful, either keep the effectful portion on every candidate that would have reached it in the reference path, split the stage into a pure expensive computation and a required effect layer, or prove those skipped effects/errors are outside the declared contract. Do not optimize away observable behavior merely because the final data rows match. +Push selective operations toward the input boundary only when doing so is semantics-preserving. Before crossing a stage, derive and validate the predicate relation for that stage: retain the same predicate only when it provably commutes, otherwise rewrite it to an equivalent pre-stage predicate, or do not push it across the stage. For example, a post-transform filter `value > 10` cannot be naively moved before a pure `value = input * 2` transform; over a compatible numeric domain it would require the proven rewrite `input > 5` (with boundary, overflow, NaN, rounding, and type semantics handled according to the target contract). Then filter before a pure expensive join/transform, cull before pure rendering work, select candidate roots before pure DSP calculation, or eliminate simulations proven unable to affect any required result/effect only when that relation is established. + +If the expensive stage is effectful, either keep the effectful portion on every candidate that would have reached it in the reference path, split the stage into a pure expensive computation and a required effect layer, or prove those skipped effects/errors are outside the declared contract. Do not optimize away observable behavior merely because the final data rows match. ## Before / after evidence @@ -48,16 +50,18 @@ Push selective operations toward the input boundary only when doing so is semant ## Validation -Differential-test reordered pipelines against the reference, with emphasis on ties, null/missing values, boundary ordering and rare candidates. Also compare observable side effects and error behavior: writes/mutations, audit/log/metric events, callbacks, exception/error surfaces and their relevant ordering/counts. Include a deliberately effectful fixture to prove the optimization is rejected or preserves the effects, and a pure-stage fixture where early reduction is admissible. +Differential-test reordered pipelines against the reference, with emphasis on ties, null/missing values, boundary ordering and rare candidates. For **each crossed stage**, validate the declared commutation law or rewritten predicate over representative, boundary, randomized, and adversarial inputs; compare both candidate membership and final downstream values. Include a negative fixture where `T(x) = 2*x` and the reference applies `value > 10`: prove that naively applying `input > 10` before `T` is rejected because it drops values such as `x = 6`, and prove that any proposed `input > 5` rewrite is accepted only for a domain whose overflow, numeric, and boundary semantics make the equivalence valid. + +Also compare observable side effects and error behavior: writes/mutations, audit/log/metric events, callbacks, exception/error surfaces and their relevant ordering/counts. Include a deliberately effectful fixture to prove the optimization is rejected or preserves the effects, and a pure-but-noncommuting fixture to prove purity alone never authorizes predicate motion. ## Target-repo adaptation -Measure selectivity and reduction cost. Classify the expensive stage as pure or effectful before reordering; inventory contractually significant side effects/errors and define how each is preserved. A cheap filter with low selectivity may simply add another pass. +Measure selectivity and reduction cost. For every candidate reorder, enumerate the crossed stages, classify each stage as pure or effectful, and record the predicate-commutation proof or explicit predicate rewrite required for that stage. Inventory contractually significant side effects/errors and define how each is preserved. Do not infer predicate mobility from purity alone. A cheap filter with low selectivity may simply add another pass. ## Failure modes -Illegal predicate reordering, changed top-k semantics, underestimated filtering cost, loss of vectorization, duplicated scans, skipped writes/audit events/mutations, changed exceptions or validation failures, and reordered side effects can all make an apparently equivalent final result semantically wrong. +Illegal predicate reordering, assuming purity implies predicate commutation, an incorrect or domain-incomplete predicate rewrite, changed top-k semantics, underestimated filtering cost, loss of vectorization, duplicated scans, skipped writes/audit events/mutations, changed exceptions or validation failures, and reordered side effects can all make an apparently equivalent final result semantically wrong. ## Rollback trigger -Immediately revert if any output, ordering, error, or contractually significant side effect differs from the reference path. Also revert if total measured cost does not fall on representative workloads. +Immediately revert if any crossed stage lacks a valid commutation/rewrite proof, if differential testing finds different candidate membership or downstream values, or if any output, ordering, error, or contractually significant side effect differs from the reference path. Also revert if total measured cost does not fall on representative workloads. From 8cd5cc927f9f167c9c2fef6031c640f7a1e64c27 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 21:54:36 +0930 Subject: [PATCH 037/229] Harden rendered Markdown schema validation --- scripts/check_catalog.py | 93 +++++++++++++++++++++++++--------------- 1 file changed, 58 insertions(+), 35 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 3396eb8..899225e 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -102,6 +102,7 @@ r"^ {0,3}]*)?/?>[ \t]*$" ) EMPHASIS_WRAPPERS = ("**", "__", "~~", "*", "_") +STATUS_WRAPPERS = ("**", "__", "~~", "*", "_", "`") CANONICAL_DEFINITION_PATTERNS = { "X": re.compile(r"^- `X` — \S"), "F": re.compile(r"^- `F(?: ⊆ X)?` — \S"), @@ -117,10 +118,13 @@ def die(msg: str) -> None: raise SystemExit(f"catalog-integrity: {msg}") -def strip_html_comments_from_visible_line( - raw: str, in_comment: bool -) -> tuple[str, bool]: - """Remove HTML comments from a non-fenced line, carrying unmatched state.""" +def is_indented_code_line(raw: str) -> bool: + """Return whether a non-fenced line is an indented Markdown code line.""" + return raw.startswith("\t") or raw.startswith(" ") + + +def strip_inline_html_comments(raw: str, in_comment: bool) -> tuple[str, bool]: + """Strip inline HTML comments while carrying a mid-line unmatched comment.""" out: list[str] = [] cursor = 0 @@ -147,12 +151,13 @@ def strip_html_comments_from_visible_line( def raw_html_block_start(raw: str) -> tuple[str, str | None] | None: - """Return a raw-HTML block mode for CommonMark-style block starts. + """Return the raw-HTML block mode for a CommonMark-style block start.""" + # A comment beginning at the start of a block line is raw HTML type 2. The + # complete terminating line belongs to the raw block, even when text follows + # the --> token, so callers must discard that whole line. + if re.match(r"^ {0,3}" - Markdown inside a raw HTML block is not parsed as Markdown, so it must not - satisfy schema headings or fields. Modes are: tag (until matching close), - token (until literal terminator), and blank (until the first blank line). - """ type1 = RAW_HTML_TYPE1_OPEN_RE.match(raw) if type1 is not None: return "tag", type1.group("tag").lower() @@ -172,18 +177,17 @@ def raw_html_tag_closes(raw: str, tag: str) -> bool: def visible_nonfenced_lines(lines: list[str]) -> list[str]: - """Return rendered-ish Markdown lines, excluding non-Markdown constructs. + """Return Markdown-visible lines used by schema validation. - Fence state is determined from the original Markdown line before HTML comments - are removed, so a fence-looking line with trailing comment text cannot become - a valid closer after preprocessing. Raw HTML blocks are also excluded because - Markdown-looking source inside them is not rendered as Markdown headings, - fields, links, or tables. + The pass excludes fenced code, indented code, raw HTML blocks, and HTML + comments before headings, fields, links, and tables are interpreted. + Fence state is derived from the original source line so preprocessing cannot + turn a non-closer into a closer. """ visible: list[str] = [] fence_char: str | None = None fence_len = 0 - in_comment = False + inline_comment = False html_mode: str | None = None html_end: str | None = None @@ -207,6 +211,7 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: if html_end is not None and html_end in raw: html_mode = None html_end = None + # The whole terminator line belongs to the raw HTML block. continue if html_mode == "blank": if raw.strip() == "": @@ -215,14 +220,16 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: visible.append("") continue - if in_comment: - rendered, in_comment = strip_html_comments_from_visible_line(raw, True) - if in_comment: + if inline_comment: + rendered, inline_comment = strip_inline_html_comments(raw, True) + if inline_comment: continue raw_for_parse = rendered else: - # A fence opener is recognized from the original line. This matters - # because HTML comment syntax in a fence info string is literal text. + # Four-space/tab-indented lines are code blocks, not headings/tables. + if is_indented_code_line(raw): + continue + opener = FENCE_OPEN_RE.match(raw) if opener is not None: run = opener.group(1) @@ -241,13 +248,18 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: elif html_mode == "token" and html_end is not None and html_end in raw: html_mode = None html_end = None + # Raw HTML owns the complete source line, including a terminator. continue - raw_for_parse, in_comment = strip_html_comments_from_visible_line(raw, False) + raw_for_parse, inline_comment = strip_inline_html_comments(raw, False) + # An inline comment can expose a remainder, but that remainder still must + # not be accepted if it is indented as code after comment removal. + if raw_for_parse and is_indented_code_line(raw_for_parse): + continue if raw_for_parse: visible.append(raw_for_parse) - elif not in_comment and raw == "": + elif not inline_comment and raw == "": visible.append("") return visible @@ -273,6 +285,10 @@ def section_lines(text: str, heading: str) -> list[str]: def markdown_table_cells(line: str) -> list[str] | None: + # Preserve the four-space/tab code-block distinction even if a caller passes + # a line that did not come through visible_nonfenced_lines. + if is_indented_code_line(line): + return None stripped = line.strip() if not stripped.startswith("|"): return None @@ -315,22 +331,27 @@ def extract_markdown_table( die(f"{context} is missing the expected Markdown table") -def unwrap_markdown_emphasis(cell: str) -> str: - """Remove balanced outer emphasis wrappers; do not unwrap code spans.""" - value = cell.strip() +def unwrap_outer_formatting(value: str, wrappers: tuple[str, ...]) -> str: + """Remove only balanced formatting that wraps the complete value.""" + result = value.strip() changed = True while changed: changed = False - for marker in EMPHASIS_WRAPPERS: + for marker in wrappers: if ( - len(value) > 2 * len(marker) - and value.startswith(marker) - and value.endswith(marker) + len(result) > 2 * len(marker) + and result.startswith(marker) + and result.endswith(marker) ): - value = value[len(marker) : -len(marker)].strip() + result = result[len(marker) : -len(marker)].strip() changed = True break - return value + return result + + +def unwrap_markdown_emphasis(cell: str) -> str: + """Remove balanced outer emphasis wrappers; do not unwrap code spans.""" + return unwrap_outer_formatting(cell, EMPHASIS_WRAPPERS) def parse_record_link_cell(cell: str, context: str) -> tuple[str, str]: @@ -362,8 +383,10 @@ def section_has_content(lines: list[str]) -> bool: def normalized_status_category(raw: str) -> str: - plain = re.sub(r"[*_`]", "", raw).strip() - return plain.split(";", 1)[0].strip() + # Notes after ';' are outside the category. Preserve literal marker characters + # inside the category and unwrap only balanced formatting around the whole one. + category = raw.split(";", 1)[0].strip() + return unwrap_outer_formatting(category, STATUS_WRAPPERS) def require_prefixed_fields( @@ -471,7 +494,7 @@ def require_prefixed_fields( target_id = record_paths.get(rel) if target_id is None: die(f"record link in {doc_name} is not a discovered OPT record: {rel}") - if re.sub(r"[*_~]", "", label).strip() != target_id: + if unwrap_markdown_emphasis(label) != target_id: die( f"record link label mismatch in {doc_name}: '{label}' points to " f"{target_id} ({rel})" From ec33fdcc064514ad1e21f6900fdd1bb00dbc68fe Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:09:57 +0930 Subject: [PATCH 038/229] Align canonical classification field name --- OPTIMIZATION-PROBLEM.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/OPTIMIZATION-PROBLEM.md b/OPTIMIZATION-PROBLEM.md index af9a424..6958dec 100644 --- a/OPTIMIZATION-PROBLEM.md +++ b/OPTIMIZATION-PROBLEM.md @@ -26,13 +26,13 @@ A candidate is admissible only if it lies in `F` **and** satisfies `C`. A faster ## Required classification -Record the following before tuning: +Record the following before tuning. The dimension names below are the canonical field names used by `templates/OPTIMIZATION-RECORD.md` and `scripts/check_catalog.py`: | Dimension | Typical values | | --- | --- | | Variables | continuous / integer / categorical / conditional / mixed | | Search scope | local / global | -| Objective | deterministic / noisy / stochastic | +| Objective behavior | deterministic / noisy / stochastic | | Information | gradient available / derivative-free / black-box | | Evaluation cost | cheap / moderate / expensive | | Constraints | bounds / equality / inequality / semantic / resource | From 8bb1f7026398b6696c97ae42969abf9db153cd7b Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:10:45 +0930 Subject: [PATCH 039/229] Separate peak memory from cumulative pruning budgets --- ...E-001-bound-driven-search-space-pruning.md | 30 +++++++++++-------- 1 file changed, 18 insertions(+), 12 deletions(-) diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index de2d4bb..4ffa34f 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -19,16 +19,16 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F - f: a scalar real-valued target objective `f : F → R` evaluated on feasible candidates only - d: exactly one of scalar `minimize` or scalar `maximize`; vector, Pareto, lexicographic, or other partial-order objectives are outside this record unless a separately specified and validated frontier-bound mechanism is introduced -- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, the target's observable tie semantics are preserved, parallel dispatch cannot oversubscribe the declared hard budget, **every unit of resource consumption covered by a hard wall-time/compute B is accounted for or enclosed by an enforceable whole-search cap**, and frontier exhaustion is declared only after all queued **and leased/in-flight** regions are accounted for -- B: a finite, predeclared target-specific **enforceable** cap. Evaluation-count budgets may count only the declared candidate/bound evaluations. A hard wall-time/compute/resource B must cover the **entire search**, including candidate/bound evaluation, branching, child generation, frontier coordination, serialization, incumbent maintenance, synchronization, cleanup and any other algorithm work that consumes the bounded resource; enforce that with a whole-search deadline/quota or complete metering/reservation. A resource dimension that cannot be capped over the complete search must be labeled observational/best-effort rather than advertised as hard B +- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, the target's observable tie semantics are preserved, parallel dispatch cannot oversubscribe the declared hard budget, **every unit of resource consumption covered by a hard wall-time/compute B is accounted for or enclosed by an enforceable whole-search cap**, every hard peak-memory B is enforced over live allocated/reserved memory rather than cumulative historical allocation, and frontier exhaustion is declared only after all queued **and leased/in-flight** regions are accounted for +- B: a finite, predeclared target-specific **enforceable** cap with its accounting semantics declared explicitly. Evaluation count, money/provider spend, CPU/GPU-seconds, energy, bytes transferred, or other cumulative-flow resources use cumulative accounting. Elapsed wall time uses one shared whole-search deadline. **Peak memory is a stock constraint, not a cumulative flow:** enforce `live_allocated + live_reserved + proposed <= B`, release live capacity when memory is freed, and retain only a recorded `peak_observed` for evidence. If a target instead wants cumulative allocation traffic, it must declare that as a distinct cumulative metric rather than calling it peak memory. Any resource dimension that cannot be hard-capped under its declared semantics must be labeled observational/best-effort rather than advertised as hard B - S: stop immediately when the required optimality/tie contract is proven, or when the **global frontier is exhausted**, meaning there are no queued regions, no leased/in-flight regions still capable of producing candidates/children, and no unpublished child/frontier updates owned by active work. Otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality - Variables: integer / categorical / discrete / mixed - Search scope: global over the declared candidate space - Objective behavior: deterministic unless uncertainty/noise is incorporated into a separately sound bound model - Information: derivative-free; bound/relaxation information is target-specific - Evaluation cost: moderate to expensive when exhaustive evaluation is infeasible -- Constraints: feasibility, semantic correctness, scalar-bound soundness, tie semantics, global-frontier accounting, and enforceable finite-resource constraints -- Parallelism: sequential, or parallel only with synchronized incumbent/frontier/bound state, **leased/in-flight region accounting**, linearizable budget reservation/completion accounting where used, and an enforceable whole-search cap for every wall-time/compute dimension advertised as hard +- Constraints: feasibility, semantic correctness, scalar-bound soundness, tie semantics, global-frontier accounting, and enforceable finite-resource constraints including correctly typed cumulative, deadline, and peak/live-capacity budgets +- Parallelism: sequential, or parallel only with synchronized incumbent/frontier/bound state, **leased/in-flight region accounting**, linearizable reservation/completion accounting for cumulative resources, a shared absolute deadline for elapsed wall time, and live-allocation reservation accounting for hard peak-memory caps - Exactness: exact only when the declared optimality and observable-tie contract is proven within B, including proof that no queued or leased region can still affect the answer; otherwise anytime/incomplete result semantics apply For each unexplored region `R`, define a bound `b(R)` separately from `f`: @@ -47,7 +47,7 @@ An independently proven infeasible region may also be pruned. A heuristic estima ## Preserved contract -A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, exhausting B without a feasible incumbent does not permit an infeasibility claim, and parallel execution must preserve the same hard resource ceiling as sequential execution rather than oversubscribing work in flight. A temporarily empty shared queue is **not** frontier exhaustion while any worker owns a leased region that may still produce a candidate, proof obligation, or child region. Likewise, a hard wall-time/compute B applies to the whole search, not just its explicit evaluation calls. +A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, exhausting B without a feasible incumbent does not permit an infeasibility claim, and parallel execution must preserve the same hard resource ceiling as sequential execution rather than oversubscribing work in flight. A temporarily empty shared queue is **not** frontier exhaustion while any worker owns a leased region that may still produce a candidate, proof obligation, or child region. Likewise, a hard wall-time/compute B applies to the whole search, while a hard peak-memory B applies to current live/reserved memory and must not be converted into irreversible historical consumption after memory is freed. ## Optimization @@ -55,11 +55,15 @@ Maintain an incumbent when one exists, partition the search space, compute a che For **parallel** search, define one global frontier lifecycle. A region remains part of the frontier from enqueue until it is either (a) soundly pruned/closed, or (b) replaced by its child regions through an atomic/linearizable completion transition. Dequeuing for worker ownership therefore changes a region from `queued` to `leased/in-flight`; it does **not** remove that region from the global frontier. A worker that branches a leased region must publish all resulting children and close/release the parent as one frontier-accounting transition, or use another protocol that cannot expose a moment where the queue is empty even though unpublished descendants still exist. Worker failure/cancellation must return or recover the lease so unexplored work is not silently lost. -For evaluation-count budgets, candidate evaluation and nontrivial bound/relaxation evaluation consume/reserve the declared slots before dispatch. For hard wall-time, compute, memory, provider-cost or equivalent resource budgets, the cap must apply to **all search work**, not merely those evaluations. Acceptable designs include an enforceable whole-search deadline/quota/cgroup/job/provider cap, or complete metering where branching, child construction, queue/frontier operations, serialization, incumbent updates, synchronization and cleanup are all charged under the same global ledger. Per-evaluation reservations may still be used internally, but they do not by themselves prove a whole-search wall-time/compute bound. +Treat hard resources according to their physical/accounting semantics rather than forcing them through one ledger shape: -When using reservations, atomically reserve the applicable budget before covered work starts. If `consumed + reserved + proposed_reservation > B`, do not start that work. Completion/failure/cancellation converts the reservation to consumed usage and releases only demonstrably unconsumed capacity under the same linearizable accounting boundary. For a whole-search deadline/quota design, every worker and coordinator path must be subordinate to that cap, including non-evaluation overhead and cleanup required before returning a result. +- **Cumulative-flow budgets** such as evaluation count, money/provider spend, CPU/GPU-seconds, energy, or declared cumulative bytes use linearizable reservations. Atomically reserve before covered work starts; if `consumed + reserved + proposed_reservation > B`, do not start it. Completion/failure/cancellation moves actual consumed usage into permanent `consumed` and releases only demonstrably unconsumed reservation. +- **Elapsed wall time** uses one enforceable monotonic whole-search deadline shared by workers and coordinator paths. Parallel overlap is not double-counted as if seconds were additive reservations, and retries/workers cannot reset or extend the deadline. +- **Peak/live-capacity resources** such as peak memory use live state rather than permanent consumption. Atomically reserve enough live capacity before an allocation-producing operation starts; require `live_allocated + live_reserved + proposed <= B`; on successful allocation move reserved capacity to `live_allocated`; on free/reclamation release the live allocation so later non-overlapping work may reuse the capacity. Maintain `peak_observed = max(peak_observed, live_allocated + live_reserved)` for audit/evidence. Freed memory must not remain in cumulative `consumed` unless the separately declared metric is cumulative allocation traffic rather than peak memory. -If the target meters only candidate/bound evaluations, then only **evaluation count** may be claimed as a hard B from that accounting. Nominal wall-time/compute targets in that design are observational/best-effort and cannot justify finite-cap correctness claims. Similarly, if branching/frontier/serialization overhead can escape an otherwise claimed resource cap, that resource dimension is not hard-bounded. +The relevant hard cap must cover all search work that can consume that resource, including candidate/bound evaluation, branching, child construction, queue/frontier operations, serialization, incumbent updates, synchronization and cleanup. Per-evaluation controls do not by themselves prove a whole-search wall-time, compute, spend, or peak-memory bound. + +If the target meters only candidate/bound evaluations, then only **evaluation count** may be claimed as a hard B from that accounting. Nominal wall-time/compute/memory targets in that design are observational/best-effort and cannot justify finite-cap correctness claims. Similarly, if branching/frontier/serialization overhead can escape an otherwise claimed resource cap, that resource dimension is not hard-bounded. A relaxed solution is evidence for a bound, not automatically a feasible final answer. @@ -77,18 +81,20 @@ For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness i Add **parallel frontier-exhaustion races**. Use a fixture where the last queued region is leased by one worker, making the shared queue empty, then pause that worker before it publishes one or more child regions. Prove the coordinator does not declare exhaustion or exact optimality while that lease remains live. Resume the worker and verify the children become searchable and the final result matches exhaustive/scalar search. Also inject worker failure/cancellation while holding the last lease and verify the region is recovered/requeued or otherwise completed without losing unexplored work. Test simultaneous parent-close/child-publish transitions and prove there is no observation in which both queued and leased frontier counts reach zero before all descendants are durably accounted for. -Add **parallel budget-boundary fixtures**. Race multiple workers against one remaining evaluation slot and prove only one reservation succeeds for an evaluation-count B. Race bound evaluations and candidate evaluations against the same final evaluation capacity and prove both charge the declared ledger. For a hard wall-time/compute/resource B, add fixtures where evaluation itself is cheap but branching, child generation, frontier coordination, serialization or incumbent maintenance deliberately dominates resource use; prove the whole-search quota/deadline stops or charges that overhead before the cap is exceeded. Race completion/cancellation with new work and verify global accounting is linearizable where a ledger is used. For every hard dimension, prove total covered search consumption remains within B—not merely `candidate_eval + bound_eval` consumption. +Add **parallel budget-boundary fixtures**. Race multiple workers against one remaining evaluation slot and prove only one reservation succeeds for an evaluation-count B. Race bound evaluations and candidate evaluations against the same final cumulative capacity and prove both charge the declared ledger. For elapsed wall time, run overlapping workers to the same absolute deadline and prove overlap is not double-counted while no worker/coordinator survives past the enforced deadline. For cumulative compute/spend, deliberately make branching/frontier/serialization work expensive and prove it is charged before the cap is exceeded. + +Add a **peak-memory reuse fixture**. Under a hard peak-memory B, run many sequential/non-overlapping branches that each allocate and then free a large work buffer. Prove each live allocation/reservation is admitted only while `live_allocated + live_reserved <= B`, freed capacity becomes reusable, `peak_observed <= B`, and the search does **not** exhaust merely because the sum of historical allocations exceeds B. Then overlap enough workers to exceed the peak if all allocations were admitted and prove the final reservation is rejected/blocked before live memory can cross B. Race allocation, free, cancellation and cleanup to verify live-memory accounting remains linearizable and no capacity is released before the corresponding memory is actually reclaimable. Add **equal-objective tie fixtures**. For an any-one-optimum contract, prove equality pruning cannot alter any observable result. For deterministic tie-winner contracts, construct regions containing equal-objective candidates with better/worse tie ranks and prove equality-bound regions are retained until the declared tie winner is established. For all-optima contracts, prove every equal-objective optimum is enumerated. If using a stronger total-order bound, validate its soundness independently against exhaustive fixtures. ## Target-repo adaptation -The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. For parallel implementations, define the global frontier state machine, lease ownership/recovery rules, parent-close/child-publish atomicity, and the exact exhaustion predicate over queued plus leased/in-flight work. Also define the budget accounting unit and enforcement boundary. If B is evaluation count, reserve candidate/bound slots linearly. If B is wall time, compute, memory, money or another resource, specify the **whole-search** enforcement mechanism or the complete ledger coverage for evaluations plus branching, frontier work, serialization, incumbent maintenance and coordination. Downgrade any dimension that can escape that enforcement boundary to best-effort/observational rather than calling it hard B. +The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. For parallel implementations, define the global frontier state machine, lease ownership/recovery rules, parent-close/child-publish atomicity, and the exact exhaustion predicate over queued plus leased/in-flight work. Also define each budget's accounting type and enforcement boundary: cumulative flow (`consumed + reserved`), elapsed deadline, or peak/live capacity (`live_allocated + live_reserved`, plus `peak_observed`). If a target says “memory budget,” state whether it means peak live memory or cumulative allocation traffic. Downgrade any dimension that can escape its correct enforcement boundary to best-effort/observational rather than calling it hard B. ## Failure modes -Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; treating queue-empty as frontier-empty can declare exact completion while a leased region still owns unexplored descendants; losing a worker lease can silently drop search regions; non-atomic parent-close/child-publication can create false exhaustion; parallel workers without linearizable evaluation reservations can oversubscribe the last evaluation slot; branching/child/frontier/serialization/incumbent overhead can exceed a nominal wall-time/compute B if only evaluations are charged; an unenforced coordinator/cleanup path can outlive a claimed whole-search deadline; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. +Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; treating queue-empty as frontier-empty can declare exact completion while a leased region still owns unexplored descendants; losing a worker lease can silently drop search regions; non-atomic parent-close/child-publication can create false exhaustion; parallel workers without linearizable evaluation reservations can oversubscribe the last evaluation slot; branching/child/frontier/serialization/incumbent overhead can exceed a nominal wall-time/compute B if only evaluations are charged; an unenforced coordinator/cleanup path can outlive a claimed whole-search deadline; treating peak memory as cumulative consumed spend can falsely exhaust a valid search and prevent reuse of freed capacity; releasing live-memory capacity before actual reclamation can instead oversubscribe the peak; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. ## Rollback trigger -Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel/exact mode if frontier exhaustion can be observed while any leased/in-flight region may still produce work, if parent-close/child-publication or lease recovery can lose unexplored regions, if workers can oversubscribe an evaluation-count budget, or if any resource consumption path can escape a dimension advertised as a hard whole-search B. Abort exact-mode claims whenever B is exhausted before the full objective/tie/frontier contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. +Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel/exact mode if frontier exhaustion can be observed while any leased/in-flight region may still produce work, if parent-close/child-publication or lease recovery can lose unexplored regions, if workers can oversubscribe an evaluation-count/cumulative budget, if an elapsed deadline can be reset/escaped, if peak live memory can exceed B, if freed peak-memory capacity is incorrectly made permanently unavailable, or if any resource-consumption path can escape a dimension advertised as hard. Abort exact-mode claims whenever B is exhausted before the full objective/tie/frontier contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. From ae29237b276e9d7e1fdb44c1917bb759a40a6548 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:11:52 +0930 Subject: [PATCH 040/229] Make coalesced waiter outcome publication atomic --- ...01-concurrent-duplicate-work-coalescing.md | 48 +++++++++++-------- 1 file changed, 27 insertions(+), 21 deletions(-) diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index 9891edd..e672af7 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,11 +14,11 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime and launch-state policies, per-generation waiter limits, **global/per-tenant in-flight generation and waiter budgets**, overflow/backpressure policies, per-waiter cancellation/deadline/terminal-claim policies, result-preparation/clone-failure policies, retry/error-sharing policies, and result-ownership policies -- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, preparation-failure, launch-cancellation, and error semantics for every joined caller, linearize cancellation/deadline against terminal delivery for each waiter, linearize unstarted-to-running launch against closing/last-waiter cancellation, **bound total registry/generation/waiter memory across all keys and retained closing generations**, and never admit new waiters to a closing or terminal generation -- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, global/per-tenant admission accounting, generation/waiter memory, launch-state synchronization, result preparation/cloning, atomic terminal-claim, overflow/backpressure, and result-copy overhead +- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime and launch-state policies, per-generation waiter limits, **global/per-tenant in-flight generation and waiter budgets**, overflow/backpressure policies, per-waiter cancellation/deadline/**atomic terminal-outcome publication** policies, result-preparation/clone-failure policies, retry/error-sharing policies, and result-ownership policies +- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, preparation-failure, launch-cancellation, and error semantics for every joined caller, linearize cancellation/deadline against **irrevocable terminal-outcome publication** for each waiter, linearize unstarted-to-running launch against closing/last-waiter cancellation, **bound total registry/generation/waiter memory across all keys and retained closing generations**, and never admit new waiters to a closing or terminal generation +- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, global/per-tenant admission accounting, generation/waiter memory, launch-state synchronization, result preparation/cloning, atomic outcome publication, overflow/backpressure, and result-copy overhead - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, deadline, launch state, and result-preparation outcome; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; no upstream operation may start after its generation has already become closing due to loss of all live waiters; closing/terminal generations are not joinable; **global/per-tenant generation and waiter admission limits are never exceeded**, including under high-cardinality keys and retained closing generations; overload has an explicit bounded result +- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, deadline, launch state, and result-preparation outcome; **a waiter may become terminal-success/error only when the corresponding outcome is already irrevocably stored/published for that waiter**; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; no upstream operation may start after its generation has already become closing due to loss of all live waiters; closing/terminal generations are not joinable; **global/per-tenant generation and waiter admission limits are never exceeded**, including under high-cardinality keys and retained closing generations; overload has an explicit bounded result - B: target-specific concurrent-load test budget plus explicit global/per-tenant coalescer admission limits (maximum in-flight generations, waiter records, and any bounded overflow queue); no portable request count, duration, or memory cap is supplied here - S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C - Variables: categorical / integer / mixed @@ -26,13 +26,17 @@ Many callers request the same expensive computation concurrently before any call - Objective behavior: noisy under concurrent load; semantic equivalence remains deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive concurrent-load testing -- Constraints: semantic equivalence, authorization, ownership, launch-state linearizability, result-preparation failure, **global/per-tenant registry and waiter memory**, cancellation, deadline, terminal-claim, timeout, and resource constraints +- Constraints: semantic equivalence, authorization, ownership, launch-state linearizability, result-preparation failure, **global/per-tenant registry and waiter memory**, cancellation, deadline, atomic outcome publication, timeout, and resource constraints - Parallelism: asynchronous / concurrent - Exactness: exact request/result semantics; no approximation is introduced ## Preserved contract -Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout/deadline, result-ownership, preparation-failure, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. **Memory/admission bounds apply across the whole coalescer, not only inside one generation:** high-cardinality keys, fresh generations, and retained closing generations must all consume explicit global/per-tenant generation and waiter capacity until their state is actually retired. No request may bypass those caps merely because it is the first waiter for a new key. Each waiter reaches exactly one linearized terminal state; a waiter that has already cancelled or timed out cannot later receive the shared value/error, and a mutable-result clone failure cannot leave a waiter marked successful without a deliverable value. The shared operation also has a linearized launch lifecycle: a generation closed before launch may never subsequently start ownerless upstream work. +Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout/deadline, result-ownership, preparation-failure, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. **Memory/admission bounds apply across the whole coalescer, not only inside one generation:** high-cardinality keys, fresh generations, and retained closing generations must all consume explicit global/per-tenant generation and waiter capacity until their state is actually retired. No request may bypass those caps merely because it is the first waiter for a new key. + +Each waiter has exactly one atomic completion cell/terminal record, initially `pending`. Terminal completion is not a two-step “claim then notify” protocol: the winning transition must atomically compare `pending` and **publish/store the complete terminal outcome**—for example `success(immutable-or-private-result-handle)`, `error(error-record)`, `cancelled`, or `timed-out`—before that terminal state becomes visible. A promise/future `set_result`/`set_exception`, transactional queue/envelope insert, or equivalent single linearization point is acceptable. A subsequent wakeup/signal/callback may tell the caller to inspect the terminal record, but that notification is advisory; failure/interruption of the notifier cannot erase an already published outcome or leave a waiter terminal with nothing retrievable. Cleanup may not retire the waiter/outcome until the target's delivery/acknowledgement/retention contract makes that outcome safely consumable or no longer required. + +The shared operation also has a linearized launch lifecycle: a generation closed before launch may never subsequently start ownerless upstream work. ## Optimization @@ -48,17 +52,19 @@ Equivalent later callers may register as independent waiters only while the gene **Closing and terminal-but-not-yet-retired generations continue to consume their generation slot and any still-live waiter/cleanup capacity until deterministic retirement actually releases those resources.** This prevents a churn attack from repeatedly canceling callers, leaving expensive upstream operations closing, and creating unlimited fresh generations that evade the advertised memory bound. Resource release must be atomic with retirement so admission cannot observe capacity before the corresponding registry state is gone. -Represent each admitted waiter with an atomic terminal state, initially `pending`. Cancellation attempts atomically claim `pending -> cancelled`; timeout/deadline handling atomically claims `pending -> timed-out`. A terminal notifier may claim a delivery outcome only if the waiter's declared deadline has not expired at the claim point. If the deadline is already expired, the notifier must instead leave/transition that waiter to the target's timed-out state and must not deliver the shared terminal value/error. For explicit cancellation racing completion, whichever atomic transition claims `pending` first wins; the losing transition is a no-op for that waiter. These claim semantics are part of the public request contract and must not depend on scheduler timing after the claim. +Cancellation and timeout publish terminal outcomes through the same waiter completion cell. Cancellation attempts atomically compare-and-publish `pending -> cancelled`; timeout/deadline handling atomically compare-and-publishes `pending -> timed-out`. A completion path may publish success/error only if the waiter's declared deadline has not already expired at that same linearization point. If the deadline is expired, completion must instead publish/leave the target's timed-out outcome and must not publish the shared value/error. For explicit cancellation racing completion, whichever atomic compare-and-publish wins `pending` first defines the public outcome; losing transitions are no-ops. Because the winning terminal state already contains the outcome itself, scheduler or notifier timing after that transition cannot change request semantics. + +Cancellation and timeout are otherwise per waiter: when one waiter leaves through a winning cancellation/timeout publication, remove only that waiter from the live-waiter accounting. If live waiters remain, keep the shared generation joinable. If the **last** live waiter leaves, atomically make the generation closing/non-joinable. If it is still `unstarted`, this closure permanently prevents the launch transition. If it is already `running`, set/trigger the sticky upstream cancellation token according to the declared policy. A new caller arriving after the closing transition must create a fresh generation **only if global/per-tenant admission capacity permits it** rather than attach to work being prevented/canceled. The closing generation may remain internally tracked until its launch-prevention or terminal cleanup is complete, and it continues consuming its generation budget during that interval. -Cancellation and timeout are otherwise per waiter: when one waiter leaves through a winning cancellation/timeout claim, remove only that waiter from the live-waiter accounting. If live waiters remain, keep the shared generation joinable. If the **last** live waiter leaves, atomically make the generation closing/non-joinable. If it is still `unstarted`, this closure permanently prevents the launch transition. If it is already `running`, set/trigger the sticky upstream cancellation token according to the declared policy. A new caller arriving after the closing transition must create a fresh generation **only if global/per-tenant admission capacity permits it** rather than attach to work being prevented/canceled. The closing generation may remain internally tracked until its launch-prevention or terminal cleanup is complete, and it continues consuming its generation budget during that interval. +On upstream success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the candidate waiter set. New callers arriving after that terminal transition must create a fresh generation only through normal bounded admission and cannot attach to the completed one. Snapshotting only identifies candidate waiter records; each waiter still resolves through its own atomic outcome publication. -On upstream success or failure, atomically transition the generation to **terminal/non-joinable** (or remove it from the joinable map) **before** snapshotting the candidate waiter set or notifying any waiter. New callers arriving after that terminal transition must create a fresh generation only through normal bounded admission and cannot attach to the completed one. Snapshot the waiter records, but do not treat membership in that snapshot as entitlement to delivery: each waiter still competes through its atomic terminal state. +For an upstream **failure**, prepare the immutable/shared error record or per-caller wrapped error representation as required by the target API while the waiter remains `pending`, then atomically compare-and-publish that complete error outcome into the waiter completion cell. If cancellation/timeout already won, discard any per-caller wrapper and publish nothing else. There is no separate “delivered-error claim” followed by a fallible notification step. -For an upstream **failure**, no mutable success value must be prepared. Immediately before delivering the shared failure (or a per-caller wrapped equivalent where the API requires ownership/context), attempt the waiter's `pending -> delivered-error` claim subject to the same deadline rule. A waiter whose cancellation/timeout claim already won is skipped. +For an upstream **success**, define result ownership explicitly. If the terminal value is immutable/share-safe under the target API, atomically compare-and-publish `success(shared-immutable-handle)` into each eligible pending waiter. If callers normally receive mutable or caller-owned results, **prepare the independent defensive clone/copy/copy-on-write handle while the waiter is still `pending`**. Preparation is not entitlement to delivery: cancellation or timeout may win while preparation is in progress, in which case discard/release the prepared value. If preparation succeeds, atomically compare-and-publish `success(prepared-private-handle)`; if that publication loses to cancellation/timeout, discard the prepared handle. If preparation fails because of allocation, serialization, quota, or another declared preparation error, prepare the corresponding error representation and atomically compare-and-publish `error(preparation-error)` only if the waiter is still pending. Thus no waiter can become terminal-success before a deliverable value exists, and no terminal success/error can exist without its outcome already stored. -For an upstream **success**, define result ownership explicitly. If the terminal value is immutable/share-safe under the target API, no per-waiter clone is needed; immediately before delivery, attempt `pending -> delivered-success` subject to the same deadline/cancellation race, and deliver only if that claim wins. If callers normally receive mutable or caller-owned results, **prepare the independent defensive clone/copy/copy-on-write handle while the waiter is still `pending`, before claiming successful delivery**. Preparation is not entitlement to delivery: cancellation or timeout may win while preparation is in progress, in which case discard/release the prepared value and deliver nothing. If preparation succeeds, attempt `pending -> delivered-success`; deliver the prepared value only when that claim wins, otherwise discard it. If preparation fails because of allocation, serialization, quota, or another declared preparation error, attempt `pending -> delivered-error` with that preparation failure (again subject to deadline/cancellation). If cancellation/timeout already won, discard the preparation error for delivery purposes. This ensures no waiter is permanently marked successful before a deliverable isolated value exists and every waiter still observes exactly one terminal outcome. +After outcome publication, wake/signal/callback delivery may be retried independently. A lost wakeup, interrupted notifier task, callback exception after publication, or scheduler failure must not make the terminal outcome inaccessible: the waiter/future/queue record remains the source of truth. If the target API cannot provide such an irrevocable completion cell or transactional delivery record, it must keep the waiter nonterminal until delivery itself is irrevocably committed; it may not mark success/error first and hope a later notification succeeds. -Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, result-preparation behavior, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired and must pass normal global/per-tenant admission. Retire/clean up the generation deterministically after all candidate waiters have either reached their terminal claim or been safely removed under the target cleanup policy, then atomically release the associated admission capacity. +Do not silently retry for only some joined callers; if shared retry is supported, its attempt limit, backoff, budget charging, authorization scope, result-preparation behavior, and terminal error semantics must be part of the declared policy. Otherwise, a retry starts a new generation after the failed generation is retired and must pass normal global/per-tenant admission. Retire/clean up the generation deterministically only after every candidate waiter has either published one terminal outcome or been safely removed under the target delivery/retention policy, then atomically release the associated admission capacity. This differs from caching: the reusable result does not exist yet. @@ -72,32 +78,32 @@ This differs from caching: the reusable result does not exist yet. ## Validation -Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; race a new caller against the last-waiter cancellation transition and prove it never joins the closing generation; race a new caller against success/failure completion and prove the terminal generation is made non-joinable before waiter snapshot/notification; test waiter-specific deadlines; verify shared failure delivery and retry accounting; prove only one upstream evaluation occurs per joinable generation while all surviving callers terminate correctly. +Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; race a new caller against the last-waiter cancellation transition and prove it never joins the closing generation; race a new caller against success/failure completion and prove the terminal generation is made non-joinable before waiter snapshot/outcome publication; test waiter-specific deadlines; verify shared failure publication and retry accounting; prove only one upstream evaluation occurs per joinable generation while all surviving callers terminate correctly. Add a **cross-key/global-admission saturation fixture**. Generate many distinct equivalence keys so every request attempts to create its own generation, and separately churn through keys whose prior generations remain `closing` because upstream cancellation/cleanup is delayed. Fill the global and per-tenant generation/waiter budgets to their limits, race additional first callers and later waiters, and prove admission is linearizable: counts never exceed the configured bounds, retained closing generations continue to occupy capacity until retirement, no first waiter bypasses the global cap, and every non-admitted caller receives exactly the declared overload/backpressure behavior. Repeat with multiple tenants to verify one tenant cannot consume capacity reserved for another when per-tenant isolation is part of C. Add an explicit **registration-to-launch cancellation race**. Pause after the generation and initiating waiter have been registered but before `unstarted -> running`. Cancel or time out that initiating waiter as the last live waiter, then release the launcher. Prove the generation becomes closing/non-joinable and upstream work is never started. In the opposite interleaving, let `unstarted -> running` win but pause before the external invocation exists; then cancel the last waiter and prove the sticky token is already set/observable so the subsequent invocation is suppressed or immediately canceled according to policy. Repeat under high contention and prove no zero-waiter generation can leak a running/hung upstream operation. -Add **terminal-delivery races** for both upstream success and upstream failure. Pause after the terminal waiter snapshot, then race explicit cancellation and deadline expiry against each waiter's delivery claim. Prove exactly one `pending -> terminal` transition wins, cancelled/timed-out waiters never receive a later value/error, completion that legitimately claims before cancellation preserves the declared completion result, and an already-expired deadline cannot be bypassed merely because the timeout worker has not run yet. Repeat under high concurrency and verify no waiter observes two terminal outcomes. +Add **terminal-publication races** for upstream success, upstream failure, cancellation, and deadline expiry. Pause immediately before each waiter compare-and-publish and prove exactly one `pending -> terminal(outcome)` transition wins. Then pause **after** success/error publication but before any wakeup/callback; kill/fail/interupt the notifier and prove the waiter can still retrieve exactly the already-published result/error from its completion cell, cancellation/timeout cannot replace it, and cleanup does not retire it prematurely. Repeat for immutable success, mutable private clones, shared upstream errors, preparation errors, lost wakeups, callback exceptions, and high concurrency. Verify no terminal waiter ever lacks a retrievable outcome and no waiter observes two outcomes. -Add **clone/preparation-failure fixtures** for mutable results. Force allocation, serialization, copy-on-write setup, or quota failure while preparing a per-waiter value. Verify a waiter is still `pending` until preparation succeeds; successful preparation followed by a winning cancellation/timeout causes the prepared value to be discarded; failed preparation can atomically resolve to exactly one `delivered-error` only if cancellation/timeout has not already won; and no clone failure can leave a waiter in `delivered-success` without an actual value. Race clone success/failure against cancellation and deadline expiry repeatedly under load. +Add **clone/preparation-failure fixtures** for mutable results. Force allocation, serialization, copy-on-write setup, or quota failure while preparing a per-waiter value. Verify a waiter is still `pending` until preparation succeeds or a preparation-error outcome is atomically published; successful preparation followed by a winning cancellation/timeout causes the prepared value to be discarded; failed preparation can resolve to exactly one published error only if cancellation/timeout has not already won; and no clone failure can leave a waiter in terminal success without an actual value. Race clone success/failure against cancellation and deadline expiry repeatedly under load. -Add an **immediate synchronous-completion** fixture where the upstream operation can finish inline once launch actually begins. Prove the initiating caller was already registered before launch and always receives the terminal result/error unless its own cancellation/deadline claim wins under the same rules. +Add an **immediate synchronous-completion** fixture where the upstream operation can finish inline once launch actually begins. Prove the initiating caller was already registered before launch and always obtains the terminal result/error from its completion cell unless its own cancellation/deadline publication wins under the same rules. -Add authorization-boundary fixtures: issue syntactically identical requests under different tenants, principals, roles, ACL/visibility scopes, or other authorization context. Prove they either map to different equivalence keys **or** that the shared upstream result is explicitly safe to reuse and each caller is independently authorized before delivery. Verify that a result produced under one authorization scope can never leak to another merely because the resource parameters match. +Add authorization-boundary fixtures: issue syntactically identical requests under different tenants, principals, roles, ACL/visibility scopes, or other authorization context. Prove they either map to different equivalence keys **or** that the shared upstream result is explicitly safe to reuse and each caller is independently authorized before outcome publication. Verify that a result produced under one authorization scope can never leak to another merely because the resource parameters match. -Add ownership-isolation fixtures for mutable results: deliver one coalesced computation to multiple callers, mutate one caller's returned object, and prove every other caller's result remains unchanged. If the API declares the shared value immutable, attempt prohibited mutation through all exposed aliases and verify the immutability/share-safety contract. +Add ownership-isolation fixtures for mutable results: publish one coalesced computation to multiple callers, mutate one caller's returned object, and prove every other caller's result remains unchanged. If the API declares the shared value immutable, attempt prohibited mutation through all exposed aliases and verify the immutability/share-safety contract. Add per-generation waiter-overflow races: fill one generation's waiter list to one slot below the maximum, launch multiple equivalent callers concurrently for the final slot, and prove admission is linearizable, local and global capacity are never exceeded, non-admitted callers receive exactly the documented backpressure/overflow behavior, and cancellation/timeouts of queued or rejected callers remain correct. ## Target-repo adaptation -Define key canonicalization, the authorization/visibility context that participates in equivalence, **global and per-tenant maximum in-flight generation counts, total waiter-record limits, whether closing/terminal cleanup consumes those limits, any separately bounded overflow queue**, per-generation waiter count, bounded overload/backpressure semantics, result ownership/share-safety policy, how mutable per-waiter results are prepared and how preparation failures surface, the per-waiter atomic terminal-state representation, cancellation/deadline winning semantics, the generation launch states and serialization primitive for `unstarted -> running` versus `closing`, the sticky cancellation-token/handle semantics used before an external operation object exists, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup/retirement and capacity release, and whether failures are shared as terminal or retried under one explicit shared retry policy. +Define key canonicalization, the authorization/visibility context that participates in equivalence, **global and per-tenant maximum in-flight generation counts, total waiter-record limits, whether closing/terminal cleanup consumes those limits, any separately bounded overflow queue**, per-generation waiter count, bounded overload/backpressure semantics, result ownership/share-safety policy, how mutable per-waiter results are prepared and how preparation failures surface, the waiter-owned atomic completion cell/transactional delivery representation, cancellation/deadline winning semantics, outcome-retention/acknowledgement and notifier retry semantics, the generation launch states and serialization primitive for `unstarted -> running` versus `closing`, the sticky cancellation-token/handle semantics used before an external operation object exists, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup/retirement and capacity release, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent or authorization-distinct work; **per-generation-only waiter caps can still permit unbounded total memory under high-cardinality keys or churned closing generations**; releasing admission capacity before a closing generation is truly retired can let registry state exceed the advertised bound; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; registering first but launching from stale creator state after the last waiter already closed an unstarted generation can leak ownerless work; cancellation issued before an external operation exists can be lost without a sticky token or atomic launch state; non-linearized cancellation/deadline versus delivery can produce late values/errors or double terminal outcomes; claiming success before a mutable per-waiter value is successfully prepared can strand a waiter with no deliverable result; clone/preparation failure can race cancellation and create inconsistent outcomes if not claimed atomically; treating terminal snapshot membership as delivery entitlement can notify a waiter after it has timed out; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. +Over-broad keys merge non-equivalent or authorization-distinct work; **per-generation-only waiter caps can still permit unbounded total memory under high-cardinality keys or churned closing generations**; releasing admission capacity before a closing generation is truly retired can let registry state exceed the advertised bound; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; registering first but launching from stale creator state after the last waiter already closed an unstarted generation can leak ownerless work; cancellation issued before an external operation exists can be lost without a sticky token or atomic launch state; non-linearized cancellation/deadline versus terminal publication can produce late or double outcomes; **marking a waiter terminal before atomically storing/enqueuing its outcome can strand it if notification fails**; cleanup that retires a published outcome before it is safely consumable can lose delivery; claiming success before a mutable per-waiter value is successfully prepared can strand a waiter with no deliverable result; clone/preparation failure can race cancellation and create inconsistent outcomes if not published atomically; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/preparation-error semantics; if **global/per-tenant in-flight generation, waiter-record, or overflow-queue limits can be exceeded across many keys or retained closing generations**; if admission capacity can be reused before the corresponding generation/waiter state is actually retired; if an upstream operation can start after its generation has become closing with no live waiters; if a pre-launch cancellation can be lost because no cancellation token/operation object existed yet; if a waiter can enter `delivered-success` before an isolated deliverable result exists; if clone/preparation failure can produce no terminal outcome or a second terminal outcome; if a cancelled/timed-out waiter can receive a later terminal value/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. +Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/preparation-error semantics; if **global/per-tenant in-flight generation, waiter-record, or overflow-queue limits can be exceeded across many keys or retained closing generations**; if admission capacity can be reused before the corresponding generation/waiter state is actually retired; if an upstream operation can start after its generation has become closing with no live waiters; if a pre-launch cancellation can be lost because no cancellation token/operation object existed yet; if any success/error terminal state can become visible before its complete outcome is irrevocably stored/enqueued; if notifier/wakeup/callback failure after terminal publication can make the outcome inaccessible; if cleanup can retire an undelivered/unacknowledged outcome contrary to the target retention contract; if a waiter can enter terminal success before an isolated deliverable result exists; if clone/preparation failure can produce no terminal outcome or a second terminal outcome; if a cancelled/timed-out waiter can later be replaced by success/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. From d5754232075b9d9b5fdd5ad27105807a990deedd Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:13:02 +0930 Subject: [PATCH 041/229] Reject non-rendering link definitions as section content --- scripts/check_catalog.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 899225e..8d5d587 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -83,6 +83,9 @@ STATUS_RE = re.compile(r"^\*\*Status:\*\*\s*(.*?)\s*$") OPT_TOKEN_RE = re.compile(r"\bOPT-[A-Z]+-\d{3}\b") EMPTY_LABEL_RE = re.compile(r"^-\s+[^:]+:\s*$") +LINK_REFERENCE_DEFINITION_RE = re.compile( + r"^\[(?:\\.|[^\[\]\\])+\]:[ \t]+\S.*$" +) HEADING_RE = re.compile(r"^#{1,6}(?:\s|$)") THEMATIC_BREAK_RE = re.compile( r"^(?:\*(?:[ \t]*\*){2,}|-(?:[ \t]*-){2,}|_(?:[ \t]*_){2,})[ \t]*$" @@ -367,6 +370,8 @@ def is_structural_only_line(line: str) -> bool: return True if LIST_MARKER_ONLY_RE.fullmatch(line) or line == ">": return True + if LINK_REFERENCE_DEFINITION_RE.fullmatch(line): + return True cells = markdown_table_cells(line) return bool(cells and all(TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in cells)) From b22854f11b928bab2d2e271b940dab1bc04a1357 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:32:33 +0930 Subject: [PATCH 042/229] Reject markup-only contract field values --- scripts/check_catalog.py | 33 +++++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 8d5d587..e5e8365 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -86,6 +86,11 @@ LINK_REFERENCE_DEFINITION_RE = re.compile( r"^\[(?:\\.|[^\[\]\\])+\]:[ \t]+\S.*$" ) +INLINE_IMAGE_RE = re.compile(r"!\[([^\]]*)\]\([^)]*\)") +INLINE_LINK_RE = re.compile(r"\[([^\]]*)\]\([^)]*\)") +REFERENCE_IMAGE_RE = re.compile(r"!\[([^\]]*)\]\[[^\]]*\]") +REFERENCE_LINK_RE = re.compile(r"\[([^\]]*)\]\[[^\]]*\]") +INLINE_HTML_TAG_RE = re.compile(r"]*>") HEADING_RE = re.compile(r"^#{1,6}(?:\s|$)") THEMATIC_BREAK_RE = re.compile( r"^(?:\*(?:[ \t]*\*){2,}|-(?:[ \t]*-){2,}|_(?:[ \t]*_){2,})[ \t]*$" @@ -365,6 +370,29 @@ def parse_record_link_cell(cell: str, context: str) -> tuple[str, str]: return match.group(1), match.group(2) +def rendered_inline_text(value: str) -> str: + """Approximate visible inline text for required field-value validation. + + Contract fields must contain textual substance after non-rendering Markdown + constructs are removed. Link/image destinations, formatting markers, and HTML + tags therefore cannot make an otherwise empty field count as populated. + """ + text = value + text = INLINE_IMAGE_RE.sub(lambda m: m.group(1), text) + text = INLINE_LINK_RE.sub(lambda m: m.group(1), text) + text = REFERENCE_IMAGE_RE.sub(lambda m: m.group(1), text) + text = REFERENCE_LINK_RE.sub(lambda m: m.group(1), text) + text = INLINE_HTML_TAG_RE.sub("", text) + text = re.sub(r"[`*_~]", "", text) + text = re.sub(r"\\(.)", r"\1", text) + return text.strip() + + +def has_substantive_rendered_text(value: str) -> bool: + """Require at least one visible alphanumeric character after inline parsing.""" + return any(ch.isalnum() for ch in rendered_inline_text(value)) + + def is_structural_only_line(line: str) -> bool: if HEADING_RE.match(line) or THEMATIC_BREAK_RE.fullmatch(line): return True @@ -413,6 +441,11 @@ def require_prefixed_fields( value = matches[0][len(prefix) :].strip() if not value: die(f"{path.relative_to(ROOT)} has empty field {field} in {section}") + if not has_substantive_rendered_text(value): + die( + f"{path.relative_to(ROOT)} has markup-only/non-substantive field " + f"{field} in {section}: '{value}'" + ) if rejected_values is not None and value == rejected_values.get(field): die( f"{path.relative_to(ROOT)} has unselected template placeholder " From 1a45293a9a610618eb82676a6439c7503700d786 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:33:31 +0930 Subject: [PATCH 043/229] Cover coordinator work in hard search compute budgets --- ...SEARCH-001-budget-aware-adaptive-search.md | 24 +++++++++++-------- 1 file changed, 14 insertions(+), 10 deletions(-) diff --git a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md index dfcb504..1ec474c 100644 --- a/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md +++ b/optimizations/OPT-SEARCH-001-budget-aware-adaptive-search.md @@ -20,9 +20,9 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra - F: candidates in X that satisfy all hard resource, platform, semantic, and correctness constraints before objective ranking - f: the target-measured objective or objective vector for each feasible candidate, including declared noise/statistical treatment - d: the target's predeclared minimize, maximize, lexicographic, or Pareto ordering -- C: search may choose where to evaluate but may not weaken correctness, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must preserve the declared budget model under concurrency; additive resources such as evaluation count, compute, and spend use linearizable consumed/reserved accounting with enforceable per-trial caps, while elapsed wall-time budgets use one enforceable absolute search deadline shared by every worker; and targets that require deterministic search outcomes must use deterministic observation assimilation **and deterministic proposal/dispatch/refill scheduling** independent of wall-clock completion order -- B: an explicit target-specific hard maximum declared before the search starts together with its **budget semantics**: additive resources (for example evaluation count, compute units, or money) use conservative enforceable reservations from one shared ledger, whereas elapsed wall time uses one absolute monotonic search deadline that bounds the whole concurrent search rather than summing overlapping worker seconds; the accounting unit, enforcement mechanism, atomic boundary, and failure/cancellation charging policy are fixed before dispatch begins -- S: stop proposing/dispatching when no additional work is admissible under B, when the absolute wall-time deadline has arrived, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial ledger and apply proposal, dispatch/refill, assimilation, and stopping decisions to the declared deterministic schedule when determinism is required +- C: search may choose where to evaluate but may not weaken correctness, evidence, API, trust, or other target semantics to improve f; asynchronous dispatch must preserve the declared budget model under concurrency; additive resources such as evaluation count, compute, and spend use linearizable consumed/reserved accounting with enforceable caps, **and any hard compute-unit budget advertised for the search must also cover coordinator/search-control work that consumes that compute resource—including surrogate fitting, acquisition/proposal generation, model updates, frontier/portfolio coordination, assimilation, and stopping logic—through the same ledger or an enforceable whole-search compute quota**; elapsed wall-time budgets use one enforceable absolute search deadline shared by every worker; and targets that require deterministic search outcomes must use deterministic observation assimilation **and deterministic proposal/dispatch/refill scheduling** independent of wall-clock completion order +- B: an explicit target-specific hard maximum declared before the search starts together with its **budget semantics**: additive resources (for example evaluation count, compute units, or money) use conservative enforceable reservations from one shared ledger, but a hard compute-unit B is valid only if **all search work consuming the bounded compute resource, including coordinator/model/proposal work outside trials, is charged to that ledger or enclosed by one enforceable whole-search compute quota**; elapsed wall time uses one absolute monotonic search deadline that bounds the whole concurrent search rather than summing overlapping worker seconds; the accounting unit, enforcement mechanism, covered work, atomic boundary, and failure/cancellation charging policy are fixed before dispatch begins +- S: stop proposing/dispatching when no additional work is admissible under B, when the absolute wall-time deadline has arrived, when a predeclared objective/quality target is met, or when a predeclared stagnation/convergence rule fires; preserve the reason for stopping in the trial/search ledger and apply proposal, dispatch/refill, assimilation, and stopping decisions to the declared deterministic schedule when determinism is required - Variables: mixed search spaces; may include continuous, integer, categorical, and conditional dimensions as explicitly declared by the target - Search scope: local or global, explicitly declared for the target - Objective behavior: deterministic, noisy, or stochastic as declared by the target; noise treatment must be explicit @@ -34,7 +34,7 @@ Optimization knobs are selected by folklore, exhaustive sweeps, or a few arbitra ## Preserved contract -Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound according to its declared semantics. For **additive** resources, actual consumed resources plus all still-reserved in-flight capacity must remain within B, no individual trial may consume beyond its reserved cap, and concurrent dispatch/completion transitions must not transiently expose phantom free capacity. For **elapsed wall time**, all workers share one absolute search deadline; overlapping trials do not consume duplicate elapsed seconds, but no proposal, trial, retry, assimilation step, or cleanup that is part of the bounded search may continue past the enforceable deadline except target-declared bounded termination cleanup. If the target requires deterministic selected configurations or trial traces, **both the observation prefix used to create each proposal and the schedule that decides when a new proposal may be generated/dispatched must be deterministic**; worker completion timing may not change the proposal sequence. +Search may choose *where to evaluate* but may not weaken correctness constraints to improve the objective. Under asynchronous execution, the declared maximum budget remains a hard bound according to its declared semantics. For **additive** resources, actual consumed resources plus all still-reserved in-flight capacity must remain within B, no covered operation may consume beyond its enforceable reservation/quota, and concurrent dispatch/completion transitions must not transiently expose phantom free capacity. If the additive resource is **compute**, the bounded search includes not only trial execution but every coordinator/search-control operation that consumes that compute resource; per-trial reservations alone do not prove a hard whole-search compute cap. For **elapsed wall time**, all workers share one absolute search deadline; overlapping trials do not consume duplicate elapsed seconds, but no proposal, trial, retry, assimilation step, or cleanup that is part of the bounded search may continue past the enforceable deadline except target-declared bounded termination cleanup. If the target requires deterministic selected configurations or trial traces, **both the observation prefix used to create each proposal and the schedule that decides when a new proposal may be generated/dispatched must be deterministic**; worker completion timing may not change the proposal sequence. ## Optimization @@ -42,9 +42,11 @@ Use observations to adapt future evaluations: surrogate/acquisition search for e First classify each hard budget dimension. **Additive budgets**—for example evaluation slots, billable compute, accelerator-seconds, or monetary spend—use one atomic/serializable accounting ledger. Before dispatching a trial against an additive budget, reserve a conservative amount and record the pending trial. If `consumed + reserved + proposed_reservation > B`, do not dispatch. The reservation must be an **enforceable upper limit** for that trial, not merely an estimate: use an evaluation-slot token, provider spending cap, cgroup/job compute quota, or another mechanism that prevents actual additive consumption from exceeding the reservation. If the target cannot enforce such a cap for an additive resource dimension, that dimension cannot be advertised as a hard maximum B; define a different enforceable budget or classify the quantity as observational. +A **hard compute-unit budget for the search is a whole-search resource contract**, not merely a sum of trial quotas. Any surrogate/model fit, acquisition optimization, candidate/proposal generation, portfolio/frontier update, observation assimilation, scheduler/coordination step, stopping-rule computation, serialization, or other coordinator work that consumes the bounded compute unit must be inside the same enforcement boundary. Two acceptable designs are: (1) an enforceable whole-search process/job/cgroup/provider compute quota that contains both workers and coordinator, or (2) complete shared-ledger accounting where coordinator operations reserve/charge compute before execution just as trials do, with no unmetered control path. If coordinator compute cannot be bounded or completely metered under the claimed unit, do not call that dimension a hard search B; scope the hard claim more narrowly (for example trial accelerator-seconds only) and label coordinator/total compute observational. + An **elapsed wall-time budget is different**. At search start, compute one absolute deadline from a monotonic clock and make every worker, trial, retry, proposal, model update, and stopping decision subordinate to that same deadline. Do not add overlapping worker durations into `consumed + reserved`; two trials that run concurrently until the same ten-minute deadline consume at most ten minutes of elapsed search time, not twenty. A trial-specific timeout may be shorter, but never later than the remaining global deadline. Dispatch must stop when insufficient time remains for the target's declared safe launch/termination policy, and the runtime must be able to cancel/terminate in-flight work at the global deadline if wall time is claimed as hard. -Completion, failure, cancellation, and forced termination for **additive** resources use the same atomic accounting boundary as dispatch reservation. For one terminal transition, atomically: (1) read the trial's reservation, (2) meter/record the amount actually consumed, (3) move that consumed amount into permanent `consumed`, (4) release only the demonstrably unconsumed remainder from `reserved`, and (5) mark the trial terminal. No dispatcher may observe released reservation capacity before the corresponding consumed charge is committed, and concurrent terminal updates must not lose increments. Completion must not double-charge the same usage. A failed or cancelled trial never erases additive resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute budgets, release only the measured or otherwise provable unused portion of the enforceable reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. For elapsed wall time, record start/finish/cancellation times for audit but enforce B via the shared absolute deadline rather than a refundable additive reservation. Every reservation, deadline/cap enforcement action, consumption adjustment, release, failure, cancellation, forced termination, and terminal accounting transaction is recorded in the ledger. +Completion, failure, cancellation, and forced termination for **additive** resources use the same atomic accounting boundary as dispatch reservation. For one terminal trial transition, atomically: (1) read the trial's reservation, (2) meter/record the amount actually consumed, (3) move that consumed amount into permanent `consumed`, (4) release only the demonstrably unconsumed remainder from `reserved`, and (5) mark the trial terminal. Coordinator/search-control compute charged through the ledger follows the same principle: reserve or otherwise atomically debit the bounded compute before covered work begins, then commit actual consumption and release only provably unused capacity. No dispatcher or coordinator may observe released capacity before the corresponding consumed charge is committed, and concurrent updates must not lose increments. Completion must not double-charge the same usage. A failed or cancelled trial never erases additive resources already consumed. For an evaluation-count budget, dispatch consumes the evaluation slot and it is not refunded merely because the trial later fails or is cancelled. For money/compute budgets, release only the measured or otherwise provable unused portion of the enforceable reservation. If unconsumed capacity cannot be established safely, retain the conservative charge. For elapsed wall time, record start/finish/cancellation times for audit but enforce B via the shared absolute deadline rather than a refundable additive reservation. Every reservation, coordinator charge, whole-search quota action, deadline/cap enforcement action, consumption adjustment, release, failure, cancellation, forced termination, and terminal accounting transaction is recorded in the ledger/audit trail. For targets that require deterministic search behavior, assign deterministic trial IDs and define a **deterministic proposal frontier**. A new proposal may be generated only from a declared ordered observation prefix that is the same in every replay. Buffer out-of-order completions until that prefix is available. Do **not** immediately refill whichever worker happens to become free if doing so would let wall-clock completion order choose the model state used for the next proposal. Acceptable deterministic designs include fixed deterministic batches/barriers, or an ordered-prefix scheduler where proposal `k+1` is generated only after the exact predeclared prefix needed for that proposal has been assimilated and its dispatch slot/order is determined independently of worker-speed races. Surrogate/model updates, acquisition decisions, domain contraction, portfolio-selection state, proposal generation, dispatch/refill decisions, and stopping criteria must consume the same deterministic state sequence. A fixed random seed plus buffered assimilation alone is not sufficient if worker availability can still change which proposal is generated next. If a target chooses immediate completion-driven refill for throughput, declare the resulting nondeterminism as an explicit contract change rather than claiming deterministic replay. @@ -60,22 +62,24 @@ Parallelism has an information cost: very wide batches receive less feedback bet ## Validation -Keep a deterministic search seed where practical, preserve the full trial ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. For asynchronous search with **additive** budgets, test the boundary with multiple workers contending for the last remaining reservation and prove no dispatch can make `consumed + reserved` exceed B. Deliberately run trials that attempt to exceed their per-trial money/compute reservation and prove the quota mechanism prevents the overrun. Inject early failures, late failures, partial consumption, and cancellation after measurable work; verify that only demonstrably unconsumed reservation is released, evaluation-count slots are not resurrected after dispatch, and repeated failures cannot create extra budget capacity. +Keep a deterministic search seed where practical, preserve the full trial/search ledger, re-evaluate finalists, and validate the selected candidate against the reference contract on held-out/repeated workloads. For asynchronous search with **additive** budgets, test the boundary with multiple workers contending for the last remaining reservation and prove no dispatch can make `consumed + reserved` exceed B. Deliberately run trials that attempt to exceed their per-trial money/compute reservation and prove the quota mechanism prevents the overrun. Inject early failures, late failures, partial consumption, and cancellation after measurable work; verify that only demonstrably unconsumed reservation is released, evaluation-count slots are not resurrected after dispatch, and repeated failures cannot create extra budget capacity. + +For a **hard compute-unit B**, add coordinator-dominant fixtures rather than validating only trial quotas. Make trial evaluations cheap while surrogate fitting, acquisition optimization, proposal generation, model/portfolio updates, assimilation, serialization, and stopping logic deliberately consume most of the compute. Under the ledger design, prove those operations reserve/charge the same compute budget and cannot start when capacity is unavailable; under a whole-search-quota design, prove coordinator and workers are all contained by the same enforceable quota. Stress repeated model refits and failed proposal attempts and verify they cannot create uncharged compute. The run must remain within B even when coordinator work dominates total compute. For **elapsed wall-time** B, use a controlled monotonic clock and launch multiple workers concurrently under one shared deadline. Verify two trials each permitted to run until the same ten-minute deadline are admissible without requiring twenty minutes of additive reservation. Permute worker count, start order, and completion times; prove the search stops launching work as the deadline approaches, every in-flight worker observes the same deadline, forced termination completes within the declared enforcement bound, and total elapsed search lifetime never exceeds B plus only the explicitly declared bounded termination-cleanup allowance. Ensure no retry or model/proposal step can reset or extend the original deadline. -Race multiple additive-resource trial completions/cancellations against one another and against workers attempting the final dispatch slot. Verify the accounting transaction is linearizable: no consumed increment is lost, no reservation is released before its corresponding consumption is charged, and a dispatcher never observes capacity that would make the post-transaction invariant `consumed + reserved <= B` false. +Race multiple additive-resource trial/coordinator completions and cancellations against one another and against workers attempting the final dispatch slot. Verify the accounting transaction is linearizable: no consumed increment is lost, no reservation is released before its corresponding consumption is charged, and no participant observes capacity that would make the post-transaction invariant `consumed + reserved <= B` false. For deterministic targets, run the same seeded search with deliberately permuted worker speeds and completion orders, including the case where trial 2 finishes before trial 1 and frees a worker first. Verify out-of-order completion **does not permit proposal 3 to be generated from a different observation prefix**. The complete proposal sequence, parameter values, deterministic trial IDs, logical dispatch/refill order, surrogate/search states, selected candidate, and stopping reason must match the deterministic reference. Test both fixed-batch/barrier scheduling and any ordered-prefix scheduler the target claims to support. Where sequential/parallel equivalence is part of C, compare the asynchronous execution with its deterministic sequential or batch replay. If completion-driven refill is intentionally retained, verify the target explicitly labels the search trace nondeterministic instead of claiming replay equivalence. ## Target-repo adaptation -Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define each budget dimension as either **additive** or **elapsed wall time**. For additive resources, define the accounting unit, conservative per-trial reservation amount, enforcement mechanism, one atomic/serializable reservation/completion ledger, metering source, and failure/cancellation charging policy. For elapsed wall time, define the monotonic absolute search deadline, maximum bounded termination-cleanup interval, worker cancellation/termination mechanism, and the minimum remaining-time rule for new dispatch. Also define the deterministic observation-assimilation policy, **deterministic proposal frontier and dispatch/refill schedule** (when required), and the exact condition under which a freed worker may receive new work before enabling asynchronous dispatch. +Do not copy acquisition constants, trial counts, domain contraction rates or parallel widths. Treat them as optimizer parameters with their own evidence boundary. Define each budget dimension as **additive**, **elapsed wall time**, or another explicitly modeled resource. For additive resources, define the accounting unit, conservative per-operation/trial reservation amount, enforcement mechanism, one atomic/serializable reservation/completion ledger, metering source, and failure/cancellation charging policy. If compute is advertised as a hard whole-search B, explicitly inventory the coordinator/search-control paths that consume it and either place the complete optimizer (workers plus coordinator) under one enforceable quota or define how surrogate fits, acquisition/proposal work, model updates, coordination, assimilation, serialization, and stopping logic reserve/charge the shared ledger. For elapsed wall time, define the monotonic absolute search deadline, maximum bounded termination-cleanup interval, worker cancellation/termination mechanism, and the minimum remaining-time rule for new dispatch. Also define the deterministic observation-assimilation policy, **deterministic proposal frontier and dispatch/refill schedule** (when required), and the exact condition under which a freed worker may receive new work before enabling asynchronous dispatch. ## Failure modes -Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an additive evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced additive reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; **treating elapsed wall time as an additive per-worker resource can falsely reject valid overlapping trials and serialize the search**; conversely, a nominal wall-time limit without one enforceable shared deadline can let work continue past B; wall-clock completion-order assimilation can make supposedly deterministic search traces irreproducible; **immediate worker refill can also make proposals nondeterministic even when assimilation itself is buffered**. +Noisy objectives, nonstationary machines, weak surrogates, excessive dimensionality and too much concurrency can waste evaluations or overfit benchmark noise. Non-atomic reservation can oversubscribe an additive evaluation or monetary cap; non-atomic completion/release can transiently undercount consumed plus reserved or lose concurrent increments; an unenforced additive reservation can let a single trial exceed B before accounting observes it; refunding consumed resources can let repeated late failures exceed B; **charging only trials while leaving surrogate fitting, proposal/acquisition work, model updates, coordination, assimilation, or stopping logic outside a claimed hard compute budget can exceed B even when every trial quota is correct**; treating elapsed wall time as an additive per-worker resource can falsely reject valid overlapping trials and serialize the search; conversely, a nominal wall-time limit without one enforceable shared deadline can let work continue past B; wall-clock completion-order assimilation can make supposedly deterministic search traces irreproducible; **immediate worker refill can also make proposals nondeterministic even when assimilation itself is buffered**. ## Rollback trigger -Stop adaptive search when its overhead exceeds evaluation savings, the declared budget is exhausted, repeated validation does not confirm the selected improvement, any additive trial can consume beyond its enforceable reservation, additive accounting/concurrency tests can violate B, any hard elapsed-wall-time run can exceed its shared absolute deadline beyond the declared bounded cleanup allowance, a retry/worker can extend or reset that deadline, or any target that requires deterministic search produces different proposals, logical dispatch/refill order, model states, selected candidates, or stopping reasons under permuted asynchronous completion orders. +Stop adaptive search when its overhead exceeds evaluation savings, the declared budget is exhausted, repeated validation does not confirm the selected improvement, any additive covered operation can consume beyond its enforceable reservation/quota, additive accounting/concurrency tests can violate B, **coordinator/search-control work can escape a hard whole-search compute ledger/quota**, any hard elapsed-wall-time run can exceed its shared absolute deadline beyond the declared bounded cleanup allowance, a retry/worker can extend or reset that deadline, or any target that requires deterministic search produces different proposals, logical dispatch/refill order, model states, selected candidates, or stopping reasons under permuted asynchronous completion orders. From 15e703ecfd1f794eea1aa4f1edf5037164812570 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:34:19 +0930 Subject: [PATCH 044/229] Require proof for hard approximation envelopes --- ...PROX-001-contract-bounded-approximation.md | 32 +++++++++++-------- 1 file changed, 18 insertions(+), 14 deletions(-) diff --git a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md index cddcc1d..ad77266 100644 --- a/optimizations/OPT-APPROX-001-contract-bounded-approximation.md +++ b/optimizations/OPT-APPROX-001-contract-bounded-approximation.md @@ -15,25 +15,25 @@ Exact processing has unbounded or unacceptable cost even though the product/scie ## Optimization problem contract -- X: target-supported approximation policies, quality/resource ceilings, sampling/culling/LOD policies, update frequencies, state-reset rules, evaluation horizons, exact-mode fallback choices, and stochastic tuning/certification procedures -- F: policies whose declared error/degradation metric remains within the target's explicit envelope over the declared state/composition horizon and whose resource/semantic constraints are satisfied; when a stochastic policy is selected from multiple candidates, feasibility certification is based on independent held-out conformance data or a predeclared selection-aware simultaneous-confidence/multiple-testing procedure rather than naive reuse of the tuning samples +- X: target-supported approximation policies, quality/resource ceilings, sampling/culling/LOD policies, update frequencies, state-reset rules, evaluation horizons, exact-mode fallback choices, hard-envelope proof/enforcement methods, and stochastic tuning/certification procedures +- F: policies whose declared error/degradation metric remains within the target's explicit envelope over the declared state/composition horizon and whose resource/semantic constraints are satisfied; **a hard pointwise/worst-case envelope must be established by an analytic/formal bound, exhaustive checking over a finite declared domain, or runtime enforcement that proves the bound for each produced approximate result and falls back to the exact path whenever the proof/guard cannot establish it**; sample-based certification alone is admissible only for explicitly statistical contracts; when a stochastic policy is selected from multiple candidates, feasibility certification is based on independent held-out conformance data or a predeclared selection-aware simultaneous-confidence/multiple-testing procedure rather than naive reuse of the tuning samples - f: target-measured resource or latency cost, optionally paired with the declared quality/error metric - d: minimize resource/latency cost subject to feasibility in F, or use the target's predeclared multi-objective ordering when quality is ranked rather than hard-bounded -- C: approximation is permitted only by an explicit contract; exact callers are not silently weakened; the error norm, aggregation rule, sequence/composition horizon, reset boundaries, whether the envelope is hard worst-case or statistical/confidence/tail-based, and the stochastic tuning-versus-certification procedure are declared before evaluation; an exact reference path or exact fixture remains available where practical -- B: target-specific benchmark/quality-evaluation budget over predeclared ordinary, boundary, adversarial, repeated-application, and long-horizon fixtures; for stochastic policy search, tuning/selection evaluations and independent certification evaluations (or the budget used by the predeclared simultaneous-confidence procedure) are accounted separately -- S: stop when the evaluation budget is exhausted or a selected policy meets the target resource objective and passes the declared conformance certification while remaining inside the quality envelope over the entire declared horizon +- C: approximation is permitted only by an explicit contract; exact callers are not silently weakened; the error norm, aggregation rule, sequence/composition horizon, reset boundaries, whether the envelope is hard worst-case or statistical/confidence/tail-based, **the proof/enforcement basis for any hard envelope**, and the stochastic tuning-versus-certification procedure are declared before evaluation; an exact reference path or exact fixture remains available where practical, and a runtime-enforced hard envelope must route any unproved/unsafe case to that exact path before an out-of-envelope approximate result becomes observable +- B: target-specific benchmark/quality-evaluation budget over predeclared ordinary, boundary, adversarial, repeated-application, and long-horizon fixtures; proof construction/exhaustive finite-domain checking/runtime-guard validation for hard envelopes is budgeted separately from statistical sampling where applicable; for stochastic policy search, tuning/selection evaluations and independent certification evaluations (or the budget used by the predeclared simultaneous-confidence procedure) are accounted separately +- S: stop when the evaluation/proof budget is exhausted or a selected policy meets the target resource objective and passes the declared conformance certification/proof while remaining inside the quality envelope over the entire declared horizon - Variables: continuous / integer / categorical / conditional / mixed, depending on approximation policy - Search scope: local or global, explicitly declared for the target - Objective behavior: deterministic, noisy, or stochastic depending on the quality/resource metric - Information: derivative-free / black-box by default -- Evaluation cost: moderate to expensive when exact references, held-out certification, or long-horizon trajectories are required -- Constraints: explicit error envelope, semantic/API, resource, horizon/reset, selection-aware statistical certification, and exact-fallback constraints +- Evaluation cost: moderate to expensive when exact references, held-out certification, hard-envelope proof, exhaustive checking, runtime guards, or long-horizon trajectories are required +- Constraints: explicit error envelope, semantic/API, resource, horizon/reset, hard-envelope proof/enforcement, selection-aware statistical certification, and exact-fallback constraints - Parallelism: sequential, synchronous batch, or asynchronous according to target evaluation; stateful validation must preserve trajectory semantics -- Exactness: approximation explicitly permitted only inside the declared measurable envelope +- Exactness: approximation explicitly permitted only inside the declared measurable/provable envelope ## Preserved contract -Approximation is admissible only when the contract explicitly permits it. A previously exact API cannot be silently weakened and still be called correctness-preserving. For stateful or repeatedly composed approximations, the contract applies over an explicitly declared horizon—not merely to each isolated step—so bounded per-step error is insufficient if drift can accumulate beyond the allowed envelope. The contract must also state whether compliance is pointwise/worst-case or statistical; a stochastic envelope is judged by its declared aggregation, confidence, exceedance-probability, quantile, or tail criterion rather than by silently substituting a hard per-sample limit. When multiple stochastic policies are tuned or screened, choosing the apparent winner changes the sampling distribution: the data used to optimize/select a policy cannot be treated as independent nominal-confidence certification evidence unless the declared procedure explicitly accounts for that selection. +Approximation is admissible only when the contract explicitly permits it. A previously exact API cannot be silently weakened and still be called correctness-preserving. For stateful or repeatedly composed approximations, the contract applies over an explicitly declared horizon—not merely to each isolated step—so bounded per-step error is insufficient if drift can accumulate beyond the allowed envelope. The contract must also state whether compliance is pointwise/worst-case or statistical; a stochastic envelope is judged by its declared aggregation, confidence, exceedance-probability, quantile, or tail criterion rather than by silently substituting a hard per-sample limit. **A hard pointwise/worst-case guarantee is a universal claim over its declared domain: ordinary, boundary, adversarial, or randomized fixtures are evidence but cannot by themselves prove that universal claim over a non-finite domain.** Such a hard envelope requires a sound analytic/formal bound, exhaustive coverage of a finite declared domain, or runtime enforcement that proves/guards each approximate output and invokes the exact reference path whenever the guard cannot certify the bound. When multiple stochastic policies are tuned or screened, choosing the apparent winner changes the sampling distribution: the data used to optimize/select a policy cannot be treated as independent nominal-confidence certification evidence unless the declared procedure explicitly accounts for that selection. ## Optimization @@ -41,6 +41,8 @@ Introduce a resource ceiling and degrade only along a declared dimension: sample For stateful streaming, simulation, DSP, iterative numerical work, or any repeatedly applied approximation, define the error model before benchmarking: the norm/metric (for example absolute, relative, L2, perceptual, state-distance, or domain-specific), how error composes or is aggregated through time, the maximum sequence length or physical/time horizon over which the envelope must hold, and any reset/checkpoint/re-synchronization boundaries that legitimately restart the horizon. If the system can run longer than the validated horizon without reset, either extend validation to that operational horizon or define a separate long-run drift bound; do not infer long-run safety from one-step ε alone. +For a **hard pointwise or worst-case envelope**, choose the proof/enforcement mode before deployment. If the input/state domain is finite and tractable, exhaustively evaluate every declared case (including every relevant trajectory/state when the contract is stateful). Otherwise establish a sound analytic or formal bound that covers the complete declared domain/horizon, **or** install a runtime guard that derives a sound per-instance error bound/invariant before publishing the approximate result and falls back to the exact path whenever that guard cannot prove compliance. A test suite—no matter how adversarial or large—does not convert a non-finite-domain hard guarantee into a proof. Sample-based testing may still regression-test the proof/guard implementation, but it is not the certification basis for the universal claim. + For stochastic/noisy approximations, also define the statistical compliance rule before evaluation: the sampling unit and workload distribution, aggregation statistic, confidence level or interval procedure, tolerated exceedance probability, quantile/tail bound, and the sample/evaluation budget used to decide compliance. Do not reinterpret a statistical guarantee as a pointwise worst-case guarantee, and do not weaken a declared hard worst-case envelope into an average-case claim after observing data. If more than one stochastic approximation policy is tuned, compared, adaptively searched, thresholded, or screened using sampled error data, **separate selection from certification**. The default pattern is to use one predeclared tuning/selection set (or stream) to choose the candidate and then evaluate that frozen candidate on an independent held-out conformance set drawn from the declared operational distribution. If independent holdout is impractical, use a predeclared selection-aware method that preserves the advertised guarantee across the entire candidate-selection procedure—for example simultaneous confidence bounds, family-wise/multiple-testing correction, valid selective-inference/e-process machinery, or another target-justified method. A nominal per-policy confidence interval computed on the same samples used to select the best-looking policy is not certification. Record exactly which evaluations influenced policy selection and which evaluations supported the final compliance claim. @@ -55,22 +57,24 @@ If more than one stochastic approximation policy is tuned, compared, adaptively ## Validation -Measure error/degradation and resource savings together across ordinary, boundary and adversarial workloads. Keep an exact reference for differential evaluation where practical. Declare and test the error norm/metric, aggregation rule, sequence/composition horizon, reset boundaries, hard-versus-statistical envelope semantics, and tuning-versus-certification procedure explicitly. +Measure error/degradation and resource savings together across ordinary, boundary and adversarial workloads. Keep an exact reference for differential evaluation where practical. Declare and test the error norm/metric, aggregation rule, sequence/composition horizon, reset boundaries, hard-versus-statistical envelope semantics, **hard-envelope proof/enforcement mode**, and tuning-versus-certification procedure explicitly. For stateful/repeated use, run differential trajectories against the exact path across short, nominal, maximum-supported, and adversarially long sequences. Include biased-error fixtures where each individual step remains within the local ε but errors accumulate in the same direction; verify the cumulative/state error still respects the declared horizon envelope. Test reset/checkpoint boundaries before, at, and after the limit; verify resets actually restore the assumptions used by the next horizon. -Where stochastic approximation is used, evaluate the declared expected, quantile, exceedance-probability, confidence, tail, or worst-case criterion against the exact path as specified by C. Include fixtures where individual samples exceed a nominal pointwise value while the declared statistical envelope remains satisfied, and fixtures where the configured tail/confidence/exceedance criterion truly fails. Verify rollback decisions distinguish those cases rather than triggering on one sample unless the contract explicitly declares a hard single-sample/worst-case bound. +For a **hard pointwise/worst-case envelope**, validation must verify the claimed proof mechanism rather than merely accumulate examples. For a finite declared domain, exhaustively enumerate every input/state/trajectory covered by C and compare with the exact path. For a non-finite domain, review/check the analytic or formal bound against the implementation and all assumptions it depends on, or validate a runtime guard whose sound per-instance bound/invariant is checked before output publication. Inject cases where the runtime guard cannot establish safety and prove the approximate result is suppressed and the exact fallback is used. Random, adversarial, property-based, and boundary samples remain valuable regression tests, but passing them is not sufficient evidence for a universal hard guarantee. + +Where stochastic approximation is used, evaluate the declared expected, quantile, exceedance-probability, confidence, or tail criterion against the exact path as specified by C. Include fixtures where individual samples exceed a nominal pointwise value while the declared statistical envelope remains satisfied, and fixtures where the configured tail/confidence/exceedance criterion truly fails. Verify rollback decisions distinguish those cases rather than triggering on one sample unless the contract explicitly declares a hard single-sample/worst-case bound. **Sample-based certification is reserved for these explicitly statistical contracts; it must not be cited as proof of a non-finite-domain hard worst-case envelope.** Add **selection-bias fixtures** whenever multiple stochastic policies are considered. Generate several candidate policies whose apparent sampled errors vary by chance, select the best-looking candidate using the declared tuning procedure, and prove that the final compliance decision uses either fresh held-out samples unavailable to selection or the declared simultaneous/selection-aware inference procedure. Verify the tuning samples alone cannot certify the selected winner at nominal per-policy confidence. Include repeated/adaptive candidate selection, early stopping, and candidate-count changes; confirm the advertised confidence/tail/exceedance guarantee remains valid under the complete selection procedure. Persist an audit trail labeling each evaluation as tuning/selection, certification, or both only when the declared selection-aware method formally permits dual use. ## Target-repo adaptation -Define `ε`, the exact quality/error norm, aggregation rule, workload distribution, maximum state/composition horizon, reset/checkpoint semantics, long-run drift policy, escape hatch and exact-mode availability locally. Explicitly classify the quality envelope as hard pointwise/worst-case or statistical, and for statistical contracts specify the confidence/tail/exceedance rule and decision sample budget. If multiple policies are searched or compared, predeclare the tuning/selection dataset or stream, the independent certification dataset/budget, **or** the exact simultaneous-confidence/multiple-testing/selective-inference method that makes data reuse valid; record which observations affected selection versus certification. If the target has no finite operational horizon, establish a justified asymptotic/stability bound or periodic re-synchronization rule instead of copying a finite benchmark horizon from another system. +Define `ε`, the exact quality/error norm, aggregation rule, workload distribution, maximum state/composition horizon, reset/checkpoint semantics, long-run drift policy, escape hatch and exact-mode availability locally. Explicitly classify the quality envelope as **hard pointwise/worst-case** or **statistical**. For a hard envelope, declare one sound certification mechanism: analytic/formal proof over the complete domain/horizon, exhaustive checking over an explicitly finite domain, or runtime per-instance enforcement with exact fallback whenever the guard cannot prove the bound. Do not use sampled conformance data as the sole certification basis for a universal hard claim. For statistical contracts specify the confidence/tail/exceedance rule and decision sample budget. If multiple policies are searched or compared, predeclare the tuning/selection dataset or stream, the independent certification dataset/budget, **or** the exact simultaneous-confidence/multiple-testing/selective-inference method that makes data reuse valid; record which observations affected selection versus certification. If the target has no finite operational horizon, establish a justified asymptotic/stability bound or periodic re-synchronization rule instead of copying a finite benchmark horizon from another system. ## Failure modes -Unmeasured quality loss, biased sampling, hidden rare-case failures, cumulative drift that is invisible to one-step checks, reset boundaries that fail to restore reference assumptions, state-dependent amplification, unstable feedback loops, selecting the best-looking stochastic policy and then certifying it on the same data with naive per-policy confidence, undisclosed adaptive candidate search/early stopping that invalidates nominal error guarantees, misclassifying a statistical envelope as a hard pointwise bound (or vice versa), and callers incorrectly assuming exact semantics. +Unmeasured quality loss, biased sampling, hidden rare-case failures, **claiming a universal hard worst-case envelope from finite sampled/adversarial fixtures over a non-finite domain**, unsound analytic/formal assumptions, incomplete finite-domain enumeration, a runtime guard that can publish before proving the per-instance bound or fails to fall back exactly, cumulative drift that is invisible to one-step checks, reset boundaries that fail to restore reference assumptions, state-dependent amplification, unstable feedback loops, selecting the best-looking stochastic policy and then certifying it on the same data with naive per-policy confidence, undisclosed adaptive candidate search/early stopping that invalidates nominal error guarantees, misclassifying a statistical envelope as a hard pointwise bound (or vice versa), and callers incorrectly assuming exact semantics. ## Rollback trigger -Evaluate rollback against the **declared envelope semantics and certification procedure**. For a hard pointwise/worst-case contract, disable immediately when any supported-horizon observation exceeds the declared bound. For a stochastic/statistical contract, disable when the predeclared aggregation, confidence, exceedance-probability, quantile, or tail criterion fails under its stated **selection-aware certification** procedure; an isolated sample beyond a nominal pointwise value is not by itself a contract violation unless the contract says it is. Treat a selected policy as uncertified—and disable or fall back—if held-out certification fails, if tuning and certification evidence are mixed contrary to the declared procedure, or if the simultaneous/multiple-testing/selective-inference assumptions required for data reuse are violated. In all cases, disable when cumulative/state drift violates its declared bound, reset/checkpoint validation fails, reference comparisons violate C, a catastrophic semantic/safety constraint is breached, or resource savings are not material. +Evaluate rollback against the **declared envelope semantics and certification procedure**. For a hard pointwise/worst-case contract, disable/fall back immediately on any proven bound violation, **any failure of the analytic/formal/exhaustive proof obligations, or any runtime case in which the guard cannot prove the bound before publication**; a runtime-enforced design must use the exact path for such unproved cases rather than emit the approximation. A sampled counterexample to a hard envelope is an immediate violation, but absence of sampled counterexamples is never sufficient certification of a non-finite-domain hard guarantee. For a stochastic/statistical contract, disable when the predeclared aggregation, confidence, exceedance-probability, quantile, or tail criterion fails under its stated **selection-aware certification** procedure; an isolated sample beyond a nominal pointwise value is not by itself a contract violation unless the contract says it is. Treat a selected policy as uncertified—and disable or fall back—if held-out certification fails, if tuning and certification evidence are mixed contrary to the declared procedure, or if the simultaneous/multiple-testing/selective-inference assumptions required for data reuse are violated. In all cases, disable when cumulative/state drift violates its declared bound, reset/checkpoint validation fails, reference comparisons violate C, a catastrophic semantic/safety constraint is breached, or resource savings are not material. From 73f1a04fec62fa96628a247847558047311fae52 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:57:20 +0930 Subject: [PATCH 045/229] Reject markup-only mandatory sections --- scripts/check_catalog.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index e5e8365..d73e2f5 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -373,9 +373,10 @@ def parse_record_link_cell(cell: str, context: str) -> tuple[str, str]: def rendered_inline_text(value: str) -> str: """Approximate visible inline text for required field-value validation. - Contract fields must contain textual substance after non-rendering Markdown - constructs are removed. Link/image destinations, formatting markers, and HTML - tags therefore cannot make an otherwise empty field count as populated. + Contract fields and mandatory section bodies must contain textual substance + after non-rendering Markdown constructs are removed. Link/image destinations, + formatting markers, and HTML tags therefore cannot make empty source count as + populated rendered content. """ text = value text = INLINE_IMAGE_RE.sub(lambda m: m.group(1), text) @@ -411,6 +412,8 @@ def section_has_content(lines: list[str]) -> bool: continue if EMPTY_LABEL_RE.match(line) or is_structural_only_line(line): continue + if not has_substantive_rendered_text(line): + continue return True return False @@ -501,7 +504,7 @@ def require_prefixed_fields( for heading in sorted(REQUIRED_V2): if not section_has_content(section_lines(text, heading)): die( - f"{path.relative_to(ROOT)} has empty/template/structural-only mandatory section {heading}" + f"{path.relative_to(ROOT)} has empty/template/structural/markup-only mandatory section {heading}" ) contract = section_lines(text, "## Optimization problem contract") From a657811ae9960d3db51606399a928264b60b8ac7 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:58:13 +0930 Subject: [PATCH 046/229] Linearize reuse hits with mutable input identity --- ...1-signature-bound-incremental-execution.md | 26 +++++++++++-------- 1 file changed, 15 insertions(+), 11 deletions(-) diff --git a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md index d583812..2d0b1b9 100644 --- a/optimizations/OPT-INC-001-signature-bound-incremental-execution.md +++ b/optimizations/OPT-INC-001-signature-bound-incremental-execution.md @@ -14,30 +14,32 @@ Expensive work is rerun even though every input capable of affecting its result ## Optimization problem contract -- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity/consumption policies, immutable-output handles, immutable-input snapshot/mutation-control policies, monotonic mutation epochs, compare-and-publish activation policies, and crash-consistent state-publication mechanisms -- F: configurations whose signature covers every output-affecting input, whose execution consumes one immutable effective-input snapshot or is protected by a mutation lock/monotonic mutation witness that detects every intervening change, whose reuse validates required outputs and binds downstream consumption to the exact validated output versions, whose authoritative generation switch is linearized with the witnessed input state, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted/raced executions never publish reusable partial or stale-current state -- f: measured repeated-work cost including stage runtime plus signature/snapshot/mutation-tracking/metadata/output-validation/output-pinning/compare-and-publish/publication I/O overhead +- X: target-supported signature definitions, persistence scopes, invalidation granularities, output-validity/consumption policies, immutable-output handles, immutable-input snapshot/mutation-control policies, monotonic mutation epochs, **reuse-hit input-binding policies**, compare-and-publish activation policies, and crash-consistent state-publication mechanisms +- F: configurations whose signature covers every output-affecting input, whose execution consumes one immutable effective-input snapshot or is protected by a mutation lock/monotonic mutation witness that detects every intervening change, whose **reuse-hit decision is bound to one coherent effective-input identity and linearized against that same mutation witness**, whose reuse validates required outputs and binds downstream consumption to the exact validated output versions, whose authoritative generation switch is linearized with the witnessed input state, whose persisted signature/output metadata form one committed generation, and whose failed/interrupted/raced executions never publish reusable partial or stale-current state +- f: measured repeated-work cost including stage runtime plus signature/snapshot/mutation-tracking/metadata/output-validation/output-pinning/**reuse-hit binding**/compare-and-publish/publication I/O overhead - d: minimize -- C: every reused output consumed downstream is the same immutable/versioned output instance whose validity predicate passed, and is semantically equivalent to a fresh execution for the same effective inputs with the same failure semantics; reuse metadata cannot mix fields from different generations; a committed generation cannot bind a pre-execution signature to output produced from changed or mixed inputs; A→B→A mutations during execution are detected rather than erased by endpoint equality; and no input mutation may linearize between the final accepted input witness and activation of that generation as authoritative for those inputs +- C: every reused output consumed downstream is the same immutable/versioned output instance whose validity predicate passed, and is semantically equivalent to a fresh execution for the **coherent effective-input identity bound to that invocation** with the same failure semantics; a reuse hit cannot be accepted by comparing against identity A and then consume A after a concurrent mutation has already made B the invocation's authoritative live identity unless the invocation was explicitly and immutably snapshot-bound to A; reuse metadata cannot mix fields from different generations; a committed generation cannot bind a pre-execution signature to output produced from changed or mixed inputs; A→B→A mutations during execution or reuse-hit qualification are detected rather than erased by endpoint equality; and no input mutation may linearize between the final accepted input witness and activation of a generation or acceptance of a reuse hit for those inputs - B: target-specific benchmark/evaluation budget declared before tuning; no portable value is supplied by this record - S: stop when the declared budget is exhausted or a validated configuration meets the predeclared improvement threshold without violating C -- Variables: categorical / mixed policy choices for signatures, snapshots, validation, output pinning, granularity, mutation control, compare-and-publish activation, and publication +- Variables: categorical / mixed policy choices for signatures, snapshots, validation, output pinning, granularity, mutation control, reuse-hit binding, compare-and-publish activation, and publication - Search scope: local to one incremental stage or pipeline boundary - Objective behavior: noisy for performance; correctness identity/mutation checks are deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive depending on stage runtime and validation cost -- Constraints: semantic equivalence, input/output integrity, crash consistency, snapshot/mutation consistency, validated-output consumption, publication linearizability, and resource constraints -- Parallelism: sequential or pipeline-specific; mutation tracking, output pinning, activation, and publication must remain race-safe under concurrent producers/consumers +- Constraints: semantic equivalence, input/output integrity, crash consistency, snapshot/mutation consistency, **reuse-hit input identity**, validated-output consumption, publication linearizability, and resource constraints +- Parallelism: sequential or pipeline-specific; mutation tracking, reuse-hit binding, output pinning, activation, and publication must remain race-safe under concurrent producers/consumers - Exactness: exact reuse semantics; no approximation is introduced ## Preserved contract -Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and mutable inputs must not change underneath execution without invalidating the candidate generation. Endpoint equality is not enough: if an input changes and later returns to its original bytes, the intervening mutation must still be observable to the publication decision. Likewise, validating a mutable output path is not enough unless downstream consumption is pinned to that exact validated version. Finally, validating the input witness and later switching the authoritative generation are not two independent steps: generation activation must linearize against the same input state that was validated. +Reused output must be semantically equivalent to a fresh execution for the same effective inputs. Failed executions must not bless a new signature, an unchanged input signature alone is insufficient when an existing output can be corrupted or overwritten externally, interrupted publication must not expose a signature paired with output identities from another generation, and mutable inputs must not change underneath execution without invalidating the candidate generation. Endpoint equality is not enough: if an input changes and later returns to its original bytes, the intervening mutation must still be observable to the publication decision. Likewise, validating a mutable output path is not enough unless downstream consumption is pinned to that exact validated version. Validating an input signature for a **reuse hit** is also not enough unless the invocation is bound to that same coherent input identity before a concurrent mutation can change which generation is authoritative for the call. Finally, validating the input witness and later switching the authoritative generation are not two independent steps: generation activation must linearize against the same input state that was validated. ## Optimization Compute a deterministic signature over the effective inputs and compare it with successfully persisted prior state. Reuse is allowed only when that signature still matches **and** every required output satisfies a declared validity predicate. Depending on the target, that predicate may be a content digest/version manifest, a trusted immutable/protected artifact identity, or another reproducible integrity check strong enough to detect external mutation. Mere file presence is not sufficient unless the target explicitly guarantees that reused outputs are immutable and protected from modification. Execute when the input signature differs, any required output is missing, or any output-validity check fails. +A **reuse hit must itself be identity-bound and linearizable**. Before treating a matching signature as permission to skip execution, the invocation must acquire one coherent effective-input identity and bind the hit to it. Preferred designs capture an immutable source snapshot/version for the invocation. Lock-based designs may instead hold the mutation/read lock while verifying that the current committed generation matches the captured identity and while binding the invocation to that generation. Epoch/version designs must use a transaction, CAS, compare-and-bind operation, or equivalent serialization boundary that atomically verifies the complete current mutation-epoch vector, verifies that the reusable generation was committed for that exact vector/signature, and records/returns the invocation's binding to that generation and its pinned output handles. A plain sequence of “signature A matches; later decide to skip/read A” is insufficient. If an input mutation wins before the hit is bound, the compare/bind must fail and the invocation must re-evaluate against the new identity. If the target's semantics are **snapshot-at-invocation**, later mutations may proceed only after the immutable A snapshot/binding is established; if the semantics require **current live inputs through consumption**, retain the appropriate lock or equivalent freshness protection through the required consumption boundary. + Treat output validation and output consumption as one identity-bound operation. A successful validity check must yield or pin the exact immutable/versioned output handle that downstream consumers will read: for example a content-addressed object, immutable artifact/version ID, snapshot handle, open file descriptor tied to a protected inode/version where the platform guarantees the needed semantics, or another target-specific stable handle. Do **not** validate bytes at a mutable pathname/object name and then later reopen that name for consumption, because another writer may replace it between validation and read. If the storage system cannot provide an immutable/versioned handle, hold an appropriate lock from validation through the downstream read/consumption, or copy the validated bytes into an immutable snapshot and consume that snapshot. Every consumer on a reuse hit must be bound to the validated handle/version, not merely to the same logical path. Bind execution to one coherent effective-input identity. The preferred design is an **immutable snapshot/version** of every mutable effective input. If a snapshot is unavailable, use a mechanism that records *intervening mutation*, not merely endpoint content equality: for example, hold a read/mutation lock for the full execution-through-activation interval, or capture a monotonically increasing version/epoch for every mutable input. Every mutation must advance its epoch durably/atomically with the mutation, including a change that later restores the original bytes. For multiple inputs, capture the snapshot/epoch vector coherently under the target's transaction/locking rules so a mixed vector cannot be mistaken for one state. A content signature recomputed at publication may supplement this check, but **must not be the sole fallback** because A→B→A can make endpoint signatures equal. Any lock violation, epoch change, incoherent snapshot, or untrackable mutable input discards the candidate generation and requires retry from a fresh identity. @@ -64,6 +66,8 @@ Wonderbuild demonstrates the mechanism and benchmark shapes, but its historical Test unchanged inputs with valid outputs, changed inputs, missing outputs, failed runs, corrupted persistent state, externally overwritten/corrupted outputs, and stale output-version metadata against a forced-fresh reference path. A mutated output must force reconstruction unless the target's immutable/protected-output contract proves such mutation impossible. +Add a **reuse-hit input-identity race**. Arrange a committed reusable generation for input identity A. Pause after the implementation has observed what would otherwise be a signature/epoch match for A but before it has irrevocably bound the invocation to the reusable generation and pinned output handles. Mutate the effective input to B in that exact interval. For snapshot-at-invocation designs, prove the call was already bound to an immutable A snapshot before the mutation and therefore legitimately consumes A. For lock-based live-input designs, prove the mutation cannot interleave until the required reuse/consumption boundary completes. For epoch/CAS designs, prove either the reuse bind or mutation wins one serialization order; if mutation wins, compare-and-bind must fail and the call must re-evaluate/rebuild for B rather than consume A as a hit. Repeat A→B→A and require the monotonic witness to reject a stale endpoint-equal hit. Include multi-input vectors and concurrent generation publishers. + Exercise an **output validation-to-consumption race**. Arrange a reuse hit for output A, validate A successfully, then have another writer replace the mutable path/object with B before the consumer reads. Prove the consumer still reads the pinned immutable/versioned A that was validated, or prove the lock prevents replacement until consumption completes. Repeat with delete/recreate, atomic rename, symlink/object-pointer replacement, version rollback, and multiple required outputs where one is swapped after validation. A test that merely corrupts output before validation is insufficient; the mutation must occur after the validity predicate succeeds and before/downstream consumption. Exercise **concurrent input mutation**, including explicit A→B→A races. Start execution from identity A, mutate one or more effective inputs during execution to B (including mixed-state multi-file/config changes), then restore the original bytes before publication. For snapshot-based targets, prove execution reads only the immutable A snapshot. For lock-based targets, prove the mutation cannot interleave with the protected execution/activation interval. For epoch/version-based targets, prove every mutation increments the monotonic witness and that the final epoch vector differs even when the final content signature returns to A. Reject/discard the candidate on any mutation witness change. Compare every accepted generation with a forced-fresh execution over the exact committed input identity. @@ -74,12 +78,12 @@ Exercise interruption/crash injection at every publication boundary: before outp ## Target-repo adaptation -Re-profile signature and output-validation cost, immutable-output handle/pinning cost, immutable-input snapshot or mutation-lock/epoch cost, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Define how a successful output-validity check returns/pins the exact version consumed downstream; if mutable storage is unavoidable, define the lock scope or immutable-copy boundary. Explicitly choose whether mutable inputs are consumed from immutable snapshots, protected by locks, or guarded by monotonic mutation epochs; define how every mutation advances the witness and how a coherent multi-input witness is captured. Define the **linearizable activation primitive**: either the mutation lock remains held through the authoritative generation switch or the system atomically compare-and-publishes against the complete witnessed epoch/version vector. Do not advertise commit-time content rehashing, an epoch read followed by a later manifest swap, or any other check-then-publish sequence as sufficient mutation protection. Also define the crash-consistency guarantee for committing the signature plus output identities. +Re-profile signature and output-validation cost, immutable-output handle/pinning cost, immutable-input snapshot or mutation-lock/epoch cost, **reuse-hit input-binding cost**, hash/version choice, metadata granularity, persistence format, and generation-publication mechanism. Include environment/toolchain inputs when they affect output. Define how every invocation captures one coherent input identity and how a reuse decision is atomically bound to the committed generation for that identity before mutable live inputs can invalidate the hit; explicitly choose snapshot-at-invocation semantics or live-through-consumption semantics and implement the corresponding snapshot/lock/compare-and-bind boundary. Define how a successful output-validity check returns/pins the exact version consumed downstream; if mutable storage is unavoidable, define the lock scope or immutable-copy boundary. Explicitly choose whether mutable inputs are consumed from immutable snapshots, protected by locks, or guarded by monotonic mutation epochs; define how every mutation advances the witness and how a coherent multi-input witness is captured. Define the **linearizable activation primitive**: either the mutation lock remains held through the authoritative generation switch or the system atomically compare-and-publishes against the complete witnessed epoch/version vector. Do not advertise commit-time content rehashing, an epoch read followed by a later manifest swap, a signature match followed by an unprotected reuse decision, or any other check-then-act sequence as sufficient mutation protection. Also define the crash-consistency guarantee for committing the signature plus output identities. ## Failure modes -Incomplete signatures create stale reuse; input mutation during execution can bind an old signature to new/mixed output; A→B→A races can defeat endpoint signature comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent per-input epoch reads can represent no real source state; a mutation that wins after the final witness read but before an unprotected manifest switch can make stale state authoritative; validating a mutable output name and reopening it later can consume different unvalidated bytes; output-version handles that are not actually immutable/pinned can create time-of-check/time-of-use reuse bugs; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. +Incomplete signatures create stale reuse; a signature/epoch match followed by an unprotected reuse decision can consume generation A after mutable live inputs have already advanced to B; input mutation during execution can bind an old signature to new/mixed output; A→B→A races can defeat endpoint signature comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent per-input epoch reads can represent no real source state; a mutation that wins after the final witness read but before an unprotected manifest switch can make stale state authoritative; validating a mutable output name and reopening it later can consume different unvalidated bytes; output-version handles that are not actually immutable/pinned can create time-of-check/time-of-use reuse bugs; existence-only output checks can return corrupted artifacts; weak output-validity predicates can miss external mutation; independently persisted signature/output metadata can create cross-generation false hits after interruption; overly broad signatures erase the benefit; persistence corruption can create false hits; timestamp-only schemes may be unsuitable where timestamp semantics are weak. ## Rollback trigger -Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference; if a consumer can read bytes/objects different from the exact output version whose validity predicate passed; if an A→B→A or other mutable-input race can publish a generation without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input identity; if an input mutation can linearize between the accepted final witness and authoritative generation activation; if external output mutation can bypass the declared validity/consumption binding; if crash/interruption testing can expose mixed-generation or stale-current state; or if signature/snapshot/mutation-tracking/output-pinning/integrity/activation/publication maintenance costs more than the avoided work. +Disable reuse immediately if any signature/output-validity hit diverges from the forced-fresh reference; if a reuse hit can be accepted for identity A and then consume A after a concurrent mutation has made B authoritative without an explicit immutable snapshot-at-invocation binding; if a consumer can read bytes/objects different from the exact output version whose validity predicate passed; if an A→B→A or other mutable-input race can publish a generation or qualify a reuse hit without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input identity; if an input mutation can linearize between an accepted reuse witness and reuse binding, or between the accepted final build witness and authoritative generation activation; if external output mutation can bypass the declared validity/consumption binding; if crash/interruption testing can expose mixed-generation or stale-current state; or if signature/snapshot/mutation-tracking/reuse-binding/output-pinning/integrity/activation/publication maintenance costs more than the avoided work. From cfb053db35fc7633f792ea25c6571d77367eb0a8 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 22:59:07 +0930 Subject: [PATCH 047/229] Require deployed-domain proof for exact pruning --- ...E-001-bound-driven-search-space-pruning.md | 22 ++++++++++--------- 1 file changed, 12 insertions(+), 10 deletions(-) diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index 4ffa34f..83031a2 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -19,7 +19,7 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F - f: a scalar real-valued target objective `f : F → R` evaluated on feasible candidates only - d: exactly one of scalar `minimize` or scalar `maximize`; vector, Pareto, lexicographic, or other partial-order objectives are outside this record unless a separately specified and validated frontier-bound mechanism is introduced -- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, the target's observable tie semantics are preserved, parallel dispatch cannot oversubscribe the declared hard budget, **every unit of resource consumption covered by a hard wall-time/compute B is accounted for or enclosed by an enforceable whole-search cap**, every hard peak-memory B is enforced over live allocated/reserved memory rather than cumulative historical allocation, and frontier exhaustion is declared only after all queued **and leased/in-flight** regions are accounted for +- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, **exact pruning is authorized only when the soundness argument for `b` covers the deployed search domain through an analytic/formal proof, exhaustive verification of the complete finite deployed domain, or a conservative runtime proof/certificate checked for each pruned region**, the target's observable tie semantics are preserved, parallel dispatch cannot oversubscribe the declared hard budget, every unit of resource consumption covered by a hard wall-time/compute B is accounted for or enclosed by an enforceable whole-search cap, every hard peak-memory B is enforced over live allocated/reserved memory rather than cumulative historical allocation, and frontier exhaustion is declared only after all queued **and leased/in-flight** regions are accounted for - B: a finite, predeclared target-specific **enforceable** cap with its accounting semantics declared explicitly. Evaluation count, money/provider spend, CPU/GPU-seconds, energy, bytes transferred, or other cumulative-flow resources use cumulative accounting. Elapsed wall time uses one shared whole-search deadline. **Peak memory is a stock constraint, not a cumulative flow:** enforce `live_allocated + live_reserved + proposed <= B`, release live capacity when memory is freed, and retain only a recorded `peak_observed` for evidence. If a target instead wants cumulative allocation traffic, it must declare that as a distinct cumulative metric rather than calling it peak memory. Any resource dimension that cannot be hard-capped under its declared semantics must be labeled observational/best-effort rather than advertised as hard B - S: stop immediately when the required optimality/tie contract is proven, or when the **global frontier is exhausted**, meaning there are no queued regions, no leased/in-flight regions still capable of producing candidates/children, and no unpublished child/frontier updates owned by active work. Otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality - Variables: integer / categorical / discrete / mixed @@ -27,15 +27,17 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - Objective behavior: deterministic unless uncertainty/noise is incorporated into a separately sound bound model - Information: derivative-free; bound/relaxation information is target-specific - Evaluation cost: moderate to expensive when exhaustive evaluation is infeasible -- Constraints: feasibility, semantic correctness, scalar-bound soundness, tie semantics, global-frontier accounting, and enforceable finite-resource constraints including correctly typed cumulative, deadline, and peak/live-capacity budgets +- Constraints: feasibility, semantic correctness, **deployed-domain scalar-bound soundness**, tie semantics, global-frontier accounting, and enforceable finite-resource constraints including correctly typed cumulative, deadline, and peak/live-capacity budgets - Parallelism: sequential, or parallel only with synchronized incumbent/frontier/bound state, **leased/in-flight region accounting**, linearizable reservation/completion accounting for cumulative resources, a shared absolute deadline for elapsed wall time, and live-allocation reservation accounting for hard peak-memory caps -- Exactness: exact only when the declared optimality and observable-tie contract is proven within B, including proof that no queued or leased region can still affect the answer; otherwise anytime/incomplete result semantics apply +- Exactness: exact only when the declared optimality and observable-tie contract is proven within B **and the pruning bound's soundness is justified over the complete deployed domain or certified conservatively at runtime for every pruned region**, including proof that no queued or leased region can still affect the answer; otherwise anytime/incomplete result semantics apply For each unexplored region `R`, define a bound `b(R)` separately from `f`: - minimizing: `b(R) ≤ inf { f(x) | x ∈ F ∩ R }`; - maximizing: `b(R) ≥ sup { f(x) | x ∈ F ∩ R }`. +**The inequality above is a proof obligation, not merely a test expectation.** Before using `b(R)` to make an exact pruning decision in a deployed search, establish one target-specific soundness basis that covers the actual deployed domain: (1) an analytic or formal derivation proving the bound relation for every admissible region/candidate under the target assumptions; (2) exhaustive verification over the complete declared finite deployed domain and every region shape the implementation may prune; or (3) a conservative runtime proof/certificate/guard whose premises are checked before each prune and which falls back to retaining/evaluating the region whenever the certificate cannot be established. Small exhaustive fixtures, randomized tests, and adversarial examples remain required regression evidence, but **they are not by themselves certification of global bound soundness** when the deployed domain is larger or non-finite. Without one of these deployed-domain soundness bases, `b(R)` may prioritize search order only and exact pruning must be disabled or explicitly treated as heuristic/incomplete. + If the target contract accepts **any one scalar optimum** and equal-objective candidates are not observably distinct, minimization may prune `R` when `b(R) >= f(x_incumbent)` and maximization may prune when `b(R) <= f(x_incumbent)`. If equal-objective candidates remain observable—for example the target requires a deterministic tie winner, a secondary total ordering, or enumeration of all optimal candidates—equality is not enough to discard a region under the scalar bound alone. In that case either: @@ -47,11 +49,11 @@ An independently proven infeasible region may also be pruned. A heuristic estima ## Preserved contract -A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**. Heuristic guesses are not proof-based pruning. Exhausting B without an optimality proof does not permit an exactness claim, exhausting B without a feasible incumbent does not permit an infeasibility claim, and parallel execution must preserve the same hard resource ceiling as sequential execution rather than oversubscribing work in flight. A temporarily empty shared queue is **not** frontier exhaustion while any worker owns a leased region that may still produce a candidate, proof obligation, or child region. Likewise, a hard wall-time/compute B applies to the whole search, while a hard peak-memory B applies to current live/reserved memory and must not be converted into irreversible historical consumption after memory is freed. +A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**, and the mechanism used to justify that bound is valid over the deployed domain/region being pruned. Passing a finite regression suite does not turn a heuristic estimate into a proof bound. Exhausting B without an optimality proof does not permit an exactness claim, exhausting B without a feasible incumbent does not permit an infeasibility claim, and parallel execution must preserve the same hard resource ceiling as sequential execution rather than oversubscribing work in flight. A temporarily empty shared queue is **not** frontier exhaustion while any worker owns a leased region that may still produce a candidate, proof obligation, or child region. Likewise, a hard wall-time/compute B applies to the whole search, while a hard peak-memory B applies to current live/reserved memory and must not be converted into irreversible historical consumption after memory is freed. ## Optimization -Maintain an incumbent when one exists, partition the search space, compute a cheap sound `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific bound plus the target's tie semantics prove the region cannot affect the required answer. Before the first incumbent exists, sound bounds may prioritize regions or prove individual regions infeasible, but incumbent-based objective pruning is unavailable. +Maintain an incumbent when one exists, partition the search space, compute a cheap **sound and deployed-domain-justified** `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific bound plus the target's tie semantics prove the region cannot affect the required answer. Before the first incumbent exists, sound bounds may prioritize regions or prove individual regions infeasible, but incumbent-based objective pruning is unavailable. For **parallel** search, define one global frontier lifecycle. A region remains part of the frontier from enqueue until it is either (a) soundly pruned/closed, or (b) replaced by its child regions through an atomic/linearizable completion transition. Dequeuing for worker ownership therefore changes a region from `queued` to `leased/in-flight`; it does **not** remove that region from the global frontier. A worker that branches a leased region must publish all resulting children and close/release the parent as one frontier-accounting transition, or use another protocol that cannot expose a moment where the queue is empty even though unpublished descendants still exist. Worker failure/cancellation must return or recover the lease so unexplored work is not silently lost. @@ -77,7 +79,7 @@ A relaxed solution is evidence for a bound, not automatically a feasible final a ## Validation -For small fixtures, compare with exhaustive enumeration. Test `b(R)` soundness independently by checking the direction-specific inequality against exhaustive feasible values inside each test region. Test pruning separately from search ordering. Include fixtures where the first feasible candidate is found late and where B expires before any feasible candidate exists; verify that the latter returns `no-incumbent / feasibility-unknown`, reports only independently valid global-bound information, and makes no infeasibility, optimality, or incumbent-based gap claim. Verify that budget exhaustion with an incumbent returns an anytime result without an exactness claim. +For small fixtures, compare with exhaustive enumeration and test `b(R)` independently by checking the direction-specific inequality against exhaustive feasible values inside each test region. These small-case tests are **regression/adversarial evidence only** unless they exhaust the complete deployed finite domain. Before enabling exact pruning on a larger or non-finite deployed domain, additionally verify the declared soundness basis: inspect/check the analytic or formal derivation and its assumptions; or exhaustively enumerate the complete finite deployed domain and every supported region construction; or exercise the conservative runtime certificate/guard and prove that every actual prune is accompanied by a valid certificate and that certificate failure retains/evaluates the region rather than pruning it. Deliberately inject a defective bound that passes the small fixtures but violates one held-out/deployed region and prove exact mode rejects it or the runtime guard refuses that prune. Test pruning separately from search ordering. Include fixtures where the first feasible candidate is found late and where B expires before any feasible candidate exists; verify that the latter returns `no-incumbent / feasibility-unknown`, reports only independently valid global-bound information, and makes no infeasibility, optimality, or incumbent-based gap claim. Verify that budget exhaustion with an incumbent returns an anytime result without an exactness claim. Add **parallel frontier-exhaustion races**. Use a fixture where the last queued region is leased by one worker, making the shared queue empty, then pause that worker before it publishes one or more child regions. Prove the coordinator does not declare exhaustion or exact optimality while that lease remains live. Resume the worker and verify the children become searchable and the final result matches exhaustive/scalar search. Also inject worker failure/cancellation while holding the last lease and verify the region is recovered/requeued or otherwise completed without losing unexplored work. Test simultaneous parent-close/child-publish transitions and prove there is no observation in which both queued and leased frontier counts reach zero before all descendants are durably accounted for. @@ -85,16 +87,16 @@ Add **parallel budget-boundary fixtures**. Race multiple workers against one rem Add a **peak-memory reuse fixture**. Under a hard peak-memory B, run many sequential/non-overlapping branches that each allocate and then free a large work buffer. Prove each live allocation/reservation is admitted only while `live_allocated + live_reserved <= B`, freed capacity becomes reusable, `peak_observed <= B`, and the search does **not** exhaust merely because the sum of historical allocations exceeds B. Then overlap enough workers to exceed the peak if all allocations were admitted and prove the final reservation is rejected/blocked before live memory can cross B. Race allocation, free, cancellation and cleanup to verify live-memory accounting remains linearizable and no capacity is released before the corresponding memory is actually reclaimable. -Add **equal-objective tie fixtures**. For an any-one-optimum contract, prove equality pruning cannot alter any observable result. For deterministic tie-winner contracts, construct regions containing equal-objective candidates with better/worse tie ranks and prove equality-bound regions are retained until the declared tie winner is established. For all-optima contracts, prove every equal-objective optimum is enumerated. If using a stronger total-order bound, validate its soundness independently against exhaustive fixtures. +Add **equal-objective tie fixtures**. For an any-one-optimum contract, prove equality pruning cannot alter any observable result. For deterministic tie-winner contracts, construct regions containing equal-objective candidates with better/worse tie ranks and prove equality-bound regions are retained until the declared tie winner is established. For all-optima contracts, prove every equal-objective optimum is enumerated. If using a stronger total-order bound, validate its soundness independently against exhaustive fixtures and establish the same deployed-domain proof basis before using it for exact pruning. ## Target-repo adaptation -The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. For parallel implementations, define the global frontier state machine, lease ownership/recovery rules, parent-close/child-publish atomicity, and the exact exhaustion predicate over queued plus leased/in-flight work. Also define each budget's accounting type and enforcement boundary: cumulative flow (`consumed + reserved`), elapsed deadline, or peak/live capacity (`live_allocated + live_reserved`, plus `peak_observed`). If a target says “memory budget,” state whether it means peak live memory or cumulative allocation traffic. Downgrade any dimension that can escape its correct enforcement boundary to best-effort/observational rather than calling it hard B. +The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. **Document the deployed-domain soundness basis for every bound used to prune in exact mode:** identify the analytic/formal theorem and assumptions, the complete finite domain exhausted by verification, or the runtime certificate/guard and its conservative fallback semantics. Treat small fixture results as regression evidence, not as the proof basis for a larger domain. For parallel implementations, define the global frontier state machine, lease ownership/recovery rules, parent-close/child-publish atomicity, and the exact exhaustion predicate over queued plus leased/in-flight work. Also define each budget's accounting type and enforcement boundary: cumulative flow (`consumed + reserved`), elapsed deadline, or peak/live capacity (`live_allocated + live_reserved`, plus `peak_observed`). If a target says “memory budget,” state whether it means peak live memory or cumulative allocation traffic. Downgrade any dimension that can escape its correct enforcement boundary to best-effort/observational rather than calling it hard B. ## Failure modes -Unsound bounds can remove the true optimum; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; treating queue-empty as frontier-empty can declare exact completion while a leased region still owns unexplored descendants; losing a worker lease can silently drop search regions; non-atomic parent-close/child-publication can create false exhaustion; parallel workers without linearizable evaluation reservations can oversubscribe the last evaluation slot; branching/child/frontier/serialization/incumbent overhead can exceed a nominal wall-time/compute B if only evaluations are charged; an unenforced coordinator/cleanup path can outlive a claimed whole-search deadline; treating peak memory as cumulative consumed spend can falsely exhaust a valid search and prevent reuse of freed capacity; releasing live-memory capacity before actual reclamation can instead oversubscribe the peak; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. +Unsound bounds can remove the true optimum; **small exhaustive fixtures can all pass while a bound remains unsound elsewhere in a larger deployed domain**; a runtime certificate that is not conservative or whose failure path still prunes destroys exactness; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; treating queue-empty as frontier-empty can declare exact completion while a leased region still owns unexplored descendants; losing a worker lease can silently drop search regions; non-atomic parent-close/child-publication can create false exhaustion; parallel workers without linearizable evaluation reservations can oversubscribe the last evaluation slot; branching/child/frontier/serialization/incumbent overhead can exceed a nominal wall-time/compute B if only evaluations are charged; an unenforced coordinator/cleanup path can outlive a claimed whole-search deadline; treating peak memory as cumulative consumed spend can falsely exhaust a valid search and prevent reuse of freed capacity; releasing live-memory capacity before actual reclamation can instead oversubscribe the peak; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. ## Rollback trigger -Disable any pruning rule that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel/exact mode if frontier exhaustion can be observed while any leased/in-flight region may still produce work, if parent-close/child-publication or lease recovery can lose unexplored regions, if workers can oversubscribe an evaluation-count/cumulative budget, if an elapsed deadline can be reset/escaped, if peak live memory can exceed B, if freed peak-memory capacity is incorrectly made permanently unavailable, or if any resource-consumption path can escape a dimension advertised as hard. Abort exact-mode claims whenever B is exhausted before the full objective/tie/frontier contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. +Disable any pruning rule that lacks a valid deployed-domain soundness basis for exact mode, whose analytic/formal assumptions fail, whose exhaustive finite-domain verification no longer covers the deployed domain/region construction, whose runtime certificate can authorize an unsound prune or fails open, that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel/exact mode if frontier exhaustion can be observed while any leased/in-flight region may still produce work, if parent-close/child-publication or lease recovery can lose unexplored regions, if workers can oversubscribe an evaluation-count/cumulative budget, if an elapsed deadline can be reset/escaped, if peak live memory can exceed B, if freed peak-memory capacity is incorrectly made permanently unavailable, or if any resource-consumption path can escape a dimension advertised as hard. Abort exact-mode claims whenever B is exhausted before the full objective/tie/frontier contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. From 78f6928d5e857eb44e009e3dd8bbccaed6e87e8b Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Tue, 15 Sep 2026 23:35:51 +0930 Subject: [PATCH 048/229] Harden Markdown visibility and placeholder parsing --- scripts/check_catalog.py | 32 ++++++++++++++++++++++---------- 1 file changed, 22 insertions(+), 10 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index d73e2f5..7f682d5 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -3,6 +3,7 @@ from __future__ import annotations +import html import re from collections import Counter from pathlib import Path @@ -92,6 +93,7 @@ REFERENCE_LINK_RE = re.compile(r"\[([^\]]*)\]\[[^\]]*\]") INLINE_HTML_TAG_RE = re.compile(r"]*>") HEADING_RE = re.compile(r"^#{1,6}(?:\s|$)") +SECTION_BOUNDARY_RE = re.compile(r"^#{1,2}(?:\s|$)") THEMATIC_BREAK_RE = re.compile( r"^(?:\*(?:[ \t]*\*){2,}|-(?:[ \t]*-){2,}|_(?:[ \t]*_){2,})[ \t]*$" ) @@ -107,7 +109,13 @@ re.IGNORECASE, ) RAW_HTML_COMPLETE_TAG_RE = re.compile( - r"^ {0,3}]*)?/?>[ \t]*$" + r"^ {0,3}(?:" + r"" + r"|<[A-Za-z][A-Za-z0-9-]*" + r"(?:[ \t]+[A-Za-z_:][A-Za-z0-9_.:-]*" + r"(?:[ \t]*=[ \t]*(?:\"[^\"]*\"|'[^']*'|[^ \t\n\"'=<>`]+))?)*" + r"[ \t]*/?>" + r")[ \t]*$" ) EMPHASIS_WRAPPERS = ("**", "__", "~~", "*", "_") STATUS_WRAPPERS = ("**", "__", "~~", "*", "_", "`") @@ -286,7 +294,7 @@ def section_lines(text: str, heading: str) -> list[str]: return [] end = len(lines) for i in range(start, len(lines)): - if lines[i].startswith("## "): + if SECTION_BOUNDARY_RE.match(lines[i]): end = i break return lines[start:end] @@ -375,8 +383,8 @@ def rendered_inline_text(value: str) -> str: Contract fields and mandatory section bodies must contain textual substance after non-rendering Markdown constructs are removed. Link/image destinations, - formatting markers, and HTML tags therefore cannot make empty source count as - populated rendered content. + formatting markers, HTML tags, and character-reference spelling therefore + cannot make empty rendered source count as populated content. """ text = value text = INLINE_IMAGE_RE.sub(lambda m: m.group(1), text) @@ -386,7 +394,7 @@ def rendered_inline_text(value: str) -> str: text = INLINE_HTML_TAG_RE.sub("", text) text = re.sub(r"[`*_~]", "", text) text = re.sub(r"\\(.)", r"\1", text) - return text.strip() + return html.unescape(text).strip() def has_substantive_rendered_text(value: str) -> bool: @@ -449,11 +457,15 @@ def require_prefixed_fields( f"{path.relative_to(ROOT)} has markup-only/non-substantive field " f"{field} in {section}: '{value}'" ) - if rejected_values is not None and value == rejected_values.get(field): - die( - f"{path.relative_to(ROOT)} has unselected template placeholder " - f"for {field} in {section}: '{value}'" - ) + if rejected_values is not None: + normalized_value = html.unescape( + unwrap_outer_formatting(value, STATUS_WRAPPERS) + ).strip() + if normalized_value == rejected_values.get(field): + die( + f"{path.relative_to(ROOT)} has unselected template placeholder " + f"for {field} in {section}: '{value}'" + ) records: dict[str, Path] = {} From be40613e656a08a31436e0390fcc4d00dda029cb Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 00:30:38 +0930 Subject: [PATCH 049/229] Reserve critical capacity from speculation --- ...T-CRIT-001-critical-path-prioritization.md | 32 +++++++++++-------- 1 file changed, 18 insertions(+), 14 deletions(-) diff --git a/optimizations/OPT-CRIT-001-critical-path-prioritization.md b/optimizations/OPT-CRIT-001-critical-path-prioritization.md index 519c98a..f0312bd 100644 --- a/optimizations/OPT-CRIT-001-critical-path-prioritization.md +++ b/optimizations/OPT-CRIT-001-critical-path-prioritization.md @@ -15,29 +15,31 @@ Non-critical work competes with the dependency chain that determines user-visibl ## Optimization problem contract -- X: target-supported task-priority, prefetch/precompute, lazy/deferred-work, speculation, speculative-input identity, mutation-control, and commitment policies -- F: policies that preserve all semantic deadlines, avoid externally visible speculative side effects before commitment, commit speculative results only from one stable effective-input generation, and satisfy starvation/resource constraints +- X: target-supported task-priority, prefetch/precompute, lazy/deferred-work, speculation, speculative-input identity, mutation-control, commitment, critical-capacity reservation, preemption/cancellation, and speculation-admission policies +- F: policies that preserve all semantic deadlines, avoid externally visible speculative side effects before commitment, commit speculative results only from one stable effective-input generation, satisfy starvation/resource constraints, and enforce enough protected or promptly reclaimable capacity that speculative work cannot occupy every resource a newly arriving critical task may need - f: measured end-to-end latency of the declared critical dependency path, including resource pressure introduced by speculation/deferment - d: minimize -- C: critical outputs and semantic deadlines are preserved; speculative work is safely discardable; any speculative result is bound to a complete immutable snapshot or full-duration mutation witness, and validation of that witness is linearized with commitment so intervening or final-window A→B→A/input changes cannot be erased before visibility; deferred work completes before it becomes semantically required -- B: target-specific trace/benchmark budget covering cold/warm, hit/miss, wrong-speculation, stale-speculation, and change/revert cases; no portable prediction horizon is supplied here +- C: critical outputs and semantic deadlines are preserved; speculative work is safely discardable; any speculative result is bound to a complete immutable snapshot or full-duration mutation witness, and validation of that witness is linearized with commitment so intervening or final-window A→B→A/input changes cannot be erased before visibility; deferred work completes before it becomes semantically required; and speculative occupancy cannot delay newly arriving critical work beyond the declared critical-start/latency bound because critical capacity is reserved, speculative work is preemptible/cancellable within a bounded reclaim latency, or speculation admission is hard-limited to leave sufficient headroom +- B: target-specific trace/benchmark budget covering cold/warm, hit/miss, wrong-speculation, stale-speculation, change/revert, and critical-arrival-under-saturation cases; no portable prediction horizon is supplied here - S: stop when the declared budget is exhausted or a validated policy materially reduces critical-path latency without violating C -- Variables: categorical / conditional / mixed priority, deferment, prefetch, and speculation policies +- Variables: categorical / conditional / mixed priority, deferment, prefetch, speculation, capacity-reservation, preemption, and admission policies - Search scope: local critical-path policy tuning -- Objective behavior: noisy under realistic workload timing; semantic identity/deadline checks are deterministic +- Objective behavior: noisy under realistic workload timing; semantic identity/deadline/capacity checks are deterministic - Information: derivative-free / black-box latency measurements - Evaluation cost: moderate to expensive end-to-end tracing/benchmarking -- Constraints: semantic deadlines, starvation, side effects, input identity/mutation freshness, commitment linearizability, memory/CPU/I/O, and target resource constraints -- Parallelism: asynchronous / concurrent execution is common -- Exactness: exact target semantics; speculative work may be discarded but not committed stale +- Constraints: semantic deadlines, starvation, side effects, input identity/mutation freshness, commitment linearizability, protected/reclaimable critical capacity, memory/CPU/I/O, and target resource constraints +- Parallelism: asynchronous / concurrent execution is common; speculative dispatch must preserve enforceable critical headroom or bounded preemption +- Exactness: exact target semantics; speculative work may be discarded but not committed stale or allowed to violate critical-capacity guarantees ## Preserved contract -Deferred work must still complete before its semantic deadline. Speculative work must be discardable and must not create externally visible side effects before commitment. A speculative result may be committed/delivered only if it was produced from one coherent effective-input generation equivalent to the non-speculative reference path; endpoint equality after an intervening mutation is not sufficient, and a successful freshness check is not sufficient unless the checked identity remains authoritative through the commit that makes the result visible. +Deferred work must still complete before its semantic deadline. Speculative work must be discardable and must not create externally visible side effects before commitment. A speculative result may be committed/delivered only if it was produced from one coherent effective-input generation equivalent to the non-speculative reference path; endpoint equality after an intervening mutation is not sufficient, and a successful freshness check is not sufficient unless the checked identity remains authoritative through the commit that makes the result visible. Priority must also remain operational rather than nominal: speculation may not consume all capacity needed by a critical request that arrives after speculative work has started. The target must preserve a declared critical-start/latency bound using reserved capacity, bounded-latency preemption/cancellation, or hard admission limits that leave sufficient headroom for non-preemptible speculation. ## Optimization -Execute critical dependencies first; prefetch/precompute likely-soon work only when probability and spare resources justify it; lazily defer non-critical work; avoid work with no demonstrated demand. +Execute critical dependencies first; prefetch/precompute likely-soon work only when probability and resource policy justify it; lazily defer non-critical work; avoid work with no demonstrated demand. + +Treat "spare at dispatch" as insufficient evidence that speculation is safe. Before launching speculative work, perform an **enforceable critical-capacity admission check**. For non-preemptible speculation, reserve the worker, I/O, accelerator, memory, connection, queue, or other resource capacity required by the declared critical workload, or hard-limit speculative concurrency/occupancy so worst-case admitted speculation cannot consume that headroom. For preemptible/cancellable speculation, define and enforce a maximum reclaim latency and ensure cancellation/preemption returns enough capacity before the critical-start/deadline bound can be violated. Capacity accounting/admission must be atomic enough that concurrent speculative launches cannot each observe the same final spare slot and collectively consume protected headroom. Bind every speculative/precomputed result to a complete effective-input identity for the **full speculation-to-commit interval**. Prefer speculation against an immutable snapshot/version. If snapshots are unavailable, use a full-duration mutation/read lock or capture a monotonically increasing, non-reusable version/epoch for every mutable effective input. Every relevant mutation must advance its witness, including A→B→A changes that restore original bytes. A commit-time hash/identity comparison may supplement the mutation witness but must not be the sole freshness proof. @@ -57,18 +59,20 @@ Commitment is the semantic boundary: no stale speculative result may become exte Trace the true dependency path and measure end-to-end latency, not only individual task duration. Test cold/warm, cache-hit/miss and wrong-speculation cases. Explicitly test semantic deadlines, starvation, cancellation, and that speculative work cannot expose side effects before commitment. +Add a **critical-arrival-under-speculation-saturation** race. Fill speculative work to the maximum admitted occupancy, then introduce a critical request requiring each protected resource class (for example a worker plus I/O or accelerator capacity). For reserved-capacity designs, prove the critical request can acquire the reserved capacity without waiting for non-preemptible speculative completion. For preemptible/cancellable designs, prove enough speculation is reclaimed within the declared maximum reclaim latency to satisfy the critical-start/deadline bound. For admission-bound designs, prove concurrent speculative launches cannot race past the headroom limit. Repeat with non-preemptible long-running speculation, simultaneous critical arrivals, and mixed-resource bottlenecks; a policy that merely observed spare capacity before dispatch must fail this fixture if it can later block the critical path. + Add stale-speculation fixtures with explicit **A→B→A** races. Start speculation from identity A, mutate the effective inputs to B while speculation reads/runs, then restore original bytes before demand/commitment. For snapshot-based targets, prove speculation consumed only immutable A. For lock-based targets, prove the mutation cannot interleave. For epoch/version-based targets, prove every mutation increments the monotonic witness and that the final witness exposes the intervening change even though endpoint content equals A. Also test delayed speculative completion, version rollback, and concurrent config/schema changes. Compare every committed speculative result against the non-speculative reference path for the exact committed identity. Add a **final validation-to-commit race**. Pause immediately after the last ordinary witness comparison but before the result would become visible, then mutate an effective input. For lock-based designs, prove the mutation is blocked until after commitment. For epoch/version designs, prove the atomic compare-and-commit rejects the stale speculative result rather than publishing it. Repeat with A→B→A and multi-input epoch-vector changes. No fixture may pass by doing an ordinary comparison followed by a separate publication step. ## Target-repo adaptation -Criticality and prediction horizons are workload-specific. Re-profile after topology or user-flow changes. Define the complete effective-input identity for each speculative result and choose immutable snapshots, full-duration mutation locks, or monotonic epochs that record every intervening change. Define the linearization boundary that couples freshness validation to external commitment: lock-through-commit or atomic compare-and-commit. Specify exactly when a stale speculative result is discarded. Do not rely on commit-time endpoint revalidation alone, or on check-then-publish epoch validation, to establish freshness. +Criticality and prediction horizons are workload-specific. Re-profile after topology or user-flow changes. Define the complete effective-input identity for each speculative result and choose immutable snapshots, full-duration mutation locks, or monotonic epochs that record every intervening change. Define the linearization boundary that couples freshness validation to external commitment: lock-through-commit or atomic compare-and-commit. Specify exactly when a stale speculative result is discarded. Also define the **critical-capacity invariant** for every contended resource: how much capacity is reserved, what speculative occupancy ceiling applies, or which work is preemptible/cancellable and the maximum reclaim latency. Make admission/concurrency accounting race-safe, and do not treat currently idle capacity as sufficient if non-preemptible speculation can consume it before future critical arrivals. Do not rely on commit-time endpoint revalidation alone, or on check-then-publish epoch validation, to establish freshness. ## Failure modes -Speculation steals resources from critical work, lazy work causes later latency cliffs, priorities become stale, deferred tasks starve, semantic deadlines are missed, speculative side effects escape before commitment, A→B→A mutations can fool endpoint-only freshness checks, non-monotonic/reused epochs can erase intervening changes, a check-then-publish window can expose stale speculation after a successful freshness check, or stale speculative output is committed after its effective inputs changed. +Speculation steals resources from critical work; non-preemptible speculation can fill every worker, I/O slot, accelerator slot, connection, or other bottleneck before a new critical request arrives; concurrent speculative launches can oversubscribe supposedly reserved headroom; preemption/cancellation can be too slow to protect the critical-start/deadline bound; lazy work causes later latency cliffs; priorities become stale; deferred tasks starve; semantic deadlines are missed; speculative side effects escape before commitment; A→B→A mutations can fool endpoint-only freshness checks; non-monotonic/reused epochs can erase intervening changes; a check-then-publish window can expose stale speculation after a successful freshness check; or stale speculative output is committed after its effective inputs changed. ## Rollback trigger -Immediately disable/revert the policy on any violation of C, including a required task missing its semantic deadline, speculative work exposing an externally visible side effect before commitment, a speculative result being committed/delivered without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input generation, or any test showing freshness validation can be separated from commitment so a mutation can win in between. Also disable it if critical-path latency or resource pressure worsens materially. +Immediately disable/revert the policy on any violation of C, including a required task missing its semantic deadline; a critical request being delayed beyond the declared start/latency bound because speculative work consumed protected capacity; a reservation/admission race allowing speculation to exceed its occupancy ceiling; preemption/cancellation failing to reclaim capacity within its declared bound; speculative work exposing an externally visible side effect before commitment; a speculative result being committed/delivered without an immutable snapshot/lock/monotonic mutation witness proving one coherent effective-input generation; or any test showing freshness validation can be separated from commitment so a mutation can win in between. Also disable it if critical-path latency or resource pressure worsens materially. From 6ce8cf568694d9c3b1a34d88712e2c2cd11211b0 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 00:32:05 +0930 Subject: [PATCH 050/229] Harden Markdown section parsing --- scripts/check_catalog.py | 65 +++++++++++++++++++--------------------- 1 file changed, 31 insertions(+), 34 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 7f682d5..2112524 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -91,9 +91,16 @@ INLINE_LINK_RE = re.compile(r"\[([^\]]*)\]\([^)]*\)") REFERENCE_IMAGE_RE = re.compile(r"!\[([^\]]*)\]\[[^\]]*\]") REFERENCE_LINK_RE = re.compile(r"\[([^\]]*)\]\[[^\]]*\]") -INLINE_HTML_TAG_RE = re.compile(r"]*>") +INLINE_HTML_TAG_RE = re.compile( + r"`]+))?)*" + r"[ \t]*/?>" +) HEADING_RE = re.compile(r"^#{1,6}(?:\s|$)") SECTION_BOUNDARY_RE = re.compile(r"^#{1,2}(?:\s|$)") +SETEXT_H1_RE = re.compile(r"^ {0,3}=+[ \t]*$") +SETEXT_H2_RE = re.compile(r"^ {0,3}-+[ \t]*$") THEMATIC_BREAK_RE = re.compile( r"^(?:\*(?:[ \t]*\*){2,}|-(?:[ \t]*-){2,}|_(?:[ \t]*_){2,})[ \t]*$" ) @@ -134,6 +141,11 @@ def die(msg: str) -> None: raise SystemExit(f"catalog-integrity: {msg}") +def markdown_source_lines(text: str) -> list[str]: + """Split only on CommonMark line endings (LF, CRLF, or CR).""" + return text.replace("\r\n", "\n").replace("\r", "\n").split("\n") + + def is_indented_code_line(raw: str) -> bool: """Return whether a non-fenced line is an indented Markdown code line.""" return raw.startswith("\t") or raw.startswith(" ") @@ -168,9 +180,6 @@ def strip_inline_html_comments(raw: str, in_comment: bool) -> tuple[str, bool]: def raw_html_block_start(raw: str) -> tuple[str, str | None] | None: """Return the raw-HTML block mode for a CommonMark-style block start.""" - # A comment beginning at the start of a block line is raw HTML type 2. The - # complete terminating line belongs to the raw block, even when text follows - # the --> token, so callers must discard that whole line. if re.match(r"^ {0,3}" @@ -193,13 +202,7 @@ def raw_html_tag_closes(raw: str, tag: str) -> bool: def visible_nonfenced_lines(lines: list[str]) -> list[str]: - """Return Markdown-visible lines used by schema validation. - - The pass excludes fenced code, indented code, raw HTML blocks, and HTML - comments before headings, fields, links, and tables are interpreted. - Fence state is derived from the original source line so preprocessing cannot - turn a non-closer into a closer. - """ + """Return Markdown-visible lines used by schema validation.""" visible: list[str] = [] fence_char: str | None = None fence_len = 0 @@ -227,7 +230,6 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: if html_end is not None and html_end in raw: html_mode = None html_end = None - # The whole terminator line belongs to the raw HTML block. continue if html_mode == "blank": if raw.strip() == "": @@ -242,7 +244,6 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: continue raw_for_parse = rendered else: - # Four-space/tab-indented lines are code blocks, not headings/tables. if is_indented_code_line(raw): continue @@ -264,13 +265,10 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: elif html_mode == "token" and html_end is not None and html_end in raw: html_mode = None html_end = None - # Raw HTML owns the complete source line, including a terminator. continue raw_for_parse, inline_comment = strip_inline_html_comments(raw, False) - # An inline comment can expose a remainder, but that remainder still must - # not be accepted if it is indented as code after comment removal. if raw_for_parse and is_indented_code_line(raw_for_parse): continue if raw_for_parse: @@ -282,27 +280,36 @@ def visible_nonfenced_lines(lines: list[str]) -> list[str]: def visible_text(text: str) -> str: - return "\n".join(visible_nonfenced_lines(text.splitlines())) + return "\n".join(visible_nonfenced_lines(markdown_source_lines(text))) + + +def is_setext_section_boundary(lines: list[str], index: int) -> bool: + """Recognize a visible Setext H1/H2 beginning at lines[index].""" + if index + 1 >= len(lines): + return False + candidate = lines[index] + if not candidate.strip() or HEADING_RE.match(candidate): + return False + underline = lines[index + 1] + return bool(SETEXT_H1_RE.fullmatch(underline) or SETEXT_H2_RE.fullmatch(underline)) def section_lines(text: str, heading: str) -> list[str]: """Return one exact visible level-2 Markdown section.""" - lines = visible_nonfenced_lines(text.splitlines()) + lines = visible_nonfenced_lines(markdown_source_lines(text)) try: start = lines.index(heading) + 1 except ValueError: return [] end = len(lines) for i in range(start, len(lines)): - if SECTION_BOUNDARY_RE.match(lines[i]): + if SECTION_BOUNDARY_RE.match(lines[i]) or is_setext_section_boundary(lines, i): end = i break return lines[start:end] def markdown_table_cells(line: str) -> list[str] | None: - # Preserve the four-space/tab code-block distinction even if a caller passes - # a line that did not come through visible_nonfenced_lines. if is_indented_code_line(line): return None stripped = line.strip() @@ -366,7 +373,6 @@ def unwrap_outer_formatting(value: str, wrappers: tuple[str, ...]) -> str: def unwrap_markdown_emphasis(cell: str) -> str: - """Remove balanced outer emphasis wrappers; do not unwrap code spans.""" return unwrap_outer_formatting(cell, EMPHASIS_WRAPPERS) @@ -379,13 +385,7 @@ def parse_record_link_cell(cell: str, context: str) -> tuple[str, str]: def rendered_inline_text(value: str) -> str: - """Approximate visible inline text for required field-value validation. - - Contract fields and mandatory section bodies must contain textual substance - after non-rendering Markdown constructs are removed. Link/image destinations, - formatting markers, HTML tags, and character-reference spelling therefore - cannot make empty rendered source count as populated content. - """ + """Approximate visible inline text for required field-value validation.""" text = value text = INLINE_IMAGE_RE.sub(lambda m: m.group(1), text) text = INLINE_LINK_RE.sub(lambda m: m.group(1), text) @@ -398,7 +398,6 @@ def rendered_inline_text(value: str) -> str: def has_substantive_rendered_text(value: str) -> bool: - """Require at least one visible alphanumeric character after inline parsing.""" return any(ch.isalnum() for ch in rendered_inline_text(value)) @@ -427,8 +426,6 @@ def section_has_content(lines: list[str]) -> bool: def normalized_status_category(raw: str) -> str: - # Notes after ';' are outside the category. Preserve literal marker characters - # inside the category and unwrap only balanced formatting around the whole one. category = raw.split(";", 1)[0].strip() return unwrap_outer_formatting(category, STATUS_WRAPPERS) @@ -472,7 +469,7 @@ def require_prefixed_fields( status_categories: dict[str, str] = {} for path in sorted(OPT_DIR.glob("OPT-*.md")): text = path.read_text(encoding="utf-8") - lines = visible_nonfenced_lines(text.splitlines()) + lines = visible_nonfenced_lines(markdown_source_lines(text)) first = lines[0] if lines else "" match = ID_RE.match(first) if not match: @@ -657,7 +654,7 @@ def require_prefixed_fields( if not problem_contract.is_file(): die("OPTIMIZATION-PROBLEM.md is missing") problem_text = problem_contract.read_text(encoding="utf-8") -problem_visible = visible_nonfenced_lines(problem_text.splitlines()) +problem_visible = visible_nonfenced_lines(markdown_source_lines(problem_text)) if not problem_visible or problem_visible[0] != "# Optimization Problem Contract": die("OPTIMIZATION-PROBLEM.md has missing/hidden/invalid title") if "## Canonical contract" not in problem_visible: From 1e56fb212246c7f3ccc87facd3099555377e8787 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 01:03:44 +0930 Subject: [PATCH 051/229] Align performance-budget objective direction --- .../OPT-BUDGET-001-performance-regression-budgets.md | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md index c36ca85..9ce7c06 100644 --- a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md +++ b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md @@ -17,8 +17,8 @@ Small performance regressions accumulate because performance is measured occasio - X: target-supported metric/fixture/statistic/threshold configurations for a performance-regression gate - F: gate configurations based on a sufficiently characterized environment and workload, with statistically justified tolerance, selection-aware validation, and no weakening of functional correctness or workload realism -- f: target-measured regression-detection quality together with CI noise/false-alarm rate and measurement overhead -- d: minimize missed material regressions and flaky/false failures under the target's predeclared multi-objective ordering +- f: target-measured **loss vector** comprising missed-material-regression loss (for example false-negative rate and, where relevant, severity-weighted miss cost), flaky/false-failure loss (false-positive rate), and measurement/CI overhead +- d: minimize every component of the declared loss vector under the target's predeclared scalar, weighted, Pareto, or lexicographic ordering; if detection quality is reported separately, it is a diagnostic complement such as `1 - false-negative-rate`, not an oppositely oriented coordinate inside `f` - C: the performance gate must not incentivize weakening tests, assertions, evidence, semantic coverage, or representative workload inputs; once a candidate gate is selected, its claimed false-positive/false-negative performance must be established on independent control executions or under a predeclared selection-aware procedure that accounts for every configuration tried - B: target-specific calibration and certification budget specifying repetitions, environment samples, held-out/control executions, and allowable CI/runtime measurement cost - S: stop calibration when the declared sample budget is exhausted or the baseline/noise estimate is stable enough to freeze one candidate gate for independent certification; promote it only if the certification contract passes @@ -55,16 +55,18 @@ Calibrate variance before setting the threshold. During tuning, compare candidat Certify the frozen gate on independent known-good and known-regressed control executions that were not used to select it. Measure false positives, false negatives, and gate overhead against predeclared acceptance limits. If independent controls are unavailable, use a predeclared nested-resampling or selection-aware procedure that accounts for every candidate/configuration examined, and report the resulting adjusted uncertainty/error rates rather than reusing naive in-sample estimates. +Verify objective orientation explicitly: construct one candidate with fewer missed regressions but more false alarms and another with the opposite tradeoff, compute the declared loss coordinates, and prove the configured scalar/Pareto/lexicographic ordering ranks them exactly as documented. A separately reported positive detection-quality score must never be fed into a minimization coordinate without an explicit monotone conversion to loss. + Record which executions were used for calibration/selection versus certification. Re-run independent controls after runner/toolchain changes and periodically enough to detect stale fixtures or sensitivity drift. Add an explicit overfitting fixture where several candidate gates are tuned on one noisy control sample set; prove the gate cannot be promoted merely because one candidate looked best on those same samples. ## Target-repo adaptation -Never copy another project's milliseconds, bundle sizes or thresholds. Establish the target's own baseline and noise envelope, define control fixtures, and declare acceptable false-positive/false-negative rates plus a maximum measurement-overhead budget. Predeclare how calibration/selection is separated from certification: held-out controls by default, or a justified nested/selection-aware alternative. Preserve the candidate-search history needed to audit the claimed certification error rates. +Never copy another project's milliseconds, bundle sizes or thresholds. Establish the target's own baseline and noise envelope, define control fixtures, and declare acceptable false-positive/false-negative rates plus a maximum measurement-overhead budget. Define `f` using consistently oriented loss coordinates and predeclare how those coordinates are ordered or scalarized; if the target also reports a positive detection-quality score, document its conversion to the minimized loss coordinate. Predeclare how calibration/selection is separated from certification: held-out controls by default, or a justified nested/selection-aware alternative. Preserve the candidate-search history needed to audit the claimed certification error rates. ## Failure modes -Flaky gates from uncontrolled runners, benchmark gaming, stale fixtures, hardware drift, thresholds so loose they miss real regressions, thresholds so tight they block good changes, selection bias from evaluating a chosen gate on the same controls used to tune it, unreported configuration search that invalidates nominal error rates, and measurement overhead large enough to damage CI usability or distort the workload under test. +Flaky gates from uncontrolled runners, benchmark gaming, stale fixtures, hardware drift, thresholds so loose they miss real regressions, thresholds so tight they block good changes, an objective vector mixing maximized quality with minimized costs without an explicit per-coordinate direction/conversion, selection bias from evaluating a chosen gate on the same controls used to tune it, unreported configuration search that invalidates nominal error rates, and measurement overhead large enough to damage CI usability or distort the workload under test. ## Rollback trigger -Disable or demote the gate to non-blocking and recalibrate whenever its measurement environment is invalid, its fixture is stale/nonrepresentative, independent/selection-aware certification no longer meets the declared false-positive/false-negative limits, known regressions are no longer detected, known-good controls fail above the declared false-positive limit, observed false negatives exceed the declared limit, or measurement overhead exceeds the predeclared budget. Do **not** disable merely because product code legitimately regressed; in that case keep the valid gate and fix or explicitly accept the regression through the target's normal review process. +Disable or demote the gate to non-blocking and recalibrate whenever its measurement environment is invalid, its fixture is stale/nonrepresentative, its objective orientation/scalarization is ambiguous or ranks a worse detector as better, independent/selection-aware certification no longer meets the declared false-positive/false-negative limits, known regressions are no longer detected, known-good controls fail above the declared false-positive limit, observed false negatives exceed the declared limit, or measurement overhead exceeds the predeclared budget. Do **not** disable merely because product code legitimately regressed; in that case keep the valid gate and fix or explicitly accept the regression through the target's normal review process. From 3cb785beadcf681155e2ed9776b01232b6ee33c4 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 01:04:46 +0930 Subject: [PATCH 052/229] Preserve coalescer admission availability --- ...01-concurrent-duplicate-work-coalescing.md | 32 +++++++++++-------- 1 file changed, 19 insertions(+), 13 deletions(-) diff --git a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md index e672af7..17b1bef 100644 --- a/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md +++ b/optimizations/OPT-COAL-001-concurrent-duplicate-work-coalescing.md @@ -14,19 +14,19 @@ Many callers request the same expensive computation concurrently before any call ## Optimization problem contract -- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime and launch-state policies, per-generation waiter limits, **global/per-tenant in-flight generation and waiter budgets**, overflow/backpressure policies, per-waiter cancellation/deadline/**atomic terminal-outcome publication** policies, result-preparation/clone-failure policies, retry/error-sharing policies, and result-ownership policies -- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, preparation-failure, launch-cancellation, and error semantics for every joined caller, linearize cancellation/deadline against **irrevocable terminal-outcome publication** for each waiter, linearize unstarted-to-running launch against closing/last-waiter cancellation, **bound total registry/generation/waiter memory across all keys and retained closing generations**, and never admit new waiters to a closing or terminal generation -- f: measured duplicate upstream evaluations and end-to-end/tail latency, including coalescer synchronization, global/per-tenant admission accounting, generation/waiter memory, launch-state synchronization, result preparation/cloning, atomic outcome publication, overflow/backpressure, and result-copy overhead -- d: minimize under the target's predeclared scalar or lexicographic ordering -- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, deadline, launch state, and result-preparation outcome; **a waiter may become terminal-success/error only when the corresponding outcome is already irrevocably stored/published for that waiter**; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; no upstream operation may start after its generation has already become closing due to loss of all live waiters; closing/terminal generations are not joinable; **global/per-tenant generation and waiter admission limits are never exceeded**, including under high-cardinality keys and retained closing generations; overload has an explicit bounded result -- B: target-specific concurrent-load test budget plus explicit global/per-tenant coalescer admission limits (maximum in-flight generations, waiter records, and any bounded overflow queue); no portable request count, duration, or memory cap is supplied here -- S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C +- X: target-supported request-key canonicalizations, authorization/equivalence scopes, shared-operation lifetime and launch-state policies, per-generation waiter limits, **global/per-tenant in-flight generation and waiter budgets**, bounded-overflow/backpressure policies, **baseline-availability/successful-admission/useful-throughput floors**, per-waiter cancellation/deadline/**atomic terminal-outcome publication** policies, result-preparation/clone-failure policies, retry/error-sharing policies, and result-ownership policies +- F: policies that coalesce only requests equivalent in both computation semantics and authorization/visibility scope, preserve authorization, timeout, cancellation, result, ownership, preparation-failure, launch-cancellation, and error semantics for every joined caller, linearize cancellation/deadline against **irrevocable terminal-outcome publication** for each waiter, linearize unstarted-to-running launch against closing/last-waiter cancellation, **bound total registry/generation/waiter memory across all keys and retained closing generations**, never admit new waiters to a closing or terminal generation, and **preserve the target's predeclared normal-load availability/successful-admission and useful-throughput floor rather than satisfying the resource bound by rejecting ordinary service demand** +- f: measured **loss vector** comprising duplicate upstream evaluations, end-to-end/tail latency, overload/rejection above the target's allowed baseline envelope, useful-throughput loss, and coalescer overhead (including synchronization, global/per-tenant admission accounting, generation/waiter memory, launch-state synchronization, result preparation/cloning, atomic outcome publication, backpressure, and result-copy cost) +- d: minimize every declared loss coordinate under the target's predeclared scalar, weighted, Pareto, or lexicographic ordering while F remains satisfied +- C: every joined caller receives exactly one terminal outcome valid for its original request semantics, authorization scope, ownership contract, cancellation state, deadline, launch state, and result-preparation outcome; **a waiter may become terminal-success/error only when the corresponding outcome is already irrevocably stored/published for that waiter**; non-equivalent or authorization-distinct requests are never merged; one caller leaving cannot incorrectly cancel work still required by another caller; no upstream operation may start after its generation has already become closing due to loss of all live waiters; closing/terminal generations are not joinable; **global/per-tenant generation and waiter admission limits are never exceeded**, including under high-cardinality keys and retained closing generations; overload has an explicit bounded result; and **normal-load successful-admission/availability and useful throughput remain at or above the target's declared baseline floor** +- B: target-specific concurrent-load test budget plus explicit global/per-tenant coalescer admission limits (maximum in-flight generations, waiter records, and any bounded overflow queue); no portable request count, duration, memory cap, or availability floor is supplied here +- S: stop when the declared load-test budget is exhausted or further policy changes fail to produce a validated material improvement without violating C or the declared availability/throughput floor - Variables: categorical / integer / mixed - Search scope: local policy tuning within one coalescing boundary -- Objective behavior: noisy under concurrent load; semantic equivalence remains deterministic +- Objective behavior: noisy under concurrent load; semantic equivalence and admission-floor checks remain deterministic for a fixed trace - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive concurrent-load testing -- Constraints: semantic equivalence, authorization, ownership, launch-state linearizability, result-preparation failure, **global/per-tenant registry and waiter memory**, cancellation, deadline, atomic outcome publication, timeout, and resource constraints +- Constraints: semantic equivalence, authorization, ownership, launch-state linearizability, result-preparation failure, **global/per-tenant registry and waiter memory**, **baseline availability/successful admission and useful throughput**, cancellation, deadline, atomic outcome publication, timeout, and resource constraints - Parallelism: asynchronous / concurrent - Exactness: exact request/result semantics; no approximation is introduced @@ -34,6 +34,8 @@ Many callers request the same expensive computation concurrently before any call Coalescing may merge only requests that are equivalent for the same **joinable generation** of the shared operation, including any tenant/principal/visibility context that affects whether the computation or its result may be shared. Each caller retains independent authorization, cancellation, timeout/deadline, result-ownership, preparation-failure, and error semantics. A caller abandoning its wait must not by itself terminate a shared operation that still has live waiters. Once a generation enters cancellation, closure, success, or failure handling, it becomes non-joinable before later callers can attach. **Memory/admission bounds apply across the whole coalescer, not only inside one generation:** high-cardinality keys, fresh generations, and retained closing generations must all consume explicit global/per-tenant generation and waiter capacity until their state is actually retired. No request may bypass those caps merely because it is the first waiter for a new key. +Bounded admission is not permission to stop serving the workload. For the target's declared normal-load/reference demand envelope, the optimized coalescer must preserve at least the predeclared successful-admission/availability and useful-throughput floor of the reference path (or another explicitly approved service-level floor). Overload/backpressure may be part of the contract outside that envelope, but a zero-cap or tiny-cap policy that merely rejects ordinary requests is infeasible even if it minimizes duplicate work and observed latency among the few requests that remain. + Each waiter has exactly one atomic completion cell/terminal record, initially `pending`. Terminal completion is not a two-step “claim then notify” protocol: the winning transition must atomically compare `pending` and **publish/store the complete terminal outcome**—for example `success(immutable-or-private-result-handle)`, `error(error-record)`, `cancelled`, or `timed-out`—before that terminal state becomes visible. A promise/future `set_result`/`set_exception`, transactional queue/envelope insert, or equivalent single linearization point is acceptable. A subsequent wakeup/signal/callback may tell the caller to inspect the terminal record, but that notification is advisory; failure/interruption of the notifier cannot erase an already published outcome or leave a waiter terminal with nothing retrievable. Cleanup may not retire the waiter/outcome until the target's delivery/acknowledgement/retention contract makes that outcome safely consumable or no longer required. The shared operation also has a linearized launch lifecycle: a generation closed before launch may never subsequently start ownerless upstream work. @@ -42,13 +44,15 @@ The shared operation also has a linearized launch lifecycle: a generation closed Create an in-flight registry entry for a canonical equivalence key. The key must include every request attribute required to establish safe sharing, including authorization-relevant tenant/principal/visibility scope unless the target instead proves that the upstream result is globally shareable and independently authorizes each delivered result. +Before tuning admission caps or overflow policy, freeze the target's **normal-load service envelope and admission/throughput floor** from the reference path or an explicitly approved service-level objective. Cap reductions are admissible only while that floor remains satisfied. Rejections/backpressure inside the declared normal-load envelope count as objective loss and, once the hard floor is crossed, make the candidate infeasible rather than artificially “fast.” + Before creating a **new generation**, atomically reserve both (a) one generation slot from the applicable global/per-tenant in-flight-generation budget and (b) one waiter slot for the initiating caller. The reservation and registry insertion must be one linearizable admission decision. If either capacity is exhausted, do not allocate a partial generation or untracked waiter; return/block/queue according to the declared bounded overload policy. A distinct equivalence key does not get a free first waiter merely because no matching generation exists yet. Atomically create the joinable generation **with the initiating caller already registered as its first waiter** and with an explicit launch state such as `unstarted`. Do not invoke, schedule, or otherwise permit upstream work yet. This prevents an immediately/synchronously completing operation from reaching terminal state with an empty waiter set while also giving early cancellation a state it can close before any work exists. Linearize upstream launch against the generation's live-waiter and closing state. Under the same registry lock/CAS/transactional boundary used for generation state, permit `unstarted -> running` only while the generation remains joinable and has at least one live `pending` waiter. Install or bind a **sticky upstream cancellation token/handle** as part of that transition, before releasing the serialization boundary. If the last waiter cancels/times out while the generation is still `unstarted`, transition it to `closing/non-joinable` and make any later launch attempt fail; no upstream work is started. If `unstarted -> running` wins first but actual invocation/scheduling occurs immediately afterward, any last-waiter cancellation that races in that interval must set the already-bound sticky cancellation token. The launcher must check/attach that token before or atomically with invocation so a cancellation that has already won cannot be lost merely because the external operation object did not yet exist. There must be no path where the generation is closed with zero live waiters and a creator later launches uncancelled work from stale local state. -Equivalent later callers may register as independent waiters only while the generation is joinable and **both** the per-generation waiter capacity and applicable global/per-tenant waiter capacity remain. Waiter admission is atomic with both counters. When any required capacity is exhausted, apply one explicit target policy rather than silently exceeding the bound: reject/return a documented overload or retryable-backpressure result, block/queue the caller behind a separately bounded admission mechanism, or use another bounded policy with explicit timeout/cancellation semantics. Starting an unconstrained parallel generation for the same equivalence key is not the default overflow behavior because it recreates the duplicate upstream load this pattern is intended to prevent. If a target deliberately permits overflow generations, those generations still consume the global/per-tenant generation budget, and their concurrency/duplicate-work tradeoff must be part of C/B and validated separately. +Equivalent later callers may register as independent waiters only while the generation is joinable and **both** the per-generation waiter capacity and applicable global/per-tenant waiter capacity remain. Waiter admission is atomic with both counters. When any required capacity is exhausted, apply one explicit target policy rather than silently exceeding the bound: reject/return a documented overload or retryable-backpressure result, block/queue the caller behind a separately bounded admission mechanism, or use another bounded policy with explicit timeout/cancellation semantics. Starting an unconstrained parallel generation for the same equivalence key is not the default overflow behavior because it recreates the duplicate upstream load this pattern is intended to prevent. If a target deliberately permits overflow generations, those generations still consume the global/per-tenant generation budget, and their concurrency/duplicate-work tradeoff must be part of C/B and validated separately. Any such overload behavior must still satisfy the declared normal-load availability/throughput floor. **Closing and terminal-but-not-yet-retired generations continue to consume their generation slot and any still-live waiter/cleanup capacity until deterministic retirement actually releases those resources.** This prevents a churn attack from repeatedly canceling callers, leaving expensive upstream operations closing, and creating unlimited fresh generations that evade the advertised memory bound. Resource release must be atomic with retirement so admission cannot observe capacity before the corresponding registry state is gone. @@ -80,6 +84,8 @@ This differs from caching: the reusable result does not exist yet. Stress simultaneous identical and non-identical keys; inject upstream failures/timeouts; cancel the first caller while other waiters remain; cancel all waiters and verify the declared upstream-cancellation policy; race a new caller against the last-waiter cancellation transition and prove it never joins the closing generation; race a new caller against success/failure completion and prove the terminal generation is made non-joinable before waiter snapshot/outcome publication; test waiter-specific deadlines; verify shared failure publication and retry accounting; prove only one upstream evaluation occurs per joinable generation while all surviving callers terminate correctly. +Add an **availability/admission-floor fixture** before accepting any tuned cap. Replay the declared normal-load/reference trace against the non-coalesced or previously accepted baseline and record successful-admission/availability plus useful completed throughput. Then evaluate candidate caps, including deliberately degenerate zero/tiny generation or waiter limits that return overload quickly. Prove such candidates are rejected as infeasible whenever they fall below the declared service floor, even if they report excellent latency for admitted requests or near-zero duplicate work. Separately measure overload/rejection above the normal-load envelope as an explicit loss coordinate rather than silently excluding rejected requests from latency statistics. + Add a **cross-key/global-admission saturation fixture**. Generate many distinct equivalence keys so every request attempts to create its own generation, and separately churn through keys whose prior generations remain `closing` because upstream cancellation/cleanup is delayed. Fill the global and per-tenant generation/waiter budgets to their limits, race additional first callers and later waiters, and prove admission is linearizable: counts never exceed the configured bounds, retained closing generations continue to occupy capacity until retirement, no first waiter bypasses the global cap, and every non-admitted caller receives exactly the declared overload/backpressure behavior. Repeat with multiple tenants to verify one tenant cannot consume capacity reserved for another when per-tenant isolation is part of C. Add an explicit **registration-to-launch cancellation race**. Pause after the generation and initiating waiter have been registered but before `unstarted -> running`. Cancel or time out that initiating waiter as the last live waiter, then release the launcher. Prove the generation becomes closing/non-joinable and upstream work is never started. In the opposite interleaving, let `unstarted -> running` win but pause before the external invocation exists; then cancel the last waiter and prove the sticky token is already set/observable so the subsequent invocation is suppressed or immediately canceled according to policy. Repeat under high contention and prove no zero-waiter generation can leak a running/hung upstream operation. @@ -98,12 +104,12 @@ Add per-generation waiter-overflow races: fill one generation's waiter list to o ## Target-repo adaptation -Define key canonicalization, the authorization/visibility context that participates in equivalence, **global and per-tenant maximum in-flight generation counts, total waiter-record limits, whether closing/terminal cleanup consumes those limits, any separately bounded overflow queue**, per-generation waiter count, bounded overload/backpressure semantics, result ownership/share-safety policy, how mutable per-waiter results are prepared and how preparation failures surface, the waiter-owned atomic completion cell/transactional delivery representation, cancellation/deadline winning semantics, outcome-retention/acknowledgement and notifier retry semantics, the generation launch states and serialization primitive for `unstarted -> running` versus `closing`, the sticky cancellation-token/handle semantics used before an external operation object exists, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup/retirement and capacity release, and whether failures are shared as terminal or retried under one explicit shared retry policy. +Define key canonicalization, the authorization/visibility context that participates in equivalence, the **normal-load/reference demand envelope and minimum successful-admission/availability plus useful-throughput floor**, **global and per-tenant maximum in-flight generation counts, total waiter-record limits, whether closing/terminal cleanup consumes those limits, any separately bounded overflow queue**, per-generation waiter count, bounded overload/backpressure semantics, result ownership/share-safety policy, how mutable per-waiter results are prepared and how preparation failures surface, the waiter-owned atomic completion cell/transactional delivery representation, cancellation/deadline winning semantics, outcome-retention/acknowledgement and notifier retry semantics, the generation launch states and serialization primitive for `unstarted -> running` versus `closing`, the sticky cancellation-token/handle semantics used before an external operation object exists, the exact condition for canceling upstream work, the atomic create-with-first-waiter rule, the atomic closing/terminal non-joinable transitions, cleanup/retirement and capacity release, and whether failures are shared as terminal or retried under one explicit shared retry policy. ## Failure modes -Over-broad keys merge non-equivalent or authorization-distinct work; **per-generation-only waiter caps can still permit unbounded total memory under high-cardinality keys or churned closing generations**; releasing admission capacity before a closing generation is truly retired can let registry state exceed the advertised bound; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; registering first but launching from stale creator state after the last waiter already closed an unstarted generation can leak ownerless work; cancellation issued before an external operation exists can be lost without a sticky token or atomic launch state; non-linearized cancellation/deadline versus terminal publication can produce late or double outcomes; **marking a waiter terminal before atomically storing/enqueuing its outcome can strand it if notification fails**; cleanup that retires a published outcome before it is safely consumable can lose delivery; claiming success before a mutable per-waiter value is successfully prepared can strand a waiter with no deliverable result; clone/preparation failure can race cancellation and create inconsistent outcomes if not published atomically; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. +Over-broad keys merge non-equivalent or authorization-distinct work; **admission caps can game the objective by rejecting normal demand unless baseline availability/successful admission and useful throughput are constrained**; per-generation-only waiter caps can still permit unbounded total memory under high-cardinality keys or churned closing generations; releasing admission capacity before a closing generation is truly retired can let registry state exceed the advertised bound; launching upstream work before registering the initiating waiter can strand that caller on synchronous completion; registering first but launching from stale creator state after the last waiter already closed an unstarted generation can leak ownerless work; cancellation issued before an external operation exists can be lost without a sticky token or atomic launch state; non-linearized cancellation/deadline versus terminal publication can produce late or double outcomes; **marking a waiter terminal before atomically storing/enqueuing its outcome can strand it if notification fails**; cleanup that retires a published outcome before it is safely consumable can lose delivery; claiming success before a mutable per-waiter value is successfully prepared can strand a waiter with no deliverable result; clone/preparation failure can race cancellation and create inconsistent outcomes if not published atomically; coupling shared lifetime to the first caller can terminate valid waiters; leaving a canceled or terminal generation joinable can attach new callers to doomed/completed work; omitting authorization scope can leak results across principals/tenants; sharing a mutable result object can create cross-caller aliasing; undefined overflow semantics can exceed memory bounds, drop callers, or recreate duplicate upstream load; never canceling after all waiters leave can leak work; a hung upstream operation can stall many callers; ambiguous retry/error policy can cause correlated or duplicated work. ## Rollback trigger -Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/preparation-error semantics; if **global/per-tenant in-flight generation, waiter-record, or overflow-queue limits can be exceeded across many keys or retained closing generations**; if admission capacity can be reused before the corresponding generation/waiter state is actually retired; if an upstream operation can start after its generation has become closing with no live waiters; if a pre-launch cancellation can be lost because no cancellation token/operation object existed yet; if any success/error terminal state can become visible before its complete outcome is irrevocably stored/enqueued; if notifier/wakeup/callback failure after terminal publication can make the outcome inaccessible; if cleanup can retire an undelivered/unacknowledged outcome contrary to the target retention contract; if a waiter can enter terminal success before an isolated deliverable result exists; if clone/preparation failure can produce no terminal outcome or a second terminal outcome; if a cancelled/timed-out waiter can later be replaced by success/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. +Disable if coalescing changes any caller's authorization/cancellation/deadline/result/ownership/preparation-error semantics; if normal-load successful-admission/availability or useful throughput falls below the declared baseline/service floor; if objective reporting excludes overload/rejection in a way that rewards denying service; if **global/per-tenant in-flight generation, waiter-record, or overflow-queue limits can be exceeded across many keys or retained closing generations**; if admission capacity can be reused before the corresponding generation/waiter state is actually retired; if an upstream operation can start after its generation has become closing with no live waiters; if a pre-launch cancellation can be lost because no cancellation token/operation object existed yet; if any success/error terminal state can become visible before its complete outcome is irrevocably stored/enqueued; if notifier/wakeup/callback failure after terminal publication can make the outcome inaccessible; if cleanup can retire an undelivered/unacknowledged outcome contrary to the target retention contract; if a waiter can enter terminal success before an isolated deliverable result exists; if clone/preparation failure can produce no terminal outcome or a second terminal outcome; if a cancelled/timed-out waiter can later be replaced by success/error; if one waiter can observe two terminal outcomes; if an expired deadline can lose merely because timeout processing was delayed; if authorization-distinct requests are merged without independent delivery authorization; if one caller can cancel work required by another; if the initiating caller is stranded on immediate completion; if a new caller joins a closing/terminal generation; if mutable-result aliasing is possible; if shared operations leak; or if tail latency/failure amplification becomes unacceptable. From 3ceb5b9f82d213e781dd890a2b87397c87b9a1a1 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 01:06:32 +0930 Subject: [PATCH 053/229] Harden inline and table Markdown parsing --- scripts/check_catalog.py | 235 +++++++++++++++++++++++++++++++++++++-- 1 file changed, 228 insertions(+), 7 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 2112524..8535524 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -87,8 +87,6 @@ LINK_REFERENCE_DEFINITION_RE = re.compile( r"^\[(?:\\.|[^\[\]\\])+\]:[ \t]+\S.*$" ) -INLINE_IMAGE_RE = re.compile(r"!\[([^\]]*)\]\([^)]*\)") -INLINE_LINK_RE = re.compile(r"\[([^\]]*)\]\([^)]*\)") REFERENCE_IMAGE_RE = re.compile(r"!\[([^\]]*)\]\[[^\]]*\]") REFERENCE_LINK_RE = re.compile(r"\[([^\]]*)\]\[[^\]]*\]") INLINE_HTML_TAG_RE = re.compile( @@ -309,13 +307,235 @@ def section_lines(text: str, heading: str) -> list[str]: return lines[start:end] +def is_backslash_escaped(text: str, index: int) -> bool: + count = 0 + cursor = index - 1 + while cursor >= 0 and text[cursor] == "\\": + count += 1 + cursor -= 1 + return count % 2 == 1 + + +def backtick_run_length(text: str, index: int) -> int: + cursor = index + while cursor < len(text) and text[cursor] == "`": + cursor += 1 + return cursor - index + + +def protect_code_spans(text: str) -> tuple[str, dict[str, str]]: + """Replace parsed code spans with private-use sentinels and preserve their text.""" + out: list[str] = [] + protected: dict[str, str] = {} + i = 0 + while i < len(text): + if text[i] != "`" or is_backslash_escaped(text, i): + out.append(text[i]) + i += 1 + continue + + run_len = backtick_run_length(text, i) + j = i + run_len + close_start: int | None = None + close_end: int | None = None + while j < len(text): + if text[j] != "`": + j += 1 + continue + candidate_len = backtick_run_length(text, j) + if candidate_len == run_len: + close_start = j + close_end = j + candidate_len + break + j += candidate_len + + if close_start is None or close_end is None: + out.append(text[i : i + run_len]) + i += run_len + continue + + token = chr(0xE000 + len(protected)) + protected[token] = text[i + run_len : close_start] + out.append(token) + i = close_end + + return "".join(out), protected + + +def find_label_close(text: str, open_index: int) -> int | None: + depth = 1 + i = open_index + 1 + while i < len(text): + if text[i] == "\\" and i + 1 < len(text): + i += 2 + continue + if text[i] == "[": + depth += 1 + elif text[i] == "]": + depth -= 1 + if depth == 0: + return i + i += 1 + return None + + +def parse_link_title_and_close(text: str, index: int) -> int | None: + """Parse whitespace plus an optional CommonMark-style title and outer close.""" + i = index + while i < len(text) and text[i] in " \t\n": + i += 1 + if i < len(text) and text[i] == ")": + return i + 1 + if i >= len(text): + return None + + opener = text[i] + if opener not in ('"', "'", "("): + return None + closer = ")" if opener == "(" else opener + i += 1 + while i < len(text): + if text[i] == "\\" and i + 1 < len(text): + i += 2 + continue + if text[i] == closer: + i += 1 + break + if text[i] == "\n": + return None + i += 1 + else: + return None + + while i < len(text) and text[i] in " \t\n": + i += 1 + if i < len(text) and text[i] == ")": + return i + 1 + return None + + +def find_inline_link_end(text: str, open_paren: int) -> int | None: + """Return the end of a valid inline-link destination/title, or None.""" + i = open_paren + 1 + while i < len(text) and text[i] in " \t\n": + i += 1 + if i >= len(text): + return None + if text[i] == ")": + return i + 1 + + if text[i] == "<": + i += 1 + while i < len(text): + if text[i] == "\\" and i + 1 < len(text): + i += 2 + continue + if text[i] == ">": + return parse_link_title_and_close(text, i + 1) + if text[i] in "\n<": + return None + i += 1 + return None + + depth = 0 + while i < len(text): + char = text[i] + if char == "\\" and i + 1 < len(text): + i += 2 + continue + if char == "(": + depth += 1 + i += 1 + continue + if char == ")": + if depth == 0: + return i + 1 + depth -= 1 + i += 1 + continue + if char in " \t\n" and depth == 0: + return parse_link_title_and_close(text, i) + if char in "<>" or ord(char) < 0x20: + return None + i += 1 + return None + + +def strip_inline_links(text: str) -> str: + """Keep rendered labels while discarding valid inline-link/image destinations.""" + out: list[str] = [] + i = 0 + while i < len(text): + image = text.startswith("![", i) + if image: + label_open = i + 1 + elif text[i] == "[": + label_open = i + else: + out.append(text[i]) + i += 1 + continue + + label_close = find_label_close(text, label_open) + if label_close is None or label_close + 1 >= len(text) or text[label_close + 1] != "(": + out.append(text[i]) + i += 1 + continue + link_end = find_inline_link_end(text, label_close + 1) + if link_end is None: + out.append(text[i]) + i += 1 + continue + + out.append(text[label_open + 1 : label_close]) + i = link_end + return "".join(out) + + def markdown_table_cells(line: str) -> list[str] | None: + """Split a pipe table on unescaped delimiters outside backtick code spans.""" if is_indented_code_line(line): return None stripped = line.strip() if not stripped.startswith("|"): return None - return [cell.strip() for cell in stripped.strip("|").split("|")] + + cells: list[str] = [] + current: list[str] = [] + code_run_len: int | None = None + i = 1 + while i < len(stripped): + char = stripped[i] + if char == "`" and not is_backslash_escaped(stripped, i): + run_len = backtick_run_length(stripped, i) + if code_run_len is None: + code_run_len = run_len + elif run_len == code_run_len: + code_run_len = None + current.append(stripped[i : i + run_len]) + i += run_len + continue + + if char == "|" and code_run_len is None: + if is_backslash_escaped(stripped, i): + if current and current[-1] == "\\": + current.pop() + current.append("|") + else: + cells.append("".join(current).strip()) + current = [] + i += 1 + continue + + current.append(char) + i += 1 + + trailing_pipe_is_delimiter = ( + stripped.endswith("|") and not is_backslash_escaped(stripped, len(stripped) - 1) + ) + if current or not trailing_pipe_is_delimiter: + cells.append("".join(current).strip()) + return cells def extract_markdown_table( @@ -385,15 +605,16 @@ def parse_record_link_cell(cell: str, context: str) -> tuple[str, str]: def rendered_inline_text(value: str) -> str: - """Approximate visible inline text for required field-value validation.""" - text = value - text = INLINE_IMAGE_RE.sub(lambda m: m.group(1), text) - text = INLINE_LINK_RE.sub(lambda m: m.group(1), text) + """Approximate rendered inline text for required field-value validation.""" + text, protected_code = protect_code_spans(value) + text = strip_inline_links(text) text = REFERENCE_IMAGE_RE.sub(lambda m: m.group(1), text) text = REFERENCE_LINK_RE.sub(lambda m: m.group(1), text) text = INLINE_HTML_TAG_RE.sub("", text) text = re.sub(r"[`*_~]", "", text) text = re.sub(r"\\(.)", r"\1", text) + for token, code_text in protected_code.items(): + text = text.replace(token, code_text) return html.unescape(text).strip() From 8c33ea9b671efa0721a817a9cc791be45410e88f Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 01:25:46 +0930 Subject: [PATCH 054/229] Harden catalog provenance and Markdown structure checks --- scripts/check_catalog.py | 144 +++++++++++++++++++++++++++++++++++---- 1 file changed, 129 insertions(+), 15 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 8535524..68a7b8f 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -44,11 +44,11 @@ "Variables": "continuous / integer / categorical / conditional / mixed", "Search scope": "local / global", "Objective behavior": "deterministic / noisy / stochastic", - "Information": "gradient / derivative-free / black-box", + "Information": "gradient available / derivative-free / black-box", "Evaluation cost": "cheap / moderate / expensive", "Constraints": "bounds / equality / inequality / semantic / resource", "Parallelism": "sequential / synchronous batch / asynchronous", - "Exactness": "exact / approximation permitted under explicit error contract", + "Exactness": "exact / approximation permitted under an explicit error contract", } ALLOWED_V2_STATUS_CATEGORIES = { "Verified", @@ -87,6 +87,17 @@ LINK_REFERENCE_DEFINITION_RE = re.compile( r"^\[(?:\\.|[^\[\]\\])+\]:[ \t]+\S.*$" ) +LINK_REFERENCE_TITLE_CONTINUATION_RE = re.compile( + r"^ {0,3}(?:\"(?:\\.|[^\"\\])*\"|'(?:\\.|[^'\\])*'|\((?:\\.|[^)\\])*\))[ \t]*$" +) +SOURCE_URL_RE = re.compile(r"https?://\S+", re.IGNORECASE) +SOURCE_DOI_RE = re.compile(r"\b(?:doi:\s*)?10\.\d{4,9}/\S+", re.IGNORECASE) +SOURCE_COMMIT_RE = re.compile(r"\b[0-9a-f]{7,40}\b", re.IGNORECASE) +SOURCE_LOCAL_NOTE_RE = re.compile(r"`?(sources/[A-Za-z0-9._/-]+\.md)`?") +SOURCE_REPOSITORY_RE = re.compile(r"`[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+`") +SOURCE_PLACEHOLDER_RE = re.compile( + r"^(?:[-*+]\s*)?(?:unknown|tbd|todo|n/?a|none|pending)\.?$", re.IGNORECASE +) REFERENCE_IMAGE_RE = re.compile(r"!\[([^\]]*)\]\[[^\]]*\]") REFERENCE_LINK_RE = re.compile(r"\[([^\]]*)\]\[[^\]]*\]") INLINE_HTML_TAG_RE = re.compile( @@ -281,15 +292,31 @@ def visible_text(text: str) -> str: return "\n".join(visible_nonfenced_lines(markdown_source_lines(text))) -def is_setext_section_boundary(lines: list[str], index: int) -> bool: - """Recognize a visible Setext H1/H2 beginning at lines[index].""" - if index + 1 >= len(lines): - return False - candidate = lines[index] - if not candidate.strip() or HEADING_RE.match(candidate): - return False - underline = lines[index + 1] - return bool(SETEXT_H1_RE.fullmatch(underline) or SETEXT_H2_RE.fullmatch(underline)) +def setext_heading_start( + lines: list[str], underline_index: int, minimum_index: int +) -> int | None: + """Return the first source line of a Setext heading paragraph.""" + if underline_index <= minimum_index: + return None + underline = lines[underline_index] + if not ( + SETEXT_H1_RE.fullmatch(underline) or SETEXT_H2_RE.fullmatch(underline) + ): + return None + + candidate = underline_index - 1 + if candidate < minimum_index or not lines[candidate].strip(): + return None + if HEADING_RE.match(lines[candidate]): + return None + + start = candidate + while start > minimum_index: + previous = lines[start - 1] + if not previous.strip() or SECTION_BOUNDARY_RE.match(previous): + break + start -= 1 + return start def section_lines(text: str, heading: str) -> list[str]: @@ -301,9 +328,13 @@ def section_lines(text: str, heading: str) -> list[str]: return [] end = len(lines) for i in range(start, len(lines)): - if SECTION_BOUNDARY_RE.match(lines[i]) or is_setext_section_boundary(lines, i): + if SECTION_BOUNDARY_RE.match(lines[i]): end = i break + setext_start = setext_heading_start(lines, i, start) + if setext_start is not None: + end = setext_start + break return lines[start:end] @@ -622,6 +653,21 @@ def has_substantive_rendered_text(value: str) -> bool: return any(ch.isalnum() for ch in rendered_inline_text(value)) +def reference_definition_hidden_indexes(lines: list[str]) -> set[int]: + """Return lines consumed by non-rendering reference definitions/titles.""" + hidden: set[int] = set() + for i, raw in enumerate(lines): + if not LINK_REFERENCE_DEFINITION_RE.fullmatch(raw.strip()): + continue + hidden.add(i) + if ( + i + 1 < len(lines) + and LINK_REFERENCE_TITLE_CONTINUATION_RE.fullmatch(lines[i + 1]) + ): + hidden.add(i + 1) + return hidden + + def is_structural_only_line(line: str) -> bool: if HEADING_RE.match(line) or THEMATIC_BREAK_RE.fullmatch(line): return True @@ -634,7 +680,11 @@ def is_structural_only_line(line: str) -> bool: def section_has_content(lines: list[str]) -> bool: - for raw in visible_nonfenced_lines(lines): + visible = visible_nonfenced_lines(lines) + hidden_reference_lines = reference_definition_hidden_indexes(visible) + for index, raw in enumerate(visible): + if index in hidden_reference_lines: + continue line = raw.strip() if not line or line in TEMPLATE_PLACEHOLDER_LINES: continue @@ -646,6 +696,22 @@ def section_has_content(lines: list[str]) -> bool: return False +def source_section_has_identity(lines: list[str]) -> bool: + """Require at least one concrete, non-placeholder provenance identity.""" + for raw in visible_nonfenced_lines(lines): + line = raw.strip() + if not line or SOURCE_PLACEHOLDER_RE.fullmatch(line): + continue + if SOURCE_URL_RE.search(line) or SOURCE_DOI_RE.search(line) or SOURCE_COMMIT_RE.search(line): + return True + for match in SOURCE_LOCAL_NOTE_RE.finditer(line): + if (ROOT / match.group(1)).is_file(): + return True + if SOURCE_REPOSITORY_RE.search(line): + return True + return False + + def normalized_status_category(raw: str) -> str: category = raw.split(";", 1)[0].strip() return unwrap_outer_formatting(category, STATUS_WRAPPERS) @@ -688,7 +754,7 @@ def require_prefixed_fields( records: dict[str, Path] = {} status_categories: dict[str, str] = {} -for path in sorted(OPT_DIR.glob("OPT-*.md")): +for path in sorted(OPT_DIR.glob("*.md")): text = path.read_text(encoding="utf-8") lines = visible_nonfenced_lines(markdown_source_lines(text)) first = lines[0] if lines else "" @@ -699,7 +765,10 @@ def require_prefixed_fields( filename_match = FILENAME_ID_RE.match(path.name) if not filename_match: - die(f"record filename does not begin with an OPT ID: {path.relative_to(ROOT)}") + die( + f"record Markdown filename does not follow OPT---... convention: " + f"{path.relative_to(ROOT)}" + ) if filename_match.group(1) != record_id: die( f"record ID mismatch: {path.relative_to(ROOT)} declares {record_id} " @@ -737,6 +806,13 @@ def require_prefixed_fields( f"{path.relative_to(ROOT)} has empty/template/structural/markup-only mandatory section {heading}" ) + source_evidence = section_lines(text, "## Source evidence") + if not source_section_has_identity(source_evidence): + die( + f"{path.relative_to(ROOT)} ## Source evidence lacks a concrete source identity " + "(URL, DOI, pinned commit, existing sources/*.md note, or repository identity)" + ) + contract = section_lines(text, "## Optimization problem contract") require_prefixed_fields( path, contract, REQUIRED_CONTRACT_FIELDS, "## Optimization problem contract" @@ -888,4 +964,42 @@ def require_prefixed_fields( if not any(pattern.match(line) for line in canonical): die(f"OPTIMIZATION-PROBLEM.md is missing visible canonical definition for {field}") +classification_lines = section_lines(problem_text, "## Required classification") +if not classification_lines: + die("OPTIMIZATION-PROBLEM.md is missing visible ## Required classification") +classification_rows = extract_markdown_table( + classification_lines, + ("Dimension", "Typical values"), + "OPTIMIZATION-PROBLEM.md ## Required classification", +) +canonical_classification: dict[str, str] = {} +for row in classification_rows: + dimension, typical_values = row + if dimension in canonical_classification: + die(f"OPTIMIZATION-PROBLEM.md has duplicate classification dimension {dimension}") + canonical_classification[dimension] = typical_values + +expected_dimensions = set(REQUIRED_CLASSIFICATION_FIELDS) +observed_dimensions = set(canonical_classification) +if observed_dimensions != expected_dimensions: + missing_dimensions = sorted(expected_dimensions - observed_dimensions) + unknown_dimensions = sorted(observed_dimensions - expected_dimensions) + details: list[str] = [] + if missing_dimensions: + details.append(f"missing={','.join(missing_dimensions)}") + if unknown_dimensions: + details.append(f"unknown={','.join(unknown_dimensions)}") + die( + "OPTIMIZATION-PROBLEM.md classification dimensions do not match the checker: " + + "; ".join(details) + ) +for dimension in REQUIRED_CLASSIFICATION_FIELDS: + canonical_value = canonical_classification[dimension] + checker_value = CLASSIFICATION_TEMPLATE_VALUES[dimension] + if canonical_value != checker_value: + die( + f"classification placeholder drift for {dimension}: " + f"canonical='{canonical_value}' checker='{checker_value}'" + ) + print(f"CATALOG_INTEGRITY_OK records={len(records)} frozen_v1={len(FROZEN_V1)}") From 246a26e83b41aa72205eacbfc09cd640203079d5 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 02:02:24 +0930 Subject: [PATCH 055/229] Harden catalog HTML and provenance checks --- scripts/check_catalog.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 68a7b8f..0142fb0 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -207,7 +207,8 @@ def raw_html_block_start(raw: str) -> tuple[str, str | None] | None: def raw_html_tag_closes(raw: str, tag: str) -> bool: - return re.search(rf"", raw, re.IGNORECASE) is not None + """Match CommonMark type-1 block terminators exactly (case-insensitive).""" + return re.search(rf"", raw, re.IGNORECASE) is not None def visible_nonfenced_lines(lines: list[str]) -> list[str]: @@ -698,6 +699,7 @@ def section_has_content(lines: list[str]) -> bool: def source_section_has_identity(lines: list[str]) -> bool: """Require at least one concrete, non-placeholder provenance identity.""" + sources_root = (ROOT / "sources").resolve() for raw in visible_nonfenced_lines(lines): line = raw.strip() if not line or SOURCE_PLACEHOLDER_RE.fullmatch(line): @@ -705,7 +707,12 @@ def source_section_has_identity(lines: list[str]) -> bool: if SOURCE_URL_RE.search(line) or SOURCE_DOI_RE.search(line) or SOURCE_COMMIT_RE.search(line): return True for match in SOURCE_LOCAL_NOTE_RE.finditer(line): - if (ROOT / match.group(1)).is_file(): + candidate = (ROOT / match.group(1)).resolve() + try: + candidate.relative_to(sources_root) + except ValueError: + continue + if candidate.is_file(): return True if SOURCE_REPOSITORY_RE.search(line): return True From 30719ec4a97dc2db63c478ab19a9665d2493c96a Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 02:04:24 +0930 Subject: [PATCH 056/229] Determinize budget-limited parallel pruning --- ...E-001-bound-driven-search-space-pruning.md | 24 ++++++++++++------- 1 file changed, 16 insertions(+), 8 deletions(-) diff --git a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md index 83031a2..6b89627 100644 --- a/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md +++ b/optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md @@ -19,16 +19,16 @@ A discrete or mixed search space is too large for exhaustive evaluation, but who - F: candidates in X satisfying every original hard constraint; relaxed/bounding solutions are not feasible final answers unless they also lie in F - f: a scalar real-valued target objective `f : F → R` evaluated on feasible candidates only - d: exactly one of scalar `minimize` or scalar `maximize`; vector, Pareto, lexicographic, or other partial-order objectives are outside this record unless a separately specified and validated frontier-bound mechanism is introduced -- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, **exact pruning is authorized only when the soundness argument for `b` covers the deployed search domain through an analytic/formal proof, exhaustive verification of the complete finite deployed domain, or a conservative runtime proof/certificate checked for each pruned region**, the target's observable tie semantics are preserved, parallel dispatch cannot oversubscribe the declared hard budget, every unit of resource consumption covered by a hard wall-time/compute B is accounted for or enclosed by an enforceable whole-search cap, every hard peak-memory B is enforced over live allocated/reserved memory rather than cumulative historical allocation, and frontier exhaustion is declared only after all queued **and leased/in-flight** regions are accounted for -- B: a finite, predeclared target-specific **enforceable** cap with its accounting semantics declared explicitly. Evaluation count, money/provider spend, CPU/GPU-seconds, energy, bytes transferred, or other cumulative-flow resources use cumulative accounting. Elapsed wall time uses one shared whole-search deadline. **Peak memory is a stock constraint, not a cumulative flow:** enforce `live_allocated + live_reserved + proposed <= B`, release live capacity when memory is freed, and retain only a recorded `peak_observed` for evidence. If a target instead wants cumulative allocation traffic, it must declare that as a distinct cumulative metric rather than calling it peak memory. Any resource dimension that cannot be hard-capped under its declared semantics must be labeled observational/best-effort rather than advertised as hard B -- S: stop immediately when the required optimality/tie contract is proven, or when the **global frontier is exhausted**, meaning there are no queued regions, no leased/in-flight regions still capable of producing candidates/children, and no unpublished child/frontier updates owned by active work. Otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality +- C: every returned incumbent satisfies the original feasibility/semantic contract, every pruning decision is justified by a separately defined sound scalar region-bound function `b`, **exact pruning is authorized only when the soundness argument for `b` covers the deployed search domain through an analytic/formal proof, exhaustive verification of the complete finite deployed domain, or a conservative runtime proof/certificate checked for each pruned region**, the target's observable tie semantics are preserved, parallel dispatch cannot oversubscribe the declared hard budget, every unit of resource consumption covered by a hard wall-time/compute B is accounted for or enclosed by an enforceable whole-search cap, every hard peak-memory B is enforced over live allocated/reserved memory rather than cumulative historical allocation, frontier exhaustion is declared only after all queued **and leased/in-flight** regions are accounted for, and **when deterministic budget-limited output is part of C, worker timing may not decide which logical regions receive the final budget entitlement or which completed results become the returned anytime state** +- B: a finite, predeclared target-specific **enforceable** cap with its accounting semantics declared explicitly. Evaluation count, money/provider spend, CPU/GPU-seconds, energy, bytes transferred, or other cumulative-flow resources use cumulative accounting. Elapsed wall time uses one shared whole-search deadline. **Peak memory is a stock constraint, not a cumulative flow:** enforce `live_allocated + live_reserved + proposed <= B`, release live capacity when memory is freed, and retain only a recorded `peak_observed` for evidence. If a target instead wants cumulative allocation traffic, it must declare that as a distinct cumulative metric rather than calling it peak memory. Any resource dimension that cannot be hard-capped under its declared semantics must be labeled observational/best-effort rather than advertised as hard B. If deterministic anytime output is required, also declare the deterministic logical work/budget boundary used to select the returned result; elapsed wall time alone is not a deterministic selection boundary under variable worker timing +- S: stop immediately when the required optimality/tie contract is proven, or when the **global frontier is exhausted**, meaning there are no queued regions, no leased/in-flight regions still capable of producing candidates/children, and no unpublished child/frontier updates owned by active work. Otherwise stop when B is exhausted. If a validated incumbent exists, return it plus any remaining valid global bound/optimality gap. If no feasible incumbent exists, return `no-incumbent / feasibility-unknown` and only a separately valid global bound if one is available; do not report an optimality gap that requires an incumbent, and do not claim infeasibility or optimality. **For deterministic targets, budget exhaustion must expose the result of a canonical logical prefix/batch of search work rather than an arbitrary prefix induced by completion order; if the only stopping boundary is physical elapsed time, budget-limited anytime results must be explicitly declared nondeterministic or selected from a separately declared deterministic checkpoint that was committed before the deadline** - Variables: integer / categorical / discrete / mixed - Search scope: global over the declared candidate space - Objective behavior: deterministic unless uncertainty/noise is incorporated into a separately sound bound model - Information: derivative-free; bound/relaxation information is target-specific - Evaluation cost: moderate to expensive when exhaustive evaluation is infeasible -- Constraints: feasibility, semantic correctness, **deployed-domain scalar-bound soundness**, tie semantics, global-frontier accounting, and enforceable finite-resource constraints including correctly typed cumulative, deadline, and peak/live-capacity budgets -- Parallelism: sequential, or parallel only with synchronized incumbent/frontier/bound state, **leased/in-flight region accounting**, linearizable reservation/completion accounting for cumulative resources, a shared absolute deadline for elapsed wall time, and live-allocation reservation accounting for hard peak-memory caps +- Constraints: feasibility, semantic correctness, **deployed-domain scalar-bound soundness**, tie semantics, global-frontier accounting, deterministic logical budget-boundary semantics when required, and enforceable finite-resource constraints including correctly typed cumulative, deadline, and peak/live-capacity budgets +- Parallelism: sequential, or parallel only with synchronized incumbent/frontier/bound state, **leased/in-flight region accounting**, linearizable reservation/completion accounting for cumulative resources, a shared absolute deadline for elapsed wall time, live-allocation reservation accounting for hard peak-memory caps, and—when deterministic output is required—**stable region/work IDs plus a deterministic leasing/reservation/assimilation boundary whose logical order is independent of worker completion timing** - Exactness: exact only when the declared optimality and observable-tie contract is proven within B **and the pruning bound's soundness is justified over the complete deployed domain or certified conservatively at runtime for every pruned region**, including proof that no queued or leased region can still affect the answer; otherwise anytime/incomplete result semantics apply For each unexplored region `R`, define a bound `b(R)` separately from `f`: @@ -51,12 +51,18 @@ An independently proven infeasible region may also be pruned. A heuristic estima A region may be discarded only when its sound bound proves it cannot contain any candidate that remains observably preferable or required under the target's scalar objective **and tie contract**, and the mechanism used to justify that bound is valid over the deployed domain/region being pruned. Passing a finite regression suite does not turn a heuristic estimate into a proof bound. Exhausting B without an optimality proof does not permit an exactness claim, exhausting B without a feasible incumbent does not permit an infeasibility claim, and parallel execution must preserve the same hard resource ceiling as sequential execution rather than oversubscribing work in flight. A temporarily empty shared queue is **not** frontier exhaustion while any worker owns a leased region that may still produce a candidate, proof obligation, or child region. Likewise, a hard wall-time/compute B applies to the whole search, while a hard peak-memory B applies to current live/reserved memory and must not be converted into irreversible historical consumption after memory is freed. +When the target requires deterministic output, that requirement also applies to an **incomplete/anytime result produced at B**. The parallel implementation must therefore make the logical search prefix independent of worker speed: the same input, seed, contract and logical budget must commit the same ordered work set, incumbent/tie winner, valid global bound/gap and stop reason as the declared canonical reference. Physical completion may occur out of order, but it cannot grant a later region the last logical budget entitlement merely because that worker finished first. If deterministic budget-limited output is not required, the target must state that nondeterminism explicitly instead of inheriting a deterministic objective classification by implication. + ## Optimization Maintain an incumbent when one exists, partition the search space, compute a cheap **sound and deployed-domain-justified** `b(R)` for each region (often from a relaxation), prioritize promising regions, and prune only when the direction-specific bound plus the target's tie semantics prove the region cannot affect the required answer. Before the first incumbent exists, sound bounds may prioritize regions or prove individual regions infeasible, but incumbent-based objective pruning is unavailable. For **parallel** search, define one global frontier lifecycle. A region remains part of the frontier from enqueue until it is either (a) soundly pruned/closed, or (b) replaced by its child regions through an atomic/linearizable completion transition. Dequeuing for worker ownership therefore changes a region from `queued` to `leased/in-flight`; it does **not** remove that region from the global frontier. A worker that branches a leased region must publish all resulting children and close/release the parent as one frontier-accounting transition, or use another protocol that cannot expose a moment where the queue is empty even though unpublished descendants still exist. Worker failure/cancellation must return or recover the lease so unexplored work is not silently lost. +For targets requiring **deterministic budget-limited results**, give every frontier item/work operation a stable logical ID and define a canonical total leasing/commit order, including deterministic tie-breaks. Reserve logical budget entitlement in that canonical order **before dispatch**, or dispatch fixed deterministic batches whose membership is independent of completion timing. Results may finish physically out of order, but logical incumbent/bound/frontier assimilation must occur through the same canonical ordered prefix (or an equivalent deterministic batch barrier). A later completed result is buffered or remains speculative until every earlier entitled item needed for the prefix has been resolved; a free worker does not opportunistically admit a new logical item if doing so would change which work fits inside B. For variable-cost cumulative resources, conservative enforceable reservations participate in the canonical entitlement decision, so the set of admitted work is determined by the declared ordering and reservation amounts rather than by which prior worker happened to finish first. + +A hard **elapsed wall-time** deadline is different: scheduler and worker speed determine which physical operations finish before the clock expires. If the target nevertheless requires deterministic anytime output, use the deadline only as a safety envelope and select the public result from the most recent fully committed **deterministic logical checkpoint/prefix** defined by a separate logical work budget or batch schedule. Otherwise explicitly classify the wall-time-cutoff anytime result as nondeterministic while retaining deterministic exact/full-exhaustion semantics where those are otherwise required. Do not claim scalar/parallel budget-exhaustion equivalence when the public cutoff itself is completion-time-driven. + Treat hard resources according to their physical/accounting semantics rather than forcing them through one ledger shape: - **Cumulative-flow budgets** such as evaluation count, money/provider spend, CPU/GPU-seconds, energy, or declared cumulative bytes use linearizable reservations. Atomically reserve before covered work starts; if `consumed + reserved + proposed_reservation > B`, do not start it. Completion/failure/cancellation moves actual consumed usage into permanent `consumed` and releases only demonstrably unconsumed reservation. @@ -85,18 +91,20 @@ Add **parallel frontier-exhaustion races**. Use a fixture where the last queued Add **parallel budget-boundary fixtures**. Race multiple workers against one remaining evaluation slot and prove only one reservation succeeds for an evaluation-count B. Race bound evaluations and candidate evaluations against the same final cumulative capacity and prove both charge the declared ledger. For elapsed wall time, run overlapping workers to the same absolute deadline and prove overlap is not double-counted while no worker/coordinator survives past the enforced deadline. For cumulative compute/spend, deliberately make branching/frontier/serialization work expensive and prove it is charged before the cap is exceeded. +For a target that declares **deterministic anytime output**, repeat budget-exhaustion fixtures under deliberately permuted worker speeds, completion orders, pauses and wakeups. Hold the same input, seed, logical B and worker-count policy constant. Prove the same stable region/work IDs receive logical budget entitlement in the same canonical order or deterministic batches, the same logical prefix is assimilated, and the returned incumbent/tie winner, valid bound/gap and stop reason match the scalar/canonical reference. Include the exact last-slot race where a later canonical region finishes before an earlier one: the later completion must not steal the final logical entitlement or become publicly assimilated ahead of the declared prefix. For a hard elapsed deadline, verify either (a) the public result comes from the same last fully committed deterministic checkpoint despite completion-order changes, or (b) the target explicitly declares wall-time-cutoff anytime results nondeterministic and tests only the deterministic guarantees it actually claims. + Add a **peak-memory reuse fixture**. Under a hard peak-memory B, run many sequential/non-overlapping branches that each allocate and then free a large work buffer. Prove each live allocation/reservation is admitted only while `live_allocated + live_reserved <= B`, freed capacity becomes reusable, `peak_observed <= B`, and the search does **not** exhaust merely because the sum of historical allocations exceeds B. Then overlap enough workers to exceed the peak if all allocations were admitted and prove the final reservation is rejected/blocked before live memory can cross B. Race allocation, free, cancellation and cleanup to verify live-memory accounting remains linearizable and no capacity is released before the corresponding memory is actually reclaimable. Add **equal-objective tie fixtures**. For an any-one-optimum contract, prove equality pruning cannot alter any observable result. For deterministic tie-winner contracts, construct regions containing equal-objective candidates with better/worse tie ranks and prove equality-bound regions are retained until the declared tie winner is established. For all-optima contracts, prove every equal-objective optimum is enumerated. If using a stronger total-order bound, validate its soundness independently against exhaustive fixtures and establish the same deployed-domain proof basis before using it for exact pruning. ## Target-repo adaptation -The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. **Document the deployed-domain soundness basis for every bound used to prune in exact mode:** identify the analytic/formal theorem and assumptions, the complete finite domain exhausted by verification, or the runtime certificate/guard and its conservative fallback semantics. Treat small fixture results as regression evidence, not as the proof basis for a larger domain. For parallel implementations, define the global frontier state machine, lease ownership/recovery rules, parent-close/child-publish atomicity, and the exact exhaustion predicate over queued plus leased/in-flight work. Also define each budget's accounting type and enforcement boundary: cumulative flow (`consumed + reserved`), elapsed deadline, or peak/live capacity (`live_allocated + live_reserved`, plus `peak_observed`). If a target says “memory budget,” state whether it means peak live memory or cumulative allocation traffic. Downgrade any dimension that can escape its correct enforcement boundary to best-effort/observational rather than calling it hard B. +The quality/cost of bounds determines whether pruning helps. Develop target-specific scalar relaxations, branch ordering, feasible-candidate discovery strategy, **tie/secondary-order semantics**, and a finite resource cap before execution; do not assume one bound or budget is universally appropriate. **Document the deployed-domain soundness basis for every bound used to prune in exact mode:** identify the analytic/formal theorem and assumptions, the complete finite domain exhausted by verification, or the runtime certificate/guard and its conservative fallback semantics. Treat small fixture results as regression evidence, not as the proof basis for a larger domain. For parallel implementations, define the global frontier state machine, lease ownership/recovery rules, parent-close/child-publish atomicity, and the exact exhaustion predicate over queued plus leased/in-flight work. Also define each budget's accounting type and enforcement boundary: cumulative flow (`consumed + reserved`), elapsed deadline, or peak/live capacity (`live_allocated + live_reserved`, plus `peak_observed`). If a target says “memory budget,” state whether it means peak live memory or cumulative allocation traffic. **Declare whether budget-limited anytime output is required to be deterministic.** If it is, specify stable region/work IDs, the canonical lease/reservation and assimilation order (or deterministic batches), the scalar/canonical reference used for equivalence, and—when elapsed wall time is a hard safety envelope—the separate deterministic logical checkpoint/budget from which the public result is selected. If wall-time-cutoff anytime results are intentionally nondeterministic, state that contract change explicitly. Downgrade any dimension that can escape its correct enforcement boundary to best-effort/observational rather than calling it hard B. ## Failure modes -Unsound bounds can remove the true optimum; **small exhaustive fixtures can all pass while a bound remains unsound elsewhere in a larger deployed domain**; a runtime certificate that is not conservative or whose failure path still prunes destroys exactness; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; treating queue-empty as frontier-empty can declare exact completion while a leased region still owns unexplored descendants; losing a worker lease can silently drop search regions; non-atomic parent-close/child-publication can create false exhaustion; parallel workers without linearizable evaluation reservations can oversubscribe the last evaluation slot; branching/child/frontier/serialization/incumbent overhead can exceed a nominal wall-time/compute B if only evaluations are charged; an unenforced coordinator/cleanup path can outlive a claimed whole-search deadline; treating peak memory as cumulative consumed spend can falsely exhaust a valid search and prevent reuse of freed capacity; releasing live-memory capacity before actual reclamation can instead oversubscribe the peak; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. +Unsound bounds can remove the true optimum; **small exhaustive fixtures can all pass while a bound remains unsound elsewhere in a larger deployed domain**; a runtime certificate that is not conservative or whose failure path still prunes destroys exactness; weak bounds provide little pruning; expensive bounds can cost more than evaluation; numeric tolerance errors can create incorrect pruning; heuristic scores mislabeled as bounds invalidate the proof obligation; equality pruning can discard a required deterministic tie winner or additional optimum; applying scalar pruning logic to vector/Pareto objectives can discard nondominated candidates; treating queue-empty as frontier-empty can declare exact completion while a leased region still owns unexplored descendants; losing a worker lease can silently drop search regions; non-atomic parent-close/child-publication can create false exhaustion; parallel workers without linearizable evaluation reservations can oversubscribe the last evaluation slot; **worker completion timing can choose the last admitted region or anytime incumbent even though C requires deterministic output**; completion-driven assimilation can make the budget-limited parallel prefix differ from the scalar/canonical prefix; using elapsed wall time itself as a deterministic result-selection boundary can make output scheduler-dependent; branching/child/frontier/serialization/incumbent overhead can exceed a nominal wall-time/compute B if only evaluations are charged; an unenforced coordinator/cleanup path can outlive a claimed whole-search deadline; treating peak memory as cumulative consumed spend can falsely exhaust a valid search and prevent reuse of freed capacity; releasing live-memory capacity before actual reclamation can instead oversubscribe the peak; treating an unenforceable resource target as hard B makes the stopping contract false; treating budget exhaustion without an incumbent as evidence of infeasibility is unsound. ## Rollback trigger -Disable any pruning rule that lacks a valid deployed-domain soundness basis for exact mode, whose analytic/formal assumptions fail, whose exhaustive finite-domain verification no longer covers the deployed domain/region construction, whose runtime certificate can authorize an unsound prune or fails open, that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel/exact mode if frontier exhaustion can be observed while any leased/in-flight region may still produce work, if parent-close/child-publication or lease recovery can lose unexplored regions, if workers can oversubscribe an evaluation-count/cumulative budget, if an elapsed deadline can be reset/escaped, if peak live memory can exceed B, if freed peak-memory capacity is incorrectly made permanently unavailable, or if any resource-consumption path can escape a dimension advertised as hard. Abort exact-mode claims whenever B is exhausted before the full objective/tie/frontier contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. +Disable any pruning rule that lacks a valid deployed-domain soundness basis for exact mode, whose analytic/formal assumptions fail, whose exhaustive finite-domain verification no longer covers the deployed domain/region construction, whose runtime certificate can authorize an unsound prune or fails open, that fails exhaustive small-case validation, violates the declared scalar/tie-bound relation, is applied to an unsupported objective ordering, discards an equal-objective candidate required by C, or whose bound cost exceeds the work it eliminates. Abort parallel/exact mode if frontier exhaustion can be observed while any leased/in-flight region may still produce work, if parent-close/child-publication or lease recovery can lose unexplored regions, if workers can oversubscribe an evaluation-count/cumulative budget, if an elapsed deadline can be reset/escaped, if peak live memory can exceed B, if freed peak-memory capacity is incorrectly made permanently unavailable, or if any resource-consumption path can escape a dimension advertised as hard. **Abort deterministic parallel anytime mode if worker timing can change logical reservation/lease entitlement, assimilation order, the committed budget prefix, returned incumbent/tie winner, valid bound/gap, or stop reason relative to the declared canonical reference.** Abort exact-mode claims whenever B is exhausted before the full objective/tie/frontier contract is proven, and reject any implementation that converts a no-incumbent budget timeout into an infeasibility or optimality claim without a separate proof. From bb580dbaad021124b56b71c29e10fad8463e369f Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 02:54:09 +0930 Subject: [PATCH 057/229] Fix optimization record classification placeholders --- templates/OPTIMIZATION-RECORD.md | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/templates/OPTIMIZATION-RECORD.md b/templates/OPTIMIZATION-RECORD.md index 419b5c6..13dbe30 100644 --- a/templates/OPTIMIZATION-RECORD.md +++ b/templates/OPTIMIZATION-RECORD.md @@ -33,11 +33,11 @@ Then record every required problem-classification dimension from `OPTIMIZATION-P - Variables: continuous / integer / categorical / conditional / mixed - Search scope: local / global - Objective behavior: deterministic / noisy / stochastic -- Information: gradient / derivative-free / black-box +- Information: gradient available / derivative-free / black-box - Evaluation cost: cheap / moderate / expensive - Constraints: bounds / equality / inequality / semantic / resource - Parallelism: sequential / synchronous batch / asynchronous -- Exactness: exact / approximation permitted under explicit error contract +- Exactness: exact / approximation permitted under an explicit error contract ## Preserved contract @@ -81,4 +81,4 @@ Define the measured or semantic condition that disables/reverts the optimization ## Composition notes -Which other OPT records compose safely, and which resource/semantic interactions must be re-measured? +Which other OPT records compose safely, and which resource/semantic interactions must be re-measured? \ No newline at end of file From 5e87a130213abdea8342f74ec7de48f649bf6570 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 02:54:50 +0930 Subject: [PATCH 058/229] Bind materialization reuse hits to source identity --- ...T-FAN-001-shared-materialization-fanout.md | 24 +++++++++++-------- 1 file changed, 14 insertions(+), 10 deletions(-) diff --git a/optimizations/OPT-FAN-001-shared-materialization-fanout.md b/optimizations/OPT-FAN-001-shared-materialization-fanout.md index 1a2bb98..b464d17 100644 --- a/optimizations/OPT-FAN-001-shared-materialization-fanout.md +++ b/optimizations/OPT-FAN-001-shared-materialization-fanout.md @@ -14,11 +14,11 @@ The same deterministic transformation is repeated independently for each consume ## Optimization problem contract -- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, immutable-source snapshot/mutation-control policies, monotonic source/config epochs, source-witness compare-and-publish policies, artifact-version pinning/consumption policies, and crash-consistent publication schemes -- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, source-snapshot/mutation-consistency, **source-validation-to-publication linearizability**, validation-to-consumption identity, and publication-atomicity requirement +- X: target-supported materialization boundaries, representation formats/versions, persistence policies, raw-versus-materialized retention policies, complete materialization-key definitions, immutable-source snapshot/mutation-control policies, monotonic source/config epochs, source-witness compare-and-publish policies, reuse-hit source-binding policies, artifact-version pinning/consumption policies, and crash-consistent publication schemes +- F: configurations whose materialized representation satisfies every declared consumer semantic, versioning, integrity, trust, materialization-equivalence, source-snapshot/mutation-consistency, **source-validation-to-publication linearizability**, **reuse-hit source-binding linearizability**, validation-to-consumption identity, and publication-atomicity requirement - f: measured transformation CPU, replay CPU, fan-out latency, and storage/I/O overhead under the target's declared objective ordering - d: minimize under the target's predeclared scalar or lexicographic ordering -- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state binds the artifact bytes to one coherent effective source/transform identity, no intervening mutable-input change can be erased by endpoint equality, the final source witness is linearized with the materialization commit, and every consumer reads the **same immutable/versioned artifact instance that was validated** rather than re-resolving a mutable alias after validation; verification/security metadata may be removed only under an explicit contract change +- C: consumers receive the declared representation semantics exactly; reuse is allowed only when one committed state binds the artifact bytes to one coherent effective source/transform identity, no intervening mutable-input change can be erased by endpoint equality, the final source witness is linearized with the materialization commit, **each reuse-hit decision is linearized against the invocation's effective source identity before delivery**, and every consumer reads the **same immutable/versioned artifact instance that was validated** rather than re-resolving a mutable alias after validation; verification/security metadata may be removed only under an explicit contract change - B: target-specific fan-out/replay benchmark budget declared before tuning; no portable subscriber count, replay size, or retention duration is supplied here - S: stop when the declared budget is exhausted or a validated materialization policy materially improves the target objective without violating C - Variables: categorical / integer / mixed @@ -26,13 +26,13 @@ The same deterministic transformation is repeated independently for each consume - Objective behavior: noisy for performance; transformation identity/equivalence is deterministic - Information: derivative-free / black-box performance measurements - Evaluation cost: moderate to expensive depending on transform/replay size -- Constraints: semantic equivalence, source-snapshot/mutation consistency, source-validation/publication linearizability, artifact-version pinning, integrity, versioning, trust/security, storage, and crash-consistency constraints -- Parallelism: concurrent fan-out/replay; source publication, artifact publication, and consumption pinning must remain race-safe +- Constraints: semantic equivalence, source-snapshot/mutation consistency, source-validation/publication linearizability, reuse-hit source binding, artifact-version pinning, integrity, versioning, trust/security, storage, and crash-consistency constraints +- Parallelism: concurrent fan-out/replay; source publication, reuse-hit binding, artifact publication, and consumption pinning must remain race-safe - Exactness: exact representation semantics; no approximation is introduced ## Preserved contract -Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive source mutation, change-and-revert races, final-check-to-commit races, crashes, interrupted publication, and concurrent replacement of mutable aliases. Validation is meaningful only if both publication and downstream consumption remain bound to the exact source/artifact identities that were validated. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. +Consumers must receive the same declared representation semantics. A persisted representation is reusable only under a named **materialization-equivalence invariant** that binds the artifact to every effective input capable of changing its bytes or semantics, and that binding must survive source mutation, change-and-revert races, final-check-to-commit races, reuse-hit races, crashes, interrupted publication, and concurrent replacement of mutable aliases. Validation is meaningful only if publication, the caller's reuse-hit decision, and downstream consumption remain bound to the exact source/artifact identities that were validated. Removing verification/security metadata is **not** a correctness-preserving optimization unless the interface contract explicitly changes. ## Optimization @@ -46,7 +46,9 @@ For lock/epoch-based targets, make the **final witness validation and materializ Publish the artifact and its identity as **one committed state**. Acceptable designs include content-addressed storage where the artifact digest is itself part of the committed key, an atomically replaced manifest that contains both the full materialization key and the artifact digest/location, or another crash-consistent transaction that makes old state or new state visible but never a mixed pair. Do not update artifact bytes and their key independently in a way that can expose a new artifact with stale metadata or stale bytes with a new key after a crash. -Before reuse, require: (1) exact agreement with the current effective-input materialization key and immutable snapshot/monotonic mutation identity, (2) a committed manifest/content-address relation that binds that identity to the artifact identity, and (3) artifact integrity/format validity. **Validation must return or retain an immutable/versioned artifact handle/snapshot that uniquely identifies the validated bytes. Every fan-out/replay consumer must read through that same pinned handle/version.** Do not validate mutable pathname/object-name A and later re-resolve that alias for consumption. If the target cannot provide immutable/versioned handles, hold an appropriate read/replacement lock from the integrity check through the complete consumer read, or first create an immutable snapshot and validate/consume that snapshot. A key mismatch, mutation-epoch mismatch, missing/incomplete publication marker, digest mismatch, unverifiable artifact, or inability to bind consumption to the validated bytes is a cache miss/fail-closed condition and requires regeneration or safe fallback. Do not use format validation, endpoint key equality, or a mutable location name alone as evidence that the bytes consumed are the bytes validated. +Before reuse, require: (1) exact agreement with the current effective-input materialization key and immutable snapshot/monotonic mutation identity, (2) a committed manifest/content-address relation that binds that identity to the artifact identity, and (3) artifact integrity/format validity. **The invocation must then bind the reuse hit to that same effective-input identity in one linearizable step before the artifact is delivered.** Prefer capturing an immutable invocation snapshot/version before lookup. Otherwise hold the source/config mutation lock through the hit decision, or use an atomic compare-and-bind/CAS that verifies the complete current epoch vector and records the hit only if it is still current in the same serialization domain that advances those epochs. A check-then-deliver sequence is insufficient: if the source advances from A to B after the key/epoch check but before the caller is bound to the hit, the invocation must miss/retry against B rather than receiving A merely because A's artifact remains valid for its own historical source version. A caller may deliberately consume A only when its invocation was already bound to immutable source identity A before the mutation. + +**Validation must return or retain an immutable/versioned artifact handle/snapshot that uniquely identifies the validated bytes. Every fan-out/replay consumer must read through that same pinned handle/version.** Do not validate mutable pathname/object-name A and later re-resolve that alias for consumption. If the target cannot provide immutable/versioned handles, hold an appropriate read/replacement lock from the integrity check through the complete consumer read, or first create an immutable snapshot and validate/consume that snapshot. A key mismatch, mutation-epoch mismatch, failed reuse-hit compare-and-bind, missing/incomplete publication marker, digest mismatch, unverifiable artifact, or inability to bind consumption to the validated bytes is a cache miss/fail-closed condition and requires regeneration or safe fallback. Do not use format validation, endpoint key equality, or a mutable location name alone as evidence that the bytes consumed are the bytes validated. For multiple consumers, each may hold its own reference to the same immutable validated artifact version, or the system may retain one immutable snapshot for the fan-out/replay lifetime. Reclamation/retention must not invalidate a pinned consumer handle before that consumer completes. A mutable alias may advance to a newer committed materialization for later callers without changing the version already pinned by an in-flight consumer. @@ -66,18 +68,20 @@ Exercise **concurrent source mutation**, including explicit A→B→A races. Sta Add a **final source-validation-to-publication race**. Pause after the last source epoch/vector check but before the new materialization becomes authoritative, mutate one source/config/transform input, then resume publication. For lock-based targets, prove the mutation cannot occur until after the authoritative switch. For epoch/CAS-based targets, prove the compare-and-publish fails because the current epoch vector no longer equals the witnessed vector; no consumer may observe the candidate as committed. Repeat with A→B→A content restoration, multiple inputs, and a concurrent manifest reader. A successful endpoint rehash after the mutation is not sufficient evidence. +Add a **reuse-hit source-binding race**. Start an invocation against source/config identity A and pause after the committed manifest, current epoch/key, and artifact integrity checks succeed but before the caller is irrevocably bound to that hit. Mutate one effective input so the current identity becomes B, then resume. Snapshot-based targets must prove the invocation was already bound to immutable A before lookup; lock-based targets must prove the mutation cannot pass the hit-binding boundary; epoch/CAS targets must prove the compare-and-bind fails and the invocation misses/retries against B. The test must never deliver pinned artifact A as a cache hit for an invocation whose effective identity became B before binding. Repeat with A→B→A mutation, multiple inputs, and concurrent callers straddling the identity transition. + Add a **post-validation replacement race**. Validate committed artifact A, pause before a consumer reads it, replace the mutable alias/path/object name with a different valid artifact B, then resume consumption. Prove a pinned immutable/versioned handle still yields exactly A (or fails closed if A was invalidated by the target's retention contract), never unvalidated B. Repeat with fan-out consumers at staggered start times, concurrent manifest advancement, replay after alias replacement, reclamation pressure, and mutable object-store/version aliases. For lock-based targets, prove replacement cannot occur until the protected consumer read completes. For snapshot-based targets, prove validation and consumption address the same snapshot digest/version. Inject crashes/interruption at every publication boundary: after artifact write but before manifest commit, after provisional metadata write, during atomic replacement, and immediately after commit. After restart, prove that readers see either the previous valid committed materialization or the new valid committed materialization, never a mixed key/artifact state. Verify digest/key mismatch is rejected even when the artifact is otherwise parseable. ## Target-repo adaptation -Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, specify whether mutable inputs are consumed from immutable snapshots, protected by full-duration locks, or guarded by monotonic mutation epochs, and define how a coherent multi-input witness is captured. For epoch/lock targets, define the serialization domain that makes final witness validation atomic with authoritative materialization publication: lock-through-commit or atomic compare-and-publish against the complete epoch vector. Specify representation versioning, invalidation, integrity checking, **the immutable/versioned artifact handle or lock/snapshot that binds validation through consumption**, retention/reclamation semantics for pinned consumers, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. Do not advertise commit-time endpoint rehashing or mutable-path validation alone as sufficient identity protection. +Define the complete materialization-equivalence invariant for the target, choose the identity primitive for each effective input, specify whether mutable inputs are consumed from immutable snapshots, protected by full-duration locks, or guarded by monotonic mutation epochs, and define how a coherent multi-input witness is captured. For epoch/lock targets, define both serialization boundaries explicitly: the mechanism that makes final witness validation atomic with authoritative materialization publication, and the mechanism that makes a reuse-hit decision atomic with the invocation's current source/config identity (immutable invocation snapshot, lock-through-hit-binding, or atomic compare-and-bind against the complete epoch vector). Specify representation versioning, invalidation, integrity checking, **the immutable/versioned artifact handle or lock/snapshot that binds validation through consumption**, retention/reclamation semantics for pinned consumers, **crash-consistent publication/commit mechanics**, storage-vs-CPU trade-offs and whether both raw and materialized forms are retained. Do not advertise commit-time endpoint rehashing, check-then-deliver reuse, or mutable-path validation alone as sufficient identity protection. ## Failure modes -Incomplete keys can serve stale representations after source or transform changes; mutable sources can change during transformation and produce mixed-state output under a stale key; A→B→A races can defeat endpoint key comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent epoch vectors can describe no real source state; a check-then-publish gap can authorize A-derived bytes after the source already advanced to B; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; a mutable alias can be replaced after validation and before consumption, delivering unvalidated bytes; reclamation can invalidate a pinned artifact prematurely; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. +Incomplete keys can serve stale representations after source or transform changes; mutable sources can change during transformation and produce mixed-state output under a stale key; A→B→A races can defeat endpoint key comparisons; non-monotonic/reused mutation versions can erase intervening changes; incoherent epoch vectors can describe no real source state; a check-then-publish gap can authorize A-derived bytes after the source already advanced to B; a reuse check followed by a later hit/delivery can bind artifact A to an invocation whose source already advanced to B; non-atomic publication can pair new bytes with an old key or vice versa after a crash; metadata can match while artifact bytes are corrupted; a mutable alias can be replaced after validation and before consumption, delivering unvalidated bytes; reclamation can invalidate a pinned artifact prematurely; materializing unused forms wastes storage; format changes create invalidation/migration costs; mutable consumer-specific transformations cannot safely share one artifact. ## Rollback trigger -Disable reuse immediately if any materialization-key hit, source-mutation race, final source-validation-to-publication race, publication-recovery path, integrity check, or validation-to-consumption race can return output that differs from a fresh transform for the same exact committed effective inputs; if an A→B→A race can evade the snapshot/lock/monotonic mutation witness; if source epochs can change after a successful final check yet the candidate can still become authoritative; if a mutable alias replacement can make a consumer read bytes other than the exact artifact version that passed validation; if a pinned artifact can be reclaimed before consumption completes; or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. +Disable reuse immediately if any materialization-key hit, source-mutation race, final source-validation-to-publication race, reuse-hit source-binding race, publication-recovery path, integrity check, or validation-to-consumption race can return output that differs from a fresh transform for the same exact committed effective inputs; if an A→B→A race can evade the snapshot/lock/monotonic mutation witness; if source epochs can change after a successful final check yet the candidate can still become authoritative; if a source/config identity can change after a successful reuse check yet the invocation can still be bound to the stale hit; if a mutable alias replacement can make a consumer read bytes other than the exact artifact version that passed validation; if a pinned artifact can be reclaimed before consumption completes; or if interrupted publication can expose a mixed key/artifact state. Also disable when storage/invalidations outweigh avoided transform work or representation equivalence fails. \ No newline at end of file From 3ed7c07ae3b8a5885d2db5ff19ee8a10952dbe91 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 02:55:21 +0930 Subject: [PATCH 059/229] Define population for performance detector error rates --- ...DGET-001-performance-regression-budgets.md | 26 +++++++++++-------- 1 file changed, 15 insertions(+), 11 deletions(-) diff --git a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md index 9ce7c06..3cd935e 100644 --- a/optimizations/OPT-BUDGET-001-performance-regression-budgets.md +++ b/optimizations/OPT-BUDGET-001-performance-regression-budgets.md @@ -16,30 +16,32 @@ Small performance regressions accumulate because performance is measured occasio ## Optimization problem contract - X: target-supported metric/fixture/statistic/threshold configurations for a performance-regression gate -- F: gate configurations based on a sufficiently characterized environment and workload, with statistically justified tolerance, selection-aware validation, and no weakening of functional correctness or workload realism +- F: gate configurations based on a sufficiently characterized environment and workload, with statistically justified tolerance, selection-aware validation, a declared good-run/regression population or explicitly enumerated control scope, and no weakening of functional correctness or workload realism - f: target-measured **loss vector** comprising missed-material-regression loss (for example false-negative rate and, where relevant, severity-weighted miss cost), flaky/false-failure loss (false-positive rate), and measurement/CI overhead - d: minimize every component of the declared loss vector under the target's predeclared scalar, weighted, Pareto, or lexicographic ordering; if detection quality is reported separately, it is a diagnostic complement such as `1 - false-negative-rate`, not an oppositely oriented coordinate inside `f` -- C: the performance gate must not incentivize weakening tests, assertions, evidence, semantic coverage, or representative workload inputs; once a candidate gate is selected, its claimed false-positive/false-negative performance must be established on independent control executions or under a predeclared selection-aware procedure that accounts for every configuration tried -- B: target-specific calibration and certification budget specifying repetitions, environment samples, held-out/control executions, and allowable CI/runtime measurement cost +- C: the performance gate must not incentivize weakening tests, assertions, evidence, semantic coverage, or representative workload inputs; once a candidate gate is selected, its claimed false-positive/false-negative performance must be established on independent control executions or under a predeclared selection-aware procedure that accounts for every configuration tried, **and every reported detector error rate must name the population/generator and sampling scheme it estimates or be explicitly scoped to the enumerated controls only** +- B: target-specific calibration and certification budget specifying repetitions, environment samples, held-out/control executions, population/generator coverage where general error rates are claimed, and allowable CI/runtime measurement cost - S: stop calibration when the declared sample budget is exhausted or the baseline/noise estimate is stable enough to freeze one candidate gate for independent certification; promote it only if the certification contract passes - Variables: continuous / integer / categorical / mixed metric, statistic, fixture, and threshold choices - Search scope: local gate/calibration tuning - Objective behavior: noisy / stochastic measurement distributions - Information: derivative-free statistical observations - Evaluation cost: moderate to expensive depending on repetitions and fixture scale -- Constraints: functional correctness, representative workload, statistical tolerance, runner/environment characterization, false-positive/false-negative, selection bias, and CI-overhead constraints +- Constraints: functional correctness, representative workload, declared population/control scope, statistical tolerance, runner/environment characterization, false-positive/false-negative, selection bias, and CI-overhead constraints - Parallelism: sequential or synchronous-batch calibration; parallel sampling only when runner interference is characterized - Exactness: no semantic approximation; statistical tolerance/noise handling is explicit ## Preserved contract -A performance gate may not incentivize weakening functional tests, correctness, evidence or workload realism. The gate is valid only while its fixture, environment characterization, detection sensitivity and measurement overhead remain inside their declared contract. Calibration evidence used to choose among competing gates is not automatically valid certification evidence for the selected gate. +A performance gate may not incentivize weakening functional tests, correctness, evidence or workload realism. The gate is valid only while its fixture, environment characterization, declared detection population/control scope, detection sensitivity and measurement overhead remain inside their declared contract. Calibration evidence used to choose among competing gates is not automatically valid certification evidence for the selected gate. A rate measured on a handpicked or finite control suite must not be generalized to unseen production regressions unless the target has declared and sampled from a population/generator that supports that inference. ## Optimization Turn a stable, reproducible performance expectation into a regression gate. Compare distributions or robust summaries where noise matters; separate machine/environment drift from code regression; keep cold/warm claims distinct. Keep known-fast and known-regressed control fixtures (or equivalent calibration cases) so the gate can periodically prove it still distinguishes acceptable from materially regressed behavior. -When multiple metric/fixture/statistic/threshold configurations are explored, treat that search as model selection. Use calibration/tuning data to choose the candidate, then freeze its complete configuration before certification. The default certification path is an independent held-out set of known-good and known-regressed executions that played no role in choosing the gate. If holding out controls is impractical, use a predeclared nested-resampling, simultaneous-confidence, multiple-testing, or other selection-aware procedure whose error guarantees cover the full configuration search—not nominal per-candidate estimates computed after selecting the best one. +Before claiming false-positive or false-negative rates beyond those exact controls, define the estimand. Declare the good-run population and the material-regression population or generator, including the target workloads, regression classes, severity range, environment distribution, and any exclusions. Predeclare how certification cases are sampled or generated from that population and how repeated executions are grouped. If the repository cannot justify a broader population model, use the controls strictly as an enumerated conformance suite and report control-suite detection/failure rates without implying a general production error rate. + +When multiple metric/fixture/statistic/threshold configurations are explored, treat that search as model selection. Use calibration/tuning data to choose the candidate, then freeze its complete configuration before certification. The default certification path is an independent held-out sample of known-good and known-regressed executions drawn under the declared sampling scheme and playing no role in choosing the gate. If holding out controls is impractical, use a predeclared nested-resampling, simultaneous-confidence, multiple-testing, or other selection-aware procedure whose error guarantees cover the full configuration search—not nominal per-candidate estimates computed after selecting the best one. ## Before / after evidence @@ -53,20 +55,22 @@ When multiple metric/fixture/statistic/threshold configurations are explored, tr Calibrate variance before setting the threshold. During tuning, compare candidate metric/fixture/statistic/threshold configurations using explicitly designated calibration data and preserve raw samples where practical. Once one gate is selected, **freeze the entire gate configuration before measuring its claimed detection performance**. -Certify the frozen gate on independent known-good and known-regressed control executions that were not used to select it. Measure false positives, false negatives, and gate overhead against predeclared acceptance limits. If independent controls are unavailable, use a predeclared nested-resampling or selection-aware procedure that accounts for every candidate/configuration examined, and report the resulting adjusted uncertainty/error rates rather than reusing naive in-sample estimates. +Before certification, write down the exact error-rate scope: either (a) a declared good-run/regression population or generator plus sampling scheme, including regression classes/severities and environment/workload strata that the rate is intended to represent, or (b) a finite enumerated control suite to which the reported rates are explicitly limited. For population claims, draw the independent certification sample according to that scheme and record coverage/counts by the predeclared strata; do not substitute a convenient handpicked set after seeing gate behavior. For control-only claims, label the result as control-suite performance and prohibit extrapolation to unrepresented regression classes. + +Certify the frozen gate on independent known-good and known-regressed executions that were not used to select it. Measure false positives, false negatives, and gate overhead against predeclared acceptance limits **within the declared scope**. If independent controls are unavailable, use a predeclared nested-resampling or selection-aware procedure that accounts for every candidate/configuration examined, and report the resulting adjusted uncertainty/error rates rather than reusing naive in-sample estimates. If the declared population/generator changes, previous rates do not automatically transfer. Verify objective orientation explicitly: construct one candidate with fewer missed regressions but more false alarms and another with the opposite tradeoff, compute the declared loss coordinates, and prove the configured scalar/Pareto/lexicographic ordering ranks them exactly as documented. A separately reported positive detection-quality score must never be fed into a minimization coordinate without an explicit monotone conversion to loss. -Record which executions were used for calibration/selection versus certification. Re-run independent controls after runner/toolchain changes and periodically enough to detect stale fixtures or sensitivity drift. Add an explicit overfitting fixture where several candidate gates are tuned on one noisy control sample set; prove the gate cannot be promoted merely because one candidate looked best on those same samples. +Record which executions were used for calibration/selection versus certification, along with the population/control scope and sampling provenance for each certification case. Re-run independent controls after runner/toolchain changes and periodically enough to detect stale fixtures or sensitivity drift. Add an explicit overfitting fixture where several candidate gates are tuned on one noisy control sample set; prove the gate cannot be promoted merely because one candidate looked best on those same samples. Also include at least one deliberately omitted regression class in a test report to prove the tooling labels that class as outside the estimated scope rather than silently counting the observed controls as universal evidence. ## Target-repo adaptation -Never copy another project's milliseconds, bundle sizes or thresholds. Establish the target's own baseline and noise envelope, define control fixtures, and declare acceptable false-positive/false-negative rates plus a maximum measurement-overhead budget. Define `f` using consistently oriented loss coordinates and predeclare how those coordinates are ordered or scalarized; if the target also reports a positive detection-quality score, document its conversion to the minimized loss coordinate. Predeclare how calibration/selection is separated from certification: held-out controls by default, or a justified nested/selection-aware alternative. Preserve the candidate-search history needed to audit the claimed certification error rates. +Never copy another project's milliseconds, bundle sizes or thresholds. Establish the target's own baseline and noise envelope, define control fixtures, and declare acceptable false-positive/false-negative rates plus a maximum measurement-overhead budget. **Define what population those rates refer to:** specify representative workloads, regression classes and severities, environment strata, exclusions, and the sampling/generation process; or explicitly limit the claim to a named finite control suite. Define `f` using consistently oriented loss coordinates and predeclare how those coordinates are ordered or scalarized; if the target also reports a positive detection-quality score, document its conversion to the minimized loss coordinate. Predeclare how calibration/selection is separated from certification: held-out controls by default, or a justified nested/selection-aware alternative. Preserve the candidate-search history and sampling provenance needed to audit the claimed certification error rates. ## Failure modes -Flaky gates from uncontrolled runners, benchmark gaming, stale fixtures, hardware drift, thresholds so loose they miss real regressions, thresholds so tight they block good changes, an objective vector mixing maximized quality with minimized costs without an explicit per-coordinate direction/conversion, selection bias from evaluating a chosen gate on the same controls used to tune it, unreported configuration search that invalidates nominal error rates, and measurement overhead large enough to damage CI usability or distort the workload under test. +Flaky gates from uncontrolled runners, benchmark gaming, stale fixtures, hardware drift, thresholds so loose they miss real regressions, thresholds so tight they block good changes, an objective vector mixing maximized quality with minimized costs without an explicit per-coordinate direction/conversion, selection bias from evaluating a chosen gate on the same controls used to tune it, **sampling bias or undefined detector populations that turn control-suite performance into an unjustified general false-positive/false-negative claim**, unrepresented regression classes/severities, unreported configuration search that invalidates nominal error rates, and measurement overhead large enough to damage CI usability or distort the workload under test. ## Rollback trigger -Disable or demote the gate to non-blocking and recalibrate whenever its measurement environment is invalid, its fixture is stale/nonrepresentative, its objective orientation/scalarization is ambiguous or ranks a worse detector as better, independent/selection-aware certification no longer meets the declared false-positive/false-negative limits, known regressions are no longer detected, known-good controls fail above the declared false-positive limit, observed false negatives exceed the declared limit, or measurement overhead exceeds the predeclared budget. Do **not** disable merely because product code legitimately regressed; in that case keep the valid gate and fix or explicitly accept the regression through the target's normal review process. +Disable or demote the gate to non-blocking and recalibrate whenever its measurement environment is invalid, its fixture is stale/nonrepresentative, its declared regression/good-run population or control scope no longer matches the deployment claim, its sampling process no longer represents the declared population, its objective orientation/scalarization is ambiguous or ranks a worse detector as better, independent/selection-aware certification no longer meets the declared false-positive/false-negative limits **within that stated scope**, known regressions are no longer detected, known-good controls fail above the declared false-positive limit, observed false negatives exceed the declared limit, or measurement overhead exceeds the predeclared budget. Do **not** disable merely because product code legitimately regressed; in that case keep the valid gate and fix or explicitly accept the regression through the target's normal review process. \ No newline at end of file From ee6700bd73e0b6c8a8a012779795d6944dad5c5d Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 02:58:44 +0930 Subject: [PATCH 060/229] Handle CommonMark heading indentation and titled record links --- scripts/check_catalog.py | 1065 +++------------------------------ scripts/check_catalog_core.py | 1012 +++++++++++++++++++++++++++++++ 2 files changed, 1105 insertions(+), 972 deletions(-) create mode 100755 scripts/check_catalog_core.py diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 0142fb0..1c6fcf7 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -1,1012 +1,133 @@ #!/usr/bin/env python3 -"""Check OPT catalog/document integrity without external dependencies.""" +"""Normalize supported CommonMark syntax, then run the hardened catalog checker. + +The core checker intentionally stays strict and source-oriented. This front end creates a +scratch copy, canonicalizes two rendering-equivalent forms that the core otherwise rejects +or overlooks, and runs the core against that copy: + +* one-to-three spaces before ATX headings (valid CommonMark indentation), and +* optional Markdown titles on links to optimization-record Markdown files. + +The repository working tree is never modified by this normalization step. +""" from __future__ import annotations -import html import re -from collections import Counter +import shutil +import subprocess +import sys +import tempfile from pathlib import Path ROOT = Path(__file__).resolve().parents[1] -OPT_DIR = ROOT / "optimizations" -FROZEN_V1 = { - "OPT-PY-001", - "OPT-INV-001", - "OPT-LEAN-001", - "OPT-PAR-001", - "OPT-DSP-001", -} -REQUIRED_V2 = { - "## Source evidence", - "## Problem", - "## Optimization problem contract", - "## Preserved contract", - "## Optimization", - "## Before / after evidence", - "## Validation", - "## Target-repo adaptation", - "## Failure modes", - "## Rollback trigger", -} -REQUIRED_CONTRACT_FIELDS = ("X", "F", "f", "d", "C", "B", "S") -REQUIRED_CLASSIFICATION_FIELDS = ( - "Variables", - "Search scope", - "Objective behavior", - "Information", - "Evaluation cost", - "Constraints", - "Parallelism", - "Exactness", -) -CLASSIFICATION_TEMPLATE_VALUES = { - "Variables": "continuous / integer / categorical / conditional / mixed", - "Search scope": "local / global", - "Objective behavior": "deterministic / noisy / stochastic", - "Information": "gradient available / derivative-free / black-box", - "Evaluation cost": "cheap / moderate / expensive", - "Constraints": "bounds / equality / inequality / semantic / resource", - "Parallelism": "sequential / synchronous batch / asynchronous", - "Exactness": "exact / approximation permitted under an explicit error contract", -} -ALLOWED_V2_STATUS_CATEGORIES = { - "Verified", - "Verified, environment-specific", - "Implemented reference", - "Implemented external reference", - "Implemented external pattern", - "Proposed / OPT synthesis", - "Source candidate", -} -TEMPLATE_PLACEHOLDER_LINES = { - "- Repository / publication / article:", - "- Release/commit/PR/DOI/date:", - "- Exact files/sections where applicable:", - "- Licensing/provenance boundary where code reuse may matter:", - "What dominates runtime, latency, memory, I/O, CI cost, quality budget or optimization-evaluation cost?", - "State exactly what must remain unchanged: output bytes, theorem targets, assertions, API, numerical tolerance, ordering, statistical guarantee, evidence boundary, trust model, etc.", - "If the optimization changes the contract (for example exact → approximate), state the new contract explicitly instead of claiming preservation.", - "Describe the reusable mechanism, not only the source-project patch.", - "If no controlled benchmark exists, say so explicitly.", - "How was equivalence, correctness, bound soundness, approximation error or other contract compliance established?", - "Which source constants, thresholds, worker counts, bit splits, cache keys, search budgets or tolerances must be re-profiled rather than copied?", - "What can make this optimization invalid, slower, less robust or misleading?", - "Define the measured or semantic condition that disables/reverts the optimization.", -} - -LINK_RE = re.compile(r"\[([^\]]+)\]\((optimizations/[^)#]+\.md)\)") -RECORD_LINK_CELL_RE = re.compile( - r"^\[(OPT-[A-Z]+-\d{3})\]\((optimizations/[^)#]+\.md)\)$" -) -ID_RE = re.compile(r"^# (OPT-[A-Z]+-\d{3}) — ") -FILENAME_ID_RE = re.compile(r"^(OPT-[A-Z]+-\d{3})-") -STATUS_RE = re.compile(r"^\*\*Status:\*\*\s*(.*?)\s*$") -OPT_TOKEN_RE = re.compile(r"\bOPT-[A-Z]+-\d{3}\b") -EMPTY_LABEL_RE = re.compile(r"^-\s+[^:]+:\s*$") -LINK_REFERENCE_DEFINITION_RE = re.compile( - r"^\[(?:\\.|[^\[\]\\])+\]:[ \t]+\S.*$" -) -LINK_REFERENCE_TITLE_CONTINUATION_RE = re.compile( - r"^ {0,3}(?:\"(?:\\.|[^\"\\])*\"|'(?:\\.|[^'\\])*'|\((?:\\.|[^)\\])*\))[ \t]*$" -) -SOURCE_URL_RE = re.compile(r"https?://\S+", re.IGNORECASE) -SOURCE_DOI_RE = re.compile(r"\b(?:doi:\s*)?10\.\d{4,9}/\S+", re.IGNORECASE) -SOURCE_COMMIT_RE = re.compile(r"\b[0-9a-f]{7,40}\b", re.IGNORECASE) -SOURCE_LOCAL_NOTE_RE = re.compile(r"`?(sources/[A-Za-z0-9._/-]+\.md)`?") -SOURCE_REPOSITORY_RE = re.compile(r"`[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+`") -SOURCE_PLACEHOLDER_RE = re.compile( - r"^(?:[-*+]\s*)?(?:unknown|tbd|todo|n/?a|none|pending)\.?$", re.IGNORECASE -) -REFERENCE_IMAGE_RE = re.compile(r"!\[([^\]]*)\]\[[^\]]*\]") -REFERENCE_LINK_RE = re.compile(r"\[([^\]]*)\]\[[^\]]*\]") -INLINE_HTML_TAG_RE = re.compile( - r"`]+))?)*" - r"[ \t]*/?>" -) -HEADING_RE = re.compile(r"^#{1,6}(?:\s|$)") -SECTION_BOUNDARY_RE = re.compile(r"^#{1,2}(?:\s|$)") -SETEXT_H1_RE = re.compile(r"^ {0,3}=+[ \t]*$") -SETEXT_H2_RE = re.compile(r"^ {0,3}-+[ \t]*$") -THEMATIC_BREAK_RE = re.compile( - r"^(?:\*(?:[ \t]*\*){2,}|-(?:[ \t]*-){2,}|_(?:[ \t]*_){2,})[ \t]*$" +CORE_NAME = "check_catalog_core.py" +ATX_INDENT_RE = re.compile(r"(?m)^ {1,3}(?=#{1,6}(?:[ \t]|$))") +RECORD_LINK_START_RE = re.compile( + r"\[([^\]\r\n]+)\]\((optimizations/[^\s)#]+\.md)" ) -LIST_MARKER_ONLY_RE = re.compile(r"^(?:[-+*]|\d+[.)])$") -TABLE_SEPARATOR_CELL_RE = re.compile(r"^:?-{3,}:?$") -FENCE_OPEN_RE = re.compile(r"^ {0,3}(`{3,}|~{3,})(.*)$") -RAW_HTML_TYPE1_OPEN_RE = re.compile( - r"^ {0,3}<(?Pscript|pre|style|textarea)(?:[ \t]|>|$)", re.IGNORECASE -) -RAW_HTML_DECLARATION_OPEN_RE = re.compile(r"^ {0,3}]|$)", - re.IGNORECASE, -) -RAW_HTML_COMPLETE_TAG_RE = re.compile( - r"^ {0,3}(?:" - r"" - r"|<[A-Za-z][A-Za-z0-9-]*" - r"(?:[ \t]+[A-Za-z_:][A-Za-z0-9_.:-]*" - r"(?:[ \t]*=[ \t]*(?:\"[^\"]*\"|'[^']*'|[^ \t\n\"'=<>`]+))?)*" - r"[ \t]*/?>" - r")[ \t]*$" -) -EMPHASIS_WRAPPERS = ("**", "__", "~~", "*", "_") -STATUS_WRAPPERS = ("**", "__", "~~", "*", "_", "`") -CANONICAL_DEFINITION_PATTERNS = { - "X": re.compile(r"^- `X` — \S"), - "F": re.compile(r"^- `F(?: ⊆ X)?` — \S"), - "f": re.compile(r"^- `f(?:\s*:[^`]*)?` — \S"), - "d": re.compile(r"^- `d` — \S"), - "C": re.compile(r"^- `C` — \S"), - "B": re.compile(r"^- `B` — \S"), - "S": re.compile(r"^- `S` — \S"), -} - - -def die(msg: str) -> None: - raise SystemExit(f"catalog-integrity: {msg}") - - -def markdown_source_lines(text: str) -> list[str]: - """Split only on CommonMark line endings (LF, CRLF, or CR).""" - return text.replace("\r\n", "\n").replace("\r", "\n").split("\n") - - -def is_indented_code_line(raw: str) -> bool: - """Return whether a non-fenced line is an indented Markdown code line.""" - return raw.startswith("\t") or raw.startswith(" ") - - -def strip_inline_html_comments(raw: str, in_comment: bool) -> tuple[str, bool]: - """Strip inline HTML comments while carrying a mid-line unmatched comment.""" - out: list[str] = [] - cursor = 0 - - if in_comment: - end = raw.find("-->") - if end < 0: - return "", True - cursor = end + 3 - in_comment = False - - while cursor < len(raw): - start = raw.find("", start + 4) - if end < 0: - in_comment = True - break - cursor = end + 3 - - return "".join(out), in_comment - - -def raw_html_block_start(raw: str) -> tuple[str, str | None] | None: - """Return the raw-HTML block mode for a CommonMark-style block start.""" - if re.match(r"^ {0,3}" - - type1 = RAW_HTML_TYPE1_OPEN_RE.match(raw) - if type1 is not None: - return "tag", type1.group("tag").lower() - if re.match(r"^ {0,3}<\?", raw): - return "token", "?>" - if re.match(r"^ {0,3}".replace(" ", "") - if RAW_HTML_DECLARATION_OPEN_RE.match(raw): - return "token", ">" - if RAW_HTML_BLOCK_TAG_RE.match(raw) or RAW_HTML_COMPLETE_TAG_RE.match(raw): - return "blank", None - return None - - -def raw_html_tag_closes(raw: str, tag: str) -> bool: - """Match CommonMark type-1 block terminators exactly (case-insensitive).""" - return re.search(rf"", raw, re.IGNORECASE) is not None - - -def visible_nonfenced_lines(lines: list[str]) -> list[str]: - """Return Markdown-visible lines used by schema validation.""" - visible: list[str] = [] - fence_char: str | None = None - fence_len = 0 - inline_comment = False - html_mode: str | None = None - html_end: str | None = None - - for raw in lines: - if fence_char is not None: - close = re.fullmatch( - rf" {{0,3}}{re.escape(fence_char)}{{{fence_len},}}[ \t]*", raw - ) - if close is not None: - fence_char = None - fence_len = 0 - continue - - if html_mode is not None: - if html_mode == "tag": - if html_end is not None and raw_html_tag_closes(raw, html_end): - html_mode = None - html_end = None - continue - if html_mode == "token": - if html_end is not None and html_end in raw: - html_mode = None - html_end = None - continue - if html_mode == "blank": - if raw.strip() == "": - html_mode = None - html_end = None - visible.append("") - continue - - if inline_comment: - rendered, inline_comment = strip_inline_html_comments(raw, True) - if inline_comment: - continue - raw_for_parse = rendered - else: - if is_indented_code_line(raw): - continue - opener = FENCE_OPEN_RE.match(raw) - if opener is not None: - run = opener.group(1) - info = opener.group(2) - if run[0] != "`" or "`" not in info: - fence_char = run[0] - fence_len = len(run) - continue - html_start = raw_html_block_start(raw) - if html_start is not None: - html_mode, html_end = html_start - if html_mode == "tag" and html_end is not None and raw_html_tag_closes(raw, html_end): - html_mode = None - html_end = None - elif html_mode == "token" and html_end is not None and html_end in raw: - html_mode = None - html_end = None - continue +def _skip_whitespace(text: str, index: int) -> int: + while index < len(text) and text[index] in " \t\r\n": + index += 1 + return index - raw_for_parse, inline_comment = strip_inline_html_comments(raw, False) - if raw_for_parse and is_indented_code_line(raw_for_parse): - continue - if raw_for_parse: - visible.append(raw_for_parse) - elif not inline_comment and raw == "": - visible.append("") - - return visible - - -def visible_text(text: str) -> str: - return "\n".join(visible_nonfenced_lines(markdown_source_lines(text))) - - -def setext_heading_start( - lines: list[str], underline_index: int, minimum_index: int -) -> int | None: - """Return the first source line of a Setext heading paragraph.""" - if underline_index <= minimum_index: - return None - underline = lines[underline_index] - if not ( - SETEXT_H1_RE.fullmatch(underline) or SETEXT_H2_RE.fullmatch(underline) - ): - return None - - candidate = underline_index - 1 - if candidate < minimum_index or not lines[candidate].strip(): - return None - if HEADING_RE.match(lines[candidate]): - return None - - start = candidate - while start > minimum_index: - previous = lines[start - 1] - if not previous.strip() or SECTION_BOUNDARY_RE.match(previous): - break - start -= 1 - return start - - -def section_lines(text: str, heading: str) -> list[str]: - """Return one exact visible level-2 Markdown section.""" - lines = visible_nonfenced_lines(markdown_source_lines(text)) - try: - start = lines.index(heading) + 1 - except ValueError: - return [] - end = len(lines) - for i in range(start, len(lines)): - if SECTION_BOUNDARY_RE.match(lines[i]): - end = i - break - setext_start = setext_heading_start(lines, i, start) - if setext_start is not None: - end = setext_start - break - return lines[start:end] - - -def is_backslash_escaped(text: str, index: int) -> bool: - count = 0 - cursor = index - 1 - while cursor >= 0 and text[cursor] == "\\": - count += 1 - cursor -= 1 - return count % 2 == 1 - - -def backtick_run_length(text: str, index: int) -> int: - cursor = index - while cursor < len(text) and text[cursor] == "`": - cursor += 1 - return cursor - index - - -def protect_code_spans(text: str) -> tuple[str, dict[str, str]]: - """Replace parsed code spans with private-use sentinels and preserve their text.""" - out: list[str] = [] - protected: dict[str, str] = {} - i = 0 - while i < len(text): - if text[i] != "`" or is_backslash_escaped(text, i): - out.append(text[i]) - i += 1 - continue - - run_len = backtick_run_length(text, i) - j = i + run_len - close_start: int | None = None - close_end: int | None = None - while j < len(text): - if text[j] != "`": - j += 1 - continue - candidate_len = backtick_run_length(text, j) - if candidate_len == run_len: - close_start = j - close_end = j + candidate_len - break - j += candidate_len - - if close_start is None or close_end is None: - out.append(text[i : i + run_len]) - i += run_len - continue - - token = chr(0xE000 + len(protected)) - protected[token] = text[i + run_len : close_start] - out.append(token) - i = close_end - - return "".join(out), protected - - -def find_label_close(text: str, open_index: int) -> int | None: - depth = 1 - i = open_index + 1 - while i < len(text): - if text[i] == "\\" and i + 1 < len(text): - i += 2 - continue - if text[i] == "[": - depth += 1 - elif text[i] == "]": - depth -= 1 - if depth == 0: - return i - i += 1 - return None - - -def parse_link_title_and_close(text: str, index: int) -> int | None: - """Parse whitespace plus an optional CommonMark-style title and outer close.""" - i = index - while i < len(text) and text[i] in " \t\n": - i += 1 - if i < len(text) and text[i] == ")": - return i + 1 - if i >= len(text): - return None - - opener = text[i] - if opener not in ('"', "'", "("): +def _parse_title(text: str, index: int) -> int | None: + """Return the index after a valid optional link title, before outer whitespace.""" + if index >= len(text) or text[index] not in ('"', "'", "("): return None + opener = text[index] closer = ")" if opener == "(" else opener - i += 1 - while i < len(text): - if text[i] == "\\" and i + 1 < len(text): - i += 2 - continue - if text[i] == closer: - i += 1 - break - if text[i] == "\n": - return None - i += 1 - else: - return None - - while i < len(text) and text[i] in " \t\n": - i += 1 - if i < len(text) and text[i] == ")": - return i + 1 - return None - - -def find_inline_link_end(text: str, open_paren: int) -> int | None: - """Return the end of a valid inline-link destination/title, or None.""" - i = open_paren + 1 - while i < len(text) and text[i] in " \t\n": - i += 1 - if i >= len(text): - return None - if text[i] == ")": - return i + 1 - - if text[i] == "<": - i += 1 - while i < len(text): - if text[i] == "\\" and i + 1 < len(text): - i += 2 - continue - if text[i] == ">": - return parse_link_title_and_close(text, i + 1) - if text[i] in "\n<": - return None - i += 1 - return None - - depth = 0 - while i < len(text): - char = text[i] - if char == "\\" and i + 1 < len(text): - i += 2 - continue - if char == "(": - depth += 1 - i += 1 - continue - if char == ")": - if depth == 0: - return i + 1 - depth -= 1 - i += 1 - continue - if char in " \t\n" and depth == 0: - return parse_link_title_and_close(text, i) - if char in "<>" or ord(char) < 0x20: - return None - i += 1 + index += 1 + while index < len(text): + char = text[index] + if char == "\\" and index + 1 < len(text): + index += 2 + continue + if char == closer: + return index + 1 + index += 1 return None -def strip_inline_links(text: str) -> str: - """Keep rendered labels while discarding valid inline-link/image destinations.""" +def canonicalize_record_link_titles(text: str) -> str: + """Drop only syntactically complete optional titles from OPT-record links.""" out: list[str] = [] - i = 0 - while i < len(text): - image = text.startswith("![", i) - if image: - label_open = i + 1 - elif text[i] == "[": - label_open = i - else: - out.append(text[i]) - i += 1 - continue - - label_close = find_label_close(text, label_open) - if label_close is None or label_close + 1 >= len(text) or text[label_close + 1] != "(": - out.append(text[i]) - i += 1 - continue - link_end = find_inline_link_end(text, label_close + 1) - if link_end is None: - out.append(text[i]) - i += 1 - continue - - out.append(text[label_open + 1 : label_close]) - i = link_end - return "".join(out) - + cursor = 0 + search_from = 0 -def markdown_table_cells(line: str) -> list[str] | None: - """Split a pipe table on unescaped delimiters outside backtick code spans.""" - if is_indented_code_line(line): - return None - stripped = line.strip() - if not stripped.startswith("|"): - return None + while True: + match = RECORD_LINK_START_RE.search(text, search_from) + if match is None: + out.append(text[cursor:]) + break - cells: list[str] = [] - current: list[str] = [] - code_run_len: int | None = None - i = 1 - while i < len(stripped): - char = stripped[i] - if char == "`" and not is_backslash_escaped(stripped, i): - run_len = backtick_run_length(stripped, i) - if code_run_len is None: - code_run_len = run_len - elif run_len == code_run_len: - code_run_len = None - current.append(stripped[i : i + run_len]) - i += run_len + after_destination = match.end() + if after_destination >= len(text) or text[after_destination] == ")": + search_from = after_destination continue - - if char == "|" and code_run_len is None: - if is_backslash_escaped(stripped, i): - if current and current[-1] == "\\": - current.pop() - current.append("|") - else: - cells.append("".join(current).strip()) - current = [] - i += 1 + if text[after_destination] not in " \t\r\n": + search_from = after_destination continue - current.append(char) - i += 1 - - trailing_pipe_is_delimiter = ( - stripped.endswith("|") and not is_backslash_escaped(stripped, len(stripped) - 1) - ) - if current or not trailing_pipe_is_delimiter: - cells.append("".join(current).strip()) - return cells - - -def extract_markdown_table( - lines: list[str], expected_headers: tuple[str, ...], context: str -) -> list[list[str]]: - """Extract one visible table and return validated data rows as cell lists.""" - visible = visible_nonfenced_lines(lines) - expected = list(expected_headers) - for i, line in enumerate(visible): - if markdown_table_cells(line) != expected: + title_start = _skip_whitespace(text, after_destination) + title_end = _parse_title(text, title_start) + if title_end is None: + search_from = after_destination continue - if i + 1 >= len(visible): - die(f"{context} table has no separator row") - separator = markdown_table_cells(visible[i + 1]) - if ( - separator is None - or len(separator) != len(expected) - or not all(TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in separator) - ): - die(f"{context} table has an invalid separator row") - - rows: list[list[str]] = [] - for row in visible[i + 2 :]: - cells = markdown_table_cells(row) - if cells is None: - break - if len(cells) != len(expected): - die( - f"{context} table row has {len(cells)} column(s); " - f"expected {len(expected)}: {row.strip()}" - ) - if any(not cell for cell in cells): - die(f"{context} table row contains an empty required cell: {row.strip()}") - rows.append(cells) - return rows - die(f"{context} is missing the expected Markdown table") - - -def unwrap_outer_formatting(value: str, wrappers: tuple[str, ...]) -> str: - """Remove only balanced formatting that wraps the complete value.""" - result = value.strip() - changed = True - while changed: - changed = False - for marker in wrappers: - if ( - len(result) > 2 * len(marker) - and result.startswith(marker) - and result.endswith(marker) - ): - result = result[len(marker) : -len(marker)].strip() - changed = True - break - return result - - -def unwrap_markdown_emphasis(cell: str) -> str: - return unwrap_outer_formatting(cell, EMPHASIS_WRAPPERS) - - -def parse_record_link_cell(cell: str, context: str) -> tuple[str, str]: - value = unwrap_markdown_emphasis(cell) - match = RECORD_LINK_CELL_RE.fullmatch(value) - if match is None: - die(f"{context} has invalid record-link cell: {cell}") - return match.group(1), match.group(2) - - -def rendered_inline_text(value: str) -> str: - """Approximate rendered inline text for required field-value validation.""" - text, protected_code = protect_code_spans(value) - text = strip_inline_links(text) - text = REFERENCE_IMAGE_RE.sub(lambda m: m.group(1), text) - text = REFERENCE_LINK_RE.sub(lambda m: m.group(1), text) - text = INLINE_HTML_TAG_RE.sub("", text) - text = re.sub(r"[`*_~]", "", text) - text = re.sub(r"\\(.)", r"\1", text) - for token, code_text in protected_code.items(): - text = text.replace(token, code_text) - return html.unescape(text).strip() - - -def has_substantive_rendered_text(value: str) -> bool: - return any(ch.isalnum() for ch in rendered_inline_text(value)) - - -def reference_definition_hidden_indexes(lines: list[str]) -> set[int]: - """Return lines consumed by non-rendering reference definitions/titles.""" - hidden: set[int] = set() - for i, raw in enumerate(lines): - if not LINK_REFERENCE_DEFINITION_RE.fullmatch(raw.strip()): - continue - hidden.add(i) - if ( - i + 1 < len(lines) - and LINK_REFERENCE_TITLE_CONTINUATION_RE.fullmatch(lines[i + 1]) - ): - hidden.add(i + 1) - return hidden - - -def is_structural_only_line(line: str) -> bool: - if HEADING_RE.match(line) or THEMATIC_BREAK_RE.fullmatch(line): - return True - if LIST_MARKER_ONLY_RE.fullmatch(line) or line == ">": - return True - if LINK_REFERENCE_DEFINITION_RE.fullmatch(line): - return True - cells = markdown_table_cells(line) - return bool(cells and all(TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in cells)) - - -def section_has_content(lines: list[str]) -> bool: - visible = visible_nonfenced_lines(lines) - hidden_reference_lines = reference_definition_hidden_indexes(visible) - for index, raw in enumerate(visible): - if index in hidden_reference_lines: - continue - line = raw.strip() - if not line or line in TEMPLATE_PLACEHOLDER_LINES: - continue - if EMPTY_LABEL_RE.match(line) or is_structural_only_line(line): - continue - if not has_substantive_rendered_text(line): - continue - return True - return False - - -def source_section_has_identity(lines: list[str]) -> bool: - """Require at least one concrete, non-placeholder provenance identity.""" - sources_root = (ROOT / "sources").resolve() - for raw in visible_nonfenced_lines(lines): - line = raw.strip() - if not line or SOURCE_PLACEHOLDER_RE.fullmatch(line): + outer_close = _skip_whitespace(text, title_end) + if outer_close >= len(text) or text[outer_close] != ")": + search_from = after_destination continue - if SOURCE_URL_RE.search(line) or SOURCE_DOI_RE.search(line) or SOURCE_COMMIT_RE.search(line): - return True - for match in SOURCE_LOCAL_NOTE_RE.finditer(line): - candidate = (ROOT / match.group(1)).resolve() - try: - candidate.relative_to(sources_root) - except ValueError: - continue - if candidate.is_file(): - return True - if SOURCE_REPOSITORY_RE.search(line): - return True - return False - - -def normalized_status_category(raw: str) -> str: - category = raw.split(";", 1)[0].strip() - return unwrap_outer_formatting(category, STATUS_WRAPPERS) - -def require_prefixed_fields( - path: Path, - lines: list[str], - fields: tuple[str, ...], - section: str, - rejected_values: dict[str, str] | None = None, -) -> None: - visible = visible_nonfenced_lines(lines) - for field in fields: - prefix = f"- {field}:" - matches = [line for line in visible if line.startswith(prefix)] - if len(matches) != 1: - die( - f"{path.relative_to(ROOT)} must contain exactly one visible field " - f"'{prefix}' in {section}" - ) - value = matches[0][len(prefix) :].strip() - if not value: - die(f"{path.relative_to(ROOT)} has empty field {field} in {section}") - if not has_substantive_rendered_text(value): - die( - f"{path.relative_to(ROOT)} has markup-only/non-substantive field " - f"{field} in {section}: '{value}'" - ) - if rejected_values is not None: - normalized_value = html.unescape( - unwrap_outer_formatting(value, STATUS_WRAPPERS) - ).strip() - if normalized_value == rejected_values.get(field): - die( - f"{path.relative_to(ROOT)} has unselected template placeholder " - f"for {field} in {section}: '{value}'" - ) - - -records: dict[str, Path] = {} -status_categories: dict[str, str] = {} -for path in sorted(OPT_DIR.glob("*.md")): - text = path.read_text(encoding="utf-8") - lines = visible_nonfenced_lines(markdown_source_lines(text)) - first = lines[0] if lines else "" - match = ID_RE.match(first) - if not match: - die(f"bad or hidden record heading: {path.relative_to(ROOT)}") - record_id = match.group(1) - - filename_match = FILENAME_ID_RE.match(path.name) - if not filename_match: - die( - f"record Markdown filename does not follow OPT---... convention: " - f"{path.relative_to(ROOT)}" - ) - if filename_match.group(1) != record_id: - die( - f"record ID mismatch: {path.relative_to(ROOT)} declares {record_id} " - f"but filename encodes {filename_match.group(1)}" - ) - if record_id in records: - die(f"duplicate record id {record_id}: {records[record_id]} and {path}") - records[record_id] = path + out.append(text[cursor:match.start()]) + out.append(f"[{match.group(1)}]({match.group(2)})") + cursor = outer_close + 1 + search_from = cursor - statuses = [m.group(1).strip() for line in lines if (m := STATUS_RE.match(line))] - if len(statuses) != 1: - die(f"{path.relative_to(ROOT)} must contain exactly one visible Status line") - if not statuses[0]: - die(f"{path.relative_to(ROOT)} has empty Status") - - if record_id in FROZEN_V1: - continue - - status_category = statuses[0].split(";", 1)[0].strip() - if status_category not in ALLOWED_V2_STATUS_CATEGORIES: - die( - f"{path.relative_to(ROOT)} uses undefined status category " - f"'{status_category}'" - ) - status_categories[record_id] = status_category - - headings = {line for line in lines if line.startswith("## ")} - missing = sorted(REQUIRED_V2 - headings) - if missing: - die(f"{path.relative_to(ROOT)} missing visible sections: {', '.join(missing)}") - - for heading in sorted(REQUIRED_V2): - if not section_has_content(section_lines(text, heading)): - die( - f"{path.relative_to(ROOT)} has empty/template/structural/markup-only mandatory section {heading}" - ) - - source_evidence = section_lines(text, "## Source evidence") - if not source_section_has_identity(source_evidence): - die( - f"{path.relative_to(ROOT)} ## Source evidence lacks a concrete source identity " - "(URL, DOI, pinned commit, existing sources/*.md note, or repository identity)" - ) - - contract = section_lines(text, "## Optimization problem contract") - require_prefixed_fields( - path, contract, REQUIRED_CONTRACT_FIELDS, "## Optimization problem contract" - ) - require_prefixed_fields( - path, - contract, - REQUIRED_CLASSIFICATION_FIELDS, - "## Optimization problem contract", - rejected_values=CLASSIFICATION_TEMPLATE_VALUES, - ) - -missing_frozen = sorted(FROZEN_V1 - records.keys()) -if missing_frozen: - die(f"frozen v1 record(s) missing: {', '.join(missing_frozen)}") + return "".join(out) -record_paths = {str(path.relative_to(ROOT)): record_id for record_id, path in records.items()} -for doc_name in ("README.md", "CATALOG.md"): - text = (ROOT / doc_name).read_text(encoding="utf-8") - rendered = visible_text(text) - for label, rel in LINK_RE.findall(rendered): - target = ROOT / rel - if not target.is_file(): - die(f"broken visible record link in {doc_name}: {rel}") - target_id = record_paths.get(rel) - if target_id is None: - die(f"record link in {doc_name} is not a discovered OPT record: {rel}") - if unwrap_markdown_emphasis(label) != target_id: - die( - f"record link label mismatch in {doc_name}: '{label}' points to " - f"{target_id} ({rel})" - ) +def canonicalize_markdown(text: str) -> str: + text = ATX_INDENT_RE.sub("", text) + return canonicalize_record_link_titles(text) - if doc_name != "README.md": - continue - catalog_lines = section_lines(text, "## Catalog") - if not catalog_lines: - die("README.md is missing a non-empty visible ## Catalog section") - catalog_rows = extract_markdown_table( - catalog_lines, - ("ID", "Optimization", "Status", "Core idea"), - "README.md ## Catalog", - ) +def markdown_inputs(root: Path) -> list[Path]: + paths = [ + root / "README.md", + root / "CATALOG.md", + root / "OPTIMIZATION-PROBLEM.md", + ] + paths.extend(sorted((root / "optimizations").glob("*.md"))) + return [path for path in paths if path.is_file()] - parsed_rows: list[tuple[str, str, str]] = [] - for cells in catalog_rows: - row_id, rel = parse_record_link_cell(cells[0], "README.md ## Catalog") - parsed_rows.append((row_id, rel, cells[2])) - counts = Counter(row_id for row_id, _rel, _status in parsed_rows) - bad_counts = sorted(record_id for record_id, count in counts.items() if count != 1) - if bad_counts: - die( - "README.md ## Catalog table must index each record exactly once; " - f"bad row counts for: {', '.join(bad_counts)}" +def main() -> int: + with tempfile.TemporaryDirectory(prefix="opt-catalog-integrity-") as temp_dir: + scratch = Path(temp_dir) / "repo" + shutil.copytree( + ROOT, + scratch, + symlinks=True, + ignore=shutil.ignore_patterns(".git", "__pycache__"), ) - missing_readme = sorted(records.keys() - counts.keys()) - if missing_readme: - die(f"README.md ## Catalog table is missing record(s): {', '.join(missing_readme)}") - unknown_rows = sorted(counts.keys() - records.keys()) - if unknown_rows: - die(f"README.md ## Catalog table references unknown record(s): {', '.join(unknown_rows)}") - row_statuses: dict[str, str] = {} - for row_id, rel, raw_status in parsed_rows: - if record_paths.get(rel) != row_id: - die(f"README.md ## Catalog row identity mismatch for {row_id}: {rel}") - if row_id in row_statuses: - die(f"README.md ## Catalog has duplicate status row for {row_id}") - row_statuses[row_id] = normalized_status_category(raw_status) + for path in markdown_inputs(scratch): + original = path.read_text(encoding="utf-8") + normalized = canonicalize_markdown(original) + if normalized != original: + path.write_text(normalized, encoding="utf-8") - for record_id, expected_status in status_categories.items(): - observed_status = row_statuses.get(record_id) - if observed_status is None: - die(f"README.md ## Catalog has no status cell for post-v1 record {record_id}") - if observed_status != expected_status: - die( - f"README.md status mismatch for {record_id}: " - f"record='{expected_status}' README='{observed_status}'" - ) + core = scratch / "scripts" / CORE_NAME + if not core.is_file(): + raise SystemExit(f"catalog-integrity: missing hardened core checker: {CORE_NAME}") -catalog = (ROOT / "CATALOG.md").read_text(encoding="utf-8") -visible_catalog = visible_text(catalog) -catalog_ids = set(OPT_TOKEN_RE.findall(visible_catalog)) -unknown_catalog_ids = sorted(catalog_ids - records.keys()) -if unknown_catalog_ids: - die(f"CATALOG.md references unknown visible record ID(s): {', '.join(unknown_catalog_ids)}") -for record_id, path in records.items(): - if record_id not in catalog_ids: - die(f"{record_id} ({path.name}) is not visibly mentioned in CATALOG.md") + completed = subprocess.run([sys.executable, str(core)], cwd=scratch, check=False) + return completed.returncode -decision_lines = section_lines(catalog, "## Quick decision table") -if not decision_lines: - die("CATALOG.md is missing a non-empty visible ## Quick decision table section") -decision_rows = extract_markdown_table( - decision_lines, - ("Bottleneck / problem shape", "First record to inspect", "Core idea"), - "CATALOG.md ## Quick decision table", -) - -parsed_decisions: list[tuple[str, str]] = [] -for cells in decision_rows: - row_id, rel = parse_record_link_cell( - cells[1], "CATALOG.md ## Quick decision table" - ) - parsed_decisions.append((row_id, rel)) - -decision_counts = Counter(record_id for record_id, _rel in parsed_decisions) -bad_decision_counts = sorted( - record_id for record_id, count in decision_counts.items() if count != 1 -) -if bad_decision_counts: - die( - "CATALOG.md ## Quick decision table must index each record exactly once; " - f"bad row counts for: {', '.join(bad_decision_counts)}" - ) -missing_decision = sorted(records.keys() - decision_counts.keys()) -if missing_decision: - die( - "CATALOG.md ## Quick decision table is missing record(s): " - f"{', '.join(missing_decision)}" - ) -unknown_decision = sorted(decision_counts.keys() - records.keys()) -if unknown_decision: - die( - "CATALOG.md ## Quick decision table references unknown record(s): " - f"{', '.join(unknown_decision)}" - ) -for row_id, rel in parsed_decisions: - if record_paths.get(rel) != row_id: - die(f"CATALOG.md ## Quick decision table row identity mismatch for {row_id}: {rel}") - -problem_contract = ROOT / "OPTIMIZATION-PROBLEM.md" -if not problem_contract.is_file(): - die("OPTIMIZATION-PROBLEM.md is missing") -problem_text = problem_contract.read_text(encoding="utf-8") -problem_visible = visible_nonfenced_lines(markdown_source_lines(problem_text)) -if not problem_visible or problem_visible[0] != "# Optimization Problem Contract": - die("OPTIMIZATION-PROBLEM.md has missing/hidden/invalid title") -if "## Canonical contract" not in problem_visible: - die("OPTIMIZATION-PROBLEM.md is missing visible ## Canonical contract") -canonical = section_lines(problem_text, "## Canonical contract") -canonical_text = "\n".join(canonical) -if "P = (X, F, f, d, C, B, S)" not in canonical_text: - die("OPTIMIZATION-PROBLEM.md is missing visible canonical P = (X, F, f, d, C, B, S) formula") -for field, pattern in CANONICAL_DEFINITION_PATTERNS.items(): - if not any(pattern.match(line) for line in canonical): - die(f"OPTIMIZATION-PROBLEM.md is missing visible canonical definition for {field}") - -classification_lines = section_lines(problem_text, "## Required classification") -if not classification_lines: - die("OPTIMIZATION-PROBLEM.md is missing visible ## Required classification") -classification_rows = extract_markdown_table( - classification_lines, - ("Dimension", "Typical values"), - "OPTIMIZATION-PROBLEM.md ## Required classification", -) -canonical_classification: dict[str, str] = {} -for row in classification_rows: - dimension, typical_values = row - if dimension in canonical_classification: - die(f"OPTIMIZATION-PROBLEM.md has duplicate classification dimension {dimension}") - canonical_classification[dimension] = typical_values - -expected_dimensions = set(REQUIRED_CLASSIFICATION_FIELDS) -observed_dimensions = set(canonical_classification) -if observed_dimensions != expected_dimensions: - missing_dimensions = sorted(expected_dimensions - observed_dimensions) - unknown_dimensions = sorted(observed_dimensions - expected_dimensions) - details: list[str] = [] - if missing_dimensions: - details.append(f"missing={','.join(missing_dimensions)}") - if unknown_dimensions: - details.append(f"unknown={','.join(unknown_dimensions)}") - die( - "OPTIMIZATION-PROBLEM.md classification dimensions do not match the checker: " - + "; ".join(details) - ) -for dimension in REQUIRED_CLASSIFICATION_FIELDS: - canonical_value = canonical_classification[dimension] - checker_value = CLASSIFICATION_TEMPLATE_VALUES[dimension] - if canonical_value != checker_value: - die( - f"classification placeholder drift for {dimension}: " - f"canonical='{canonical_value}' checker='{checker_value}'" - ) -print(f"CATALOG_INTEGRITY_OK records={len(records)} frozen_v1={len(FROZEN_V1)}") +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/check_catalog_core.py b/scripts/check_catalog_core.py new file mode 100755 index 0000000..0142fb0 --- /dev/null +++ b/scripts/check_catalog_core.py @@ -0,0 +1,1012 @@ +#!/usr/bin/env python3 +"""Check OPT catalog/document integrity without external dependencies.""" + +from __future__ import annotations + +import html +import re +from collections import Counter +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +OPT_DIR = ROOT / "optimizations" +FROZEN_V1 = { + "OPT-PY-001", + "OPT-INV-001", + "OPT-LEAN-001", + "OPT-PAR-001", + "OPT-DSP-001", +} +REQUIRED_V2 = { + "## Source evidence", + "## Problem", + "## Optimization problem contract", + "## Preserved contract", + "## Optimization", + "## Before / after evidence", + "## Validation", + "## Target-repo adaptation", + "## Failure modes", + "## Rollback trigger", +} +REQUIRED_CONTRACT_FIELDS = ("X", "F", "f", "d", "C", "B", "S") +REQUIRED_CLASSIFICATION_FIELDS = ( + "Variables", + "Search scope", + "Objective behavior", + "Information", + "Evaluation cost", + "Constraints", + "Parallelism", + "Exactness", +) +CLASSIFICATION_TEMPLATE_VALUES = { + "Variables": "continuous / integer / categorical / conditional / mixed", + "Search scope": "local / global", + "Objective behavior": "deterministic / noisy / stochastic", + "Information": "gradient available / derivative-free / black-box", + "Evaluation cost": "cheap / moderate / expensive", + "Constraints": "bounds / equality / inequality / semantic / resource", + "Parallelism": "sequential / synchronous batch / asynchronous", + "Exactness": "exact / approximation permitted under an explicit error contract", +} +ALLOWED_V2_STATUS_CATEGORIES = { + "Verified", + "Verified, environment-specific", + "Implemented reference", + "Implemented external reference", + "Implemented external pattern", + "Proposed / OPT synthesis", + "Source candidate", +} +TEMPLATE_PLACEHOLDER_LINES = { + "- Repository / publication / article:", + "- Release/commit/PR/DOI/date:", + "- Exact files/sections where applicable:", + "- Licensing/provenance boundary where code reuse may matter:", + "What dominates runtime, latency, memory, I/O, CI cost, quality budget or optimization-evaluation cost?", + "State exactly what must remain unchanged: output bytes, theorem targets, assertions, API, numerical tolerance, ordering, statistical guarantee, evidence boundary, trust model, etc.", + "If the optimization changes the contract (for example exact → approximate), state the new contract explicitly instead of claiming preservation.", + "Describe the reusable mechanism, not only the source-project patch.", + "If no controlled benchmark exists, say so explicitly.", + "How was equivalence, correctness, bound soundness, approximation error or other contract compliance established?", + "Which source constants, thresholds, worker counts, bit splits, cache keys, search budgets or tolerances must be re-profiled rather than copied?", + "What can make this optimization invalid, slower, less robust or misleading?", + "Define the measured or semantic condition that disables/reverts the optimization.", +} + +LINK_RE = re.compile(r"\[([^\]]+)\]\((optimizations/[^)#]+\.md)\)") +RECORD_LINK_CELL_RE = re.compile( + r"^\[(OPT-[A-Z]+-\d{3})\]\((optimizations/[^)#]+\.md)\)$" +) +ID_RE = re.compile(r"^# (OPT-[A-Z]+-\d{3}) — ") +FILENAME_ID_RE = re.compile(r"^(OPT-[A-Z]+-\d{3})-") +STATUS_RE = re.compile(r"^\*\*Status:\*\*\s*(.*?)\s*$") +OPT_TOKEN_RE = re.compile(r"\bOPT-[A-Z]+-\d{3}\b") +EMPTY_LABEL_RE = re.compile(r"^-\s+[^:]+:\s*$") +LINK_REFERENCE_DEFINITION_RE = re.compile( + r"^\[(?:\\.|[^\[\]\\])+\]:[ \t]+\S.*$" +) +LINK_REFERENCE_TITLE_CONTINUATION_RE = re.compile( + r"^ {0,3}(?:\"(?:\\.|[^\"\\])*\"|'(?:\\.|[^'\\])*'|\((?:\\.|[^)\\])*\))[ \t]*$" +) +SOURCE_URL_RE = re.compile(r"https?://\S+", re.IGNORECASE) +SOURCE_DOI_RE = re.compile(r"\b(?:doi:\s*)?10\.\d{4,9}/\S+", re.IGNORECASE) +SOURCE_COMMIT_RE = re.compile(r"\b[0-9a-f]{7,40}\b", re.IGNORECASE) +SOURCE_LOCAL_NOTE_RE = re.compile(r"`?(sources/[A-Za-z0-9._/-]+\.md)`?") +SOURCE_REPOSITORY_RE = re.compile(r"`[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+`") +SOURCE_PLACEHOLDER_RE = re.compile( + r"^(?:[-*+]\s*)?(?:unknown|tbd|todo|n/?a|none|pending)\.?$", re.IGNORECASE +) +REFERENCE_IMAGE_RE = re.compile(r"!\[([^\]]*)\]\[[^\]]*\]") +REFERENCE_LINK_RE = re.compile(r"\[([^\]]*)\]\[[^\]]*\]") +INLINE_HTML_TAG_RE = re.compile( + r"`]+))?)*" + r"[ \t]*/?>" +) +HEADING_RE = re.compile(r"^#{1,6}(?:\s|$)") +SECTION_BOUNDARY_RE = re.compile(r"^#{1,2}(?:\s|$)") +SETEXT_H1_RE = re.compile(r"^ {0,3}=+[ \t]*$") +SETEXT_H2_RE = re.compile(r"^ {0,3}-+[ \t]*$") +THEMATIC_BREAK_RE = re.compile( + r"^(?:\*(?:[ \t]*\*){2,}|-(?:[ \t]*-){2,}|_(?:[ \t]*_){2,})[ \t]*$" +) +LIST_MARKER_ONLY_RE = re.compile(r"^(?:[-+*]|\d+[.)])$") +TABLE_SEPARATOR_CELL_RE = re.compile(r"^:?-{3,}:?$") +FENCE_OPEN_RE = re.compile(r"^ {0,3}(`{3,}|~{3,})(.*)$") +RAW_HTML_TYPE1_OPEN_RE = re.compile( + r"^ {0,3}<(?Pscript|pre|style|textarea)(?:[ \t]|>|$)", re.IGNORECASE +) +RAW_HTML_DECLARATION_OPEN_RE = re.compile(r"^ {0,3}]|$)", + re.IGNORECASE, +) +RAW_HTML_COMPLETE_TAG_RE = re.compile( + r"^ {0,3}(?:" + r"" + r"|<[A-Za-z][A-Za-z0-9-]*" + r"(?:[ \t]+[A-Za-z_:][A-Za-z0-9_.:-]*" + r"(?:[ \t]*=[ \t]*(?:\"[^\"]*\"|'[^']*'|[^ \t\n\"'=<>`]+))?)*" + r"[ \t]*/?>" + r")[ \t]*$" +) +EMPHASIS_WRAPPERS = ("**", "__", "~~", "*", "_") +STATUS_WRAPPERS = ("**", "__", "~~", "*", "_", "`") +CANONICAL_DEFINITION_PATTERNS = { + "X": re.compile(r"^- `X` — \S"), + "F": re.compile(r"^- `F(?: ⊆ X)?` — \S"), + "f": re.compile(r"^- `f(?:\s*:[^`]*)?` — \S"), + "d": re.compile(r"^- `d` — \S"), + "C": re.compile(r"^- `C` — \S"), + "B": re.compile(r"^- `B` — \S"), + "S": re.compile(r"^- `S` — \S"), +} + + +def die(msg: str) -> None: + raise SystemExit(f"catalog-integrity: {msg}") + + +def markdown_source_lines(text: str) -> list[str]: + """Split only on CommonMark line endings (LF, CRLF, or CR).""" + return text.replace("\r\n", "\n").replace("\r", "\n").split("\n") + + +def is_indented_code_line(raw: str) -> bool: + """Return whether a non-fenced line is an indented Markdown code line.""" + return raw.startswith("\t") or raw.startswith(" ") + + +def strip_inline_html_comments(raw: str, in_comment: bool) -> tuple[str, bool]: + """Strip inline HTML comments while carrying a mid-line unmatched comment.""" + out: list[str] = [] + cursor = 0 + + if in_comment: + end = raw.find("-->") + if end < 0: + return "", True + cursor = end + 3 + in_comment = False + + while cursor < len(raw): + start = raw.find("", start + 4) + if end < 0: + in_comment = True + break + cursor = end + 3 + + return "".join(out), in_comment + + +def raw_html_block_start(raw: str) -> tuple[str, str | None] | None: + """Return the raw-HTML block mode for a CommonMark-style block start.""" + if re.match(r"^ {0,3}" + + type1 = RAW_HTML_TYPE1_OPEN_RE.match(raw) + if type1 is not None: + return "tag", type1.group("tag").lower() + if re.match(r"^ {0,3}<\?", raw): + return "token", "?>" + if re.match(r"^ {0,3}".replace(" ", "") + if RAW_HTML_DECLARATION_OPEN_RE.match(raw): + return "token", ">" + if RAW_HTML_BLOCK_TAG_RE.match(raw) or RAW_HTML_COMPLETE_TAG_RE.match(raw): + return "blank", None + return None + + +def raw_html_tag_closes(raw: str, tag: str) -> bool: + """Match CommonMark type-1 block terminators exactly (case-insensitive).""" + return re.search(rf"", raw, re.IGNORECASE) is not None + + +def visible_nonfenced_lines(lines: list[str]) -> list[str]: + """Return Markdown-visible lines used by schema validation.""" + visible: list[str] = [] + fence_char: str | None = None + fence_len = 0 + inline_comment = False + html_mode: str | None = None + html_end: str | None = None + + for raw in lines: + if fence_char is not None: + close = re.fullmatch( + rf" {{0,3}}{re.escape(fence_char)}{{{fence_len},}}[ \t]*", raw + ) + if close is not None: + fence_char = None + fence_len = 0 + continue + + if html_mode is not None: + if html_mode == "tag": + if html_end is not None and raw_html_tag_closes(raw, html_end): + html_mode = None + html_end = None + continue + if html_mode == "token": + if html_end is not None and html_end in raw: + html_mode = None + html_end = None + continue + if html_mode == "blank": + if raw.strip() == "": + html_mode = None + html_end = None + visible.append("") + continue + + if inline_comment: + rendered, inline_comment = strip_inline_html_comments(raw, True) + if inline_comment: + continue + raw_for_parse = rendered + else: + if is_indented_code_line(raw): + continue + + opener = FENCE_OPEN_RE.match(raw) + if opener is not None: + run = opener.group(1) + info = opener.group(2) + if run[0] != "`" or "`" not in info: + fence_char = run[0] + fence_len = len(run) + continue + + html_start = raw_html_block_start(raw) + if html_start is not None: + html_mode, html_end = html_start + if html_mode == "tag" and html_end is not None and raw_html_tag_closes(raw, html_end): + html_mode = None + html_end = None + elif html_mode == "token" and html_end is not None and html_end in raw: + html_mode = None + html_end = None + continue + + raw_for_parse, inline_comment = strip_inline_html_comments(raw, False) + + if raw_for_parse and is_indented_code_line(raw_for_parse): + continue + if raw_for_parse: + visible.append(raw_for_parse) + elif not inline_comment and raw == "": + visible.append("") + + return visible + + +def visible_text(text: str) -> str: + return "\n".join(visible_nonfenced_lines(markdown_source_lines(text))) + + +def setext_heading_start( + lines: list[str], underline_index: int, minimum_index: int +) -> int | None: + """Return the first source line of a Setext heading paragraph.""" + if underline_index <= minimum_index: + return None + underline = lines[underline_index] + if not ( + SETEXT_H1_RE.fullmatch(underline) or SETEXT_H2_RE.fullmatch(underline) + ): + return None + + candidate = underline_index - 1 + if candidate < minimum_index or not lines[candidate].strip(): + return None + if HEADING_RE.match(lines[candidate]): + return None + + start = candidate + while start > minimum_index: + previous = lines[start - 1] + if not previous.strip() or SECTION_BOUNDARY_RE.match(previous): + break + start -= 1 + return start + + +def section_lines(text: str, heading: str) -> list[str]: + """Return one exact visible level-2 Markdown section.""" + lines = visible_nonfenced_lines(markdown_source_lines(text)) + try: + start = lines.index(heading) + 1 + except ValueError: + return [] + end = len(lines) + for i in range(start, len(lines)): + if SECTION_BOUNDARY_RE.match(lines[i]): + end = i + break + setext_start = setext_heading_start(lines, i, start) + if setext_start is not None: + end = setext_start + break + return lines[start:end] + + +def is_backslash_escaped(text: str, index: int) -> bool: + count = 0 + cursor = index - 1 + while cursor >= 0 and text[cursor] == "\\": + count += 1 + cursor -= 1 + return count % 2 == 1 + + +def backtick_run_length(text: str, index: int) -> int: + cursor = index + while cursor < len(text) and text[cursor] == "`": + cursor += 1 + return cursor - index + + +def protect_code_spans(text: str) -> tuple[str, dict[str, str]]: + """Replace parsed code spans with private-use sentinels and preserve their text.""" + out: list[str] = [] + protected: dict[str, str] = {} + i = 0 + while i < len(text): + if text[i] != "`" or is_backslash_escaped(text, i): + out.append(text[i]) + i += 1 + continue + + run_len = backtick_run_length(text, i) + j = i + run_len + close_start: int | None = None + close_end: int | None = None + while j < len(text): + if text[j] != "`": + j += 1 + continue + candidate_len = backtick_run_length(text, j) + if candidate_len == run_len: + close_start = j + close_end = j + candidate_len + break + j += candidate_len + + if close_start is None or close_end is None: + out.append(text[i : i + run_len]) + i += run_len + continue + + token = chr(0xE000 + len(protected)) + protected[token] = text[i + run_len : close_start] + out.append(token) + i = close_end + + return "".join(out), protected + + +def find_label_close(text: str, open_index: int) -> int | None: + depth = 1 + i = open_index + 1 + while i < len(text): + if text[i] == "\\" and i + 1 < len(text): + i += 2 + continue + if text[i] == "[": + depth += 1 + elif text[i] == "]": + depth -= 1 + if depth == 0: + return i + i += 1 + return None + + +def parse_link_title_and_close(text: str, index: int) -> int | None: + """Parse whitespace plus an optional CommonMark-style title and outer close.""" + i = index + while i < len(text) and text[i] in " \t\n": + i += 1 + if i < len(text) and text[i] == ")": + return i + 1 + if i >= len(text): + return None + + opener = text[i] + if opener not in ('"', "'", "("): + return None + closer = ")" if opener == "(" else opener + i += 1 + while i < len(text): + if text[i] == "\\" and i + 1 < len(text): + i += 2 + continue + if text[i] == closer: + i += 1 + break + if text[i] == "\n": + return None + i += 1 + else: + return None + + while i < len(text) and text[i] in " \t\n": + i += 1 + if i < len(text) and text[i] == ")": + return i + 1 + return None + + +def find_inline_link_end(text: str, open_paren: int) -> int | None: + """Return the end of a valid inline-link destination/title, or None.""" + i = open_paren + 1 + while i < len(text) and text[i] in " \t\n": + i += 1 + if i >= len(text): + return None + if text[i] == ")": + return i + 1 + + if text[i] == "<": + i += 1 + while i < len(text): + if text[i] == "\\" and i + 1 < len(text): + i += 2 + continue + if text[i] == ">": + return parse_link_title_and_close(text, i + 1) + if text[i] in "\n<": + return None + i += 1 + return None + + depth = 0 + while i < len(text): + char = text[i] + if char == "\\" and i + 1 < len(text): + i += 2 + continue + if char == "(": + depth += 1 + i += 1 + continue + if char == ")": + if depth == 0: + return i + 1 + depth -= 1 + i += 1 + continue + if char in " \t\n" and depth == 0: + return parse_link_title_and_close(text, i) + if char in "<>" or ord(char) < 0x20: + return None + i += 1 + return None + + +def strip_inline_links(text: str) -> str: + """Keep rendered labels while discarding valid inline-link/image destinations.""" + out: list[str] = [] + i = 0 + while i < len(text): + image = text.startswith("![", i) + if image: + label_open = i + 1 + elif text[i] == "[": + label_open = i + else: + out.append(text[i]) + i += 1 + continue + + label_close = find_label_close(text, label_open) + if label_close is None or label_close + 1 >= len(text) or text[label_close + 1] != "(": + out.append(text[i]) + i += 1 + continue + link_end = find_inline_link_end(text, label_close + 1) + if link_end is None: + out.append(text[i]) + i += 1 + continue + + out.append(text[label_open + 1 : label_close]) + i = link_end + return "".join(out) + + +def markdown_table_cells(line: str) -> list[str] | None: + """Split a pipe table on unescaped delimiters outside backtick code spans.""" + if is_indented_code_line(line): + return None + stripped = line.strip() + if not stripped.startswith("|"): + return None + + cells: list[str] = [] + current: list[str] = [] + code_run_len: int | None = None + i = 1 + while i < len(stripped): + char = stripped[i] + if char == "`" and not is_backslash_escaped(stripped, i): + run_len = backtick_run_length(stripped, i) + if code_run_len is None: + code_run_len = run_len + elif run_len == code_run_len: + code_run_len = None + current.append(stripped[i : i + run_len]) + i += run_len + continue + + if char == "|" and code_run_len is None: + if is_backslash_escaped(stripped, i): + if current and current[-1] == "\\": + current.pop() + current.append("|") + else: + cells.append("".join(current).strip()) + current = [] + i += 1 + continue + + current.append(char) + i += 1 + + trailing_pipe_is_delimiter = ( + stripped.endswith("|") and not is_backslash_escaped(stripped, len(stripped) - 1) + ) + if current or not trailing_pipe_is_delimiter: + cells.append("".join(current).strip()) + return cells + + +def extract_markdown_table( + lines: list[str], expected_headers: tuple[str, ...], context: str +) -> list[list[str]]: + """Extract one visible table and return validated data rows as cell lists.""" + visible = visible_nonfenced_lines(lines) + expected = list(expected_headers) + for i, line in enumerate(visible): + if markdown_table_cells(line) != expected: + continue + if i + 1 >= len(visible): + die(f"{context} table has no separator row") + separator = markdown_table_cells(visible[i + 1]) + if ( + separator is None + or len(separator) != len(expected) + or not all(TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in separator) + ): + die(f"{context} table has an invalid separator row") + + rows: list[list[str]] = [] + for row in visible[i + 2 :]: + cells = markdown_table_cells(row) + if cells is None: + break + if len(cells) != len(expected): + die( + f"{context} table row has {len(cells)} column(s); " + f"expected {len(expected)}: {row.strip()}" + ) + if any(not cell for cell in cells): + die(f"{context} table row contains an empty required cell: {row.strip()}") + rows.append(cells) + return rows + die(f"{context} is missing the expected Markdown table") + + +def unwrap_outer_formatting(value: str, wrappers: tuple[str, ...]) -> str: + """Remove only balanced formatting that wraps the complete value.""" + result = value.strip() + changed = True + while changed: + changed = False + for marker in wrappers: + if ( + len(result) > 2 * len(marker) + and result.startswith(marker) + and result.endswith(marker) + ): + result = result[len(marker) : -len(marker)].strip() + changed = True + break + return result + + +def unwrap_markdown_emphasis(cell: str) -> str: + return unwrap_outer_formatting(cell, EMPHASIS_WRAPPERS) + + +def parse_record_link_cell(cell: str, context: str) -> tuple[str, str]: + value = unwrap_markdown_emphasis(cell) + match = RECORD_LINK_CELL_RE.fullmatch(value) + if match is None: + die(f"{context} has invalid record-link cell: {cell}") + return match.group(1), match.group(2) + + +def rendered_inline_text(value: str) -> str: + """Approximate rendered inline text for required field-value validation.""" + text, protected_code = protect_code_spans(value) + text = strip_inline_links(text) + text = REFERENCE_IMAGE_RE.sub(lambda m: m.group(1), text) + text = REFERENCE_LINK_RE.sub(lambda m: m.group(1), text) + text = INLINE_HTML_TAG_RE.sub("", text) + text = re.sub(r"[`*_~]", "", text) + text = re.sub(r"\\(.)", r"\1", text) + for token, code_text in protected_code.items(): + text = text.replace(token, code_text) + return html.unescape(text).strip() + + +def has_substantive_rendered_text(value: str) -> bool: + return any(ch.isalnum() for ch in rendered_inline_text(value)) + + +def reference_definition_hidden_indexes(lines: list[str]) -> set[int]: + """Return lines consumed by non-rendering reference definitions/titles.""" + hidden: set[int] = set() + for i, raw in enumerate(lines): + if not LINK_REFERENCE_DEFINITION_RE.fullmatch(raw.strip()): + continue + hidden.add(i) + if ( + i + 1 < len(lines) + and LINK_REFERENCE_TITLE_CONTINUATION_RE.fullmatch(lines[i + 1]) + ): + hidden.add(i + 1) + return hidden + + +def is_structural_only_line(line: str) -> bool: + if HEADING_RE.match(line) or THEMATIC_BREAK_RE.fullmatch(line): + return True + if LIST_MARKER_ONLY_RE.fullmatch(line) or line == ">": + return True + if LINK_REFERENCE_DEFINITION_RE.fullmatch(line): + return True + cells = markdown_table_cells(line) + return bool(cells and all(TABLE_SEPARATOR_CELL_RE.fullmatch(cell) for cell in cells)) + + +def section_has_content(lines: list[str]) -> bool: + visible = visible_nonfenced_lines(lines) + hidden_reference_lines = reference_definition_hidden_indexes(visible) + for index, raw in enumerate(visible): + if index in hidden_reference_lines: + continue + line = raw.strip() + if not line or line in TEMPLATE_PLACEHOLDER_LINES: + continue + if EMPTY_LABEL_RE.match(line) or is_structural_only_line(line): + continue + if not has_substantive_rendered_text(line): + continue + return True + return False + + +def source_section_has_identity(lines: list[str]) -> bool: + """Require at least one concrete, non-placeholder provenance identity.""" + sources_root = (ROOT / "sources").resolve() + for raw in visible_nonfenced_lines(lines): + line = raw.strip() + if not line or SOURCE_PLACEHOLDER_RE.fullmatch(line): + continue + if SOURCE_URL_RE.search(line) or SOURCE_DOI_RE.search(line) or SOURCE_COMMIT_RE.search(line): + return True + for match in SOURCE_LOCAL_NOTE_RE.finditer(line): + candidate = (ROOT / match.group(1)).resolve() + try: + candidate.relative_to(sources_root) + except ValueError: + continue + if candidate.is_file(): + return True + if SOURCE_REPOSITORY_RE.search(line): + return True + return False + + +def normalized_status_category(raw: str) -> str: + category = raw.split(";", 1)[0].strip() + return unwrap_outer_formatting(category, STATUS_WRAPPERS) + + +def require_prefixed_fields( + path: Path, + lines: list[str], + fields: tuple[str, ...], + section: str, + rejected_values: dict[str, str] | None = None, +) -> None: + visible = visible_nonfenced_lines(lines) + for field in fields: + prefix = f"- {field}:" + matches = [line for line in visible if line.startswith(prefix)] + if len(matches) != 1: + die( + f"{path.relative_to(ROOT)} must contain exactly one visible field " + f"'{prefix}' in {section}" + ) + value = matches[0][len(prefix) :].strip() + if not value: + die(f"{path.relative_to(ROOT)} has empty field {field} in {section}") + if not has_substantive_rendered_text(value): + die( + f"{path.relative_to(ROOT)} has markup-only/non-substantive field " + f"{field} in {section}: '{value}'" + ) + if rejected_values is not None: + normalized_value = html.unescape( + unwrap_outer_formatting(value, STATUS_WRAPPERS) + ).strip() + if normalized_value == rejected_values.get(field): + die( + f"{path.relative_to(ROOT)} has unselected template placeholder " + f"for {field} in {section}: '{value}'" + ) + + +records: dict[str, Path] = {} +status_categories: dict[str, str] = {} +for path in sorted(OPT_DIR.glob("*.md")): + text = path.read_text(encoding="utf-8") + lines = visible_nonfenced_lines(markdown_source_lines(text)) + first = lines[0] if lines else "" + match = ID_RE.match(first) + if not match: + die(f"bad or hidden record heading: {path.relative_to(ROOT)}") + record_id = match.group(1) + + filename_match = FILENAME_ID_RE.match(path.name) + if not filename_match: + die( + f"record Markdown filename does not follow OPT---... convention: " + f"{path.relative_to(ROOT)}" + ) + if filename_match.group(1) != record_id: + die( + f"record ID mismatch: {path.relative_to(ROOT)} declares {record_id} " + f"but filename encodes {filename_match.group(1)}" + ) + if record_id in records: + die(f"duplicate record id {record_id}: {records[record_id]} and {path}") + records[record_id] = path + + statuses = [m.group(1).strip() for line in lines if (m := STATUS_RE.match(line))] + if len(statuses) != 1: + die(f"{path.relative_to(ROOT)} must contain exactly one visible Status line") + if not statuses[0]: + die(f"{path.relative_to(ROOT)} has empty Status") + + if record_id in FROZEN_V1: + continue + + status_category = statuses[0].split(";", 1)[0].strip() + if status_category not in ALLOWED_V2_STATUS_CATEGORIES: + die( + f"{path.relative_to(ROOT)} uses undefined status category " + f"'{status_category}'" + ) + status_categories[record_id] = status_category + + headings = {line for line in lines if line.startswith("## ")} + missing = sorted(REQUIRED_V2 - headings) + if missing: + die(f"{path.relative_to(ROOT)} missing visible sections: {', '.join(missing)}") + + for heading in sorted(REQUIRED_V2): + if not section_has_content(section_lines(text, heading)): + die( + f"{path.relative_to(ROOT)} has empty/template/structural/markup-only mandatory section {heading}" + ) + + source_evidence = section_lines(text, "## Source evidence") + if not source_section_has_identity(source_evidence): + die( + f"{path.relative_to(ROOT)} ## Source evidence lacks a concrete source identity " + "(URL, DOI, pinned commit, existing sources/*.md note, or repository identity)" + ) + + contract = section_lines(text, "## Optimization problem contract") + require_prefixed_fields( + path, contract, REQUIRED_CONTRACT_FIELDS, "## Optimization problem contract" + ) + require_prefixed_fields( + path, + contract, + REQUIRED_CLASSIFICATION_FIELDS, + "## Optimization problem contract", + rejected_values=CLASSIFICATION_TEMPLATE_VALUES, + ) + +missing_frozen = sorted(FROZEN_V1 - records.keys()) +if missing_frozen: + die(f"frozen v1 record(s) missing: {', '.join(missing_frozen)}") + +record_paths = {str(path.relative_to(ROOT)): record_id for record_id, path in records.items()} + +for doc_name in ("README.md", "CATALOG.md"): + text = (ROOT / doc_name).read_text(encoding="utf-8") + rendered = visible_text(text) + for label, rel in LINK_RE.findall(rendered): + target = ROOT / rel + if not target.is_file(): + die(f"broken visible record link in {doc_name}: {rel}") + target_id = record_paths.get(rel) + if target_id is None: + die(f"record link in {doc_name} is not a discovered OPT record: {rel}") + if unwrap_markdown_emphasis(label) != target_id: + die( + f"record link label mismatch in {doc_name}: '{label}' points to " + f"{target_id} ({rel})" + ) + + if doc_name != "README.md": + continue + + catalog_lines = section_lines(text, "## Catalog") + if not catalog_lines: + die("README.md is missing a non-empty visible ## Catalog section") + catalog_rows = extract_markdown_table( + catalog_lines, + ("ID", "Optimization", "Status", "Core idea"), + "README.md ## Catalog", + ) + + parsed_rows: list[tuple[str, str, str]] = [] + for cells in catalog_rows: + row_id, rel = parse_record_link_cell(cells[0], "README.md ## Catalog") + parsed_rows.append((row_id, rel, cells[2])) + + counts = Counter(row_id for row_id, _rel, _status in parsed_rows) + bad_counts = sorted(record_id for record_id, count in counts.items() if count != 1) + if bad_counts: + die( + "README.md ## Catalog table must index each record exactly once; " + f"bad row counts for: {', '.join(bad_counts)}" + ) + missing_readme = sorted(records.keys() - counts.keys()) + if missing_readme: + die(f"README.md ## Catalog table is missing record(s): {', '.join(missing_readme)}") + unknown_rows = sorted(counts.keys() - records.keys()) + if unknown_rows: + die(f"README.md ## Catalog table references unknown record(s): {', '.join(unknown_rows)}") + + row_statuses: dict[str, str] = {} + for row_id, rel, raw_status in parsed_rows: + if record_paths.get(rel) != row_id: + die(f"README.md ## Catalog row identity mismatch for {row_id}: {rel}") + if row_id in row_statuses: + die(f"README.md ## Catalog has duplicate status row for {row_id}") + row_statuses[row_id] = normalized_status_category(raw_status) + + for record_id, expected_status in status_categories.items(): + observed_status = row_statuses.get(record_id) + if observed_status is None: + die(f"README.md ## Catalog has no status cell for post-v1 record {record_id}") + if observed_status != expected_status: + die( + f"README.md status mismatch for {record_id}: " + f"record='{expected_status}' README='{observed_status}'" + ) + +catalog = (ROOT / "CATALOG.md").read_text(encoding="utf-8") +visible_catalog = visible_text(catalog) +catalog_ids = set(OPT_TOKEN_RE.findall(visible_catalog)) +unknown_catalog_ids = sorted(catalog_ids - records.keys()) +if unknown_catalog_ids: + die(f"CATALOG.md references unknown visible record ID(s): {', '.join(unknown_catalog_ids)}") +for record_id, path in records.items(): + if record_id not in catalog_ids: + die(f"{record_id} ({path.name}) is not visibly mentioned in CATALOG.md") + +decision_lines = section_lines(catalog, "## Quick decision table") +if not decision_lines: + die("CATALOG.md is missing a non-empty visible ## Quick decision table section") +decision_rows = extract_markdown_table( + decision_lines, + ("Bottleneck / problem shape", "First record to inspect", "Core idea"), + "CATALOG.md ## Quick decision table", +) + +parsed_decisions: list[tuple[str, str]] = [] +for cells in decision_rows: + row_id, rel = parse_record_link_cell( + cells[1], "CATALOG.md ## Quick decision table" + ) + parsed_decisions.append((row_id, rel)) + +decision_counts = Counter(record_id for record_id, _rel in parsed_decisions) +bad_decision_counts = sorted( + record_id for record_id, count in decision_counts.items() if count != 1 +) +if bad_decision_counts: + die( + "CATALOG.md ## Quick decision table must index each record exactly once; " + f"bad row counts for: {', '.join(bad_decision_counts)}" + ) +missing_decision = sorted(records.keys() - decision_counts.keys()) +if missing_decision: + die( + "CATALOG.md ## Quick decision table is missing record(s): " + f"{', '.join(missing_decision)}" + ) +unknown_decision = sorted(decision_counts.keys() - records.keys()) +if unknown_decision: + die( + "CATALOG.md ## Quick decision table references unknown record(s): " + f"{', '.join(unknown_decision)}" + ) +for row_id, rel in parsed_decisions: + if record_paths.get(rel) != row_id: + die(f"CATALOG.md ## Quick decision table row identity mismatch for {row_id}: {rel}") + +problem_contract = ROOT / "OPTIMIZATION-PROBLEM.md" +if not problem_contract.is_file(): + die("OPTIMIZATION-PROBLEM.md is missing") +problem_text = problem_contract.read_text(encoding="utf-8") +problem_visible = visible_nonfenced_lines(markdown_source_lines(problem_text)) +if not problem_visible or problem_visible[0] != "# Optimization Problem Contract": + die("OPTIMIZATION-PROBLEM.md has missing/hidden/invalid title") +if "## Canonical contract" not in problem_visible: + die("OPTIMIZATION-PROBLEM.md is missing visible ## Canonical contract") +canonical = section_lines(problem_text, "## Canonical contract") +canonical_text = "\n".join(canonical) +if "P = (X, F, f, d, C, B, S)" not in canonical_text: + die("OPTIMIZATION-PROBLEM.md is missing visible canonical P = (X, F, f, d, C, B, S) formula") +for field, pattern in CANONICAL_DEFINITION_PATTERNS.items(): + if not any(pattern.match(line) for line in canonical): + die(f"OPTIMIZATION-PROBLEM.md is missing visible canonical definition for {field}") + +classification_lines = section_lines(problem_text, "## Required classification") +if not classification_lines: + die("OPTIMIZATION-PROBLEM.md is missing visible ## Required classification") +classification_rows = extract_markdown_table( + classification_lines, + ("Dimension", "Typical values"), + "OPTIMIZATION-PROBLEM.md ## Required classification", +) +canonical_classification: dict[str, str] = {} +for row in classification_rows: + dimension, typical_values = row + if dimension in canonical_classification: + die(f"OPTIMIZATION-PROBLEM.md has duplicate classification dimension {dimension}") + canonical_classification[dimension] = typical_values + +expected_dimensions = set(REQUIRED_CLASSIFICATION_FIELDS) +observed_dimensions = set(canonical_classification) +if observed_dimensions != expected_dimensions: + missing_dimensions = sorted(expected_dimensions - observed_dimensions) + unknown_dimensions = sorted(observed_dimensions - expected_dimensions) + details: list[str] = [] + if missing_dimensions: + details.append(f"missing={','.join(missing_dimensions)}") + if unknown_dimensions: + details.append(f"unknown={','.join(unknown_dimensions)}") + die( + "OPTIMIZATION-PROBLEM.md classification dimensions do not match the checker: " + + "; ".join(details) + ) +for dimension in REQUIRED_CLASSIFICATION_FIELDS: + canonical_value = canonical_classification[dimension] + checker_value = CLASSIFICATION_TEMPLATE_VALUES[dimension] + if canonical_value != checker_value: + die( + f"classification placeholder drift for {dimension}: " + f"canonical='{canonical_value}' checker='{checker_value}'" + ) + +print(f"CATALOG_INTEGRITY_OK records={len(records)} frozen_v1={len(FROZEN_V1)}") From f54d70009e64305ba360e9ec54f5497aa60d3a31 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 03:22:50 +0930 Subject: [PATCH 061/229] Harden Markdown normalization for catalog integrity --- scripts/check_catalog.py | 279 ++++++++++++++++++++++++++++++++++++++- 1 file changed, 272 insertions(+), 7 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 1c6fcf7..25a5522 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -2,17 +2,21 @@ """Normalize supported CommonMark syntax, then run the hardened catalog checker. The core checker intentionally stays strict and source-oriented. This front end creates a -scratch copy, canonicalizes two rendering-equivalent forms that the core otherwise rejects +scratch copy, canonicalizes rendering-equivalent forms that the core otherwise rejects or overlooks, and runs the core against that copy: -* one-to-three spaces before ATX headings (valid CommonMark indentation), and -* optional Markdown titles on links to optimization-record Markdown files. +* one-to-three spaces before ATX headings (valid CommonMark indentation), +* optional Markdown titles on links to optimization-record Markdown files, +* inline-code examples that resemble optimization-record links, +* block-quoted link-reference definitions that render no substantive content, and +* classification placeholders hidden behind rendering-only inline formatting. The repository working tree is never modified by this normalization step. """ from __future__ import annotations +import html import re import shutil import subprocess @@ -26,6 +30,35 @@ RECORD_LINK_START_RE = re.compile( r"\[([^\]\r\n]+)\]\((optimizations/[^\s)#]+\.md)" ) +BLOCKQUOTE_PREFIX_RE = re.compile(r"^ {0,3}>[ \t]?") +LINK_REFERENCE_DEFINITION_RE = re.compile( + r"^\[(?:\\.|[^\[\]\\])+\]:[ \t]+\S.*$" +) +LINK_REFERENCE_TITLE_CONTINUATION_RE = re.compile( + r'^ {0,3}(?:"(?:\\.|[^"\\])*"|\'(?:\\.|[^\'\\])*\'|\((?:\\.|[^)\\])*\))[ \t]*$' +) +INLINE_HTML_TAG_RE = re.compile( + r"`]+))?)*" + r"[ \t]*/?>" +) +INLINE_WRAPPERS = ("**", "__", "~~", "*", "_", "`") +CLASSIFICATION_TEMPLATE_VALUES = { + "Variables": "continuous / integer / categorical / conditional / mixed", + "Search scope": "local / global", + "Objective behavior": "deterministic / noisy / stochastic", + "Information": "gradient available / derivative-free / black-box", + "Evaluation cost": "cheap / moderate / expensive", + "Constraints": "bounds / equality / inequality / semantic / resource", + "Parallelism": "sequential / synchronous batch / asynchronous", + "Exactness": "exact / approximation permitted under an explicit error contract", +} +CLASSIFICATION_FIELD_RE = re.compile( + r"^(?P- (?P" + + "|".join(re.escape(field) for field in CLASSIFICATION_TEMPLATE_VALUES) + + r"):[ \t]*)(?P.*)$" +) def _skip_whitespace(text: str, index: int) -> int: @@ -48,10 +81,150 @@ def _parse_title(text: str, index: int) -> int | None: continue if char == closer: return index + 1 + if char in "\r\n": + return None + index += 1 + return None + + +def _find_label_close(text: str, open_index: int) -> int | None: + depth = 1 + index = open_index + 1 + while index < len(text): + if text[index] == "\\" and index + 1 < len(text): + index += 2 + continue + if text[index] == "[": + depth += 1 + elif text[index] == "]": + depth -= 1 + if depth == 0: + return index + index += 1 + return None + + +def _find_inline_link_end(text: str, open_paren: int) -> int | None: + """Return the index after a complete inline-link destination/title.""" + index = _skip_whitespace(text, open_paren + 1) + if index >= len(text): + return None + if text[index] == ")": + return index + 1 + + if text[index] == "<": + index += 1 + while index < len(text): + if text[index] == "\\" and index + 1 < len(text): + index += 2 + continue + if text[index] == ">": + after_title = _skip_whitespace(text, index + 1) + if after_title < len(text) and text[after_title] == ")": + return after_title + 1 + title_end = _parse_title(text, after_title) + if title_end is None: + return None + outer_close = _skip_whitespace(text, title_end) + if outer_close < len(text) and text[outer_close] == ")": + return outer_close + 1 + return None + if text[index] in "\r\n<": + return None + index += 1 + return None + + depth = 0 + while index < len(text): + char = text[index] + if char == "\\" and index + 1 < len(text): + index += 2 + continue + if char == "(": + depth += 1 + index += 1 + continue + if char == ")": + if depth == 0: + return index + 1 + depth -= 1 + index += 1 + continue + if char in " \t\r\n" and depth == 0: + after_title = _skip_whitespace(text, index) + if after_title < len(text) and text[after_title] == ")": + return after_title + 1 + title_end = _parse_title(text, after_title) + if title_end is None: + return None + outer_close = _skip_whitespace(text, title_end) + if outer_close < len(text) and text[outer_close] == ")": + return outer_close + 1 + return None + if char in "<>" or ord(char) < 0x20: + return None index += 1 return None +def _is_backslash_escaped(text: str, index: int) -> bool: + count = 0 + cursor = index - 1 + while cursor >= 0 and text[cursor] == "\\": + count += 1 + cursor -= 1 + return count % 2 == 1 + + +def _backtick_run_length(text: str, index: int) -> int: + cursor = index + while cursor < len(text) and text[cursor] == "`": + cursor += 1 + return cursor - index + + +def mask_inline_code_record_destinations(text: str) -> str: + """Prevent link-shaped inline-code examples from being treated as actual links.""" + out: list[str] = [] + index = 0 + while index < len(text): + if text[index] != "`" or _is_backslash_escaped(text, index): + out.append(text[index]) + index += 1 + continue + + run_len = _backtick_run_length(text, index) + cursor = index + run_len + close_start: int | None = None + while cursor < len(text): + if text[cursor] != "`": + cursor += 1 + continue + candidate_len = _backtick_run_length(text, cursor) + if candidate_len == run_len: + close_start = cursor + break + cursor += candidate_len + + if close_start is None: + out.append(text[index : index + run_len]) + index += run_len + continue + + code_text = text[index + run_len : close_start] + code_text = code_text.replace( + "(optimizations/", "(__inline_code__/optimizations/" + ) + out.append( + text[index : index + run_len] + + code_text + + text[close_start : close_start + run_len] + ) + index = close_start + run_len + + return "".join(out) + + def canonicalize_record_link_titles(text: str) -> str: """Drop only syntactically complete optional titles from OPT-record links.""" out: list[str] = [] @@ -82,7 +255,7 @@ def canonicalize_record_link_titles(text: str) -> str: search_from = after_destination continue - out.append(text[cursor:match.start()]) + out.append(text[cursor : match.start()]) out.append(f"[{match.group(1)}]({match.group(2)})") cursor = outer_close + 1 search_from = cursor @@ -90,9 +263,98 @@ def canonicalize_record_link_titles(text: str) -> str: return "".join(out) -def canonicalize_markdown(text: str) -> str: +def _strip_blockquote_prefix(line: str) -> tuple[str, bool]: + result = line + changed = False + while True: + match = BLOCKQUOTE_PREFIX_RE.match(result) + if match is None: + return result, changed + result = result[match.end() :] + changed = True + + +def canonicalize_nested_reference_definitions(text: str) -> str: + """Expose non-rendering quoted reference definitions to the strict core checker.""" + out: list[str] = [] + for raw in text.splitlines(keepends=True): + content = raw.rstrip("\r\n") + ending = raw[len(content) :] + unquoted, changed = _strip_blockquote_prefix(content) + stripped = unquoted.strip() + if changed and ( + LINK_REFERENCE_DEFINITION_RE.fullmatch(stripped) + or LINK_REFERENCE_TITLE_CONTINUATION_RE.fullmatch(unquoted) + ): + out.append(unquoted + ending) + else: + out.append(raw) + return "".join(out) + + +def _render_placeholder_candidate(value: str) -> str: + """Render the subset of inline Markdown relevant to template placeholders.""" + result = html.unescape(value.strip()) + changed = True + while changed: + changed = False + for marker in INLINE_WRAPPERS: + if ( + len(result) > 2 * len(marker) + and result.startswith(marker) + and result.endswith(marker) + ): + result = result[len(marker) : -len(marker)].strip() + changed = True + break + if changed: + continue + + if result.startswith("["): + label_close = _find_label_close(result, 0) + if ( + label_close is not None + and label_close + 1 < len(result) + and result[label_close + 1] == "(" + ): + link_end = _find_inline_link_end(result, label_close + 1) + if link_end == len(result): + result = result[1:label_close].strip() + changed = True + + result = INLINE_HTML_TAG_RE.sub("", result) + result = re.sub(r"\\(.)", r"\1", result) + return html.unescape(result).strip() + + +def canonicalize_classification_placeholders(text: str) -> str: + """Normalize rendered-but-unselected template values back to canonical source.""" + out: list[str] = [] + for raw in text.splitlines(keepends=True): + content = raw.rstrip("\r\n") + ending = raw[len(content) :] + match = CLASSIFICATION_FIELD_RE.fullmatch(content) + if match is None: + out.append(raw) + continue + + field = match.group("field") + value = match.group("value") + if _render_placeholder_candidate(value) == CLASSIFICATION_TEMPLATE_VALUES[field]: + out.append(match.group("prefix") + CLASSIFICATION_TEMPLATE_VALUES[field] + ending) + else: + out.append(raw) + return "".join(out) + + +def canonicalize_markdown(text: str, *, link_scan_document: bool) -> str: text = ATX_INDENT_RE.sub("", text) - return canonicalize_record_link_titles(text) + text = canonicalize_nested_reference_definitions(text) + text = canonicalize_classification_placeholders(text) + if link_scan_document: + text = mask_inline_code_record_destinations(text) + text = canonicalize_record_link_titles(text) + return text def markdown_inputs(root: Path) -> list[Path]: @@ -117,7 +379,10 @@ def main() -> int: for path in markdown_inputs(scratch): original = path.read_text(encoding="utf-8") - normalized = canonicalize_markdown(original) + normalized = canonicalize_markdown( + original, + link_scan_document=path.name in {"README.md", "CATALOG.md"}, + ) if normalized != original: path.write_text(normalized, encoding="utf-8") From 6f89cb061e71bf7e221beeffd92a3222d2f88d7b Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 03:52:11 +0930 Subject: [PATCH 062/229] Harden Markdown context normalization --- scripts/check_catalog.py | 202 +++++++++++++++++++++++++++++++++++++-- 1 file changed, 194 insertions(+), 8 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 25a5522..51c650f 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -8,8 +8,11 @@ * one-to-three spaces before ATX headings (valid CommonMark indentation), * optional Markdown titles on links to optimization-record Markdown files, * inline-code examples that resemble optimization-record links, -* block-quoted link-reference definitions that render no substantive content, and -* classification placeholders hidden behind rendering-only inline formatting. +* block-quoted link-reference definitions that actually parse as definitions, +* classification placeholders hidden behind rendering-only inline formatting, +* generic TODO/TBD-style required-field placeholders, +* hash-shaped source text that lacks explicit commit/revision context, and +* type-7 raw-HTML tags that CommonMark keeps inside an already-open paragraph. The repository working tree is never modified by this normalization step. """ @@ -27,6 +30,12 @@ ROOT = Path(__file__).resolve().parents[1] CORE_NAME = "check_catalog_core.py" ATX_INDENT_RE = re.compile(r"(?m)^ {1,3}(?=#{1,6}(?:[ \t]|$))") +ATX_HEADING_RE = re.compile(r"^ {0,3}#{1,6}(?:[ \t]|$)") +FENCE_LINE_RE = re.compile(r"^ {0,3}(?:`{3,}|~{3,})") +LIST_BLOCK_RE = re.compile(r"^ {0,3}(?:[-+*]|\d+[.)])[ \t]+") +THEMATIC_BREAK_RE = re.compile( + r"^ {0,3}(?:\*(?:[ \t]*\*){2,}|-(?:[ \t]*-){2,}|_(?:[ \t]*_){2,})[ \t]*$" +) RECORD_LINK_START_RE = re.compile( r"\[([^\]\r\n]+)\]\((optimizations/[^\s)#]+\.md)" ) @@ -43,6 +52,15 @@ r"(?:[ \t]*=[ \t]*(?:\"[^\"]*\"|'[^']*'|[^ \t\n\"'=<>`]+))?)*" r"[ \t]*/?>" ) +STANDALONE_HTML_TAG_RE = re.compile( + r"^ {0,3}(?:" + r"[A-Za-z][A-Za-z0-9-]*)[ \t]*>" + r"|<(?P[A-Za-z][A-Za-z0-9-]*)" + r"(?:[ \t]+[A-Za-z_:][A-Za-z0-9_.:-]*" + r"(?:[ \t]*=[ \t]*(?:\"[^\"]*\"|'[^']*'|[^ \t\n\"'=<>`]+))?)*" + r"[ \t]*/?>" + r")[ \t]*$" +) INLINE_WRAPPERS = ("**", "__", "~~", "*", "_", "`") CLASSIFICATION_TEMPLATE_VALUES = { "Variables": "continuous / integer / categorical / conditional / mixed", @@ -59,6 +77,44 @@ + "|".join(re.escape(field) for field in CLASSIFICATION_TEMPLATE_VALUES) + r"):[ \t]*)(?P.*)$" ) +REQUIRED_FIELD_NAMES = ( + "X", + "F", + "f", + "d", + "C", + "B", + "S", + *CLASSIFICATION_TEMPLATE_VALUES.keys(), +) +REQUIRED_FIELD_RE = re.compile( + r"^(?P- (?P" + + "|".join(re.escape(field) for field in REQUIRED_FIELD_NAMES) + + r"):[ \t]*)(?P.*)$" +) +GENERIC_PLACEHOLDER_RE = re.compile( + r"^(?:unknown|tbd|todo|n/?a|none|pending)(?:[.!?])?$", re.IGNORECASE +) +HEX_TOKEN_RE = re.compile(r"\b[0-9a-fA-F]{7,40}\b") +EXPLICIT_COMMIT_CONTEXT_RE = re.compile( + r"(?:" + r"\b(?:commit(?:[ \t]+sha)?|sha|revision|rev)\b[^0-9A-Za-z]{0,12}" + r"|\b(?:pinned|inspected)[ \t]+at\b[^0-9A-Za-z]{0,12}" + r"|@" + r")$", + re.IGNORECASE, +) +HTML_BLOCK_TAGS = { + "address", "article", "aside", "base", "basefont", "blockquote", "body", + "caption", "center", "col", "colgroup", "dd", "details", "dialog", "dir", + "div", "dl", "dt", "fieldset", "figcaption", "figure", "footer", "form", + "frame", "frameset", "h1", "h2", "h3", "h4", "h5", "h6", "head", + "header", "hr", "html", "iframe", "legend", "li", "link", "main", "menu", + "menuitem", "nav", "noframes", "ol", "optgroup", "option", "p", "param", + "search", "section", "summary", "table", "tbody", "td", "tfoot", "th", + "thead", "title", "tr", "track", "ul", +} +TYPE1_HTML_TAGS = {"script", "pre", "style", "textarea"} def _skip_whitespace(text: str, index: int) -> int: @@ -274,21 +330,107 @@ def _strip_blockquote_prefix(line: str) -> tuple[str, bool]: changed = True +def _standalone_html_tag_name(line: str) -> str | None: + match = STANDALONE_HTML_TAG_RE.fullmatch(line) + if match is None: + return None + return (match.group("open") or match.group("close")).lower() + + +def _is_type7_complete_tag_line(line: str) -> bool: + tag = _standalone_html_tag_name(line) + return tag is not None and tag not in HTML_BLOCK_TAGS and tag not in TYPE1_HTML_TAGS + + +def _line_can_open_or_continue_paragraph(line: str) -> bool: + if not line.strip(): + return False + if line.startswith("\t") or line.startswith(" "): + return False + if ATX_HEADING_RE.match(line) or FENCE_LINE_RE.match(line): + return False + if THEMATIC_BREAK_RE.fullmatch(line) or LIST_BLOCK_RE.match(line): + return False + if BLOCKQUOTE_PREFIX_RE.match(line): + return False + if _standalone_html_tag_name(line) is not None or line.lstrip().startswith("<"): + return False + return True + + def canonicalize_nested_reference_definitions(text: str) -> str: - """Expose non-rendering quoted reference definitions to the strict core checker.""" + """Expose only quoted reference definitions that CommonMark parses as definitions.""" out: list[str] = [] + paragraph_open = False + reference_title_expected = False + for raw in text.splitlines(keepends=True): content = raw.rstrip("\r\n") ending = raw[len(content) :] unquoted, changed = _strip_blockquote_prefix(content) + + if not changed: + paragraph_open = False + reference_title_expected = False + out.append(raw) + continue + stripped = unquoted.strip() - if changed and ( - LINK_REFERENCE_DEFINITION_RE.fullmatch(stripped) - or LINK_REFERENCE_TITLE_CONTINUATION_RE.fullmatch(unquoted) - ): + if not stripped: + paragraph_open = False + reference_title_expected = False + out.append(raw) + continue + + if reference_title_expected and LINK_REFERENCE_TITLE_CONTINUATION_RE.fullmatch(unquoted): out.append(unquoted + ending) - else: + reference_title_expected = False + paragraph_open = False + continue + + if LINK_REFERENCE_DEFINITION_RE.fullmatch(stripped): + if paragraph_open: + out.append(raw) + reference_title_expected = False + paragraph_open = True + else: + out.append(unquoted + ending) + reference_title_expected = True + paragraph_open = False + continue + + out.append(raw) + reference_title_expected = False + paragraph_open = _line_can_open_or_continue_paragraph(unquoted) + + return "".join(out) + + +def canonicalize_type7_html_paragraph_interruptions(text: str) -> str: + """Keep type-7 complete tags inline when a CommonMark paragraph is already open.""" + out: list[str] = [] + paragraph_open = False + + for raw in text.splitlines(keepends=True): + content = raw.rstrip("\r\n") + ending = raw[len(content) :] + + if not content.strip(): + paragraph_open = False out.append(raw) + continue + + if paragraph_open and _is_type7_complete_tag_line(content): + # The core is deliberately source-strict and would otherwise start a type-7 + # raw-HTML block here. Prefix only the scratch copy so it remains paragraph + # text, matching CommonMark's rule that type-7 blocks cannot interrupt one. + out.append("INLINE_HTML_CONTINUATION " + content.lstrip() + ending) + paragraph_open = True + continue + + out.append(raw) + paragraph_open = _line_can_open_or_continue_paragraph(content) + return "".join(out) @@ -347,10 +489,54 @@ def canonicalize_classification_placeholders(text: str) -> str: return "".join(out) +def canonicalize_generic_required_placeholders(text: str) -> str: + """Turn rendered generic placeholders into empty values so the strict core rejects them.""" + out: list[str] = [] + for raw in text.splitlines(keepends=True): + content = raw.rstrip("\r\n") + ending = raw[len(content) :] + match = REQUIRED_FIELD_RE.fullmatch(content) + if match is None: + out.append(raw) + continue + rendered = _render_placeholder_candidate(match.group("value")) + if GENERIC_PLACEHOLDER_RE.fullmatch(rendered): + out.append(match.group("prefix") + ending) + else: + out.append(raw) + return "".join(out) + + +def _has_explicit_commit_context(prefix: str) -> bool: + cleaned = prefix.rstrip(" \t`*_~([{<") + return EXPLICIT_COMMIT_CONTEXT_RE.search(cleaned) is not None + + +def canonicalize_ambiguous_commit_tokens(text: str) -> str: + """Break hash-shaped prose unless the token has explicit commit/revision context.""" + out: list[str] = [] + cursor = 0 + for match in HEX_TOKEN_RE.finditer(text): + out.append(text[cursor : match.start()]) + token = match.group(0) + prefix = text[max(0, match.start() - 80) : match.start()] + if _has_explicit_commit_context(prefix): + out.append(token) + else: + split_at = min(4, len(token) - 1) + out.append(token[:split_at] + "-" + token[split_at:]) + cursor = match.end() + out.append(text[cursor:]) + return "".join(out) + + def canonicalize_markdown(text: str, *, link_scan_document: bool) -> str: text = ATX_INDENT_RE.sub("", text) text = canonicalize_nested_reference_definitions(text) + text = canonicalize_type7_html_paragraph_interruptions(text) text = canonicalize_classification_placeholders(text) + text = canonicalize_generic_required_placeholders(text) + text = canonicalize_ambiguous_commit_tokens(text) if link_scan_document: text = mask_inline_code_record_destinations(text) text = canonicalize_record_link_titles(text) From 30421d76784658774ee15dab552da0e0fe3d14c6 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:02:29 +0930 Subject: [PATCH 063/229] Add GALAXY CPU optimization source note --- sources/GALAXY-CPU.md | 73 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 73 insertions(+) create mode 100644 sources/GALAXY-CPU.md diff --git a/sources/GALAXY-CPU.md b/sources/GALAXY-CPU.md new file mode 100644 index 0000000..93648a3 --- /dev/null +++ b/sources/GALAXY-CPU.md @@ -0,0 +1,73 @@ +# GALAXY CPU optimization phases + +## Source identity + +- Repository: `QSOLKCB/GALAXY` +- License: Apache-2.0 (`LICENSE` on the source repository) +- Optimization sequence: merged PRs #10 through #14 on 2026-09-15 +- Scope: CPU SIMD/autovectorization evidence, worker-local SoA tiling, guarded production integration, persistent topology-aware worker pools, and calibrated host-aware execution-path promotion. + +## Merged donor evidence + +### PR #10 — deterministic CPU SIMD probe + +- PR: https://github.com/QSOLKCB/GALAXY/pull/10 +- Merge commit: `02f26f0f3630487b7db9522e9db3a5e5536f5765` +- Key files: `cpu-runtime/src/bin/simd_probe.rs`, `scripts/bench-cpu-simd.sh`, `docs/CPU-SIMD.md` +- Mechanism: reshape an exact hot contribution/hash loop into a batch/SoA form that LLVM can autovectorize, build the same source for generic and host-native CPU targets, inspect generated assembly, and require element/checksum parity before considering integration. +- Environment-scoped donor observation reported by the next phase: Ryzen 9 5950X generic median 34,942,130 ns versus native median 10,742,437 ns, a 3.252719099x isolated speedup and 69.256491% median reduction. The donor identified SSE2 packed operations in the generic build and AVX2/VEX packed operations in the native build; AVX-512 was absent on that Zen 3 host as expected. +- Boundary: PR #10 explicitly did not claim an end-to-end GALAXY speedup. + +### PR #11 — worker-local SoA runtime probe + +- PR: https://github.com/QSOLKCB/GALAXY/pull/11 +- Merge commit: `8c69f87278d071db47d85a787905778eeec64ded` +- Key files: `cpu-runtime/src/bin/worker_soa_probe.rs`, `scripts/bench-cpu-worker-soa.sh`, `docs/CPU-WORKER-SOA.md` +- Mechanism: replace a full-resident AoS traversal in the experimental path with deterministic contiguous worker ranges, bounded compact SoA particle tiles, reusable worker-local scratch, and SIMD-friendly batch contribution hashing. Working memory scales with workers × tile instead of total resident population. +- Validation: reference/generic/native/SoA checksum parity across every benchmark matrix cell, deterministic worker-count behavior, tile-range sweeps, disassembly evidence where available, RSS evidence, and cross-platform native CI. +- Boundary: the donor requires a practical winning tile range, useful worker scaling, and acceptable RSS before promotion; one pathological fast point is insufficient. + +### PR #12 — guarded production SoA integration + +- PR: https://github.com/QSOLKCB/GALAXY/pull/12 +- Merge commit: `0e389cb4179902fae93f4a77d6ab2a539e73bc72` +- Key files: `cpu-runtime/src/bin/galaxy_cpu_dispatch.rs`, `docs/CPU-SOA-INTEGRATION.md` +- Mechanism: integrate the proven worker-local SoA path behind explicit opt-in production commands while preserving the canonical AoS path as unchanged default, correctness oracle, and fallback. +- Validation: cross-platform integrated `verify-soa`, production-identity receipts, guarded execution metadata, and exact checksum equality with the canonical BAM-LUT path. +- Boundary: guarded integration is evidence for the SoA record's safe deployment shape, not a claim that opt-in command surfaces are themselves a performance optimization. + +### PR #13 — persistent topology-aware SoA execution + +- PR: https://github.com/QSOLKCB/GALAXY/pull/13 +- Merge commit: `1966bc2595a402a2c653f2e392465224621e20fb` +- Key mechanism: create the SoA worker set once, allocate each worker's compact tile once, reuse threads/buffers across warm-up and measured repetitions, collect completion asynchronously, and reduce strictly in worker-index order. +- Topology policy: explicit `physical-first` versus `logical`; Linux uses process CPU allowance plus sysfs topology, macOS uses `sysctl`, and unsupported/unreliable physical topology falls back explicitly to logical availability. +- Timing boundary: steady-state timings exclude pool startup but record `pool_startup_ns` separately. The donor explicitly makes no CPU-affinity or NUMA-placement claim. +- Validation: canonical checksum = spawned SoA checksum = first persistent dispatch = second persistent dispatch, plus cross-platform CI and explicit topology/fallback receipts. + +### PR #14 — calibrated host-aware CPU promotion + +- PR: https://github.com/QSOLKCB/GALAXY/pull/14 +- Merge commit: `b2e860309a04d7591c86f71d2b4ab1e5eec4c4d7` +- Policy identifier in donor: `calibrated-host-auto-v1` +- Candidate families: canonical execution, spawned worker-local SoA across a finite tile set, persistent physical-first SoA, and persistent logical/SMT SoA when distinct. +- Calibration shape: preserves requested frame depth, preserves effective per-worker tile shape by expanding only within the requested resident workload, and uses an explicit projection from calibration to requested particle-frame work. +- Lifecycle accounting: persistent candidates include full requested pool startup plus teardown amortized over requested repetitions; startup includes worker creation, tile allocation, and first-touch/commitment of worker-local buffers. +- Promotion: donor uses a 5% projected advantage over canonical; near-ties remain canonical. +- Correctness: every calibration candidate and the selected full workload must match an independent streaming canonical oracle exactly. Any mismatch fails closed rather than silently falling back. +- Evidence scope: process-wide Linux `VmHWM` is labelled whole-invocation evidence and is not misrepresented as selected-engine RSS. + +## Reusable mechanisms promoted into OPT + +1. **Evidence-gated native autovectorization** — expose a compiler-friendly batch shape, compare generic and native builds from identical source, inspect actual code generation, and require exact parity plus repeatable gain before integration. +2. **Worker-local SoA tiling** — transform a large AoS traversal into bounded worker-local SoA tiles with deterministic partition/reduction and an unchanged oracle path. +3. **Persistent topology-aware worker pools** — amortize thread/buffer setup across repeated executions, choose physical/logical worker policies explicitly, and keep completion-order independence through deterministic reduction. +4. **Calibrated host/workload-aware promotion** — benchmark a bounded candidate set on workload-shaped calibration, include lifecycle/tuning costs, require a material promotion margin, and verify the selected full execution against an independent fail-closed oracle. + +## Non-transferable constants + +Do not copy the donor's worker counts, tile sizes (including 1,024 / 4,096 / 16,384 / 65,536), 65,536-particle calibration base, three calibration repeats, 5% promotion margin, or Ryzen timing observations as universal settings. They are source-environment evidence only. Re-profile candidate sets, calibration budgets, promotion margins, topology policy, and lifecycle amortization in the target repository. + +## Licensing / reuse boundary + +Both GALAXY and OPT are Apache-2.0 repositories at the time of this source note. The OPT records describe reusable mechanisms and provenance; they do not require copying GALAXY implementation code. If code is copied later, retain the applicable Apache-2.0 notices and re-check the donor license at the pinned source revision. \ No newline at end of file From f00c7330ffcc09469c9d3a3c336c809f87d7b44b Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:03:00 +0930 Subject: [PATCH 064/229] Add evidence-gated SIMD optimization record --- ...evidence-gated-native-autovectorization.md | 98 +++++++++++++++++++ 1 file changed, 98 insertions(+) create mode 100644 optimizations/OPT-SIMD-001-evidence-gated-native-autovectorization.md diff --git a/optimizations/OPT-SIMD-001-evidence-gated-native-autovectorization.md b/optimizations/OPT-SIMD-001-evidence-gated-native-autovectorization.md new file mode 100644 index 0000000..0f92ac2 --- /dev/null +++ b/optimizations/OPT-SIMD-001-evidence-gated-native-autovectorization.md @@ -0,0 +1,98 @@ +# OPT-SIMD-001 — Evidence-gated native autovectorization + +**Status:** Verified, environment-specific; donor measurements are isolated-kernel evidence, not a transferable end-to-end speedup claim. +**Domains:** CPU numerical kernels, hashing, simulation, DSP, batch transforms, compiler specialization + +## Source evidence + +- Repository: `QSOLKCB/GALAXY` +- PR: https://github.com/QSOLKCB/GALAXY/pull/10 +- Merge commit: `02f26f0f3630487b7db9522e9db3a5e5536f5765` +- Source note: `sources/GALAXY-CPU.md` +- Key donor files: `cpu-runtime/src/bin/simd_probe.rs`, `scripts/bench-cpu-simd.sh`, `docs/CPU-SIMD.md` +- Licensing boundary: donor and this repository are Apache-2.0 at the pinned revisions; this record promotes the mechanism, not copied implementation code. + +## Problem + +A deterministic hot loop performs the same scalar or narrowly vectorized operation over a large batch, but the source layout or loop shape prevents the compiler from generating the widest useful instructions for the host. Hand-written ISA intrinsics may be premature, non-portable, or unsupported by the project's compiler baseline, while a generic build may leave substantial throughput unused. + +## Optimization problem contract + +- X: Semantically equivalent loop shapes, batch layouts, compiler target settings and optional native-specialized build variants for the identified hot kernel. +- F: Candidates that preserve exact element results and aggregate checksums, compile on the supported toolchain, never expose unsupported instructions through the portable/default path, and retain an auditable reference implementation. +- f: Measured kernel runtime together with code-generation evidence and portability/deployment cost for each validated candidate. +- d: Minimize repeatable runtime after correctness gates; prefer the simpler portable candidate when performance is a near tie or code-generation evidence is ambiguous. +- C: Exact output/checksum parity, deterministic repeat stability, truthful ISA evidence, and no promotion of an isolated-kernel result into an end-to-end claim without separate full-path measurement. +- B: A bounded set of generic/native builds, benchmark repetitions, representative batch sizes and assembly inspections on the target hardware envelope. +- S: Stop after all declared candidates have passed parity and repeated measurement; retain the generic/reference path unless a native/vectorized candidate shows a useful repeatable advantage without violating deployment constraints. +- Variables: categorical and conditional +- Search scope: local +- Objective behavior: noisy +- Information: black-box +- Evaluation cost: moderate +- Constraints: semantic and resource +- Parallelism: sequential +- Exactness: exact + +## Preserved contract + +The optimized build must produce exactly the same declared element outputs and aggregate checksum as the reference path. Build-target specialization may change generated instructions, but it must not silently weaken arithmetic, hashing, ordering, determinism, or supported-machine behavior. A native binary must not become the universal default unless deployment guarantees the required ISA. + +Microbenchmark or probe wins remain probe evidence. They do not establish whole-application speedup until the transformed loop is measured inside the real memory, scheduling, setup and reduction path. + +## Optimization + +1. Isolate the suspected hot kernel behind a deterministic reference function. +2. Reshape data into a compiler-friendly batch form, commonly structure-of-arrays or separate primitive arrays, so independent iterations are visible to the optimizer. +3. Build the exact same source with a portable target and with target-specific/native code generation. +4. Compare every element and aggregate checksum against the scalar/reference path before timing. +5. Repeat timings under a controlled workload and inspect emitted assembly or equivalent compiler evidence to verify that the expected vector form actually exists. +6. Treat ISA width as evidence, not the objective: wider instructions are valuable only when the measured target workload improves. +7. Integrate only after the isolated gain survives the relevant production path, retaining a portable/reference fallback. + +The reusable idea is not “turn on AVX2/AVX-512.” It is to make the loop vectorizable, verify what the compiler emitted, and require measured semantic-preserving benefit on the machine that will run it. + +## Before / after evidence + +- Environment: GALAXY donor run on AMD Ryzen 9 5950X / Zen 3, where AVX2/FMA are available and AVX-512 is not. +- Workload/fixture: deterministic contribution/hash batch from the GALAXY CPU runtime probe. +- Cold baseline: generic x86-64 build median reported as 34,942,130 ns in the subsequent merged phase's summary. +- Warm/no-op baseline where relevant: not applicable; the donor compared identical probe work under generic and host-native compilation. +- Small invalidation / partial-work case where relevant: not applicable. +- Large invalidation / full-work case where relevant: donor documentation includes an 8M-item confirmation procedure. +- Optimized: host-native build median reported as 10,742,437 ns. +- Speedup / memory / I/O / quality change: 3.252719099x isolated speedup and 69.256491% median reduction; exact checksum parity retained. Generic code used SSE2 packed operations while the native Zen 3 build showed AVX2/VEX packed operations. +- Variance / repetitions / raw samples: the donor retains machine-readable receipts and repeated runs; this OPT record does not promote the source timings as a target expectation. + +## Validation + +Require all of the following before promotion: + +- element-by-element equality with the canonical/reference function; +- identical aggregate checksum across generic/native/vectorized candidates; +- repeat checksum stability; +- explicit host feature reporting or deployment capability guarantees; +- assembly/compiler evidence that the tested optimized build actually uses the intended vector form; +- repeated timing under the same workload and timing boundary; +- full-path validation after integration, including memory traffic, scheduling, reduction and RSS where those can erase the isolated gain. + +## Target-repo adaptation + +Re-profile the compiler version, minimum supported CPU, target-feature flags, batch size, data alignment, aliasing assumptions, integer/float semantics, hot-loop shape and deployment model. Do not copy `target-cpu=native` into distributed binaries unless the execution fleet guarantees compatibility. For floating-point kernels, separately decide whether reassociation, contraction/FMA or altered rounding is allowed; exact integer/hash evidence does not authorize floating-point semantic changes. + +## Failure modes + +- The loop is memory-bound, so wider arithmetic does not improve elapsed time. +- The compiler cannot vectorize because of aliasing, branches, gathers or unsupported operations. +- A native build improves the probe but regresses the integrated path due to cache pressure, downclocking or changed instruction mix. +- The optimized binary reaches hardware lacking the required ISA. +- Floating-point vectorization changes observable numerical behavior that the target contract requires to remain stable. +- Benchmark noise or frequency scaling makes a small apparent win non-repeatable. + +## Rollback trigger + +Disable or decline the specialized/vectorized path immediately on any parity/checksum failure, unsupported-instruction risk, reproducible full-path regression, or loss of deterministic behavior. Revert to the portable/reference implementation when the measured target workload does not retain a useful advantage after integration. + +## Composition notes + +Composes naturally with `OPT-SOA-001` when a data-layout change exposes independent batch lanes, and with `OPT-BUDGET-001` for environment-scoped regression protection after a configuration is promoted. Combine with `OPT-PAR-001` carefully: thread-level parallelism plus SIMD can shift the bottleneck to memory bandwidth or oversubscribe shared resources, so re-measure the complete path. \ No newline at end of file From 3455d052117089e5bdddfa9ebf58c5852ade3486 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:03:30 +0930 Subject: [PATCH 065/229] Add worker-local SoA tiling optimization record --- .../OPT-SOA-001-worker-local-soa-tiling.md | 99 +++++++++++++++++++ 1 file changed, 99 insertions(+) create mode 100644 optimizations/OPT-SOA-001-worker-local-soa-tiling.md diff --git a/optimizations/OPT-SOA-001-worker-local-soa-tiling.md b/optimizations/OPT-SOA-001-worker-local-soa-tiling.md new file mode 100644 index 0000000..37d17d1 --- /dev/null +++ b/optimizations/OPT-SOA-001-worker-local-soa-tiling.md @@ -0,0 +1,99 @@ +# OPT-SOA-001 — Worker-local SoA tiling + +**Status:** Implemented external reference; GALAXY demonstrates exact cross-platform parity and guarded production integration, while target tile/worker settings remain environment-specific. +**Domains:** CPU simulation, rendering, numerical kernels, batch transforms, memory-bound pipelines + +## Source evidence + +- Repository: `QSOLKCB/GALAXY` +- Prototype PR: https://github.com/QSOLKCB/GALAXY/pull/11 +- Guarded production integration PR: https://github.com/QSOLKCB/GALAXY/pull/12 +- Merge commits: `8c69f87278d071db47d85a787905778eeec64ded` and `0e389cb4179902fae93f4a77d6ab2a539e73bc72` +- Source note: `sources/GALAXY-CPU.md` +- Licensing boundary: Apache-2.0 donor; this record describes the reusable data-layout/execution pattern rather than importing source code. + +## Problem + +A large array-of-structures working set is repeatedly traversed by multiple workers and frames/stages. The representation carries fields not needed by the hot path, causes poor cache/vector access, inflates resident memory, and forces worker execution to touch more data than necessary. A whole-population SoA conversion may itself be too large or expensive. + +## Optimization problem contract + +- X: Worker-local layout, packing, tile-capacity and deterministic partition choices that transform only the hot fields required for a bounded chunk of the source population. +- F: Candidates that preserve exact source-to-output semantics, deterministic partitioning and reduction, represent every required field without lossy reinterpretation, keep worker-local memory bounded, and retain an unchanged canonical/reference path. +- f: End-to-end runtime, peak working-memory/RSS evidence and useful worker scaling over the declared workload matrix. +- d: Pareto-minimize runtime and memory footprint subject to exact parity; reject candidates whose timing gain requires unacceptable RSS growth or unstable worker scaling. +- C: Exact checksum/output equality with the reference path, deterministic worker-count behavior where required, no dropped/duplicated elements, and bounded worker-local storage independent of total resident population. +- B: A bounded worker-count × tile-size benchmark matrix with repeated runs and representative workload sizes on the target machines. +- S: Stop after a practical winning tile/worker region is identified or all candidates fail; do not promote a single pathological fast point without surrounding evidence. +- Variables: integer and categorical +- Search scope: global +- Objective behavior: noisy +- Information: black-box +- Evaluation cost: expensive +- Constraints: semantic and resource +- Parallelism: synchronous batch +- Exactness: exact + +## Preserved contract + +The tiled SoA path must compute the same declared outputs/checksum as the reference AoS path for every processed element and supported worker count. Reordering storage is allowed only when observable output order, tie behavior, reduction semantics and deterministic identity remain unchanged. + +Packing fields into narrower representations is permitted only when the representation is proven exact for the target domain or when the target contract explicitly allows approximation. This record is exact by default. + +## Optimization + +1. Keep the authoritative source representation or generation contract unchanged. +2. Partition the source population into deterministic contiguous or otherwise contract-safe worker ranges. +3. Give each worker a bounded tile containing only hot primitive fields in structure-of-arrays form. +4. Fill one tile, execute all profitable work for that tile while it is cache-resident, and reuse worker-local scratch rather than allocating a full transformed population. +5. Batch the resulting primitive lanes so compiler/native vectorization can operate on independent values. +6. Reduce worker results in a deterministic order when arithmetic/order semantics require it. +7. Sweep tile sizes and worker counts because the useful tile is a cache/memory/scheduling property of the target, not a universal constant. +8. Integrate behind an explicit guarded path until production evidence justifies any default change; preserve the canonical path as oracle/fallback. + +The key scaling property is that optimized working storage grows roughly with workers × tile capacity, not with the complete resident population. + +## Before / after evidence + +- Environment: GALAXY PRs #11–#12 exercised Linux x86-64, Linux ARM64, macOS ARM64 and Windows x86-64 parity/CI surfaces; performance evidence remained host-scoped. +- Workload/fixture: resident particle generation, BAM-LUT projection, contribution hashing and deterministic worker reduction across multiple frames. +- Cold baseline: full-resident 40-byte AoS reference shape in the donor experiment. +- Warm/no-op baseline where relevant: not promoted as a portable metric. +- Small invalidation / partial-work case where relevant: small tile/worker matrix cells validate capacity and partition behavior. +- Large invalidation / full-work case where relevant: donor sweep supports resident populations up to the full experimental workload and multiple tile sizes/workers. +- Optimized: bounded worker-local compact SoA tiles with worker-local x/y/output scratch and SIMD-friendly batch hashing. +- Speedup / memory / I/O / quality change: donor PR #12 states that PR #11 established exact cross-platform parity and strong multi-host performance/memory evidence; this record intentionally does not turn those donor observations into universal target numbers. +- Variance / repetitions / raw samples: donor sweep emits raw matrix data, comparison tables and receipts; targets must repeat the sweep locally. + +## Validation + +- Check exact reference/generic/native/SoA checksum equality for every matrix cell. +- Verify deterministic partitioning covers every source element exactly once. +- Verify worker-count invariance when the target contract requires it. +- Test tile-capacity boundaries, partial final tiles and minimum/maximum supported sizes. +- Validate packed-field encode/decode or direct arithmetic equivalence independently. +- Measure end-to-end timing with the same setup/generation boundary for baseline and optimized paths. +- Record peak-memory/RSS scope honestly; process-wide high-water marks are not per-engine measurements unless isolated. +- Retain cross-platform CI for the guarded optimized path before changing defaults. + +## Target-repo adaptation + +Re-profile hot-field selection, field widths, tile capacity, cache hierarchy, worker count, scratch-array size, alignment, source-generation cost, frame/stage reuse depth and memory-bandwidth limits. Copy neither GALAXY's tile sizes nor its compact encodings without proving they fit the target domain exactly. Consider NUMA placement separately; ordinary worker-local tiling does not imply NUMA locality. + +## Failure modes + +- Tile fill/conversion overhead dominates the saved traversal cost. +- Tiles are too small to amortize setup or too large for useful cache residency. +- Narrow packing truncates or aliases values outside the donor's domain. +- Worker-local scratch multiplies memory enough to erase the AoS savings at high worker counts. +- Memory bandwidth becomes the bottleneck after vectorization/parallelism. +- Deterministic reduction is replaced by completion-order reduction and changes results. +- A guarded experimental win is promoted globally without evidence across the supported hardware/workload envelope. + +## Rollback trigger + +Fall back to the canonical representation/path on any parity failure, missing/duplicated work, representation-range violation, unstable worker-count behavior, unacceptable RSS growth, or reproducible end-to-end slowdown. Keep the optimized path opt-in when the winning region is narrow or host-specific. + +## Composition notes + +Composes strongly with `OPT-SIMD-001` because SoA/batched primitives often unlock vector code generation, and with `OPT-POOL-001` when worker-local tiles can persist across repeated runs. Re-measure with `OPT-PAR-001`: more workers can increase local scratch and memory-bandwidth pressure even when each worker is individually faster. \ No newline at end of file From 7a0336033abfbbd299e6cbc18f3fb6eec0eb6160 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:03:52 +0930 Subject: [PATCH 066/229] Add persistent worker-pool optimization record --- ...-persistent-topology-aware-worker-pools.md | 83 +++++++++++++++++++ 1 file changed, 83 insertions(+) create mode 100644 optimizations/OPT-POOL-001-persistent-topology-aware-worker-pools.md diff --git a/optimizations/OPT-POOL-001-persistent-topology-aware-worker-pools.md b/optimizations/OPT-POOL-001-persistent-topology-aware-worker-pools.md new file mode 100644 index 0000000..068199a --- /dev/null +++ b/optimizations/OPT-POOL-001-persistent-topology-aware-worker-pools.md @@ -0,0 +1,83 @@ +# OPT-POOL-001 — Persistent topology-aware worker pools + +**Status:** Implemented external reference; persistent reuse and topology-aware selection are merged in GALAXY, while scaling remains host- and workload-specific. +**Domains:** CPU batch runtimes, repeated simulation/render passes, parallel numerical pipelines, thread-pool execution + +## Source evidence + +- Repository: `QSOLKCB/GALAXY` +- PR: https://github.com/QSOLKCB/GALAXY/pull/13 +- Merge commit: `1966bc2595a402a2c653f2e392465224621e20fb` +- Source note: `sources/GALAXY-CPU.md` +- Licensing boundary: Apache-2.0 donor; mechanism promoted without requiring copied source. + +## Problem + +A validated parallel kernel is fast enough that repeatedly creating worker threads, allocating worker-local buffers and choosing an unsuitable logical/physical worker count become material overheads. Per-run spawning also adds latency variance and can hide whether SMT helps or hurts. + +## Optimization problem contract + +- X: Persistent-pool lifetime, worker count, topology policy, reusable worker-local buffer capacity and dispatch strategy. +- F: Candidates preserving exact output/checksum parity, deterministic work ownership and reduction, bounded live resources, explicit topology fallback, and correct shutdown/error handling. +- f: Total or amortized runtime across the expected repetition horizon, including pool lifecycle cost where relevant, plus resource/scaling evidence. +- d: Minimize lifecycle-adjusted runtime while preserving deterministic semantics; prefer simpler scheduling when gains are negligible. +- C: Completion order must not alter observable results, topology claims must match detected evidence, and persistent workers must not retain stale per-dispatch state. +- B: Bounded worker/schedule/tile sweeps and repeated dispatches on the target execution environment. +- S: Stop when the expected repetition horizon and topology policy have a repeatable useful winner, or retain spawned/canonical execution when startup amortization is insufficient. +- Variables: integer, categorical and conditional +- Search scope: local +- Objective behavior: noisy +- Information: black-box +- Evaluation cost: moderate +- Constraints: semantic and resource +- Parallelism: asynchronous +- Exactness: exact + +## Preserved contract + +Persistent reuse changes worker lifetime, not computation semantics. Each dispatch must process the same logical work as the reference/spawned path, and reduction must remain deterministic where required. Buffer reuse must reset or overwrite all state that can affect a later dispatch. + +## Optimization + +Create workers once, allocate their reusable local buffers once, and dispatch repeated jobs through the persistent pool. Give each worker a stable deterministic range or identity. Allow workers to finish independently, but collect/reduce results under a deterministic ordering rule when arithmetic or output order requires it. + +Expose topology policy explicitly. A `physical-first` policy may cap workers at detected physical cores; a `logical` policy may include SMT threads. Detection must fail softly and record the fallback instead of pretending unavailable topology data is authoritative. + +Separate steady-state dispatch timing from startup/teardown, then include lifecycle cost when deciding whether persistence is worthwhile for the real repetition horizon. + +## Before / after evidence + +- Environment: GALAXY PR #13 verifies Linux x86-64, Linux ARM64, macOS ARM64 and Windows x86-64 command/parity surfaces. +- Workload/fixture: repeated worker-local SoA executions over deterministic resident ranges. +- Cold baseline: spawned worker-local SoA creates worker threads for each complete execution. +- Warm/no-op baseline where relevant: persistent steady-state dispatch excludes startup but records pool startup separately. +- Small invalidation / partial-work case where relevant: repeated first/second dispatch parity verifies reused state does not leak. +- Large invalidation / full-work case where relevant: bounded production receipts exercise persistent dispatch across selected worker counts. +- Optimized: one persistent worker set and reusable tile buffers across warm-up and measured repetitions. +- Speedup / memory / I/O / quality change: donor establishes the mechanism and verification boundary but does not provide a universal scaling claim in the PR summary. +- Variance / repetitions / raw samples: target-specific receipts and repetitions are required before promotion. + +## Validation + +Require equality among canonical/reference output, spawned optimized output, first persistent dispatch and subsequent persistent dispatches. Test repeated reuse, shutdown, worker-count changes, topology fallback and completion-order independence. Record requested/effective workers and topology source. Measure startup and teardown separately, then evaluate amortized cost for the actual repetition horizon. + +## Target-repo adaptation + +Re-profile pool lifetime, worker count, SMT policy, buffer size, task granularity, expected number of dispatches, CPU allowance/cgroup constraints and shutdown behavior. Do not infer CPU affinity or NUMA placement from topology-aware worker counting; those require separate mechanisms and evidence. + +## Failure modes + +- The workload is too infrequent to amortize pool startup and retained resources. +- Reused buffers leak stale state between dispatches. +- SMT/logical workers increase contention or memory pressure. +- Container CPU allowance or topology changes after pool creation. +- Long-lived workers hold scarce memory/resources during idle periods. +- Async completion accidentally changes reduction/output order. + +## Rollback trigger + +Use spawned/canonical execution on any parity failure, stale-state leak, shutdown/resource leak, topology mismatch, or lifecycle-adjusted slowdown for the target repetition horizon. Disable physical-first selection when topology detection is unreliable and record the fallback. + +## Composition notes + +Composes with `OPT-SOA-001` when workers own reusable local tiles and with `OPT-SIMD-001` inside each worker kernel. Re-measure with `OPT-PAR-001` because persistent worker counts can still oversubscribe libraries or nested parallel regions. \ No newline at end of file From 7df366360f28c2c1b210e906d2d862b3a17ee80c Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:04:21 +0930 Subject: [PATCH 067/229] Add host-aware promotion optimization record --- ...01-calibrated-host-aware-path-promotion.md | 98 +++++++++++++++++++ 1 file changed, 98 insertions(+) create mode 100644 optimizations/OPT-AUTO-001-calibrated-host-aware-path-promotion.md diff --git a/optimizations/OPT-AUTO-001-calibrated-host-aware-path-promotion.md b/optimizations/OPT-AUTO-001-calibrated-host-aware-path-promotion.md new file mode 100644 index 0000000..2a7d03d --- /dev/null +++ b/optimizations/OPT-AUTO-001-calibrated-host-aware-path-promotion.md @@ -0,0 +1,98 @@ +# OPT-AUTO-001 — Calibrated host-aware path promotion + +**Status:** Implemented external reference; GALAXY merges a fail-closed calibrated selector, while calibration constants and projection accuracy remain host/workload-specific. +**Domains:** multi-path CPU runtimes, heterogeneous execution strategies, production tuning, adaptive dispatch + +## Source evidence + +- Repository: `QSOLKCB/GALAXY` +- PR: https://github.com/QSOLKCB/GALAXY/pull/14 +- Merge commit: `b2e860309a04d7591c86f71d2b4ab1e5eec4c4d7` +- Source note: `sources/GALAXY-CPU.md` +- Licensing boundary: Apache-2.0 donor; record promotes policy structure, not target-specific constants. + +## Problem + +Several semantically equivalent execution paths exist, but the fastest path depends on CPU topology, workload shape, tile size, startup/teardown cost and repetition count. A universal hardware/model table becomes stale, while always choosing the apparently fastest microbenchmark can regress real workloads or violate correctness. + +## Optimization problem contract + +- X: A bounded candidate set of semantically equivalent execution paths plus target-specific calibration shape, scoring policy and promotion margin. +- F: Candidates that pass an independent correctness oracle, preserve workload-relevant calibration dimensions, expose truthful lifecycle/tuning costs, keep canonical/manual control available and fail closed on mismatch. +- f: Projected or directly measured full-work runtime including lifecycle amortization and tuning overhead, with supporting resource/evidence scope. +- d: Minimize expected full-work runtime, but keep the canonical path on near ties or insufficient evidence. +- C: Exact selected/oracle parity, no silent fallback after a correctness mismatch, explicit requested/effective topology policy, accurate evidence scope and no universal claim from one host calibration. +- B: A bounded calibration candidate matrix and repetition budget that never exceeds the requested workload/resource envelope. +- S: Select an optimized candidate only when it beats canonical by a predeclared material margin and then passes full-work oracle verification; otherwise select canonical. +- Variables: categorical, integer and conditional +- Search scope: global +- Objective behavior: noisy +- Information: black-box +- Evaluation cost: expensive +- Constraints: semantic and resource +- Parallelism: sequential +- Exactness: exact + +## Preserved contract + +Automatic selection may change which implementation executes, but not the externally declared result. Every candidate admitted to calibration and the final selected full workload must match an implementation-independent or sufficiently independent oracle under the target's exactness contract. + +Manual/canonical execution surfaces remain available for audit and recovery. A selector must not hide parity failures by silently switching paths after a mismatch; correctness failure is evidence that the candidate or calibration is invalid. + +## Optimization + +1. Define a bounded set of already-validated candidate implementations. +2. Build calibration work that preserves the workload dimensions that materially affect ranking, rather than using an arbitrary tiny microbenchmark. +3. Measure each candidate under the same calibration boundary and record requested versus effective topology/configuration. +4. Project or extrapolate only under an explicit documented model; include startup, teardown, allocation/first-touch and other lifecycle terms when they affect the requested repetition horizon. +5. Keep the canonical path when candidates are within a predeclared margin so noise and model error do not trigger unstable path switching. +6. After selection, run or validate the full requested workload against an independent oracle and fail closed on any mismatch. +7. Emit a receipt explaining candidate scores, lifecycle accounting, selection reason, oracle result and evidence scope. + +The reusable mechanism is calibrated promotion with an uncertainty margin and correctness oracle, not a static table mapping CPU names to implementations. + +## Before / after evidence + +- Environment: GALAXY PR #14 adds dedicated host-auto coverage on Linux x86-64, Linux ARM64, macOS ARM64 and Windows x86-64. +- Workload/fixture: canonical, spawned SoA and persistent SoA candidate families calibrated against requested resident/frame shape. +- Cold baseline: canonical BAM-LUT execution retained as an explicit candidate and fallback/manual path. +- Warm/no-op baseline where relevant: persistent candidate scores include amortized startup and teardown over requested repetitions rather than comparing only steady-state dispatch. +- Small invalidation / partial-work case where relevant: calibration is bounded and may use less resident work while preserving requested frame depth/effective tile shape. +- Large invalidation / full-work case where relevant: selected path is verified on the full requested workload against the streaming canonical oracle. +- Optimized: host/workload-specific candidate chosen only after calibration and margin gating. +- Speedup / memory / I/O / quality change: donor uses a 5% projected promotion margin; that number is source-specific and is not promoted as a universal OPT default. +- Variance / repetitions / raw samples: donor uses three calibration repeats and versioned receipts; targets must choose their own statistically defensible budget and margin. + +## Validation + +- Verify every calibration candidate against an independent oracle before it can compete. +- Preserve workload dimensions known to affect ranking, including depth, effective tile shape and topology where relevant. +- Test projection/scoring identities and lifecycle accounting. +- Include topology-detection and tuning time in the declared selection overhead when users pay that cost. +- Verify the selected full workload again against the oracle. +- Test near ties, canonical wins, optimized wins and intentional parity failures. +- Scope RSS/memory evidence correctly; a process-wide high-water mark covering calibration plus selection cannot be presented as isolated selected-engine memory. +- Keep receipts versioned so future policy changes are distinguishable from earlier selection behavior. + +## Target-repo adaptation + +Re-profile candidate families, calibration size, repetitions, projection model, promotion margin, lifecycle amortization, workload-shape dimensions, topology policy and recalibration cadence. Do not copy GALAXY's 65,536-particle base, tile list, three repeats or 5% margin without target evidence. Prefer direct full-work measurement when calibration cost is affordable or projection error is material. + +## Failure modes + +- Calibration shape does not preserve the feature that determines real-work ranking. +- Runtime scaling is nonlinear, making the projection misleading. +- Selection overhead exceeds the saved runtime on short-lived workloads. +- Workload phases change after calibration and invalidate the choice. +- Near-tie noise causes path thrashing because the margin is too small. +- The oracle shares the same defect or optimized primitive as the candidate and is not genuinely independent. +- Process-wide memory evidence is mislabelled as per-candidate memory. +- Static host/model assumptions replace live evidence and age badly. + +## Rollback trigger + +Immediately reject the selected path on full-work oracle mismatch. Revert automatic promotion to canonical/manual mode when lifecycle-adjusted benefit disappears, calibration becomes unstable or unrepresentative, workload drift changes rankings, or selector overhead materially outweighs expected savings. Recalibrate rather than preserving a stale winner. + +## Composition notes + +This record selects among mechanisms such as `OPT-SOA-001`, `OPT-POOL-001`, `OPT-SIMD-001` and canonical paths after those mechanisms have their own correctness gates. It complements `OPT-BUDGET-001`: regression budgets can detect when a formerly promoted path stops meeting its measured advantage. \ No newline at end of file From 328b5229abdc51fbddb3bdaa060da0a789104570 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:04:46 +0930 Subject: [PATCH 068/229] Index GALAXY-derived optimization records --- README.md | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/README.md b/README.md index c1e102a..dfcd9bc 100644 --- a/README.md +++ b/README.md @@ -33,6 +33,10 @@ The point of this repository is simple: when a future project needs to go faster | [OPT-CRIT-001](optimizations/OPT-CRIT-001-critical-path-prioritization.md) | Critical-path prioritization | **Proposed / OPT synthesis** | Do critical work now, speculate carefully, defer non-critical work | | [OPT-BUDGET-001](optimizations/OPT-BUDGET-001-performance-regression-budgets.md) | Performance regression budgets | **Proposed / OPT synthesis** | Turn performance expectations into environment-scoped regression contracts | | [OPT-PRUNE-001](optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md) | Bound-driven search-space pruning | **Proposed / OPT synthesis** | Prove whole search regions cannot improve the incumbent and skip them | +| [OPT-SIMD-001](optimizations/OPT-SIMD-001-evidence-gated-native-autovectorization.md) | Evidence-gated native autovectorization | **Verified, environment-specific** | Reshape a hot batch for vector codegen, prove parity, inspect instructions, then require measured native benefit | +| [OPT-SOA-001](optimizations/OPT-SOA-001-worker-local-soa-tiling.md) | Worker-local SoA tiling | **Implemented external reference** | Keep only hot fields in bounded per-worker SoA tiles and reuse cache-local scratch | +| [OPT-POOL-001](optimizations/OPT-POOL-001-persistent-topology-aware-worker-pools.md) | Persistent topology-aware worker pools | **Implemented external reference** | Reuse workers/buffers across dispatches and choose physical/logical topology explicitly | +| [OPT-AUTO-001](optimizations/OPT-AUTO-001-calibrated-host-aware-path-promotion.md) | Calibrated host-aware path promotion | **Implemented external reference** | Calibrate equivalent paths on the live host/workload, include lifecycle costs, and promote only with margin + oracle parity | See [CATALOG.md](CATALOG.md) for the decision map and [README4AI.md](README4AI.md) for machine-oriented usage. @@ -53,6 +57,7 @@ where `d` is the objective direction/order; the remaining components are search - [`sources/OPTIMIZATION-LIBRARIES.md`](sources/OPTIMIZATION-LIBRARIES.md) — BayesianOptimization, Hyperopt and NLopt mechanism/taxonomy notes. - [`sources/WPO.md`](sources/WPO.md) — critical-path and performance-budget discovery source. - [`sources/MATHEMATICAL-OPTIMIZATION.md`](sources/MATHEMATICAL-OPTIMIZATION.md) — mathematical/combinatorial problem vocabulary and pruning foundations. +- [`sources/GALAXY-CPU.md`](sources/GALAXY-CPU.md) — merged GALAXY CPU optimization phases covering SIMD/autovectorization, worker-local SoA tiling, persistent topology-aware pools and calibrated host-aware path promotion. - [`power_module.md`](power_module.md) — E8/qutrit DSP architecture that motivated **OPT-DSP-001**. - [`sources/SUXEN.md`](sources/SUXEN.md) — provenance and the required bounded recursive inventory procedure for the opaque `suxen.zip` source candidate. - [`scripts/inventory_zip.py`](scripts/inventory_zip.py) — bounded recursive ZIP inventory entry point; use the explicit limits documented in `sources/SUXEN.md` rather than generic/unbounded extraction. From 177e9de4fcea5d7a77eeb47d0224e3b39de65d03 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:05:13 +0930 Subject: [PATCH 069/229] Add GALAXY CPU patterns to decision catalog --- CATALOG.md | 21 ++++++++++++++++++++- 1 file changed, 20 insertions(+), 1 deletion(-) diff --git a/CATALOG.md b/CATALOG.md index 2993fa9..057e6d6 100644 --- a/CATALOG.md +++ b/CATALOG.md @@ -20,6 +20,10 @@ | Non-critical work delays the dependency chain users actually wait on | [OPT-CRIT-001](optimizations/OPT-CRIT-001-critical-path-prioritization.md) | Prioritize the critical path; speculate/defer deliberately | | Small performance regressions accumulate unnoticed | [OPT-BUDGET-001](optimizations/OPT-BUDGET-001-performance-regression-budgets.md) | Guard stable performance expectations in CI | | Discrete search space is huge but optimistic bounds are available | [OPT-PRUNE-001](optimizations/OPT-PRUNE-001-bound-driven-search-space-pruning.md) | Prune regions that provably cannot beat the incumbent | +| A deterministic hot loop is not exploiting useful host vector instructions | [OPT-SIMD-001](optimizations/OPT-SIMD-001-evidence-gated-native-autovectorization.md) | Reshape for autovectorization, prove parity, inspect codegen, then measure native benefit | +| Large AoS traversal wastes cache/memory and only a bounded hot subset is needed at once | [OPT-SOA-001](optimizations/OPT-SOA-001-worker-local-soa-tiling.md) | Transform bounded per-worker tiles into SoA and reuse cache-local scratch | +| Repeated parallel runs keep paying thread/buffer startup or misuse SMT topology | [OPT-POOL-001](optimizations/OPT-POOL-001-persistent-topology-aware-worker-pools.md) | Persist workers/buffers and choose physical/logical worker policy explicitly | +| Several exact execution paths trade places across hosts or workload shapes | [OPT-AUTO-001](optimizations/OPT-AUTO-001-calibrated-host-aware-path-promotion.md) | Calibrate bounded candidates, include lifecycle cost, require margin and fail-closed oracle parity | Before selecting a record, define the target problem using [`OPTIMIZATION-PROBLEM.md`](OPTIMIZATION-PROBLEM.md). @@ -92,6 +96,18 @@ Protect a stable benchmark expectation with an environment-scoped, variance-awar ### OPT-PRUNE-001 — Bound-driven search-space pruning Maintain a feasible incumbent, derive optimistic bounds for subregions, and discard regions that provably cannot improve the incumbent. +### OPT-SIMD-001 — Evidence-gated native autovectorization +Expose independent batch lanes to the compiler, compare portable/native builds from identical source, prove exact parity and inspect emitted instructions before integrating a measured specialized path. + +### OPT-SOA-001 — Worker-local SoA tiling +Transform only bounded worker-local chunks of a large AoS population into hot-field SoA tiles, reuse local scratch, and retain deterministic reference/reduction semantics. + +### OPT-POOL-001 — Persistent topology-aware worker pools +Create workers and local buffers once for repeated dispatches, expose physical/logical worker policy explicitly, and account for lifecycle amortization rather than timing only the steady-state kernel. + +### OPT-AUTO-001 — Calibrated host-aware path promotion +Choose among already-correct execution paths using workload-shaped live calibration, lifecycle-aware scoring, a material promotion margin and full-work fail-closed oracle verification. + ## Composition guidance Optimizations compose only when their semantic and resource models compose. @@ -103,6 +119,9 @@ Optimizations compose only when their semantic and resource models compose. - approximation must never leak into an API whose callers still assume exact semantics; - adaptive search can lose information efficiency when parallel batches are too wide; - performance budgets require controlled environments or statistically defensible noise handling; -- pruning is valid only when the bound is sound. +- pruning is valid only when the bound is sound; +- SIMD and thread-level parallelism can move the bottleneck to memory bandwidth or CPU frequency limits; +- SoA tiling and persistent pools multiply worker-local storage by worker count, so cache/RSS behavior must be re-measured together; +- host-auto selection must calibrate only candidates that already satisfy their own correctness contracts and must not convert a selector heuristic into a universal hardware ranking. Prefer one measured bottleneck removal at a time, then re-profile and reconsider the problem contract. From 2df8fe6f0dbe06bfc9616d917804a7ad284a59ba Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:05:33 +0930 Subject: [PATCH 070/229] Teach agents the new GALAXY-derived patterns --- README4AI.md | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/README4AI.md b/README4AI.md index a3cb786..ab26bf6 100644 --- a/README4AI.md +++ b/README4AI.md @@ -43,6 +43,10 @@ Before choosing an optimizer, classify: - latency-critical path competes with optional work → `OPT-CRIT-001` - gradual performance drift/regression → `OPT-BUDGET-001` - combinatorial search with valid optimistic bounds → `OPT-PRUNE-001` +- hot deterministic batch underuses vector ISA → `OPT-SIMD-001` +- large AoS traversal needs cache-local bounded hot-field chunks → `OPT-SOA-001` +- repeated parallel runs pay thread/buffer startup or need explicit physical/logical worker policy → `OPT-POOL-001` +- several exact execution paths trade places across hosts/workloads → `OPT-AUTO-001` ## Important distinctions @@ -50,6 +54,10 @@ Before choosing an optimizer, classify: - **coalescing**: result does not exist yet, but equivalent callers share one in-flight evaluation; - **async search diversification**: independent workers should intentionally avoid evaluating the same pending region; - **parallelism**: improves throughput only when resource contention and information dependencies allow it; +- **SIMD/autovectorization**: changes instruction-level execution of equivalent batch work; code-generation evidence is not itself an end-to-end speedup; +- **SoA tiling**: changes temporary data layout/working-set shape while preserving the logical source/output contract; +- **persistent pools**: change worker lifetime and lifecycle amortization, not the kernel's semantics; +- **host-auto promotion**: selects among already-correct paths using live calibration and a fail-closed oracle; it does not make one path universally best; - **approximation**: a contract choice, never a hidden optimization. ## Status vocabulary @@ -72,17 +80,21 @@ The frozen v1 records retain their historical release wording and are exempt fro - Never weaken an assertion, tolerance, theorem target, receipt, trust boundary or validation rule without an explicit contract change. - Never treat a cache hit as proof of a cold rebuild. - Never equate requested workers with observed effective execution. -- Never copy historical worker counts, thresholds, search budgets, bit partitions, cache sizes or approximation limits without target measurement. +- Never copy historical worker counts, thresholds, search budgets, bit partitions, cache sizes, tile sizes, calibration repeats, promotion margins or approximation limits without target measurement. - Never prune a search region unless the bound used for pruning is sound for the declared problem. - Never call an approximate result exact. +- Never infer end-to-end speedup from vector instructions or an isolated kernel probe alone. +- Never publish a native/ISA-specialized path as universal if deployment compatibility is not guaranteed. +- Never treat process-wide RSS gathered across calibration as isolated selected-engine memory evidence. +- Never let an auto selector hide a parity failure by silently falling back; fail closed and preserve explicit canonical/manual control. - Never optimize from stale workload assumptions when fresh measurements are available. - `suxen.zip` remains unpromoted until inventoried and inspected. ## What to copy vs what to adapt -Copy the **structure**: equivalence gates, complete signature identity, coalescing ownership, partitioned coordination, density-adaptive representation, shared materialization, adaptive trial ledgers, explicit approximation envelopes, early reduction, critical-path classification, performance budgets and sound bounds. +Copy the **structure**: equivalence gates, complete signature identity, coalescing ownership, partitioned coordination, density-adaptive representation, shared materialization, adaptive trial ledgers, explicit approximation envelopes, early reduction, critical-path classification, performance budgets, sound bounds, SIMD parity/codegen gates, bounded SoA working sets, persistent-worker lifecycle accounting, and calibrated promotion with independent oracle verification. -Adapt the **numbers and policies**: trial counts, worker caps, hashes, cache sizes, shard counts, bit splits, batch widths, domain-contraction rates, acquisition parameters, tolerances, error limits, benchmark thresholds and stopping budgets. +Adapt the **numbers and policies**: trial counts, worker caps, hashes, cache sizes, shard counts, bit splits, batch widths, tile sizes, domain-contraction rates, acquisition parameters, tolerances, error limits, benchmark thresholds, calibration sizes/repeats, promotion margins, topology policy and stopping budgets. ## Evidence expected in a new record From 3a711826a5413984c7bf5515b314885c97893047 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:05:44 +0930 Subject: [PATCH 071/229] Add agent rules for CPU specialization and auto promotion --- AGENTS.md | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index ef6eef8..d29a43b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -18,8 +18,12 @@ Machine-facing rules for agents using this repository. 14. For critical-path/speculative work, ensure speculation cannot expose side effects before commitment and does not starve the actual critical path. 15. For performance budgets, characterize benchmark noise/environment before enforcing a threshold. 16. For real-time/DSP work, separate slow control work from hot sample/block work when semantics allow it; avoid allocations and synchronization on the hot path. -17. `power_module.md` contains both implemented ideas and aspirational performance language. Check corresponding code/evidence before promoting a claim. -18. `suxen.zip` is a source candidate, not validated evidence. Inventory and read relevant source before extracting optimization claims. -19. The three pinned v1 Lean model files are immutable historical formalization. New records do not become formally proved by association; version future formal modules separately. -20. New post-v1 records must state status, source identity, optimization problem contract, preserved contract, validation, limitations and rollback conditions. -21. Run `python3 scripts/check_catalog.py` after catalog changes. +17. For SIMD/native specialization, require reference parity plus actual code-generation evidence and repeated target-host measurement. Never infer end-to-end speedup merely from wider instructions or an isolated probe. +18. For SoA/tiling, bound temporary storage per worker, prove packed-field and reduction equivalence, and re-profile tile sizes on the target cache/memory hierarchy rather than copying donor values. +19. For persistent worker pools, test repeated-dispatch state isolation and account for startup/teardown amortization. Physical/logical worker selection is not evidence of CPU affinity or NUMA placement. +20. For automatic execution-path promotion, calibrate only already-correct candidates on workload-shaped samples, include lifecycle/tuning costs, require a material promotion margin, preserve canonical/manual control, and fail closed against an independent oracle on parity mismatch. +21. `power_module.md` contains both implemented ideas and aspirational performance language. Check corresponding code/evidence before promoting a claim. +22. `suxen.zip` is a source candidate, not validated evidence. Inventory and read relevant source before extracting optimization claims. +23. The three pinned v1 Lean model files are immutable historical formalization. New records do not become formally proved by association; version future formal modules separately. +24. New post-v1 records must state status, source identity, optimization problem contract, preserved contract, validation, limitations and rollback conditions. +25. Run `python3 scripts/check_catalog.py` after catalog changes. From 42ce933d6551a7b6af1ae7cf5f0ceed8ab0522da Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:09:50 +0930 Subject: [PATCH 072/229] Handle remaining CommonMark link and paragraph edge cases --- scripts/check_catalog.py | 128 +++++++++++++++++++++++++++++++++++++-- 1 file changed, 122 insertions(+), 6 deletions(-) diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py index 51c650f..951156a 100755 --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -6,13 +6,15 @@ or overlooks, and runs the core against that copy: * one-to-three spaces before ATX headings (valid CommonMark indentation), -* optional Markdown titles on links to optimization-record Markdown files, +* optional Markdown titles and angle-bracket destinations on record links, * inline-code examples that resemble optimization-record links, * block-quoted link-reference definitions that actually parse as definitions, * classification placeholders hidden behind rendering-only inline formatting, * generic TODO/TBD-style required-field placeholders, -* hash-shaped source text that lacks explicit commit/revision context, and -* type-7 raw-HTML tags that CommonMark keeps inside an already-open paragraph. +* hash-shaped source text that lacks explicit commit/revision context, +* type-7 raw-HTML tags that CommonMark keeps inside an already-open paragraph, +* thematic breaks after paragraph-interrupting blocks that are not Setext headings, and +* heading-shaped suffixes after multiline inline comments that remain paragraph text. The repository working tree is never modified by this normalization step. """ @@ -31,14 +33,19 @@ CORE_NAME = "check_catalog_core.py" ATX_INDENT_RE = re.compile(r"(?m)^ {1,3}(?=#{1,6}(?:[ \t]|$))") ATX_HEADING_RE = re.compile(r"^ {0,3}#{1,6}(?:[ \t]|$)") +ATX_SUFFIX_RE = re.compile(r"^(?P {0,3})(?P#{1,6})(?=[ \t]|$)") FENCE_LINE_RE = re.compile(r"^ {0,3}(?:`{3,}|~{3,})") LIST_BLOCK_RE = re.compile(r"^ {0,3}(?:[-+*]|\d+[.)])[ \t]+") THEMATIC_BREAK_RE = re.compile( r"^ {0,3}(?:\*(?:[ \t]*\*){2,}|-(?:[ \t]*-){2,}|_(?:[ \t]*_){2,})[ \t]*$" ) +SETEXT_H2_LINE_RE = re.compile(r"^(?P {0,3})-{3,}[ \t]*$") RECORD_LINK_START_RE = re.compile( r"\[([^\]\r\n]+)\]\((optimizations/[^\s)#]+\.md)" ) +ANGLE_RECORD_DEST_RE = re.compile( + r"(?P\[[^\]\r\n]+\]\()<(?Poptimizations/[^\s<>#]+\.md)>" +) BLOCKQUOTE_PREFIX_RE = re.compile(r"^ {0,3}>[ \t]?") LINK_REFERENCE_DEFINITION_RE = re.compile( r"^\[(?:\\.|[^\[\]\\])+\]:[ \t]+\S.*$" @@ -281,6 +288,13 @@ def mask_inline_code_record_destinations(text: str) -> str: return "".join(out) +def canonicalize_angle_record_destinations(text: str) -> str: + """Remove CommonMark angle brackets around optimization-record destinations.""" + return ANGLE_RECORD_DEST_RE.sub( + lambda match: match.group("prefix") + match.group("dest"), text + ) + + def canonicalize_record_link_titles(text: str) -> str: """Drop only syntactically complete optional titles from OPT-record links.""" out: list[str] = [] @@ -421,9 +435,6 @@ def canonicalize_type7_html_paragraph_interruptions(text: str) -> str: continue if paragraph_open and _is_type7_complete_tag_line(content): - # The core is deliberately source-strict and would otherwise start a type-7 - # raw-HTML block here. Prefix only the scratch copy so it remains paragraph - # text, matching CommonMark's rule that type-7 blocks cannot interrupt one. out.append("INLINE_HTML_CONTINUATION " + content.lstrip() + ending) paragraph_open = True continue @@ -434,6 +445,108 @@ def canonicalize_type7_html_paragraph_interruptions(text: str) -> str: return "".join(out) +def _previous_line_interrupts_setext_paragraph(line: str) -> bool: + return bool( + LIST_BLOCK_RE.match(line) + or BLOCKQUOTE_PREFIX_RE.match(line) + or ATX_HEADING_RE.match(line) + or FENCE_LINE_RE.match(line) + or THEMATIC_BREAK_RE.fullmatch(line) + or line.startswith("\t") + or line.startswith(" ") + ) + + +def canonicalize_nonsetext_thematic_breaks(text: str) -> str: + """Keep a hyphen thematic break from being mistaken for a Setext underline.""" + out: list[str] = [] + previous_content: str | None = None + + for raw in text.splitlines(keepends=True): + content = raw.rstrip("\r\n") + ending = raw[len(content) :] + match = SETEXT_H2_LINE_RE.fullmatch(content) + if ( + match is not None + and previous_content is not None + and _previous_line_interrupts_setext_paragraph(previous_content) + ): + normalized = match.group("indent") + "- - -" + out.append(normalized + ending) + previous_content = normalized + continue + out.append(raw) + previous_content = content + + return "".join(out) + + +def canonicalize_multiline_inline_comment_context(text: str) -> str: + """Prevent comment-closing suffixes from becoming fresh block starts mid-paragraph.""" + out: list[str] = [] + comment_from_paragraph = False + fence_char: str | None = None + fence_len = 0 + + for raw in text.splitlines(keepends=True): + content = raw.rstrip("\r\n") + ending = raw[len(content) :] + + if fence_char is not None: + if re.fullmatch( + rf" {{0,3}}{re.escape(fence_char)}{{{fence_len},}}[ \t]*", content + ): + fence_char = None + fence_len = 0 + out.append(raw) + continue + + fence = re.match(r"^ {0,3}(`{3,}|~{3,})(.*)$", content) + if fence is not None: + run = fence.group(1) + info = fence.group(2) + if run[0] != "`" or "`" not in info: + fence_char = run[0] + fence_len = len(run) + out.append(raw) + continue + + if comment_from_paragraph: + close = content.find("-->") + if close < 0: + out.append(raw) + continue + suffix_start = close + 3 + suffix = content[suffix_start:] + heading = ATX_SUFFIX_RE.match(suffix) + if heading is not None: + escaped_suffix = ( + heading.group("indent") + + "\\" + + suffix[len(heading.group("indent")) :] + ) + content = content[:suffix_start] + escaped_suffix + comment_from_paragraph = False + out.append(content + ending) + continue + + cursor = 0 + while True: + start = content.find("", start + 4) + if close >= 0: + cursor = close + 3 + continue + if content[:start].strip(): + comment_from_paragraph = True + break + out.append(raw) + + return "".join(out) + + def _render_placeholder_candidate(value: str) -> str: """Render the subset of inline Markdown relevant to template placeholders.""" result = html.unescape(value.strip()) @@ -532,12 +645,15 @@ def canonicalize_ambiguous_commit_tokens(text: str) -> str: def canonicalize_markdown(text: str, *, link_scan_document: bool) -> str: text = ATX_INDENT_RE.sub("", text) + text = canonicalize_multiline_inline_comment_context(text) text = canonicalize_nested_reference_definitions(text) text = canonicalize_type7_html_paragraph_interruptions(text) + text = canonicalize_nonsetext_thematic_breaks(text) text = canonicalize_classification_placeholders(text) text = canonicalize_generic_required_placeholders(text) text = canonicalize_ambiguous_commit_tokens(text) if link_scan_document: + text = canonicalize_angle_record_destinations(text) text = mask_inline_code_record_destinations(text) text = canonicalize_record_link_titles(text) return text From b21f6cf00b8e54e0f9e1cd6526ac4920915be127 Mon Sep 17 00:00:00 2001 From: Trent Slade <73952179+EmergentMonk@users.noreply.github.com> Date: Wed, 16 Sep 2026 04:24:39 +0930 Subject: [PATCH 073/229] Harden reference placeholders and GALAXY-derived contracts --- ...01-calibrated-host-aware-path-promotion.md | 42 +- ...-persistent-topology-aware-worker-pools.md | 11 +- .../OPT-SOA-001-worker-local-soa-tiling.md | 13 +- scripts/check_catalog.py | 703 +----------------- scripts/check_catalog_normalizer.py | 700 +++++++++++++++++ 5 files changed, 777 insertions(+), 692 deletions(-) mode change 100755 => 100644 scripts/check_catalog.py create mode 100644 scripts/check_catalog_normalizer.py diff --git a/optimizations/OPT-AUTO-001-calibrated-host-aware-path-promotion.md b/optimizations/OPT-AUTO-001-calibrated-host-aware-path-promotion.md index 2a7d03d..2443425 100644 --- a/optimizations/OPT-AUTO-001-calibrated-host-aware-path-promotion.md +++ b/optimizations/OPT-AUTO-001-calibrated-host-aware-path-promotion.md @@ -18,12 +18,12 @@ Several semantically equivalent execution paths exist, but the fastest path depe ## Optimization problem contract - X: A bounded candidate set of semantically equivalent execution paths plus target-specific calibration shape, scoring policy and promotion margin. -- F: Candidates that pass an independent correctness oracle, preserve workload-relevant calibration dimensions, expose truthful lifecycle/tuning costs, keep canonical/manual control available and fail closed on mismatch. -- f: Projected or directly measured full-work runtime including lifecycle amortization and tuning overhead, with supporting resource/evidence scope. -- d: Minimize expected full-work runtime, but keep the canonical path on near ties or insufficient evidence. -- C: Exact selected/oracle parity, no silent fallback after a correctness mismatch, explicit requested/effective topology policy, accurate evidence scope and no universal claim from one host calibration. -- B: A bounded calibration candidate matrix and repetition budget that never exceeds the requested workload/resource envelope. -- S: Select an optimized candidate only when it beats canonical by a predeclared material margin and then passes full-work oracle verification; otherwise select canonical. +- F: Candidates that pass an independent correctness oracle, preserve workload-relevant calibration dimensions, expose truthful lifecycle/tuning/verification costs, keep canonical/manual control available, withhold externally visible effects until parity is established, and fail closed on mismatch. +- f: Total user-paid automatic-selection cost: calibration/tuning, candidate lifecycle terms, selected full-work execution, mandatory independent full-work oracle/verification, and any staging/commit overhead. If verification is explicitly out-of-band rather than paid per invocation, state and score that different scope explicitly. +- d: Minimize expected total user-paid full-work cost under one symmetric accounting boundary, but keep the canonical path on near ties or insufficient evidence. +- C: Exact selected/oracle parity before selected-path effects become externally visible, no silent fallback after a correctness mismatch, explicit requested/effective topology policy, accurate evidence scope and no universal claim from one host calibration. +- B: A bounded end-to-end selection budget covering the calibration candidate matrix/repetitions, selected full-work execution, mandatory full-work oracle/verification, and required staging/commit overhead without exceeding the declared workload/resource envelope. +- S: Select an optimized candidate only when its expected total paid invocation cost, including mandatory verification/staging where applicable, beats canonical by a predeclared material margin and the staged full-work result passes oracle verification; otherwise select canonical. - Variables: categorical, integer and conditional - Search scope: global - Objective behavior: noisy @@ -37,6 +37,8 @@ Several semantically equivalent execution paths exist, but the fastest path depe Automatic selection may change which implementation executes, but not the externally declared result. Every candidate admitted to calibration and the final selected full workload must match an implementation-independent or sufficiently independent oracle under the target's exactness contract. +A selected execution must either be side-effect-free until verification or stage all externally observable outputs and state changes transactionally. The staged result may become visible only after full-work oracle parity succeeds; on mismatch the staged result is discarded. A system that cannot defer irreversible effects must validate before those effects or is not feasible for this pattern. + Manual/canonical execution surfaces remain available for audit and recovery. A selector must not hide parity failures by silently switching paths after a mismatch; correctness failure is evidence that the candidate or calibration is invalid. ## Optimization @@ -44,12 +46,12 @@ Manual/canonical execution surfaces remain available for audit and recovery. A s 1. Define a bounded set of already-validated candidate implementations. 2. Build calibration work that preserves the workload dimensions that materially affect ranking, rather than using an arbitrary tiny microbenchmark. 3. Measure each candidate under the same calibration boundary and record requested versus effective topology/configuration. -4. Project or extrapolate only under an explicit documented model; include startup, teardown, allocation/first-touch and other lifecycle terms when they affect the requested repetition horizon. -5. Keep the canonical path when candidates are within a predeclared margin so noise and model error do not trigger unstable path switching. -6. After selection, run or validate the full requested workload against an independent oracle and fail closed on any mismatch. -7. Emit a receipt explaining candidate scores, lifecycle accounting, selection reason, oracle result and evidence scope. +4. Project or extrapolate only under an explicit documented model. Account for startup, teardown, allocation/first-touch, tuning, and the mandatory full-work oracle/verification and staging costs whenever the user pays them for the requested invocation horizon. Compare candidates and canonical under the same cost boundary; do not promote using an asymmetric score that omits work required only by the optimized path. +5. Keep the canonical path when candidates are within a predeclared margin so noise and model error do not trigger unstable path switching; apply that margin to the total expected paid invocation cost rather than only the selected kernel. +6. Execute the selected full workload into side-effect-free or transactional staging, run or validate the independent full-work oracle within the declared budget, and publish selected-path effects only after exact parity succeeds. Discard staged results and fail closed on any mismatch. +7. Emit a receipt explaining candidate scores, lifecycle/verification accounting, selection reason, oracle result, staging/publish result and evidence scope. -The reusable mechanism is calibrated promotion with an uncertainty margin and correctness oracle, not a static table mapping CPU names to implementations. +The reusable mechanism is calibrated promotion with a symmetric cost boundary, uncertainty margin and correctness oracle, not a static table mapping CPU names to implementations. ## Before / after evidence @@ -58,7 +60,7 @@ The reusable mechanism is calibrated promotion with an uncertainty margin and co - Cold baseline: canonical BAM-LUT execution retained as an explicit candidate and fallback/manual path. - Warm/no-op baseline where relevant: persistent candidate scores include amortized startup and teardown over requested repetitions rather than comparing only steady-state dispatch. - Small invalidation / partial-work case where relevant: calibration is bounded and may use less resident work while preserving requested frame depth/effective tile shape. -- Large invalidation / full-work case where relevant: selected path is verified on the full requested workload against the streaming canonical oracle. +- Large invalidation / full-work case where relevant: selected path is verified on the full requested workload against the streaming canonical oracle; the donor records full-oracle timing separately, and a target must include that verification cost in the user-paid objective/budget whenever it is mandatory per invocation. - Optimized: host/workload-specific candidate chosen only after calibration and margin gating. - Speedup / memory / I/O / quality change: donor uses a 5% projected promotion margin; that number is source-specific and is not promoted as a universal OPT default. - Variance / repetitions / raw samples: donor uses three calibration repeats and versioned receipts; targets must choose their own statistically defensible budget and margin. @@ -67,22 +69,26 @@ The reusable mechanism is calibrated promotion with an uncertainty margin and co - Verify every calibration candidate against an independent oracle before it can compete. - Preserve workload dimensions known to affect ranking, including depth, effective tile shape and topology where relevant. -- Test projection/scoring identities and lifecycle accounting. +- Test projection/scoring identities and lifecycle accounting, including the selected full run, mandatory full-work oracle/verification, staging and publish costs whenever those are paid by the invocation. +- Compare the promoted path's total paid cost against canonical under the same scope; test a case where a fast selected kernel is correctly rejected because verification overhead erases the advantage. - Include topology-detection and tuning time in the declared selection overhead when users pay that cost. -- Verify the selected full workload again against the oracle. -- Test near ties, canonical wins, optimized wins and intentional parity failures. +- Verify the selected full workload again against the oracle before making staged effects visible. +- Test side-effect-free and transactional staging, intentional parity failures, and prove that mismatching staged output/state is discarded rather than exposed. +- Test near ties, canonical wins and optimized wins. - Scope RSS/memory evidence correctly; a process-wide high-water mark covering calibration plus selection cannot be presented as isolated selected-engine memory. - Keep receipts versioned so future policy changes are distinguishable from earlier selection behavior. ## Target-repo adaptation -Re-profile candidate families, calibration size, repetitions, projection model, promotion margin, lifecycle amortization, workload-shape dimensions, topology policy and recalibration cadence. Do not copy GALAXY's 65,536-particle base, tile list, three repeats or 5% margin without target evidence. Prefer direct full-work measurement when calibration cost is affordable or projection error is material. +Re-profile candidate families, calibration size, repetitions, projection model, promotion margin, lifecycle amortization, full-work oracle cost, staging/publish strategy, workload-shape dimensions, topology policy and recalibration cadence. Do not copy GALAXY's 65,536-particle base, tile list, three repeats or 5% margin without target evidence. Prefer direct full-work measurement when calibration cost is affordable or projection error is material. If the target has irreversible externally visible effects, establish a pre-effect oracle/validation boundary before adopting automatic promotion. ## Failure modes - Calibration shape does not preserve the feature that determines real-work ranking. - Runtime scaling is nonlinear, making the projection misleading. -- Selection overhead exceeds the saved runtime on short-lived workloads. +- Selection or mandatory verification overhead exceeds the saved runtime on short-lived workloads. +- The scoring boundary omits oracle/staging work paid only by promoted candidates and creates a false win. +- Selected-path output or state changes escape before oracle parity is known. - Workload phases change after calibration and invalidate the choice. - Near-tie noise causes path thrashing because the margin is too small. - The oracle shares the same defect or optimized primitive as the candidate and is not genuinely independent. @@ -91,7 +97,7 @@ Re-profile candidate families, calibration size, repetitions, projection model, ## Rollback trigger -Immediately reject the selected path on full-work oracle mismatch. Revert automatic promotion to canonical/manual mode when lifecycle-adjusted benefit disappears, calibration becomes unstable or unrepresentative, workload drift changes rankings, or selector overhead materially outweighs expected savings. Recalibrate rather than preserving a stale winner. +Immediately reject the selected path on full-work oracle mismatch and discard all uncommitted staged results. Treat any already-exposed mismatching result as a contract failure. Revert automatic promotion to canonical/manual mode when the complete selection+execution+verification budget is exceeded, lifecycle-adjusted total paid benefit disappears, calibration becomes unstable or unrepresentative, workload drift changes rankings, or selector/verification overhead materially outweighs expected savings. Recalibrate rather than preserving a stale winner. ## Composition notes diff --git a/optimizations/OPT-POOL-001-persistent-topology-aware-worker-pools.md b/optimizations/OPT-POOL-001-persistent-topology-aware-worker-pools.md index 068199a..82a07ae 100644 --- a/optimizations/OPT-POOL-001-persistent-topology-aware-worker-pools.md +++ b/optimizations/OPT-POOL-001-persistent-topology-aware-worker-pools.md @@ -37,12 +37,16 @@ A validated parallel kernel is fast enough that repeatedly creating worker threa Persistent reuse changes worker lifetime, not computation semantics. Each dispatch must process the same logical work as the reference/spawned path, and reduction must remain deterministic where required. Buffer reuse must reset or overwrite all state that can affect a later dispatch. +A failed or cancelled dispatch may be followed by reuse only after every worker-local buffer, queue, completion flag and dispatch-generation marker is returned to a known clean state. If that reset cannot be proven complete, mark the pool unusable and create a fresh pool before accepting more work. + ## Optimization Create workers once, allocate their reusable local buffers once, and dispatch repeated jobs through the persistent pool. Give each worker a stable deterministic range or identity. Allow workers to finish independently, but collect/reduce results under a deterministic ordering rule when arithmetic or output order requires it. Expose topology policy explicitly. A `physical-first` policy may cap workers at detected physical cores; a `logical` policy may include SMT threads. Detection must fail softly and record the fallback instead of pretending unavailable topology data is authoritative. +Treat dispatch completion as a state transition. Successful completion must leave all reusable state ready for the next generation. Failure or cancellation must either run the same complete reset protocol or retire the pool so partial state cannot leak into a later dispatch. + Separate steady-state dispatch timing from startup/teardown, then include lifecycle cost when deciding whether persistence is worthwhile for the real repetition horizon. ## Before / after evidence @@ -59,16 +63,17 @@ Separate steady-state dispatch timing from startup/teardown, then include lifecy ## Validation -Require equality among canonical/reference output, spawned optimized output, first persistent dispatch and subsequent persistent dispatches. Test repeated reuse, shutdown, worker-count changes, topology fallback and completion-order independence. Record requested/effective workers and topology source. Measure startup and teardown separately, then evaluate amortized cost for the actual repetition horizon. +Require equality among canonical/reference output, spawned optimized output, first persistent dispatch and subsequent persistent dispatches. In addition to ordinary repeated-success cases, force success → failure → success and success → cancellation → success sequences after partial worker activity. Verify that every reusable buffer, queue, completion record and dispatch generation is reset before the final success, or verify that the affected pool is retired and replaced before reuse. Test shutdown, worker-count changes, topology fallback and completion-order independence. Record requested/effective workers and topology source. Measure startup and teardown separately, then evaluate amortized cost for the actual repetition horizon. ## Target-repo adaptation -Re-profile pool lifetime, worker count, SMT policy, buffer size, task granularity, expected number of dispatches, CPU allowance/cgroup constraints and shutdown behavior. Do not infer CPU affinity or NUMA placement from topology-aware worker counting; those require separate mechanisms and evidence. +Re-profile pool lifetime, worker count, SMT policy, buffer size, task granularity, expected number of dispatches, CPU allowance/cgroup constraints, failure-reset protocol and shutdown behavior. Do not infer CPU affinity or NUMA placement from topology-aware worker counting; those require separate mechanisms and evidence. ## Failure modes - The workload is too infrequent to amortize pool startup and retained resources. - Reused buffers leak stale state between dispatches. +- A failed/cancelled dispatch leaves partial buffers, queue entries or completion state that contaminates the next generation. - SMT/logical workers increase contention or memory pressure. - Container CPU allowance or topology changes after pool creation. - Long-lived workers hold scarce memory/resources during idle periods. @@ -76,7 +81,7 @@ Re-profile pool lifetime, worker count, SMT policy, buffer size, task granularit ## Rollback trigger -Use spawned/canonical execution on any parity failure, stale-state leak, shutdown/resource leak, topology mismatch, or lifecycle-adjusted slowdown for the target repetition horizon. Disable physical-first selection when topology detection is unreliable and record the fallback. +Use spawned/canonical execution on any parity failure, stale-state leak, failed-dispatch reset failure, shutdown/resource leak, topology mismatch, or lifecycle-adjusted slowdown for the target repetition horizon. Retire a pool immediately when a failure/cancellation leaves its reusable state uncertain. Disable physical-first selection when topology detection is unreliable and record the fallback. ## Composition notes diff --git a/optimizations/OPT-SOA-001-worker-local-soa-tiling.md b/optimizations/OPT-SOA-001-worker-local-soa-tiling.md index 37d17d1..ba9a819 100644 --- a/optimizations/OPT-SOA-001-worker-local-soa-tiling.md +++ b/optimizations/OPT-SOA-001-worker-local-soa-tiling.md @@ -22,7 +22,7 @@ A large array-of-structures working set is repeatedly traversed by multiple work - F: Candidates that preserve exact source-to-output semantics, deterministic partitioning and reduction, represent every required field without lossy reinterpretation, keep worker-local memory bounded, and retain an unchanged canonical/reference path. - f: End-to-end runtime, peak working-memory/RSS evidence and useful worker scaling over the declared workload matrix. - d: Pareto-minimize runtime and memory footprint subject to exact parity; reject candidates whose timing gain requires unacceptable RSS growth or unstable worker scaling. -- C: Exact checksum/output equality with the reference path, deterministic worker-count behavior where required, no dropped/duplicated elements, and bounded worker-local storage independent of total resident population. +- C: Exact canonical-output equality with the reference path, deterministic worker-count behavior where required, no dropped/duplicated elements, observable ordering preserved where required, and bounded worker-local storage independent of total resident population. Aggregate checksums are supplementary evidence unless the aggregate is itself the complete public output contract. - B: A bounded worker-count × tile-size benchmark matrix with repeated runs and representative workload sizes on the target machines. - S: Stop after a practical winning tile/worker region is identified or all candidates fail; do not promote a single pathological fast point without surrounding evidence. - Variables: integer and categorical @@ -36,7 +36,9 @@ A large array-of-structures working set is repeatedly traversed by multiple work ## Preserved contract -The tiled SoA path must compute the same declared outputs/checksum as the reference AoS path for every processed element and supported worker count. Reordering storage is allowed only when observable output order, tie behavior, reduction semantics and deterministic identity remain unchanged. +The tiled SoA path must compute the same declared outputs as the reference AoS path for every processed element and supported worker count. When the interface exposes per-element values or ordering, validation must compare those complete outputs element-by-element and preserve observable order. Reordering storage is allowed only when observable output order, tie behavior, reduction semantics and deterministic identity remain unchanged. + +An aggregate checksum may remain as an additional repeatability/corruption signal, but it is not a substitute for full-output comparison when richer output is observable. If the target contract exposes only the aggregate itself, state that boundary explicitly. Packing fields into narrower representations is permitted only when the representation is proven exact for the target domain or when the target contract explicitly allows approximation. This record is exact by default. @@ -62,12 +64,12 @@ The key scaling property is that optimized working storage grows roughly with wo - Small invalidation / partial-work case where relevant: small tile/worker matrix cells validate capacity and partition behavior. - Large invalidation / full-work case where relevant: donor sweep supports resident populations up to the full experimental workload and multiple tile sizes/workers. - Optimized: bounded worker-local compact SoA tiles with worker-local x/y/output scratch and SIMD-friendly batch hashing. -- Speedup / memory / I/O / quality change: donor PR #12 states that PR #11 established exact cross-platform parity and strong multi-host performance/memory evidence; this record intentionally does not turn those donor observations into universal target numbers. +- Speedup / memory / I/O / quality change: donor PR #12 states that PR #11 established exact cross-platform checksum parity and strong multi-host performance/memory evidence; this portable record additionally requires full canonical-output comparison whenever a target exposes per-element values or order. - Variance / repetitions / raw samples: donor sweep emits raw matrix data, comparison tables and receipts; targets must repeat the sweep locally. ## Validation -- Check exact reference/generic/native/SoA checksum equality for every matrix cell. +- When the target exposes per-element values, compare the complete canonical/reference and SoA outputs element-by-element for every matrix cell, including observable order. Retain exact checksum equality as an additional repeatability signal rather than the sole equivalence proof. If the aggregate is the complete public output, state that explicitly. - Verify deterministic partitioning covers every source element exactly once. - Verify worker-count invariance when the target contract requires it. - Test tile-capacity boundaries, partial final tiles and minimum/maximum supported sizes. @@ -87,12 +89,13 @@ Re-profile hot-field selection, field widths, tile capacity, cache hierarchy, wo - Narrow packing truncates or aliases values outside the donor's domain. - Worker-local scratch multiplies memory enough to erase the AoS savings at high worker counts. - Memory bandwidth becomes the bottleneck after vectorization/parallelism. +- Aggregate checksum parity masks an element-level or ordering defect on an interface that exposes richer output. - Deterministic reduction is replaced by completion-order reduction and changes results. - A guarded experimental win is promoted globally without evidence across the supported hardware/workload envelope. ## Rollback trigger -Fall back to the canonical representation/path on any parity failure, missing/duplicated work, representation-range violation, unstable worker-count behavior, unacceptable RSS growth, or reproducible end-to-end slowdown. Keep the optimized path opt-in when the winning region is narrow or host-specific. +Fall back to the canonical representation/path on any full-output parity failure, missing/duplicated work, observable ordering mismatch, representation-range violation, unstable worker-count behavior, unacceptable RSS growth, or reproducible end-to-end slowdown. Keep the optimized path opt-in when the winning region is narrow or host-specific. ## Composition notes diff --git a/scripts/check_catalog.py b/scripts/check_catalog.py old mode 100755 new mode 100644 index 951156a..f0c737d --- a/scripts/check_catalog.py +++ b/scripts/check_catalog.py @@ -1,700 +1,71 @@ #!/usr/bin/env python3 -"""Normalize supported CommonMark syntax, then run the hardened catalog checker. - -The core checker intentionally stays strict and source-oriented. This front end creates a -scratch copy, canonicalizes rendering-equivalent forms that the core otherwise rejects -or overlooks, and runs the core against that copy: - -* one-to-three spaces before ATX headings (valid CommonMark indentation), -* optional Markdown titles and angle-bracket destinations on record links, -* inline-code examples that resemble optimization-record links, -* block-quoted link-reference definitions that actually parse as definitions, -* classification placeholders hidden behind rendering-only inline formatting, -* generic TODO/TBD-style required-field placeholders, -* hash-shaped source text that lacks explicit commit/revision context, -* type-7 raw-HTML tags that CommonMark keeps inside an already-open paragraph, -* thematic breaks after paragraph-interrupting blocks that are not Setext headings, and -* heading-shaped suffixes after multiline inline comments that remain paragraph text. - -The repository working tree is never modified by this normalization step. -""" +"""Run the catalog normalizer with definition-aware placeholder rendering.""" from __future__ import annotations -import html import re -import shutil -import subprocess -import sys -import tempfile -from pathlib import Path -ROOT = Path(__file__).resolve().parents[1] -CORE_NAME = "check_catalog_core.py" -ATX_INDENT_RE = re.compile(r"(?m)^ {1,3}(?=#{1,6}(?:[ \t]|$))") -ATX_HEADING_RE = re.compile(r"^ {0,3}#{1,6}(?:[ \t]|$)") -ATX_SUFFIX_RE = re.compile(r"^(?P {0,3})(?P#{1,6})(?=[ \t]|$)") -FENCE_LINE_RE = re.compile(r"^ {0,3}(?:`{3,}|~{3,})") -LIST_BLOCK_RE = re.compile(r"^ {0,3}(?:[-+*]|\d+[.)])[ \t]+") -THEMATIC_BREAK_RE = re.compile( - r"^ {0,3}(?:\*(?:[ \t]*\*){2,}|-(?:[ \t]*-){2,}|_(?:[ \t]*_){2,})[ \t]*$" -) -SETEXT_H2_LINE_RE = re.compile(r"^(?P {0,3})-{3,}[ \t]*$") -RECORD_LINK_START_RE = re.compile( - r"\[([^\]\r\n]+)\]\((optimizations/[^\s)#]+\.md)" -) -ANGLE_RECORD_DEST_RE = re.compile( - r"(?P\[[^\]\r\n]+\]\()<(?Poptimizations/[^\s<>#]+\.md)>" -) -BLOCKQUOTE_PREFIX_RE = re.compile(r"^ {0,3}>[ \t]?") -LINK_REFERENCE_DEFINITION_RE = re.compile( - r"^\[(?:\\.|[^\[\]\\])+\]:[ \t]+\S.*$" -) -LINK_REFERENCE_TITLE_CONTINUATION_RE = re.compile( - r'^ {0,3}(?:"(?:\\.|[^"\\])*"|\'(?:\\.|[^\'\\])*\'|\((?:\\.|[^)\\])*\))[ \t]*$' -) -INLINE_HTML_TAG_RE = re.compile( - r"`]+))?)*" - r"[ \t]*/?>" -) -STANDALONE_HTML_TAG_RE = re.compile( - r"^ {0,3}(?:" - r"[A-Za-z][A-Za-z0-9-]*)[ \t]*>" - r"|<(?P[A-Za-z][A-Za-z0-9-]*)" - r"(?:[ \t]+[A-Za-z_:][A-Za-z0-9_.:-]*" - r"(?:[ \t]*=[ \t]*(?:\"[^\"]*\"|'[^']*'|[^ \t\n\"'=<>`]+))?)*" - r"[ \t]*/?>" - r")[ \t]*$" -) -INLINE_WRAPPERS = ("**", "__", "~~", "*", "_", "`") -CLASSIFICATION_TEMPLATE_VALUES = { - "Variables": "continuous / integer / categorical / conditional / mixed", - "Search scope": "local / global", - "Objective behavior": "deterministic / noisy / stochastic", - "Information": "gradient available / derivative-free / black-box", - "Evaluation cost": "cheap / moderate / expensive", - "Constraints": "bounds / equality / inequality / semantic / resource", - "Parallelism": "sequential / synchronous batch / asynchronous", - "Exactness": "exact / approximation permitted under an explicit error contract", -} -CLASSIFICATION_FIELD_RE = re.compile( - r"^(?P- (?P" - + "|".join(re.escape(field) for field in CLASSIFICATION_TEMPLATE_VALUES) - + r"):[ \t]*)(?P.*)$" -) -REQUIRED_FIELD_NAMES = ( - "X", - "F", - "f", - "d", - "C", - "B", - "S", - *CLASSIFICATION_TEMPLATE_VALUES.keys(), -) -REQUIRED_FIELD_RE = re.compile( - r"^(?P- (?P" - + "|".join(re.escape(field) for field in REQUIRED_FIELD_NAMES) - + r"):[ \t]*)(?P.*)$" -) -GENERIC_PLACEHOLDER_RE = re.compile( - r"^(?:unknown|tbd|todo|n/?a|none|pending)(?:[.!?])?$", re.IGNORECASE +import check_catalog_normalizer as normalizer + +REFERENCE_DEFINITION_RE = re.compile( + r"(?m)^ {0,3}\[(?P