diff --git a/.gitignore b/.gitignore index dadf6e5..420fcbd 100644 --- a/.gitignore +++ b/.gitignore @@ -50,6 +50,7 @@ ENV/ .venv pip-log.txt pip-delete-this-directory.txt +*.egg-info/ # Jupyter Notebooks .ipynb_checkpoints diff --git a/examples/README.md b/examples/README.md index 35b7539..11ad03b 100644 --- a/examples/README.md +++ b/examples/README.md @@ -13,7 +13,7 @@ This directory contains example data files and configurations for testing and de - `receipts.json` - Sample receipt data with participant activities - `payouts.json` - Sample payout data for index computation - `epoch_budget.json` - Sample AFI emissions budget data -- `scores_min.json` - Minimal BenchKit scores for testing +- `merit_scores.synthetic.json` - Synthetic merit scores for the `--scores` path (see "Merit scores (synthetic)" below) ### Integration Examples - `pipeline_demo.sh` - Complete end-to-end pipeline demonstration @@ -34,18 +34,18 @@ afi-econ-kit simulate --config examples/config.yaml --outdir budget_sim \ --budget examples/epoch_budget.json ``` -### 3. Simulation with BenchKit Integration +### 3. Simulation with Merit Scores (synthetic) ```bash -# Run simulation with merit scores +# Run simulation with the synthetic merit scores afi-econ-kit simulate --config examples/config.yaml --outdir merit_sim \ - --scores examples/scores_min.json + --scores examples/merit_scores.synthetic.json ``` ### 4. Full Integration (Budget + Merit) ```bash # Run simulation with both budget and merit scores afi-econ-kit simulate --config examples/config.yaml --outdir full_sim \ - --budget examples/epoch_budget.json --scores examples/scores_min.json + --budget examples/epoch_budget.json --scores examples/merit_scores.synthetic.json ``` ### 5. AFI Index Computation @@ -108,26 +108,27 @@ afi-econ-kit payouts --allocations stage_safety/safety_allocations.json \ } ``` -### scores_min.json Format (BenchKit) +### merit_scores.synthetic.json Format (merit scores, synthetic) ```json { - "participants": { - "user_001": { - "role": "reputation", - "total_score": 85.5, - "benchmark_scores": { - "latency": 90.0, - "throughput": 81.0 - } - } - }, + "reputation": {"score": 0.58}, + "poi": {"score": 0.55}, + "poinsight": {"score": 0.60}, + "n": 20, "stamp": { + "source": "synthetic", "version": "0.1.0", - "utc_ts": "2024-01-01T12:00:00Z" + "utc_ts": "2026-08-24T00:00:00Z" } } ``` +`reputation.score` is required; `poi.score` and `poinsight.score` default to 0.5 when +absent. Scores are clipped to [0, 1]. `n` is the row count behind the scores and `stamp` +is passed through unchanged into the simulation's provenance stamp (as `merit_stamp`). +These are synthetic research inputs -- see "Merit scores (synthetic)" below for what +they are and are not. + ## Expected Outputs ### Simulation Outputs @@ -171,18 +172,24 @@ afi-econ-kit simulate --config config.yaml --outdir integrated_sim \ --budget ../afi-emissions/budget_out/epoch_budget.json ``` -### With afi-benchkit -```bash -# Generate scores in afi-benchkit -cd ../afi-benchkit -afi-benchkit reputation --config config.yaml --out scores_out +### Merit scores (synthetic) + +The `--scores` merit path takes a **synthetic** merit-scores file: -# Use scores in afi-econ-kit -cd ../afi-econ-kit +```bash afi-econ-kit simulate --config config.yaml --outdir integrated_sim \ - --scores ../afi-benchkit/scores_out/scores.json + --scores examples/merit_scores.synthetic.json ``` +These scores are research-plane inputs only. The protocol source of analyst merit is +the CAL-GOV analyst calibration record (`afi.analyst-calibration.v1`) -- **not a +scalar** -- and any conversion of it into a merit value is **CHAIN-GOV reserved**. +afi-econ consumes no such value from the protocol; edit the synthetic file to explore +how merit multipliers move gauge allocations. Proof-of-Intelligence (PoI) and +Proof-of-Insight (PoInsight) remain reserved protocol reputation primitives +(CONST-GOV D-CONST-5; CAL-GOV D-CAL-5); the `poi` / `poinsight` keys here are +synthetic placeholders that stand in for no protocol value. + ## Testing and Validation All example files are designed to work with the golden test suite: diff --git a/examples/merit_scores.synthetic.json b/examples/merit_scores.synthetic.json new file mode 100644 index 0000000..97f9c2f --- /dev/null +++ b/examples/merit_scores.synthetic.json @@ -0,0 +1,12 @@ +{ + "reputation": {"score": 0.58}, + "poi": {"score": 0.55}, + "poinsight": {"score": 0.60}, + "n": 20, + "stamp": { + "source": "synthetic", + "version": "0.1.0", + "utc_ts": "2026-08-24T00:00:00Z", + "note": "Synthetic research input for afi-econ's --scores merit path. The protocol source of analyst merit is the CAL-GOV analyst calibration record (afi.analyst-calibration.v1) -- not a scalar -- and any merit conversion is CHAIN-GOV reserved." + } +} diff --git a/examples/pipeline_demo.sh b/examples/pipeline_demo.sh index f2550dc..f396b1d 100755 --- a/examples/pipeline_demo.sh +++ b/examples/pipeline_demo.sh @@ -30,7 +30,7 @@ echo "" echo "🏆 Step 3: Simulation with Merit Scores" echo "--------------------------------------" afi-econ-kit simulate --config config.yaml --outdir demo_pipeline_out/merit \ - --scores tests/fixtures/scores_min.json + --scores examples/merit_scores.synthetic.json echo "✅ Merit integration complete" # Step 4: Full integration @@ -38,7 +38,7 @@ echo "" echo "🔗 Step 4: Full Integration (Budget + Merit)" echo "-------------------------------------------" afi-econ-kit simulate --config config.yaml --outdir demo_pipeline_out/full \ - --budget examples/epoch_budget.json --scores tests/fixtures/scores_min.json + --budget examples/epoch_budget.json --scores examples/merit_scores.synthetic.json echo "✅ Full integration complete" # Step 5: AFI Index computation diff --git a/out/econ_breakers.png b/out/econ_breakers.png deleted file mode 100644 index d0362d7..0000000 Binary files a/out/econ_breakers.png and /dev/null differ diff --git a/out/econ_budget.png b/out/econ_budget.png deleted file mode 100644 index 935d24a..0000000 Binary files a/out/econ_budget.png and /dev/null differ diff --git a/out/econ_gauge_shares.png b/out/econ_gauge_shares.png deleted file mode 100644 index bdd5db4..0000000 Binary files a/out/econ_gauge_shares.png and /dev/null differ diff --git a/out/econ_summary.json b/out/econ_summary.json deleted file mode 100644 index bee3ac2..0000000 --- a/out/econ_summary.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "stamp": { - "timestamp": "2025-10-05T00:38:01.146950+00:00", - "git_short_hash": "n/a", - "seed": 1337, - "data_hash": "mc_1337_64", - "version": "0.1.0", - "git_sha_short": "n/a", - "utc_ts": "2025-10-05T00:38:01.199829+00:00", - "config_hash": "5dbfe76285c659439bcb8a7628dd646ba56079d47a9a958b3f08d2a6343b9ef8", - "params_hash": "1f7c145b1cafbcf3477f92e1e2d41dcfd9bfc2e06d8ea59ab74da2e1b29b1d82", - "rng_seed_used": 1337, - "scores_path": null, - "scores_hash": null, - "benchkit_stamp": null, - "bench_merit": null, - "budget_path": null, - "budget_hash": null, - "budget_stamp": null, - "budget_epoch": null - }, - "summary": { - "total_epochs": 64, - "total_breaker_events": 0, - "final_budget": 1000000.0, - "avg_pool_utilization": 1.0 - }, - "config_hash": -126590584610477770 -} \ No newline at end of file diff --git a/out_audit/AUDIT.md b/out_audit/AUDIT.md deleted file mode 100644 index 73726a1..0000000 --- a/out_audit/AUDIT.md +++ /dev/null @@ -1,33 +0,0 @@ -# AFI Economics End-to-End Audit Report - -> **DEPRECATED / SUPERSEDED:** This document predates AFI Settlement v1 doctrine. It may describe v0 per-signal minting, ERC-1155 receipts, direct beneficiary payouts, stale ENS/Snapshot references, or missing vault architecture. See `afi-docs/specs/AFI_SETTLEMENT_V1_DOCTRINE.md` for canonical architecture. - -## Summary - -This audit verifies the mathematical and economic integrity of the AFI Economics system -for whitepaper publication readiness. - -## Audit Results - -| Check | Status | Description | -|-------|--------|-------------| -| Provenance | ✅ PASS | Complete provenance chain across components | -| Emissions Invariants | ✅ PASS | E_t = max(0, B_t × m_t) mathematical correctness | -| Gauge Invariants | ✅ PASS | Share allocation sums to 1.0, caps enforced | -| Payouts Conservation | ✅ PASS | Pool = Payouts + Holdbacks conservation | -| Stability | ✅ PASS | Rate limiting within bounds (≤0.15) | -| Determinism | ✅ PASS | Reproducible outputs with identical inputs | -| BenchKit Influence | ✅ PASS | Measurable impact of merit scores on allocations | - -## BenchKit Influence Deltas - -- **producers**: +0.004211 -- **enrichment**: +0.002034 -- **validators**: +0.001101 -- **public_goods**: -0.007347 - -## Conclusion - -**✅ WHITEPAPER READY**: All critical checks pass. The AFI Economics system demonstrates mathematical integrity, economic conservation, and deterministic reproducibility suitable for academic publication. - -**Summary**: 7 passed, 0 N/A, 0 failed out of 7 total checks. diff --git a/out_audit/audit_report.json b/out_audit/audit_report.json deleted file mode 100644 index e433725..0000000 --- a/out_audit/audit_report.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "benchkit_influence_ok": true, - "deltas": { - "enrichment": 0.002034, - "producers": 0.004211, - "public_goods": -0.007347, - "validators": 0.001101 - }, - "determinism_ok": true, - "emissions_invariants_ok": true, - "gauge_invariants_ok": true, - "hashes": { - "emissions_series_sha256": "6a6e60fd268a364071ea65a5d765bd846af001a3a7ec5efc0623f2d01e384b1a" - }, - "paths": { - "budget": "/Users/secretservice/afi-emissions/out_emit/epoch_budget.json", - "econ_base_dir": "/Users/secretservice/afi-econ-kit/out_base", - "econ_with_dir": "/Users/secretservice/afi-econ-kit/out_with", - "scores": "/Users/secretservice/afi-benchkit/out/scores.json" - }, - "payouts_conservation_ok": true, - "provenance_ok": true, - "stability_ok": true, - "summary": { - "failed": 0, - "na": 0, - "passed": 7, - "total_checks": 7 - }, - "timestamp": "2025-10-05T17:22:19.592104+00:00" -} \ No newline at end of file diff --git a/out_base/econ_breakers.png b/out_base/econ_breakers.png deleted file mode 100644 index c8183bb..0000000 Binary files a/out_base/econ_breakers.png and /dev/null differ diff --git a/out_base/econ_budget.png b/out_base/econ_budget.png deleted file mode 100644 index 96450af..0000000 Binary files a/out_base/econ_budget.png and /dev/null differ diff --git a/out_base/econ_gauge_shares.png b/out_base/econ_gauge_shares.png deleted file mode 100644 index 5f00607..0000000 Binary files a/out_base/econ_gauge_shares.png and /dev/null differ diff --git a/out_base/econ_summary.json b/out_base/econ_summary.json deleted file mode 100644 index 35757d9..0000000 --- a/out_base/econ_summary.json +++ /dev/null @@ -1,34 +0,0 @@ -{ - "stamp": { - "timestamp": "2025-10-05T17:22:35.692713+00:00", - "git_short_hash": "n/a", - "seed": 1337, - "data_hash": "mc_1337_64", - "version": "0.1.0", - "git_sha_short": "n/a", - "utc_ts": "2025-10-05T17:22:35.741806+00:00", - "config_hash": "5dbfe76285c659439bcb8a7628dd646ba56079d47a9a958b3f08d2a6343b9ef8", - "params_hash": "1f7c145b1cafbcf3477f92e1e2d41dcfd9bfc2e06d8ea59ab74da2e1b29b1d82", - "rng_seed_used": 1337, - "scores_path": null, - "scores_hash": null, - "benchkit_stamp": null, - "bench_merit": null, - "budget_path": "/Users/secretservice/afi-emissions/out_emit/epoch_budget.json", - "budget_hash": "2017534d57f93a545ba2f37e9c3315a8cf13fda398b35df6b1ee4db262b30b26", - "budget_stamp": { - "version": "0.1.0", - "git_sha": "n/a", - "utc_ts": "2025-10-05T02:57:34.624683+00:00", - "config_hash": "c13c4a8ecffe04d71f5d71f1d36ea214d395b96c35a740b48c90273595efdf34" - }, - "budget_epoch": 25 - }, - "summary": { - "total_epochs": 64, - "total_breaker_events": 0, - "final_budget": 1000000.0, - "avg_pool_utilization": 1.0 - }, - "config_hash": -1364694410363294296 -} \ No newline at end of file diff --git a/out_with/econ_breakers.png b/out_with/econ_breakers.png deleted file mode 100644 index 51aa8a9..0000000 Binary files a/out_with/econ_breakers.png and /dev/null differ diff --git a/out_with/econ_budget.png b/out_with/econ_budget.png deleted file mode 100644 index 3dfaff9..0000000 Binary files a/out_with/econ_budget.png and /dev/null differ diff --git a/out_with/econ_gauge_shares.png b/out_with/econ_gauge_shares.png deleted file mode 100644 index 01eba39..0000000 Binary files a/out_with/econ_gauge_shares.png and /dev/null differ diff --git a/out_with/econ_summary.json b/out_with/econ_summary.json deleted file mode 100644 index 9be533a..0000000 --- a/out_with/econ_summary.json +++ /dev/null @@ -1,43 +0,0 @@ -{ - "stamp": { - "timestamp": "2025-10-05T17:22:36.885207+00:00", - "git_short_hash": "n/a", - "seed": 1337, - "data_hash": "mc_1337_64", - "version": "0.1.0", - "git_sha_short": "n/a", - "utc_ts": "2025-10-05T17:22:36.920481+00:00", - "config_hash": "5dbfe76285c659439bcb8a7628dd646ba56079d47a9a958b3f08d2a6343b9ef8", - "params_hash": "1f7c145b1cafbcf3477f92e1e2d41dcfd9bfc2e06d8ea59ab74da2e1b29b1d82", - "rng_seed_used": 1337, - "scores_path": "/Users/secretservice/afi-benchkit/out/scores.json", - "scores_hash": "7372c882873aeaa82753dcf0b8fdc5e4a6e9fb2a5d1db55a8e60500e835a3560", - "benchkit_stamp": { - "cfg_hash": null, - "data_hash": "18e9c234f2ee33be2e4a4bc4680ab1dbbc68bde23e25fe4eed6a8bfe68384307", - "git_sha": "9eb3a14ffb6c1eb8cc6c669424ec0fd6d27631f9", - "utc_ts": "2025-10-04T16:34:39Z" - }, - "bench_merit": { - "reputation": 0.5455885659232473, - "poi": 0.5516229932966229, - "poinsight": 0.5430023827632293 - }, - "budget_path": "/Users/secretservice/afi-emissions/out_emit/epoch_budget.json", - "budget_hash": "2017534d57f93a545ba2f37e9c3315a8cf13fda398b35df6b1ee4db262b30b26", - "budget_stamp": { - "version": "0.1.0", - "git_sha": "n/a", - "utc_ts": "2025-10-05T02:57:34.624683+00:00", - "config_hash": "c13c4a8ecffe04d71f5d71f1d36ea214d395b96c35a740b48c90273595efdf34" - }, - "budget_epoch": 25 - }, - "summary": { - "total_epochs": 64, - "total_breaker_events": 0, - "final_budget": 1000000.0, - "avg_pool_utilization": 1.0 - }, - "config_hash": -1364694410363294296 -} \ No newline at end of file diff --git a/params/gauge_v0.yaml b/params/gauge_v0.yaml index cf18ab1..383670f 100644 --- a/params/gauge_v0.yaml +++ b/params/gauge_v0.yaml @@ -12,7 +12,9 @@ caps: policy_merit_blend: 0.5 # 50/50 policy vs. merit -# BenchKit merit weights for role mapping +# Merit weights for role mapping (applied to merit scores -- synthetic research inputs; +# the protocol source is the CAL-GOV analyst calibration record, not a scalar, and any +# merit conversion is CHAIN-GOV reserved) bench_merit_weights: producers: poinsight: 0.7 diff --git a/scripts/end_to_end_audit.py b/scripts/end_to_end_audit.py index ee36de9..e0e8964 100644 --- a/scripts/end_to_end_audit.py +++ b/scripts/end_to_end_audit.py @@ -2,7 +2,7 @@ """End-to-end AFI Economics audit for whitepaper readiness. This script performs a comprehensive audit of the AFI Economics system by: -1. Locating or generating required inputs (emissions budget, BenchKit scores) +1. Locating or generating required inputs (emissions budget, synthetic merit scores) 2. Running baseline and with-scores economic simulations 3. Performing 7 critical checks for mathematical and economic invariants 4. Generating machine-readable and human-readable audit reports @@ -141,53 +141,20 @@ def ensure_emissions_budget() -> Tuple[Path, str]: return budget_path, "synthetic" -def ensure_benchkit_scores() -> Tuple[Optional[Path], str]: - """Ensure BenchKit scores exist, creating if necessary.""" - # Try to find existing scores - benchkit_repo = find_sibling_repo('afi-benchkit') - if benchkit_repo: - scores_path = benchkit_repo / 'out' / 'scores.json' - if scores_path.exists(): - return scores_path, "existing" - - # Try to generate scores - poi_success, _ = run_command([ - 'afi-bench', 'run', - '--suite', 'poi', - '--dataset', 'bench/poi/sample.csv', - '--seed', '1337', - '--outdir', 'out', - '--emit-scores' - ], cwd=benchkit_repo) - - poinsight_success, _ = run_command([ - 'afi-bench', 'run', - '--suite', 'poinsight', - '--dataset', 'bench/poinsight/sample.csv', - '--seed', '1337', - '--outdir', 'out', - '--emit-scores' - ], cwd=benchkit_repo) - - if poi_success and poinsight_success and scores_path.exists(): - return scores_path, "generated" - - # Fallback: create synthetic scores with visible influence - temp_dir = Path(tempfile.mkdtemp()) - scores_path = temp_dir / 'scores_min.json' - - synthetic_scores = { - "poi": {"score": 0.62}, - "poinsight": {"score": 0.68}, - "reputation": {"score": 0.65}, - "n": 42, - "n_detail": {"poi": 21, "poinsight": 21} - } - - with scores_path.open('w') as f: - json.dump(synthetic_scores, f, indent=2) - - return scores_path, "synthetic" +def ensure_merit_scores() -> Tuple[Optional[Path], str]: + """Return the repository's synthetic merit-scores file. + + The merit path is research-plane only. Its input is the documented synthetic + file examples/merit_scores.synthetic.json; the protocol source of analyst merit + is the CAL-GOV analyst calibration record -- not a scalar -- and any merit + conversion is CHAIN-GOV reserved. Nothing is generated by shelling out to a + sibling repository. + """ + repo_root = Path(__file__).resolve().parents[1] + scores_path = repo_root / 'examples' / 'merit_scores.synthetic.json' + if scores_path.exists(): + return scores_path, "synthetic" + return None, "none" def run_econ_simulations(budget_path: Path, scores_path: Optional[Path]) -> Tuple[Path, Path]: @@ -422,8 +389,8 @@ def check_determinism() -> Tuple[bool, str]: return False, "error" -def check_benchkit_influence(base_dir: Path, with_dir: Path, scores_source: str) -> Tuple[bool, Dict[str, float], str]: - """Check BenchKit influence on allocations.""" +def check_merit_influence(base_dir: Path, with_dir: Path, scores_source: str) -> Tuple[bool, Dict[str, float], str]: + """Check the influence of the synthetic merit scores on allocations.""" base_gauge_path = base_dir / 'econ_gauge_series.csv' with_gauge_path = with_dir / 'econ_gauge_series.csv' @@ -456,10 +423,6 @@ def check_benchkit_influence(base_dir: Path, with_dir: Path, scores_source: str) # With synthetic scores, we expect visible influence has_influence = any(abs(delta) > 1e-6 for delta in deltas.values()) return has_influence, deltas, "synthetic_scores" - elif scores_source in ["existing", "generated"]: - # With real scores, any influence is valid - has_influence = any(abs(delta) > 1e-6 for delta in deltas.values()) - return has_influence, deltas, "real_scores" else: # No scores - mark as N/A return True, deltas, "N/A" @@ -475,7 +438,7 @@ def generate_audit_report(checks: Dict[str, Any], paths: Dict[str, str], deltas: "payouts_conservation_ok": checks["payouts_conservation"], "stability_ok": checks["stability"], "determinism_ok": checks["determinism"], - "benchkit_influence_ok": checks["benchkit_influence"], + "merit_influence_ok": checks["merit_influence"], "hashes": { "emissions_series_sha256": emissions_hash }, @@ -518,7 +481,7 @@ def status_symbol(value): | Payouts Conservation | {payouts} | Pool = Payouts + Holdbacks conservation | | Stability | {stability} | Rate limiting within bounds (≤0.15) | | Determinism | {determinism} | Reproducible outputs with identical inputs | -| BenchKit Influence | {benchkit} | Measurable impact of merit scores on allocations | +| Merit Influence (synthetic) | {merit} | Measurable impact of the synthetic merit scores on allocations | """.format( provenance=status_symbol(checks["provenance"]), @@ -527,11 +490,11 @@ def status_symbol(value): payouts=status_symbol(checks["payouts_conservation"]), stability=status_symbol(checks["stability"]), determinism=status_symbol(checks["determinism"]), - benchkit=status_symbol(checks["benchkit_influence"]) + merit=status_symbol(checks["merit_influence"]) ) if deltas: - report += "## BenchKit Influence Deltas\n\n" + report += "## Merit Influence Deltas (synthetic scores)\n\n" for role, delta in deltas.items(): report += f"- **{role}**: {delta:+.6f}\n" report += "\n" @@ -581,7 +544,7 @@ def main(): # Ensure inputs exist print("📋 Preparing inputs...") budget_path, budget_source = ensure_emissions_budget() - scores_path, scores_source = ensure_benchkit_scores() + scores_path, scores_source = ensure_merit_scores() print(f" Budget: {budget_source} ({budget_path})") print(f" Scores: {scores_source} ({scores_path if scores_path else 'none'})") @@ -603,11 +566,11 @@ def main(): determinism_ok, emissions_hash = check_determinism() checks["determinism"] = determinism_ok - influence_ok, deltas, influence_status = check_benchkit_influence(base_dir, with_dir, scores_source) + influence_ok, deltas, influence_status = check_merit_influence(base_dir, with_dir, scores_source) if influence_status == "N/A": - checks["benchkit_influence"] = "N/A" + checks["merit_influence"] = "N/A" else: - checks["benchkit_influence"] = influence_ok + checks["merit_influence"] = influence_ok # Convert numpy types to native Python types early def convert_value(v): diff --git a/src/afi_econ_kit.egg-info/PKG-INFO b/src/afi_econ_kit.egg-info/PKG-INFO deleted file mode 100644 index 7661ece..0000000 --- a/src/afi_econ_kit.egg-info/PKG-INFO +++ /dev/null @@ -1,381 +0,0 @@ -Metadata-Version: 2.4 -Name: afi-econ-kit -Version: 0.1.0 -Summary: AFI economic model reproducibility toolkit -Author: AFI -Requires-Python: >=3.10 -Description-Content-Type: text/markdown -License-File: LICENSE -Requires-Dist: pandas>=2.0 -Requires-Dist: numpy>=1.24 -Requires-Dist: matplotlib>=3.7 -Requires-Dist: pyyaml>=6.0 -Requires-Dist: click>=8.0 -Provides-Extra: dev -Requires-Dist: pytest>=7.4; extra == "dev" -Dynamic: license-file - -# AFI Econ Kit - -**AFI Economic System Implementation** - A comprehensive toolkit for simulating and analyzing the AFI Protocol's economic mechanisms. - -## Overview - -The AFI Econ Kit provides a complete implementation of the AFI Protocol's economic system, including: - -- **Allocation Gauge System**: Policy-based and merit-based allocation mechanisms -- **Safety & Smoothing**: Temporal smoothing and safety caps for stable payouts -- **Budget Integration**: Consumption of AFI emissions budget data from afi-emissions -- **AFI Index Computation**: Multi-component index (AN_t, EG_t, Cov_t) for network health -- **Anti-Gaming Mechanisms**: Light anti-gaming helpers (Sybil detection, wash trading) -- **Monte Carlo Simulation**: Statistical analysis with deterministic reproducibility -- **BenchKit Integration**: Merit-based allocation using benchmark reputation scores -- **Stage-by-Stage Processing**: Individual CLI commands for gauge, safety, payouts stages -- **Comprehensive Testing**: Golden image tests for plot reproducibility -- **Docker Capsule**: Multi-stage builds with STRICT=1 (hash verification) and STRICT=0 (dev) modes - -## End-to-End Whitepaper Pipeline - -The AFI Econ Kit is part of a complete whitepaper-ready pipeline: - -``` -afi-emissions → afi-benchkit → afi-econ-kit - ↓ ↓ ↓ - epoch_budget scores.json final_payouts - ↓ ↓ ↓ - E_t=132 merit_scores afi_index.json -``` - -### Pipeline Flow - -1. **AFI Emissions** (`afi-emissions`): Computes epoch pool E_t using baseline + AIM -2. **AFI BenchKit** (`afi-benchkit`): Generates merit scores from benchmark performance -3. **AFI Econ Kit** (`afi-econ-kit`): Combines policy + merit → gauge → safety → payouts → index - -### Integration Example - -```bash -# Step 1: Generate epoch budget (afi-emissions) -afi-emissions emit --config params/emissions_v0.yaml --epoch 208 --out budget_out -# → budget_out/epoch_budget.json - -# Step 2: Generate merit scores (afi-benchkit) -afi-benchkit reputation --config config.yaml --out scores_out -# → scores_out/scores.json - -# Step 3: Run economic simulation (afi-econ-kit) -afi-econ-kit simulate --config config.yaml --outdir econ_out \ - --budget budget_out/epoch_budget.json --scores scores_out/scores.json -# → econ_out/econ_summary.json, econ_out/gauge_shares.png, etc. - -# Step 4: Compute AFI Index -afi-econ-kit index --receipts examples/receipts.json \ - --payouts econ_out/final_payouts.json --epoch 208 --outdir index_out -# → index_out/afi_index.json, index_out/afi_index.png -``` - -## Quickstart -```bash -python3 -m venv .venv -source .venv/bin/activate -pip install -U pip -pip install -e .[dev] -``` - -### Original Commands -With the provided `config.yaml` (which points to `./data/*.csv`) you can run: - -```bash -afi-econ-kit check --config config.yaml -afi-econ-kit plot --config config.yaml -afi-econ-kit plot-index --config config.yaml -afi-econ-kit plot-scenarios --config config.yaml -afi-econ-kit summarize-scenarios --config config.yaml -``` - -### Core Economic Pipeline Commands - -```bash -# Full Monte Carlo economic simulation -afi-econ-kit simulate --config config.yaml --outdir out \ - --budget epoch_budget.json --scores scores.json - -# Single epoch replay with provided data -afi-econ-kit replay --config config.yaml --receipts receipts.csv \ - --credits credits.csv --prev-state state.json --outdir out - -# AFI Index computation -afi-econ-kit index --receipts receipts.json --payouts payouts.json \ - --epoch 208 --outdir index_out -``` - -### Stage-by-Stage Processing Commands - -```bash -# Run individual pipeline stages -afi-econ-kit gauge --receipts receipts.json --scores scores.json \ - --config params/gauge_v0.yaml --outdir gauge_out - -afi-econ-kit safety --allocations gauge_out/gauge_allocations.json \ - --prev-allocations prev_allocations.json --outdir safety_out - -afi-econ-kit payouts --allocations safety_out/safety_allocations.json \ - --budget epoch_budget.json --outdir payouts_out -``` - -## What each command does - -### Original Commands -- `check` — validates that required files exist, resolves column mappings, and runs monotonic supply / AIM sanity checks. -- `plot` — generates timeseries figures in `./figures`: - - `emissions_by_epoch.png` - - `cumulative_supply.png` - - `cumulative_fraction.png` (only if `cum_frac` column present) - - `baseline_vs_actual_cum.png` (only if both baseline and minted columns exist) -- `plot-index` — emits `index_components.png` and `index_series.png` when the index CSV is available (computes a composite if `I_t` is absent). -- `plot-scenarios` — builds `scenario_distributions.png` and `scenario_totals.png` from the summary statistics in `scenarios_mc30k.csv`. -- `summarize-scenarios` — prints and saves (`./figures/scenarios_summary.md`) a Markdown table with count/mean/std/min/p10/p50/p90/max for every available scenario metric. - -### Core Economic Pipeline Commands - -- **`simulate`** — Full Monte Carlo economic simulation producing: - - `econ_summary.json` - Complete economic summary with provenance - - `gauge_shares.png` - Allocation gauge visualization - - `safety_smoothing.png` - Safety mechanism visualization - - `payout_distribution.png` - Final payout distribution - - `monte_carlo_stats.json` - Statistical analysis results - -- **`replay`** — Single epoch replay with provided input data producing: - - `replay_summary.json` - Epoch replay results - - `replay_allocations.json` - Computed allocations - - Same visualization outputs as simulate - -- **`index`** — AFI Index computation producing: - - `afi_index.json` - Index components and final I_t value - - `afi_index.png` - Index visualization with components - -### Stage-by-Stage Commands - -- **`gauge`** — Allocation gauge stage producing: - - `gauge_allocations.json` - Policy + merit blended allocations - -- **`safety`** — Safety smoothing stage producing: - - `safety_allocations.json` - Smoothed and capped allocations - -- **`payouts`** — Final payouts stage producing: - - `final_payouts.json` - Final payout amounts per participant - -### Anti-Gaming & Analysis - -The toolkit includes light anti-gaming mechanisms (disabled by default): -- **Sybil Detection**: Behavioral similarity clustering -- **Wash Trading Detection**: Round-trip transaction pattern analysis -- **Penalty Application**: Configurable penalty rates for flagged behavior - -### Docker Capsule Usage - -```bash -# Development mode (STRICT=0) -docker run --rm -v $(pwd):/work ghcr.io/afi-protocol/afi-econ-kit:stable \ - simulate --config config.yaml --outdir docker_out - -# Production mode with hash verification (STRICT=1) -docker run --rm -v $(pwd):/work ghcr.io/afi-protocol/afi-econ-kit:stable-strict \ - simulate --config config.yaml --outdir docker_out -``` - - `econ_pool_series.csv` - budget pool evolution per epoch (includes convenience columns) - - `econ_gauge_series.csv` - allocation gauge shares per epoch - - `econ_payouts_summary.csv` - payout summaries per epoch - - `econ_summary.json` - overall simulation summary with reproducibility stamp - - `econ_budget.png` - budget evolution plot - - `econ_gauge_shares.png` - gauge share evolution plot - - `econ_breakers.png` - circuit breaker events plot - -### econ_pool_series.csv: convenience columns - -We include two helper columns in `econ_pool_series.csv`: - -- `emission`: effective per-epoch increments (includes multipliers when present). -- `cum_supply`: a guardrailed cumulative (monotone non-decreasing) derived from `emission`. - -Quick sanity check: - - python - <<'PY' - import numpy as np, pandas as pd - ts = pd.read_csv("out/econ_pool_series.csv") - incr = pd.to_numeric(ts["emission"], errors="coerce").fillna(0.0).clip(lower=0.0) - cum = incr.cumsum().to_numpy() - print("violations:", (np.diff(cum) < -1e-9).sum()) - PY -- `replay` — deterministic single-epoch replay producing: - - `state_next.json` - next epoch state after safety controls - - `payouts_epoch.csv` - detailed payouts for the epoch - -Each figure contains a small footer of the form `afi-econ-kit | git: cfg: | ` so the provenance of every plot is explicit. - -## Configuration Overview - -The `config.yaml` file now contains both original timeseries configuration and new economic pipeline parameters: - -### Original Configuration -- `timeseries_csv`, `scenarios_csv`, `index_csv` - data file paths -- `columns` - logical-to-physical column mappings (tolerant to synonyms like `aim`, `AIM_t`, `mult_aim`) - -### Economic Pipeline Configuration -- `budget` - baseline budget and AIM (Adaptive Issuance Mechanism) settings -- `gauge` - allocation weights, caps, and policy/merit blending parameters -- `safety` - rate limits, smoothing factors, and circuit breaker thresholds -- `payouts` - maturity holdback fractions -- `monte_carlo` - simulation runs and random seed for reproducibility - -Versioned gauge parameters are stored in `params/gauge_v0.yaml` for parameter management. - -## Development Lanes - -### Prebuilt Container (Recommended) -```bash -# Use exact dependency pins for reproducible builds -pip install -r requirements.txt -``` - -### Development Lane -```bash -pip install -e .[dev] -``` - -### Strict Lane -```bash -# For CI/production with exact versions -pip install -r requirements.txt -pytest -q -``` - -## From Benchmarks to Budgets - -The complete end-to-end workflow integrates BenchKit reputation scores into the economic pipeline: - -### Step 1: Generate BenchKit Scores -```bash -# (From afi-benchkit repository) -cd ../afi-benchkit -afi-benchkit run --config benchkit_config.yaml --outdir out -# This produces out/scores.json with PoI, PoInsight, and Reputation scores -``` - -### Step 2: Run Economic Simulation with BenchKit Integration -```bash -# (In afi-econ-kit repository) -afi-econ-kit simulate --config config.yaml --outdir out --scores ../afi-benchkit/out/scores.json -``` - -This command will: -- Echo the absolute path to the scores file -- Print a summary: `Using BenchKit scores: rep=0.546, poi=0.552, poinsight=0.543 (n=21)` -- Apply BenchKit merit multipliers to gauge allocation: - - **Producers**: 70% PoInsight + 30% Reputation (benefits from high insight scores) - - **Validators**: 100% Reputation (pure reputation-based allocation) - - **Enrichment**: 60% PoInsight + 40% Reputation (mixed insight/reputation) - - **Public Goods**: Neutral (unaffected by BenchKit scores) - -### Step 3: Inspect Results -```bash -# View gauge share evolution with BenchKit influence -open out/econ_gauge_shares.png - -# Check comprehensive provenance in summary -cat out/econ_summary.json | jq '.stamp' -``` - -The `econ_summary.json` now includes comprehensive provenance: -- `version`: Package version from metadata -- `git_sha_short`: Git commit (first 7 chars) -- `utc_ts`: ISO8601 UTC timestamp -- `config_hash`: SHA256 of config.yaml -- `params_hash`: SHA256 of params/gauge_v0.yaml -- `scores_path`: Absolute path to BenchKit scores file -- `scores_hash`: SHA256 of BenchKit scores file (64-hex string) -- `benchkit_stamp`: Pass-through BenchKit stamp object (or null) -- `bench_merit`: The actual scores used (reputation, poi, poinsight) - -Example provenance with BenchKit integration: -```json -{ - "scores_path": "/abs/path/to/scores.json", - "scores_hash": "abcdef1234567890abcdef1234567890abcdef1234567890abcdef1234567890", - "benchkit_stamp": { - "timestamp": "2024-01-01T12:00:00Z", - "version": "1.0.0", - "config_hash": "..." - } -} -``` - -### Verifying Configuration Integrity -```bash -# Verify config hash matches current file -python -c " -import hashlib, json -with open('out/econ_summary.json') as f: stamp = json.load(f)['stamp'] -with open('config.yaml', 'rb') as f: current_hash = hashlib.sha256(f.read()).hexdigest() -print('Config hash matches:', stamp['config_hash'] == current_hash) -" -``` - -## Reproduce the Paper - -Generate the economic assets used in the whitepaper: - -```bash -make paper-assets -``` - -This creates `examples/whitepaper/econ/` with: -- `econ_budget.png` - budget evolution figure -- `econ_gauge_shares.png` - allocation gauge shares -- `econ_summary.json` - simulation summary -- `METADATA.txt` - reproducibility metadata - -## Testing -Comprehensive test suite including golden image tests: -```bash -pytest -q # Run all tests (should complete in <5s) -AFI_ECON_GOLDEN_TEST=1 pytest # Run with deterministic plotting -AFI_ECON_REGEN_GOLDEN=1 pytest # Regenerate golden image hashes -``` - -### Golden Test Policy -The golden image tests enforce byte-stable plot reproducibility with different behavior for local vs CI: - -**Local Development** (forgiving): -```bash -make golden # Prints digest, assertion skipped -``` - -**CI / Strict Local** (enforcing): -```bash -make golden-ci # Enforces hash assertion -``` - -### Regenerating Golden Tests -To regenerate golden hashes for plot reproducibility: -```bash -# Regenerate budget plot golden hash -AFI_ECON_REGEN_GOLDEN=1 pytest -q tests/test_budget_golden.py -s - -# Regenerate gauge shares plot golden hash -AFI_ECON_KIT_REGEN_GOLDEN=1 pytest -q tests/test_gauge_shares_golden.py -s -``` -Copy the printed hash into the respective test file's `GOLDEN_HASH` constant. - -## Makefile Targets -- `make simulate` - run economic simulation with default config -- `make sim` - run simulation with BenchKit integration (if available) -- `make replay` - run single epoch replay (requires input files) -- `make golden` - run golden image tests (local mode - prints digest, assertion skipped) -- `make golden-ci` - run golden image tests (CI enforcement mode - strict assertions) -- `make paper-assets` - generate whitepaper economic assets -- `make clean` - remove output files -- `make spotless` - remove all generated files including virtual env - -## License -MIT License. See `LICENSE` for details. diff --git a/src/afi_econ_kit.egg-info/SOURCES.txt b/src/afi_econ_kit.egg-info/SOURCES.txt deleted file mode 100644 index 13d4814..0000000 --- a/src/afi_econ_kit.egg-info/SOURCES.txt +++ /dev/null @@ -1,41 +0,0 @@ -LICENSE -README.md -pyproject.toml -src/afi_econ_kit/__init__.py -src/afi_econ_kit/anti_gaming.py -src/afi_econ_kit/cli.py -src/afi_econ_kit/config.py -src/afi_econ_kit/data_io.py -src/afi_econ_kit/emissions.py -src/afi_econ_kit/gauge.py -src/afi_econ_kit/index.py -src/afi_econ_kit/payouts.py -src/afi_econ_kit/plots.py -src/afi_econ_kit/safety.py -src/afi_econ_kit/scenarios.py -src/afi_econ_kit/schemas.py -src/afi_econ_kit.egg-info/PKG-INFO -src/afi_econ_kit.egg-info/SOURCES.txt -src/afi_econ_kit.egg-info/dependency_links.txt -src/afi_econ_kit.egg-info/entry_points.txt -src/afi_econ_kit.egg-info/requires.txt -src/afi_econ_kit.egg-info/top_level.txt -tests/test_anti_gaming.py -tests/test_anti_gaming_redundancy.py -tests/test_budget_golden.py -tests/test_budget_ingest.py -tests/test_cli_plot.py -tests/test_cli_simulate_smoke.py -tests/test_cli_smoke.py -tests/test_cum_supply_guard.py -tests/test_cum_supply_monotone.py -tests/test_gauge_caps.py -tests/test_gauge_shares_golden.py -tests/test_index_golden.py -tests/test_load_timeseries.py -tests/test_pool_series_schema.py -tests/test_replay_determinism.py -tests/test_safety_helpers.py -tests/test_safety_ops.py -tests/test_scores_ingest_default.py -tests/test_scores_ingest_with_file.py \ No newline at end of file diff --git a/src/afi_econ_kit.egg-info/dependency_links.txt b/src/afi_econ_kit.egg-info/dependency_links.txt deleted file mode 100644 index 8b13789..0000000 --- a/src/afi_econ_kit.egg-info/dependency_links.txt +++ /dev/null @@ -1 +0,0 @@ - diff --git a/src/afi_econ_kit.egg-info/entry_points.txt b/src/afi_econ_kit.egg-info/entry_points.txt deleted file mode 100644 index ce76fc5..0000000 --- a/src/afi_econ_kit.egg-info/entry_points.txt +++ /dev/null @@ -1,2 +0,0 @@ -[console_scripts] -afi-econ-kit = afi_econ_kit.cli:main diff --git a/src/afi_econ_kit.egg-info/requires.txt b/src/afi_econ_kit.egg-info/requires.txt deleted file mode 100644 index eebfaaf..0000000 --- a/src/afi_econ_kit.egg-info/requires.txt +++ /dev/null @@ -1,8 +0,0 @@ -pandas>=2.0 -numpy>=1.24 -matplotlib>=3.7 -pyyaml>=6.0 -click>=8.0 - -[dev] -pytest>=7.4 diff --git a/src/afi_econ_kit.egg-info/top_level.txt b/src/afi_econ_kit.egg-info/top_level.txt deleted file mode 100644 index 4c12c1a..0000000 --- a/src/afi_econ_kit.egg-info/top_level.txt +++ /dev/null @@ -1 +0,0 @@ -afi_econ_kit diff --git a/src/afi_econ_kit/cli.py b/src/afi_econ_kit/cli.py index 7dbc556..4914343 100644 --- a/src/afi_econ_kit/cli.py +++ b/src/afi_econ_kit/cli.py @@ -265,8 +265,20 @@ def _load_econ_config(args: argparse.Namespace) -> Dict[str, Any]: return config -def _load_benchkit_scores(scores_path: Optional[str]) -> Optional[Dict[str, Any]]: - """Load and validate BenchKit scores JSON file. +MERIT_SCORES_HELP = ( + "Merit scores JSON file (optional; synthetic -- see examples/merit_scores.synthetic.json). " + "The protocol source is the CAL-GOV analyst calibration record -- not a scalar -- " + "and any merit conversion is CHAIN-GOV reserved." +) + + +def _load_merit_scores(scores_path: Optional[str]) -> Optional[Dict[str, Any]]: + """Load and validate a merit-scores JSON file. + + The scores are SYNTHETIC research inputs (see examples/merit_scores.synthetic.json). + The protocol source of analyst merit is the CAL-GOV analyst calibration record -- + not a scalar -- and any conversion of it into a merit value is CHAIN-GOV reserved; + afi-econ consumes no such value from the protocol. Args: scores_path: Path to scores JSON file, or None @@ -327,8 +339,8 @@ def _load_benchkit_scores(scores_path: Optional[str]) -> Optional[Dict[str, Any] from afi_econ_kit.scenarios import compute_file_sha256 scores_hash = compute_file_sha256(scores_file) - # Extract BenchKit stamp if present (pass-through) - benchkit_stamp = data.get("stamp", None) + # Extract the scores file's own stamp if present (pass-through) + merit_stamp = data.get("stamp", None) return { "scores": bench_merit, @@ -336,7 +348,7 @@ def _load_benchkit_scores(scores_path: Optional[str]) -> Optional[Dict[str, Any] "warnings": warnings, "path": str(scores_file.resolve()), "hash": scores_hash, - "benchkit_stamp": benchkit_stamp + "merit_stamp": merit_stamp } @@ -391,8 +403,8 @@ def cmd_econ_simulate(args: argparse.Namespace) -> int: config = _load_econ_config(args) output_dir = _ensure_output_dir(args.outdir) - # Load BenchKit scores if provided - bench_data = _load_benchkit_scores(getattr(args, 'scores', None)) + # Load merit scores (synthetic) if provided + bench_data = _load_merit_scores(getattr(args, 'scores', None)) if bench_data: # Echo absolute path to scores file @@ -401,7 +413,7 @@ def cmd_econ_simulate(args: argparse.Namespace) -> int: # Print summary line with hash scores = bench_data['scores'] hash_short = bench_data['hash'][:8] if bench_data['hash'] != "none" else "none" - print(f"Using BenchKit scores: rep={scores['reputation']:.3f}, " + print(f"Using merit scores (synthetic): rep={scores['reputation']:.3f}, " f"poi={scores['poi']:.3f}, poinsight={scores['poinsight']:.3f} " f"(n={bench_data['n']}) hash={hash_short}") @@ -447,8 +459,8 @@ def cmd_econ_replay(args: argparse.Namespace) -> int: config = _load_econ_config(args) output_dir = _ensure_output_dir(args.outdir) - # Load BenchKit scores if provided - bench_data = _load_benchkit_scores(getattr(args, 'scores', None)) + # Load merit scores (synthetic) if provided + bench_data = _load_merit_scores(getattr(args, 'scores', None)) if bench_data: # Echo absolute path to scores file @@ -457,7 +469,7 @@ def cmd_econ_replay(args: argparse.Namespace) -> int: # Print summary line with hash scores = bench_data['scores'] hash_short = bench_data['hash'][:8] if bench_data['hash'] != "none" else "none" - print(f"Using BenchKit scores: rep={scores['reputation']:.3f}, " + print(f"Using merit scores (synthetic): rep={scores['reputation']:.3f}, " f"poi={scores['poi']:.3f}, poinsight={scores['poinsight']:.3f} " f"(n={bench_data['n']}) hash={hash_short}") @@ -668,7 +680,7 @@ def cmd_econ_gauge(args: argparse.Namespace) -> int: # Load scores if provided bench_data = None if hasattr(args, 'scores') and args.scores: - bench_data = _load_benchkit_scores(args.scores) + bench_data = _load_merit_scores(args.scores) # Load config config_path = Path(args.config) @@ -682,7 +694,7 @@ def cmd_econ_gauge(args: argparse.Namespace) -> int: print(f"Running gauge allocation...") print(f"Receipts: {len(receipts_data)} records") if bench_data and bench_data['scores']: - print(f"BenchKit scores: {len(bench_data['scores'])} participants") + print(f"Merit scores (synthetic): {len(bench_data['scores'])} components") # Compute gauge shares gauge_results = compute_gauge_shares(receipts_data, config, bench_data) @@ -873,7 +885,7 @@ def build_parser() -> argparse.ArgumentParser: parser_simulate = subparsers.add_parser("simulate", help="Run economic Monte Carlo simulation") parser_simulate.add_argument("--config", type=Path, default=Path("config.yaml")) parser_simulate.add_argument("--outdir", type=str, required=True, help="Output directory") - parser_simulate.add_argument("--scores", type=str, help="BenchKit scores JSON file (optional)") + parser_simulate.add_argument("--scores", type=str, help=MERIT_SCORES_HELP) parser_simulate.add_argument("--budget", type=str, help="AFI emissions budget JSON file (optional)") parser_simulate.set_defaults(func=cmd_econ_simulate) @@ -883,7 +895,7 @@ def build_parser() -> argparse.ArgumentParser: parser_replay.add_argument("--credits", type=str, required=True, help="Credits CSV file") parser_replay.add_argument("--prev-state", type=str, required=True, help="Previous state JSON file") parser_replay.add_argument("--outdir", type=str, required=True, help="Output directory") - parser_replay.add_argument("--scores", type=str, help="BenchKit scores JSON file (optional)") + parser_replay.add_argument("--scores", type=str, help=MERIT_SCORES_HELP) parser_replay.add_argument("--budget", type=str, help="AFI emissions budget JSON file (optional)") parser_replay.set_defaults(func=cmd_econ_replay) @@ -898,7 +910,7 @@ def build_parser() -> argparse.ArgumentParser: parser_gauge = subparsers.add_parser("gauge", help="Run gauge allocation stage") parser_gauge.add_argument("--receipts", type=str, required=True, help="Receipts JSON file") parser_gauge.add_argument("--config", type=str, default="params/gauge_v0.yaml", help="Gauge config file") - parser_gauge.add_argument("--scores", type=str, help="BenchKit scores JSON file (optional)") + parser_gauge.add_argument("--scores", type=str, help=MERIT_SCORES_HELP) parser_gauge.add_argument("--outdir", type=str, required=True, help="Output directory") parser_gauge.set_defaults(func=cmd_econ_gauge) diff --git a/src/afi_econ_kit/gauge.py b/src/afi_econ_kit/gauge.py index ee91c6a..f6c2766 100644 --- a/src/afi_econ_kit/gauge.py +++ b/src/afi_econ_kit/gauge.py @@ -73,11 +73,11 @@ def _compute_bench_merit_multipliers( bench_merit: Dict[str, float], bench_merit_weights: Dict[str, Any] ) -> Dict[str, float]: - """Compute per-role merit multipliers from BenchKit scores. + """Compute per-role merit multipliers from merit scores (synthetic inputs). Args: roles: List of role names - bench_merit: BenchKit scores (reputation, poi, poinsight) + bench_merit: Merit scores (reputation, poi, poinsight) -- synthetic research inputs bench_merit_weights: Role mapping configuration Returns: @@ -132,8 +132,8 @@ def allocation_gauge( caps: Caps configuration including per_role_max merit: Merit inputs for blending blend: Blend factor between policy (0.0) and merit (1.0) - bench_merit: BenchKit scores (reputation, poi, poinsight) - bench_merit_weights: Role mapping configuration for BenchKit scores + bench_merit: Merit scores (reputation, poi, poinsight) -- synthetic research inputs + bench_merit_weights: Role mapping configuration for merit scores Returns: Dictionary containing: @@ -156,7 +156,7 @@ def allocation_gauge( merit_weight = weights[role] * (1.0 + merit_avg) raw_allocation[role] = (1 - blend) * weights[role] + blend * merit_weight - # Apply BenchKit merit multipliers if provided + # Apply merit multipliers if provided if bench_merit and bench_merit_weights: # Set default bench_merit_weights if not provided if bench_merit_weights is None: diff --git a/src/afi_econ_kit/scenarios.py b/src/afi_econ_kit/scenarios.py index b9e0a24..8c4939d 100644 --- a/src/afi_econ_kit/scenarios.py +++ b/src/afi_econ_kit/scenarios.py @@ -80,7 +80,7 @@ def create_stamp(seed: int, data_hash: str, config_path: str = "config.yaml", data_hash: Hash of input data config_path: Path to config file for hashing params_path: Path to params file for hashing - bench_data: Optional BenchKit data + bench_data: Optional merit-scores data (synthetic) budget_data: Optional AFI emissions budget data Returns: @@ -102,7 +102,7 @@ def create_stamp(seed: int, data_hash: str, config_path: str = "config.yaml", "rng_seed_used": seed, "scores_path": bench_data['path'] if bench_data else None, "scores_hash": bench_data['hash'] if bench_data else None, - "benchkit_stamp": bench_data['benchkit_stamp'] if bench_data else None, + "merit_stamp": bench_data['merit_stamp'] if bench_data else None, "bench_merit": bench_data['scores'] if bench_data else None, "budget_path": budget_data['path'] if budget_data and budget_data['budget'] else None, "budget_hash": budget_data['hash'] if budget_data and budget_data['budget'] else None, @@ -193,7 +193,7 @@ def run_epoch_simulation( config: Full configuration prev_state: Previous epoch state rng: Random number generator - bench_data: Optional BenchKit scores data + bench_data: Optional merit-scores data (synthetic) Returns: Tuple of (new_state, epoch_results) @@ -278,7 +278,7 @@ def run_monte_carlo_simulation(config: Dict[str, Any], bench_data: Optional[Dict Args: config: Full configuration dictionary - bench_data: Optional BenchKit scores data + bench_data: Optional merit-scores data (synthetic) budget_data: Optional AFI emissions budget data Returns: diff --git a/src/afi_econ_kit/schemas.py b/src/afi_econ_kit/schemas.py index dfc8487..7fc7062 100644 --- a/src/afi_econ_kit/schemas.py +++ b/src/afi_econ_kit/schemas.py @@ -159,7 +159,7 @@ class GaugeInput(StageInput): """Schema for gauge stage input.""" receipts: List[ReceiptData] = Field(..., description="Receipt data") policy_allocation: Dict[str, float] = Field(..., description="Policy-based allocation") - merit_scores: Optional[Dict[str, float]] = Field(None, description="Merit scores from BenchKit") + merit_scores: Optional[Dict[str, float]] = Field(None, description="Merit scores (synthetic research inputs; the protocol source is the CAL-GOV analyst calibration record, not a scalar)") class GaugeOutput(StageOutput): diff --git a/tests/fixtures/scores_min.json b/tests/fixtures/scores_min.json deleted file mode 100644 index 0c37e52..0000000 --- a/tests/fixtures/scores_min.json +++ /dev/null @@ -1,7 +0,0 @@ -{ - "poi": {"score": 0.55}, - "poinsight": {"score": 0.60}, - "reputation": {"score": 0.58}, - "n": 20, - "n_detail": {"poi": 10, "poinsight": 10} -} diff --git a/tests/test_scores_ingest_with_file.py b/tests/test_scores_ingest_with_file.py index 4fb0e2d..8b61c65 100644 --- a/tests/test_scores_ingest_with_file.py +++ b/tests/test_scores_ingest_with_file.py @@ -1,4 +1,4 @@ -"""Test scores ingestion with BenchKit scores file.""" +"""Test scores ingestion with the synthetic merit-scores file (examples/merit_scores.synthetic.json).""" import json import subprocess @@ -12,8 +12,8 @@ def test_scores_ingestion_with_file(): """Test simulation with --scores file produces expected changes.""" # Load the test scores - scores_path = Path("tests/fixtures/scores_min.json") - assert scores_path.exists(), "Test scores fixture not found" + scores_path = Path("examples/merit_scores.synthetic.json") + assert scores_path.exists(), "Synthetic merit-scores example not found" with scores_path.open("r") as f: scores_data = json.load(f) @@ -30,7 +30,7 @@ def test_scores_ingestion_with_file(): "warnings": [], "path": str(scores_path.resolve()), "hash": compute_file_sha256(scores_path), - "benchkit_stamp": scores_data.get("stamp", None) + "merit_stamp": scores_data.get("stamp", None) } # Minimal config for fast test @@ -73,12 +73,13 @@ def test_scores_ingestion_with_file(): differences_found = True break - assert differences_found, "BenchKit scores should affect gauge allocation" + assert differences_found, "Merit scores should affect gauge allocation" # Check stamp includes bench_merit data stamp = result_with_scores["stamp"] assert stamp["scores_path"] == str(scores_path.resolve()) assert stamp["bench_merit"] is not None + assert stamp["merit_stamp"]["source"] == "synthetic" assert stamp["bench_merit"]["reputation"] == 0.58 assert stamp["bench_merit"]["poi"] == 0.55 assert stamp["bench_merit"]["poinsight"] == 0.60 @@ -99,7 +100,7 @@ def test_cli_simulate_with_scores(): """Test CLI simulate command with --scores parameter.""" with tempfile.TemporaryDirectory() as tmpdir: tmpdir_path = Path(tmpdir) - scores_path = Path("tests/fixtures/scores_min.json") + scores_path = Path("examples/merit_scores.synthetic.json") # Run CLI command cmd = [ @@ -116,7 +117,7 @@ def test_cli_simulate_with_scores(): # Check output contains expected messages assert str(scores_path.resolve()) in result.stdout, "Should echo absolute scores path" - assert "Using BenchKit scores:" in result.stdout, "Should print scores summary" + assert "Using merit scores (synthetic):" in result.stdout, "Should print scores summary" assert "rep=0.580" in result.stdout, "Should show reputation score" assert "poi=0.550" in result.stdout, "Should show poi score" assert "poinsight=0.600" in result.stdout, "Should show poinsight score"