From 49ab41b271a2d7c5c0b5b0ef58953c46c51fe5ed Mon Sep 17 00:00:00 2001 From: Carlos Navarro <68403497+0xnavarro@users.noreply.github.com> Date: Wed, 23 Sep 2026 20:08:15 -0600 Subject: [PATCH] feat: add machine-readable benchmark discovery surfaces --- README.md | 8 ++ assets/evaluated-on-reflexbench-v1.svg | 8 ++ benchmark.json | 29 +++++ docs/INTEGRATORS.md | 37 +++++++ leaderboards/v1.json | 143 +++++++++++++++++++++++++ tests/test_machine_surfaces.py | 36 +++++++ 6 files changed, 261 insertions(+) create mode 100644 assets/evaluated-on-reflexbench-v1.svg create mode 100644 benchmark.json create mode 100644 docs/INTEGRATORS.md create mode 100644 leaderboards/v1.json create mode 100644 tests/test_machine_surfaces.py diff --git a/README.md b/README.md index 76aa9ba..ce50934 100644 --- a/README.md +++ b/README.md @@ -73,6 +73,14 @@ Accepted public results may use the **Evaluated on ReflexBench v1** badge; accep For model labs, evaluation platforms and bulk integrations, see **[FOR_LABS.md](FOR_LABS.md)**. ReflexBench also publishes a versioned [result-submission JSON Schema](schemas/result-submission-v1.schema.json), [template](templates/result-submission-v1.json) and dependency-free validator so benchmark evidence can be produced directly from CI. +For benchmark aggregators and automated catalogs, [`benchmark.json`](benchmark.json) and [`leaderboards/v1.json`](leaderboards/v1.json) provide stable machine-readable discovery. See [`docs/INTEGRATORS.md`](docs/INTEGRATORS.md). + +Use the repository-hosted badge after an accepted public result: + +```markdown +[![Evaluated on ReflexBench v1](https://raw.githubusercontent.com/brida-ai/reflexbench/main/assets/evaluated-on-reflexbench-v1.svg)](https://github.com/brida-ai/reflexbench) +``` + ## Quick start Requires Python 3.11+ for the core harness. Core v1 uses only the standard library. diff --git a/assets/evaluated-on-reflexbench-v1.svg b/assets/evaluated-on-reflexbench-v1.svg new file mode 100644 index 0000000..4e97fe4 --- /dev/null +++ b/assets/evaluated-on-reflexbench-v1.svg @@ -0,0 +1,8 @@ + + + + + + BRIDA + Evaluated on ReflexBench v1 + diff --git a/benchmark.json b/benchmark.json new file mode 100644 index 0000000..441af07 --- /dev/null +++ b/benchmark.json @@ -0,0 +1,29 @@ +{ + "badge": "assets/evaluated-on-reflexbench-v1.svg", + "canonical_repository": "https://github.com/brida-ai/reflexbench", + "category": [ + "system-one-models", + "typed-decision-models", + "probabilistic-decision-engines" + ], + "citation": "CITATION.cff", + "contact": "https://www.brida.ai/", + "homepage": "https://www.brida.ai/blog/reflexbench-v1-system-one-models", + "id": "brida/reflexbench", + "lab_integration_guide": "FOR_LABS.md", + "license": "Apache-2.0", + "name": "ReflexBench", + "primary_leaderboard": "leaderboards/v1.json", + "primary_metric": "semantic_accuracy", + "raw_vs_harness_policy": "FOR_LABS.md#raw-model-vs-model--harness", + "schema": "reflexbench.benchmark-discovery/v1", + "status": "stable", + "submission_guide": "SUBMIT_A_MODEL.md", + "submission_schema": "schemas/result-submission-v1.schema.json", + "task_types": [ + "binary/noul", + "choice", + "score" + ], + "version": "1.0.0" +} diff --git a/docs/INTEGRATORS.md b/docs/INTEGRATORS.md new file mode 100644 index 0000000..58f2e39 --- /dev/null +++ b/docs/INTEGRATORS.md @@ -0,0 +1,37 @@ +# Integrating ReflexBench into a benchmark catalog + +ReflexBench exposes stable machine-readable discovery and leaderboard surfaces so model labs, benchmark aggregators and evaluation platforms do not need to scrape prose. + +## Discovery + +- `benchmark.json` — benchmark identity, category, version and canonical integration links. +- `leaderboards/v1.json` — canonical v1 raw leaderboard plus explicitly separate supplemental harness/policy lanes. +- `schemas/result-submission-v1.schema.json` — third-party result envelope. +- `CITATION.cff` — citation metadata. + +## Display rules + +ReflexBench intentionally distinguishes three result types: + +1. **raw engine** — canonical same-corpus model/engine quality; +2. **model + Reflex harness** — explicit harness/ensemble ablation; +3. **same-response policy** — downstream deterministic policy over an unchanged model response. + +Do not relabel lanes 2 or 3 as raw model quality. Do not combine hosted/local latency into a hardware-normalized ranking unless you rerun under a normalized environment. + +## Recommended catalog fields + +At minimum ingest: + +- benchmark version and lane; +- corpus row count / hash where supplied; +- engine owner/name/revision; +- semantic accuracy; +- completion rate; +- deployment boundary; +- receipt path/source; +- benchmark-used-for-development disclosure. + +## Labs and bulk results + +See [`FOR_LABS.md`](../FOR_LABS.md) and [`SUBMIT_A_MODEL.md`](../SUBMIT_A_MODEL.md). Bulk submissions should use the same public result schema and remain reproducible without privileged Brida infrastructure. diff --git a/leaderboards/v1.json b/leaderboards/v1.json new file mode 100644 index 0000000..fbdc892 --- /dev/null +++ b/leaderboards/v1.json @@ -0,0 +1,143 @@ +{ + "benchmark_id": "brida/reflexbench", + "benchmark_version": "1.0.0", + "generated_from_public_receipts": true, + "notes": [ + "No global aggregate score.", + "Hosted and local latency are not hardware-normalized against each other.", + "Raw engine quality and model+harness results are distinct claims." + ], + "primary_lane": { + "higher_is_better": true, + "id": "public-hard-raw", + "metric": "semantic_accuracy", + "purpose": "same-corpus raw System One engine quality", + "results": [ + { + "deployment_boundary": "hosted-api", + "display_name": "TypeSafe Jev", + "engine_id": "typesafe-jev", + "rank": 1, + "receipt": "results/v1/jevbench-public-hard-typesafe-jev-consolidated.json", + "semantic_accuracy": 0.73 + }, + { + "deployment_boundary": "local-gpu", + "display_name": "upstream Reflex / Qwen3.5-2B", + "engine_id": "upstream-reflex-qwen35-2b", + "rank": 2, + "receipt": "results/v1/jevbench-public-hard-qwen-reflex-2b-rtx3070.json", + "semantic_accuracy": 0.414 + }, + { + "deployment_boundary": "reference-kernel", + "display_name": "frozen Qwen3.5-0.8B readout control", + "engine_id": "qwen35-0.8b-readout-control", + "rank": 3, + "receipt": "results/v1/jevbench-public-hard-qwen35-08b-base-readout-rtx3070-reference-kernel.json", + "semantic_accuracy": 0.396 + }, + { + "deployment_boundary": "local-gpu", + "display_name": "jeff / GLiFormer ~400M", + "engine_id": "jeff-gliformer", + "rank": 4, + "receipt": "results/v1/jevbench-public-hard-jeff-rtx3070.json", + "semantic_accuracy": 0.378 + }, + { + "deployment_boundary": "local-gpu", + "display_name": "openJev Verdict 1.4 / 151M", + "engine_id": "openjev-verdict-1.4", + "rank": 5, + "receipt": "results/v1/jevbench-public-hard-verdict-1.4-rtx3070.json", + "semantic_accuracy": 0.369 + }, + { + "deployment_boundary": "local-gpu", + "display_name": "Laya base / 421M", + "engine_id": "laya-base", + "rank": 6, + "receipt": "results/v1/jevbench-public-hard-laya-base-rtx3070.json", + "semantic_accuracy": 0.351 + }, + { + "deployment_boundary": "local-gpu", + "display_name": "Kev-0.8B", + "engine_id": "kev-0.8b", + "rank": 7, + "receipt": "results/v1/jevbench-public-hard-kev-0.8b-rtx3070.json", + "semantic_accuracy": 0.324 + } + ], + "rows": 111 + }, + "schema": "reflexbench.leaderboard/v1", + "supplemental_lanes": { + "public110_reflex_harness_ablation": { + "max_delta_pp": 8.1818181818, + "median_delta_pp": 2.7272727273, + "purpose": "raw engine vs best measured Reflex harness mode; separate from primary leaderboard", + "receipt": "results/v1/reflex-public110-harness-uplift.json", + "results": [ + { + "best_mode": "policy-only + reverse-average", + "best_reflex_semantic_accuracy": 0.7636363636363637, + "ci95_high_pp": 14.5454545455, + "ci95_low_pp": 1.8181818182, + "delta_pp": 8.1818181818, + "engine_id": "qwen-reflex-2b", + "raw_semantic_accuracy": 0.6818181818181818 + }, + { + "best_mode": "policy-only + reverse-average", + "best_reflex_semantic_accuracy": 0.5818181818181818, + "ci95_high_pp": 9.0909090909, + "ci95_low_pp": -1.8181818182, + "delta_pp": 3.6363636364, + "engine_id": "jeff", + "raw_semantic_accuracy": 0.5454545454545454 + }, + { + "best_mode": "all questions + reverse-average", + "best_reflex_semantic_accuracy": 0.6545454545454545, + "ci95_high_pp": 6.3636363636, + "ci95_low_pp": 0.0, + "delta_pp": 2.7272727273, + "engine_id": "laya-typed", + "raw_semantic_accuracy": 0.6272727272727273 + }, + { + "best_mode": "policy-only; no ensemble", + "best_reflex_semantic_accuracy": 0.7, + "ci95_high_pp": 0.0, + "ci95_low_pp": 0.0, + "delta_pp": 0.0, + "engine_id": "kev-0.8b", + "raw_semantic_accuracy": 0.7 + }, + { + "best_mode": "policy-only; no ensemble", + "best_reflex_semantic_accuracy": 0.5818181818181818, + "ci95_high_pp": 0.0, + "ci95_low_pp": 0.0, + "delta_pp": 0.0, + "engine_id": "laya-base", + "raw_semantic_accuracy": 0.5818181818181818 + } + ], + "rows": 110, + "simple_mean_delta_pp": 2.9090909091 + }, + "typesafe_jev_same_response_policy": { + "extra_model_calls": 0, + "harms": 0, + "note": "Operational policy result; not 100% raw semantic accuracy.", + "raw_semantic_accuracy": 0.9636363636363636, + "receipt": "results/v1/reflex-core-v1-typesafe-jev-public110.json", + "reflex_operational_accuracy": 1.0, + "rescues": 4, + "rows": 110 + } + } +} diff --git a/tests/test_machine_surfaces.py b/tests/test_machine_surfaces.py new file mode 100644 index 0000000..b14aad2 --- /dev/null +++ b/tests/test_machine_surfaces.py @@ -0,0 +1,36 @@ +import json +import unittest +from pathlib import Path + +ROOT=Path(__file__).resolve().parents[1] + +class MachineSurfacesTest(unittest.TestCase): + def test_discovery_targets_exist(self): + d=json.loads((ROOT/'benchmark.json').read_text()) + self.assertEqual(d['id'],'brida/reflexbench') + self.assertEqual(d['version'],'1.0.0') + for key in ['primary_leaderboard','submission_guide','submission_schema','lab_integration_guide','citation','badge']: + self.assertTrue((ROOT/d[key]).exists(), key) + + def test_primary_leaderboard_matches_locked_claims(self): + d=json.loads((ROOT/'leaderboards/v1.json').read_text()) + rows=d['primary_lane']['results'] + self.assertEqual(d['primary_lane']['rows'],111) + self.assertEqual([round(x['semantic_accuracy'],3) for x in rows],[.730,.414,.396,.378,.369,.351,.324]) + self.assertEqual(rows[0]['display_name'],'TypeSafe Jev') + self.assertIn('jeff / GLiFormer', rows[3]['display_name']) + for x in rows: + self.assertTrue((ROOT/x['receipt']).exists(), x['receipt']) + + def test_harness_lane_is_explicitly_separate(self): + d=json.loads((ROOT/'leaderboards/v1.json').read_text()) + h=d['supplemental_lanes']['public110_reflex_harness_ablation'] + self.assertAlmostEqual(h['simple_mean_delta_pp'],2.9090909091,places=8) + self.assertAlmostEqual(h['max_delta_pp'],8.1818181818,places=8) + j=d['supplemental_lanes']['typesafe_jev_same_response_policy'] + self.assertAlmostEqual(j['raw_semantic_accuracy'],106/110) + self.assertEqual(j['reflex_operational_accuracy'],1.0) + self.assertEqual(j['extra_model_calls'],0) + self.assertIn('not 100% raw semantic accuracy',j['note']) + +if __name__=='__main__': unittest.main()