From d332a65274b6e7b7a3b690c4aead584920dfc7e5 Mon Sep 17 00:00:00 2001 From: Carlos Navarro <68403497+0xnavarro@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:45:11 -0600 Subject: [PATCH] feat: add lab result submission contract --- FOR_LABS.md | 63 +++++++++++++++++ README.md | 2 + schemas/result-submission-v1.schema.json | 86 ++++++++++++++++++++++++ templates/result-submission-v1.json | 43 ++++++++++++ tests/test_validate_submission.py | 18 +++++ tools/validate_submission.py | 73 ++++++++++++++++++++ 6 files changed, 285 insertions(+) create mode 100644 FOR_LABS.md create mode 100644 schemas/result-submission-v1.schema.json create mode 100644 templates/result-submission-v1.json create mode 100644 tests/test_validate_submission.py create mode 100755 tools/validate_submission.py diff --git a/FOR_LABS.md b/FOR_LABS.md new file mode 100644 index 0000000..432d11c --- /dev/null +++ b/FOR_LABS.md @@ -0,0 +1,63 @@ +# ReflexBench for model labs and evaluation platforms + +ReflexBench is designed to be a low-friction public evaluation target for System One models, typed decision engines and compatible hosted APIs. + +## What a lab gets + +- a frozen, versioned benchmark contract instead of a moving leaderboard; +- machine-readable task/result formats; +- probability-aware metrics, not only top-1 accuracy; +- completion/failure accounting; +- multilingual, option-order and cardinality evidence; +- explicit hosted-vs-local deployment boundaries; +- an optional Reflex harness ablation that is kept separate from raw model quality; +- public receipts that can be independently inspected and cited. + +## Minimal integration contract + +1. Pin the benchmark tag/version. +2. Map Binary/Noul, Choice and Score to your engine without changing task semantics. +3. Run the frozen lane and retain failures/retries. +4. Publish the raw receipt. +5. Fill `templates/result-submission-v1.json`. +6. Validate it locally: + +```bash +python3 tools/validate_submission.py path/to/result-submission.json +``` + +7. Open a result submission issue/PR. + +The JSON Schema is published at `schemas/result-submission-v1.schema.json` for CI systems that already use a standards-based validator. + +## Recommended public result identity + +A canonical result should be addressable as: + +```text +benchmark version + lane + corpus SHA ++ engine owner/name/revision ++ adapter revision ++ deployment boundary ++ raw receipt(s) +``` + +Do not submit an ambiguous API alias as model identity. + +## Raw model vs model + harness + +ReflexBench treats these as separate experimental claims: + +- **Raw engine lane:** the engine is evaluated directly under the frozen task contract. +- **Reflex harness lane:** the same engine is evaluated with an explicitly declared deterministic/ensemble policy. +- **Same-response policy lane:** where possible, the model response itself is frozen and only downstream policy changes. + +An aggregator may display these next to one another, but should not silently relabel a model+harness result as raw model quality. + +## Benchmark contamination + +If ReflexBench influenced model training, prompt development, model selection, calibration or thresholds, set `benchmark_used_for_development: true` and describe it. The result can still be useful, but it is development evidence rather than pristine held-out evaluation. + +## Partnership / bulk evaluation + +Evaluation platforms and labs can contribute adapters, bulk result snapshots or co-published analysis using exactly the same public evidence contract. Public submissions should remain reproducible without privileged Brida infrastructure. diff --git a/README.md b/README.md index c1041f1..bb1c536 100644 --- a/README.md +++ b/README.md @@ -54,6 +54,8 @@ Maintaining a System One or typed-decision model? Run the frozen protocol and su Accepted public results may use the **Evaluated on ReflexBench v1** badge; acceptance records reproducible evidence and is not an endorsement or universal ranking. +For model labs, evaluation platforms and bulk integrations, see **[FOR_LABS.md](FOR_LABS.md)**. ReflexBench also publishes a versioned [result-submission JSON Schema](schemas/result-submission-v1.schema.json), [template](templates/result-submission-v1.json) and dependency-free validator so benchmark evidence can be produced directly from CI. + ## Quick start Requires Python 3.11+ for the core harness. Core v1 uses only the standard library. diff --git a/schemas/result-submission-v1.schema.json b/schemas/result-submission-v1.schema.json new file mode 100644 index 0000000..83c4533 --- /dev/null +++ b/schemas/result-submission-v1.schema.json @@ -0,0 +1,86 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/brida-ai/reflexbench/schemas/result-submission-v1.schema.json", + "title": "ReflexBench v1 result submission envelope", + "type": "object", + "additionalProperties": false, + "required": ["schema", "benchmark_version", "lane", "engine", "adapter", "corpus", "deployment", "run", "metrics", "evidence"], + "properties": { + "schema": {"const": "reflexbench.result-submission/v1"}, + "benchmark_version": {"const": "1.0.0"}, + "lane": {"type": "string", "minLength": 1}, + "engine": { + "type": "object", + "additionalProperties": false, + "required": ["owner", "name", "revision"], + "properties": { + "owner": {"type": "string", "minLength": 1}, + "name": {"type": "string", "minLength": 1}, + "revision": {"type": "string", "minLength": 1}, + "url": {"type": "string"} + } + }, + "adapter": { + "type": "object", + "additionalProperties": false, + "required": ["revision"], + "properties": { + "repository": {"type": "string"}, + "revision": {"type": "string", "minLength": 1} + } + }, + "corpus": { + "type": "object", + "additionalProperties": false, + "required": ["name", "sha256", "rows"], + "properties": { + "name": {"type": "string", "minLength": 1}, + "sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "rows": {"type": "integer", "minimum": 1} + } + }, + "deployment": { + "type": "object", + "additionalProperties": false, + "required": ["kind"], + "properties": { + "kind": {"enum": ["hosted-api", "local-cpu", "local-gpu", "other"]}, + "hardware": {"type": "string"}, + "provider": {"type": "string"}, + "region": {"type": "string"} + } + }, + "run": { + "type": "object", + "additionalProperties": false, + "required": ["date", "command", "retry_policy"], + "properties": { + "date": {"type": "string", "format": "date"}, + "command": {"type": "string", "minLength": 1}, + "retry_policy": {"type": "string", "minLength": 1} + } + }, + "metrics": { + "type": "object", + "additionalProperties": true, + "required": ["completion_rate", "semantic_accuracy"], + "properties": { + "completion_rate": {"type": "number", "minimum": 0, "maximum": 1}, + "semantic_accuracy": {"type": "number", "minimum": 0, "maximum": 1}, + "operational_accuracy": {"type": "number", "minimum": 0, "maximum": 1}, + "ece": {"type": ["number", "null"], "minimum": 0}, + "p50_latency_ms": {"type": ["number", "null"], "minimum": 0} + } + }, + "evidence": { + "type": "object", + "additionalProperties": false, + "required": ["receipt_paths", "benchmark_used_for_development"], + "properties": { + "receipt_paths": {"type": "array", "minItems": 1, "items": {"type": "string", "minLength": 1}}, + "benchmark_used_for_development": {"type": "boolean"}, + "development_note": {"type": "string"} + } + } + } +} diff --git a/templates/result-submission-v1.json b/templates/result-submission-v1.json new file mode 100644 index 0000000..937ca10 --- /dev/null +++ b/templates/result-submission-v1.json @@ -0,0 +1,43 @@ +{ + "schema": "reflexbench.result-submission/v1", + "benchmark_version": "1.0.0", + "lane": "public-hard", + "engine": { + "owner": "example-lab", + "name": "example-system-one-model", + "revision": "immutable-model-or-checkpoint-revision", + "url": "https://example.com/model" + }, + "adapter": { + "repository": "https://github.com/example-lab/reflexbench-adapter", + "revision": "immutable-adapter-revision" + }, + "corpus": { + "name": "reflexbench-v1-public-hard", + "sha256": "0000000000000000000000000000000000000000000000000000000000000000", + "rows": 111 + }, + "deployment": { + "kind": "local-gpu", + "hardware": "GPU model and relevant runtime", + "provider": "", + "region": "" + }, + "run": { + "date": "2026-09-24", + "command": "exact command used to produce the attached receipt", + "retry_policy": "none" + }, + "metrics": { + "completion_rate": 1.0, + "semantic_accuracy": 0.0, + "operational_accuracy": 0.0, + "ece": null, + "p50_latency_ms": null + }, + "evidence": { + "receipt_paths": ["results/v1/example-result.json"], + "benchmark_used_for_development": false, + "development_note": "" + } +} diff --git a/tests/test_validate_submission.py b/tests/test_validate_submission.py new file mode 100644 index 0000000..0684272 --- /dev/null +++ b/tests/test_validate_submission.py @@ -0,0 +1,18 @@ +import copy, importlib.util, json, pathlib, unittest +ROOT=pathlib.Path(__file__).resolve().parents[1] +spec=importlib.util.spec_from_file_location("validate_submission",ROOT/"tools/validate_submission.py") +mod=importlib.util.module_from_spec(spec); assert spec.loader; spec.loader.exec_module(mod) + +class SubmissionEnvelopeTest(unittest.TestCase): + def setUp(self): self.valid=json.loads((ROOT/"templates/result-submission-v1.json").read_text()) + def test_template_is_valid(self): self.assertEqual(mod.validate(self.valid),[]) + def test_rejects_bad_sha(self): + d=copy.deepcopy(self.valid); d["corpus"]["sha256"]="abc"; self.assertTrue(any("sha256" in x for x in mod.validate(d))) + def test_rejects_out_of_range_accuracy(self): + d=copy.deepcopy(self.valid); d["metrics"]["semantic_accuracy"]=1.1; self.assertTrue(any("semantic_accuracy" in x for x in mod.validate(d))) + def test_rejects_missing_receipts(self): + d=copy.deepcopy(self.valid); d["evidence"]["receipt_paths"]=[]; self.assertTrue(any("receipt_paths" in x for x in mod.validate(d))) + def test_requires_development_disclosure(self): + d=copy.deepcopy(self.valid); del d["evidence"]["benchmark_used_for_development"]; self.assertTrue(any("benchmark_used" in x for x in mod.validate(d))) + +if __name__ == '__main__': unittest.main() diff --git a/tools/validate_submission.py b/tools/validate_submission.py new file mode 100755 index 0000000..a94388b --- /dev/null +++ b/tools/validate_submission.py @@ -0,0 +1,73 @@ +#!/usr/bin/env python3 +"""Validate the stable ReflexBench v1 result-submission envelope without third-party deps.""" +from __future__ import annotations +import argparse, datetime as dt, json, re, sys +from pathlib import Path + +SCHEMA = "reflexbench.result-submission/v1" +REQUIRED_TOP = {"schema","benchmark_version","lane","engine","adapter","corpus","deployment","run","metrics","evidence"} +HEX64 = re.compile(r"^[0-9a-f]{64}$") + +def fail(errors:list[str], message:str) -> None: errors.append(message) +def is_nonempty(v:object) -> bool: return isinstance(v,str) and bool(v.strip()) +def rate(errors:list[str], obj:dict, key:str, *, required:bool=False) -> None: + if key not in obj: + if required: fail(errors,f"metrics.{key} is required") + return + v=obj[key] + if v is None and not required: return + if not isinstance(v,(int,float)) or isinstance(v,bool) or not 0 <= float(v) <= 1: + fail(errors,f"metrics.{key} must be a number in [0,1]") + +def validate(d:object) -> list[str]: + e:list[str]=[] + if not isinstance(d,dict): return ["submission must be a JSON object"] + missing=sorted(REQUIRED_TOP-set(d)); + if missing: fail(e,"missing top-level keys: "+", ".join(missing)) + if d.get("schema") != SCHEMA: fail(e,f"schema must be {SCHEMA!r}") + if d.get("benchmark_version") != "1.0.0": fail(e,"benchmark_version must be '1.0.0'") + if not is_nonempty(d.get("lane")): fail(e,"lane must be a non-empty string") + for section, keys in {"engine":("owner","name","revision"),"adapter":("revision",),"corpus":("name","sha256","rows"),"deployment":("kind",),"run":("date","command","retry_policy")}.items(): + obj=d.get(section) + if not isinstance(obj,dict): fail(e,f"{section} must be an object"); continue + for k in keys: + if k not in obj: fail(e,f"{section}.{k} is required") + engine=d.get("engine",{}) if isinstance(d.get("engine"),dict) else {} + for k in ("owner","name","revision"): + if k in engine and not is_nonempty(engine[k]): fail(e,f"engine.{k} must be non-empty") + corpus=d.get("corpus",{}) if isinstance(d.get("corpus"),dict) else {} + if "sha256" in corpus and (not isinstance(corpus["sha256"],str) or not HEX64.fullmatch(corpus["sha256"])): fail(e,"corpus.sha256 must be 64 lowercase hex chars") + if "rows" in corpus and (not isinstance(corpus["rows"],int) or isinstance(corpus["rows"],bool) or corpus["rows"] < 1): fail(e,"corpus.rows must be a positive integer") + deployment=d.get("deployment",{}) if isinstance(d.get("deployment"),dict) else {} + if deployment.get("kind") not in {"hosted-api","local-cpu","local-gpu","other"}: fail(e,"deployment.kind is invalid") + run=d.get("run",{}) if isinstance(d.get("run"),dict) else {} + if "date" in run: + try: dt.date.fromisoformat(run["date"]) + except Exception: fail(e,"run.date must be YYYY-MM-DD") + for k in ("command","retry_policy"): + if k in run and not is_nonempty(run[k]): fail(e,f"run.{k} must be non-empty") + metrics=d.get("metrics") + if not isinstance(metrics,dict): fail(e,"metrics must be an object") + else: + rate(e,metrics,"completion_rate",required=True); rate(e,metrics,"semantic_accuracy",required=True); rate(e,metrics,"operational_accuracy"); + for k in ("ece","p50_latency_ms"): + if k in metrics and metrics[k] is not None and (not isinstance(metrics[k],(int,float)) or isinstance(metrics[k],bool) or metrics[k] < 0): fail(e,f"metrics.{k} must be null or non-negative") + evidence=d.get("evidence") + if not isinstance(evidence,dict): fail(e,"evidence must be an object") + else: + paths=evidence.get("receipt_paths") + if not isinstance(paths,list) or not paths or any(not is_nonempty(x) for x in paths): fail(e,"evidence.receipt_paths must be a non-empty list of paths") + if not isinstance(evidence.get("benchmark_used_for_development"),bool): fail(e,"evidence.benchmark_used_for_development must be boolean") + return e + +def main() -> int: + ap=argparse.ArgumentParser(); ap.add_argument("submission",type=Path); args=ap.parse_args() + try: data=json.loads(args.submission.read_text()) + except Exception as exc: print(f"invalid JSON: {exc}",file=sys.stderr); return 2 + errors=validate(data) + if errors: + for x in errors: print("ERROR:",x,file=sys.stderr) + return 1 + print(f"ReflexBench submission envelope: PASS ({args.submission})") + return 0 +if __name__ == "__main__": raise SystemExit(main())