Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 14 additions & 14 deletions results/comparison.json
Original file line number Diff line number Diff line change
Expand Up @@ -5,14 +5,14 @@
"answered": 430,
"declined": 11,
"hedged_but_answered": 91,
"unscoreable_units": 110,
"scoreable": 315,
"within_10pct": 45.7,
"within_50pct": 77.5,
"confidently_wrong": 22.5,
"unscoreable_units": 98,
"scoreable": 327,
"within_10pct": 46.2,
"within_50pct": 78.0,
"confidently_wrong": 22.0,
"cited_a_source": 71.7,
"citation_correct_of_cited": 89.9,
"right_source_wrong_number": 54.1
"right_source_wrong_number": 53.7
},
"n_answers": 467
},
Expand Down Expand Up @@ -56,9 +56,9 @@
"answered": 326,
"declined": 115,
"hedged_but_answered": 19,
"unscoreable_units": 65,
"scoreable": 256,
"within_10pct": 58.2,
"unscoreable_units": 64,
"scoreable": 257,
"within_10pct": 58.4,
"within_50pct": 85.2,
"confidently_wrong": 14.8,
"cited_a_source": 53.7,
Expand All @@ -73,11 +73,11 @@
"answered": 144,
"declined": 313,
"hedged_but_answered": 34,
"unscoreable_units": 75,
"scoreable": 66,
"within_10pct": 62.1,
"within_50pct": 86.4,
"confidently_wrong": 13.6,
"unscoreable_units": 74,
"scoreable": 67,
"within_10pct": 62.7,
"within_50pct": 86.6,
"confidently_wrong": 13.4,
"cited_a_source": 21.6,
"citation_correct_of_cited": 82.2,
"right_source_wrong_number": 31.6
Expand Down
6 changes: 3 additions & 3 deletions results/paired.json
Original file line number Diff line number Diff line change
Expand Up @@ -5,10 +5,10 @@
"avg_tool_calls": 1.62,
"without": {
"answered": 86,
"scoreable": 62,
"within10": 37.1,
"scoreable": 64,
"within10": 35.9,
"correct_of_all": 25.6,
"wrong50": 25.8
"wrong50": 25.0
},
"with": {
"answered": 89,
Expand Down
104 changes: 93 additions & 11 deletions score.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,66 @@
reported separately, never counted as a wrong answer.
"""
import json, re, unicodedata
from units import reconcile
from units import CURRENCY, reconcile

# Tokens recognised immediately before a number. CURRENCY already has
# usd/us$/$/eur/gbp/sek; € and £ are listed there in spirit but stripped by
# _clean, so they are canonicalised to EUR/GBP before they reach reconcile.
# Extra ISO/local codes are recognised so "NT$" is not stolen by "$" and so a
# foreign-only figure stays attached to its own code — never given an FX rate.
_EXTRA_CURRENCY = ("€", "£", "nt$", "nok", "aud", "huf", "zar", "jpy", "twd", "mxn")
_SYMBOL_CANON = {"€": "EUR", "£": "GBP"}
_CURRENCY_FAMILY = {
"usd": "usd", "us$": "usd", "$": "usd",
"eur": "eur", "€": "eur",
"gbp": "gbp", "£": "gbp",
"sek": "sek", "nok": "nok",
"nt$": "twd", "twd": "twd",
"aud": "aud", "huf": "huf", "zar": "zar", "jpy": "jpy", "mxn": "mxn",
}


def _currency_token_alts():
toks = sorted(set(CURRENCY) | set(_EXTRA_CURRENCY), key=len, reverse=True)
parts = []
for tok in toks:
esc = re.escape(tok)
parts.append(rf"(?<![\w]){esc}(?![\w])" if tok.isalpha() else esc)
return parts


_CUR_ALTS = _currency_token_alts()
_CUR_BEFORE = re.compile(r"(?:" + "|".join(_CUR_ALTS) + r")\s*$", re.I)
_CUR_LEADING = re.compile(r"^(?:" + "|".join(_CUR_ALTS) + r")", re.I)


def _leading_currency_family(unit):
"""Numerator currency family, or None if the unit does not start with one."""
if not unit:
return None
m = _CUR_LEADING.match(unit.lstrip())
if not m:
return None
return _CURRENCY_FAMILY.get(m.group(0).rstrip().lower())


def _trailing_takes_currency(unit):
"""True when the captured tail is a unit (``/tCO2e``, ``per tCO2e``), not a scale word."""
raw = (unit or "").lstrip().lower()
return (not raw) or raw.startswith("/") or raw.startswith("per")


def _with_currency_prefix(text, start, unit):
"""Carry a currency written immediately before the number into `unit`."""
m = _CUR_BEFORE.search(text[:start])
if not m or not _trailing_takes_currency(unit):
return unit, None
cur = _SYMBOL_CANON.get(m.group(0).strip(), m.group(0).strip())
fam = _CURRENCY_FAMILY.get(m.group(0).strip().lower())
unit = unit or ""
if not unit:
return cur, fam
return f"{cur} {unit.lstrip()}", fam

# The unit is whatever immediately follows the number, but it must stop at the
# first separator — an em-dash, comma or bracket usually introduces the source
Expand Down Expand Up @@ -63,27 +122,50 @@ def norm(s):
return unicodedata.normalize("NFKD", (s or "")).lower()


def candidates(text):
"""[(value, trailing_unit_text)] — ranges collapse to their midpoint."""
out, seen = [], set()
for m in RANGE.finditer(text or ""):
def _iter_candidates(text):
"""Yield (value, unit, prefix_family_or_None). Ranges collapse to midpoint."""
text = text or ""
seen = set()
for m in RANGE.finditer(text):
lo, hi = float(m.group(1).replace(",", "")), float(m.group(2).replace(",", ""))
out.append(((lo + hi) / 2.0, m.group(3)))
unit, pref_fam = _with_currency_prefix(text, m.start(), m.group(3))
yield (lo + hi) / 2.0, unit, pref_fam
seen.update(range(m.start(), m.end()))
for m in SINGLE.finditer(text or ""):
for m in SINGLE.finditer(text):
if m.start() in seen:
continue
whole, exp, unit = m.group(1).replace(",", ""), m.group(2), m.group(3)
if "." not in whole and not exp and re.fullmatch(r"(19|20)\d{2}", whole):
continue # a bare year is not a value
v = float(whole) * (10 ** int(exp) if exp else 1)
out.append((v, unit))
return out
unit, pref_fam = _with_currency_prefix(text, m.start(), unit)
yield v, unit, pref_fam


def candidates(text):
"""[(value, unit_text)] — ranges collapse to their midpoint.

The unit is the text that trails the number, plus a currency token written
immediately before it. Models write ``USD 5 /tCO2e``; without the prefix
the extractor only sees ``/tCO2e``, which never reaches reconcile as
``USD/tCO2e``.
"""
return [(v, unit) for v, unit, _ in _iter_candidates(text)]


def best_value(answer, truth_unit):
"""The first candidate whose unit reconciles with the truth's unit."""
for v, unit in candidates(answer):
"""The first candidate whose unit reconciles with the truth's unit.

A currency we lifted from immediately before the number is not FX-converted
into a different currency. ``EUR 65 /tCO2e`` against ``USD/tCO2e`` stays
unscoreable; if the same sentence also states a USD figure, that is scored.
Trailing currencies already in the unit text (``85 EUR per tonne``) keep
their existing reconcile behaviour.
"""
want = _leading_currency_family(truth_unit)
for v, unit, pref_fam in _iter_candidates(answer):
if want and pref_fam and pref_fam != want:
continue
conv, note = reconcile(v, unit, truth_unit)
if conv is not None:
return conv, unit, note
Expand Down
11 changes: 10 additions & 1 deletion verify/allowlist.json
Original file line number Diff line number Diff line change
Expand Up @@ -48,6 +48,15 @@
"46.7%": "headline at the first scorer version",
"46.9%": "headline at an intermediate scorer version",
"47%": "hand adjudication of 45 pilot answers",
"35.6%": "citation rate at an earlier scorer version"
"35.6%": "citation rate at an earlier scorer version",
"45.7%": "Claude headline before the #23 currency-prefix fix; current value is results.comparison.claude-opus-5.summary.within_10pct",
"77.5%": "Claude within-50pct before #23",
"22.5%": "Claude confidently-wrong before #23",
"54.1%": "Claude right-source-wrong-number before #23",
"58.2%": "GPT-5.5 headline before #23",
"62.1%": "Grok 4.6 headline before #23",
"86.4%": "Grok 4.6 within-50pct before #23",
"37.1%": "Claude paired-unaided within-10pct before #23",
"25.8%": "Claude paired-unaided >50pct-off before #23"
}
}
81 changes: 81 additions & 0 deletions verify/test_currency_prefix.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,81 @@
#!/usr/bin/env python3
"""Issue #23: a currency written immediately before the number must reach reconcile.

python3 verify/test_currency_prefix.py
"""
import os, sys

ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, ROOT)

import score
import units


def _unit_for(text, value):
hits = [(v, u) for v, u in score.candidates(text) if v == value]
assert hits, f"no candidate {value} in {score.candidates(text)!r} from {text!r}"
return hits[0][1]


def test_usd_prefix_reaches_reconcile():
text = "Best estimate: about USD 5 /tCO2e."
cands = score.candidates(text)
assert cands, text
unit = _unit_for(text, 5.0)
assert "usd" in unit.lower(), unit
got, note = units.reconcile(5, unit, "USD/tCO2e")
assert got == 5.0 and note is None, (got, note)
assert units.reconcile(5, "/tCO2e", "USD/tCO2e")[0] is None


def test_other_usd_spellings():
for text in ("about US$ 5 /tCO2e.", "about $5 /tCO2e."):
unit = _unit_for(text, 5.0)
got, _ = units.reconcile(5, unit, "USD/tCO2e")
assert got == 5.0, (text, unit, got)


def test_foreign_only_is_unscoreable():
cases = (
"About NT$300 per tCO2e.",
"About NOK 1,000 per tCO2e in 2025.",
"About EUR 65 /tCO2e.",
"About SEK 1,450 per tCO2.",
)
for text in cases:
unit = score.candidates(text)[0][1]
# Currency is carried so "$" cannot steal "NT$", but there is no FX rate.
got, _, _ = score.best_value(text, "USD/tCO2e")
assert got is None, (text, unit, got)


def test_mixed_sentence_scores_the_usd_figure():
text = (
"About EUR 65–75 per tCO2e in 2024–2025, i.e. roughly USD 76 /tCO2e."
)
got, unit, _ = score.best_value(text, "USD/tCO2e")
assert got == 76.0, (got, unit)
assert "usd" in unit.lower(), unit


def test_euro_symbol_against_eur_truth():
text = "Approximately €4.5 per tCO2e average EUA price in 2013."
unit = _unit_for(text, 4.5)
got, _ = units.reconcile(4.5, unit, "EUR/tCO2e")
assert got == 4.5, (unit, got)


def test_nt_dollar_is_not_usd():
text = "About NT$300 per tCO2e."
unit = _unit_for(text, 300.0)
assert "nt$" in unit.lower(), unit
assert units.reconcile(300, unit, "USD/tCO2e")[0] is None


if __name__ == "__main__":
for name, fn in list(globals().items()):
if name.startswith("test_") and callable(fn):
fn()
print(f"ok {name}")
print("all currency-prefix tests passed")
Loading