diff --git a/results/comparison.json b/results/comparison.json index abfee23..ec83fcc 100644 --- a/results/comparison.json +++ b/results/comparison.json @@ -5,14 +5,14 @@ "answered": 430, "declined": 11, "hedged_but_answered": 91, - "unscoreable_units": 110, - "scoreable": 315, - "within_10pct": 45.7, - "within_50pct": 77.5, - "confidently_wrong": 22.5, + "unscoreable_units": 98, + "scoreable": 327, + "within_10pct": 46.2, + "within_50pct": 78.0, + "confidently_wrong": 22.0, "cited_a_source": 71.7, "citation_correct_of_cited": 89.9, - "right_source_wrong_number": 54.1 + "right_source_wrong_number": 53.7 }, "n_answers": 467 }, @@ -56,9 +56,9 @@ "answered": 326, "declined": 115, "hedged_but_answered": 19, - "unscoreable_units": 65, - "scoreable": 256, - "within_10pct": 58.2, + "unscoreable_units": 64, + "scoreable": 257, + "within_10pct": 58.4, "within_50pct": 85.2, "confidently_wrong": 14.8, "cited_a_source": 53.7, @@ -73,11 +73,11 @@ "answered": 144, "declined": 313, "hedged_but_answered": 34, - "unscoreable_units": 75, - "scoreable": 66, - "within_10pct": 62.1, - "within_50pct": 86.4, - "confidently_wrong": 13.6, + "unscoreable_units": 74, + "scoreable": 67, + "within_10pct": 62.7, + "within_50pct": 86.6, + "confidently_wrong": 13.4, "cited_a_source": 21.6, "citation_correct_of_cited": 82.2, "right_source_wrong_number": 31.6 diff --git a/results/paired.json b/results/paired.json index f696192..bd87122 100644 --- a/results/paired.json +++ b/results/paired.json @@ -5,10 +5,10 @@ "avg_tool_calls": 1.62, "without": { "answered": 86, - "scoreable": 62, - "within10": 37.1, + "scoreable": 64, + "within10": 35.9, "correct_of_all": 25.6, - "wrong50": 25.8 + "wrong50": 25.0 }, "with": { "answered": 89, diff --git a/score.py b/score.py index c2b458b..a0fc2f2 100644 --- a/score.py +++ b/score.py @@ -16,7 +16,66 @@ reported separately, never counted as a wrong answer. """ import json, re, unicodedata -from units import reconcile +from units import CURRENCY, reconcile + +# Tokens recognised immediately before a number. CURRENCY already has +# usd/us$/$/eur/gbp/sek; € and £ are listed there in spirit but stripped by +# _clean, so they are canonicalised to EUR/GBP before they reach reconcile. +# Extra ISO/local codes are recognised so "NT$" is not stolen by "$" and so a +# foreign-only figure stays attached to its own code — never given an FX rate. +_EXTRA_CURRENCY = ("€", "£", "nt$", "nok", "aud", "huf", "zar", "jpy", "twd", "mxn") +_SYMBOL_CANON = {"€": "EUR", "£": "GBP"} +_CURRENCY_FAMILY = { + "usd": "usd", "us$": "usd", "$": "usd", + "eur": "eur", "€": "eur", + "gbp": "gbp", "£": "gbp", + "sek": "sek", "nok": "nok", + "nt$": "twd", "twd": "twd", + "aud": "aud", "huf": "huf", "zar": "zar", "jpy": "jpy", "mxn": "mxn", +} + + +def _currency_token_alts(): + toks = sorted(set(CURRENCY) | set(_EXTRA_CURRENCY), key=len, reverse=True) + parts = [] + for tok in toks: + esc = re.escape(tok) + parts.append(rf"(?50pct-off before #23" } } diff --git a/verify/test_currency_prefix.py b/verify/test_currency_prefix.py new file mode 100644 index 0000000..a0c6bed --- /dev/null +++ b/verify/test_currency_prefix.py @@ -0,0 +1,81 @@ +#!/usr/bin/env python3 +"""Issue #23: a currency written immediately before the number must reach reconcile. + + python3 verify/test_currency_prefix.py +""" +import os, sys + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, ROOT) + +import score +import units + + +def _unit_for(text, value): + hits = [(v, u) for v, u in score.candidates(text) if v == value] + assert hits, f"no candidate {value} in {score.candidates(text)!r} from {text!r}" + return hits[0][1] + + +def test_usd_prefix_reaches_reconcile(): + text = "Best estimate: about USD 5 /tCO2e." + cands = score.candidates(text) + assert cands, text + unit = _unit_for(text, 5.0) + assert "usd" in unit.lower(), unit + got, note = units.reconcile(5, unit, "USD/tCO2e") + assert got == 5.0 and note is None, (got, note) + assert units.reconcile(5, "/tCO2e", "USD/tCO2e")[0] is None + + +def test_other_usd_spellings(): + for text in ("about US$ 5 /tCO2e.", "about $5 /tCO2e."): + unit = _unit_for(text, 5.0) + got, _ = units.reconcile(5, unit, "USD/tCO2e") + assert got == 5.0, (text, unit, got) + + +def test_foreign_only_is_unscoreable(): + cases = ( + "About NT$300 per tCO2e.", + "About NOK 1,000 per tCO2e in 2025.", + "About EUR 65 /tCO2e.", + "About SEK 1,450 per tCO2.", + ) + for text in cases: + unit = score.candidates(text)[0][1] + # Currency is carried so "$" cannot steal "NT$", but there is no FX rate. + got, _, _ = score.best_value(text, "USD/tCO2e") + assert got is None, (text, unit, got) + + +def test_mixed_sentence_scores_the_usd_figure(): + text = ( + "About EUR 65–75 per tCO2e in 2024–2025, i.e. roughly USD 76 /tCO2e." + ) + got, unit, _ = score.best_value(text, "USD/tCO2e") + assert got == 76.0, (got, unit) + assert "usd" in unit.lower(), unit + + +def test_euro_symbol_against_eur_truth(): + text = "Approximately €4.5 per tCO2e average EUA price in 2013." + unit = _unit_for(text, 4.5) + got, _ = units.reconcile(4.5, unit, "EUR/tCO2e") + assert got == 4.5, (unit, got) + + +def test_nt_dollar_is_not_usd(): + text = "About NT$300 per tCO2e." + unit = _unit_for(text, 300.0) + assert "nt$" in unit.lower(), unit + assert units.reconcile(300, unit, "USD/tCO2e")[0] is None + + +if __name__ == "__main__": + for name, fn in list(globals().items()): + if name.startswith("test_") and callable(fn): + fn() + print(f"ok {name}") + print("all currency-prefix tests passed")