-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy patheval_model.py
More file actions
700 lines (641 loc) · 31.6 KB
/
Copy patheval_model.py
File metadata and controls
700 lines (641 loc) · 31.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
#!/usr/bin/env python3
"""Drive `eval_tasks.py` with a real model and record what it costs to reach green.
`eval_repair.py` measures the compiler alone. `eval_tasks.py` defines the tasks
and the floor. This file closes the loop: it hands a model the language card and
nothing else, lets it submit programs, feeds back exactly what the compiler
says, and counts rounds.
The protocol is deliberately narrow. The model gets `LLM.md`, the task brief,
the starter, and the `main` its code will be called from. It never gets the
expected value and never gets a solution. Between rounds it gets the verdict —
diagnostics with `rule` and `fixes`, or a value mismatch — because that is all a
real user would have. A language that needs the answer in the prompt has not
taught the model anything.
python3 eval_model.py --self-test # no credentials needed
python3 eval_model.py --models claude-opus-5
python3 eval_model.py --models claude-opus-5,claude-sonnet-5,claude-haiku-4-5
python3 eval_model.py --set large --models claude-opus-5
Two sets. `core` is the original twelve, one idea each. `large` is 50+ lines of
Zerdali per task, some of it across modules, because the core set stopped
discriminating — every model finished it in about one round, and a set that
cannot fail cannot validate a change.
Results are written to measurements/eval-<model>.json so a later change can be compared
against an earlier run rather than argued about. Keep a run as a baseline with
`--label`; a partial run (`--tasks`) writes a `.partial` file so it can never
overwrite a full record.
"""
import argparse
import json
import os
import sys
import time
from eval_tasks import LARGE_TASKS, TASKS, score
TASK_SETS = {"core": TASKS, "large": LARGE_TASKS}
CARD = "LLM.md"
MAX_ROUNDS = 6
# Per-model request shape. Thinking and effort are not uniform across tiers, and
# sending an unsupported combination is a 400 rather than a graceful degrade.
#
# `cache_min` is the shortest prefix that model will cache at all. Below it the
# breakpoint is silently ignored — no error, just no hit — and the card is ~2.7k
# tokens, so on Haiku it does not cache. That is worth printing rather than
# quietly paying for.
# Üç satıcı, üç kademe. Fiyatlar liste fiyatı, milyon belirteç başına dolar, ve
# *değişiyor* -- karşılaştırmanın sert olan yarısı belirteç sayıları, para onların
# bir çarpımı. Bir fiyat eskidiyse burada düzeltilir ve rapor yeniden üretilir.
#
# `effort` üç satıcıda üç ayrı ad taşıyor ve aynı şeyi soruyor: ne kadar
# düşünsün. Anthropic'te `output_config.effort`, OpenAI'de `reasoning_effort`,
# Gemini'de `thinking_config.thinking_level`. Eşitlenmesi şart, çünkü eşitlenmemiş
# bir düşünme bütçesi ölçülen şeyi modelden alıp ayara veriyor.
MODELS = {
"claude-opus-5": {"thinking": "adaptive", "effort": "high",
"in": 5.00, "out": 25.00, "cache_min": 512},
"claude-opus-4-8": {"thinking": "adaptive", "effort": "high",
"in": 5.00, "out": 25.00, "cache_min": 1024},
"claude-sonnet-5": {"thinking": "adaptive", "effort": "high",
"in": 3.00, "out": 15.00, "cache_min": 1024},
"claude-haiku-4-5": {"thinking": None, "effort": None,
"in": 1.00, "out": 5.00, "cache_min": 4096},
# Bir önceki kuşak, aynı ayarlarla. Karşılaştırmanın ikinci ekseni: aynı
# satıcı içinde kuşaklar arası fark, satıcılar arası farkın yanında ne
# kadar büyük.
"claude-opus-4-7": {"thinking": "adaptive", "effort": "high",
"in": 5.00, "out": 25.00, "cache_min": 1024},
"claude-opus-4-6": {"thinking": "adaptive", "effort": "high",
"in": 5.00, "out": 25.00, "cache_min": 1024},
"claude-sonnet-4-6": {"thinking": "adaptive", "effort": "high",
"in": 3.00, "out": 15.00, "cache_min": 1024},
# Fiyatı bilinmiyor, ve `None` onu *uydurmamanın* yolu: belirteç sayıları
# ölçülen şey, para onların bir çarpımı, ve bilinmeyen bir çarpan sıfır
# değil -- boş.
"claude-fable-5": {"thinking": "adaptive", "effort": "high",
"in": None, "out": None, "cache_min": 1024},
"gpt-5.1": {"thinking": None, "effort": "high",
"in": 1.25, "out": 10.00, "cache_min": 1024},
"gpt-5-mini": {"thinking": None, "effort": "high",
"in": 0.25, "out": 2.00, "cache_min": 1024},
"gpt-5-nano": {"thinking": None, "effort": "high",
"in": 0.05, "out": 0.40, "cache_min": 1024},
"gemini-3.1-pro-preview": {"thinking": None, "effort": "high",
"in": 2.00, "out": 12.00, "cache_min": 1024},
"gemini-3-flash-preview": {"thinking": None, "effort": "high",
"in": 0.30, "out": 2.50, "cache_min": 1024},
"gemini-3.1-flash-lite": {"thinking": None, "effort": "low",
"in": 0.10, "out": 0.40, "cache_min": 1024},
}
SUBMIT = {
"name": "submit",
"description": (
"Submit the complete Zerdali source for this task. It is compiled and "
"run immediately. You get back either {\"green\": true} or the exact "
"compiler verdict: diagnostics carrying `rule` and `fixes`, or the "
"value the program produced. Submit as often as you need."
),
"input_schema": {
"type": "object",
"properties": {
"source": {
"type": "string",
"description": (
"The whole program, including any effect/data/label "
"declarations. Do not write `main` — it is appended."
),
},
},
"required": ["source"],
"additionalProperties": False,
},
}
# Multi-module tasks take one source per module, the same shape `zerdali serve`
# takes. A dict with model-chosen keys is awkward to state in JSON Schema, so
# the wire form is a list and the harness folds it back into {module: text}.
SUBMIT_MODULES = {
"name": "submit",
"description": SUBMIT["description"] + (
" This task spans several modules: send one entry per module. Do not "
"write `main` in any of them — it is appended to the entry module for "
"you."
),
"input_schema": {
"type": "object",
"properties": {
"modules": {
"type": "array",
"items": {
"type": "object",
"properties": {
"name": {
"type": "string",
"description": (
"Module name. The empty string is the entry "
"module, where `main` will be appended."
),
},
"source": {"type": "string"},
},
"required": ["name", "source"],
"additionalProperties": False,
},
},
},
"required": ["modules"],
"additionalProperties": False,
},
}
class BadSubmission(Exception):
"""A tool call whose arguments do not have the declared shape.
A model can send `modules` as a list of strings even though the schema says
objects, and `m.get` then raises. Letting that escape ends the whole run:
a five-repeat measurement stopped after one, and the record on disk said
`1/1 green` -- which reads as a result rather than as a crash.
A malformed submission is a *failed round*, not a failed run. The model is
told what was wrong and gets to try again, which is exactly what happens
when the compiler rejects anything else it sends.
"""
def submitted_source(task: dict, payload: dict):
"""Normalise a tool call into whatever `score` expects for this task."""
if not isinstance(payload, dict):
raise BadSubmission("the tool argument was not an object")
if "modules" in task:
mods = payload.get("modules")
if not isinstance(mods, list):
raise BadSubmission("`modules` must be a list of objects")
out = {}
for m in mods:
if not isinstance(m, dict):
raise BadSubmission(
"each entry of `modules` must be an object with `name` and "
"`source`, not a bare string")
out[m.get("name", "")] = m.get("source", "")
return out
src = payload.get("source", "")
if not isinstance(src, str):
raise BadSubmission("`source` must be a string")
return src
def system_prompt() -> str:
card = open(CARD, encoding="utf-8").read()
return (
card
+ "\n\n---\n\n"
"You are writing Zerdali. The card above is the whole language.\n\n"
"Call `submit` with a complete program. It is compiled and run against a "
"`main` you are shown but do not write. On failure you get the compiler's "
"own verdict — branch on `rule`, apply any `fixes[].edit` verbatim, and "
"treat a fix without an `edit` as a judgement call you must make.\n\n"
"Submit rather than explain. Text outside a tool call is not scored."
)
def task_prompt(task: dict) -> str:
if "modules" in task:
start = "\n\n".join(
f"module {name or '(entry)'}:\n```\n{text}\n```"
for name, text in sorted(task["modules"].items(),
key=lambda kv: (kv[0] != "", kv[0])))
else:
start = f"```\n{task['starter']}\n```"
return (
f"Task: {task['brief']}\n\n"
f"Starting point:\n{start}\n\n"
f"Your code will be called from the entry module:\n"
f"```\n{task['expect']}\n```\n\n"
"Submit a program that compiles clean and produces the intended value."
)
def verdict_message(verdict: dict) -> str:
"""What the model sees after a failed submission — the compiler, not a hint."""
stage = verdict["stage"]
if stage == "answer":
return json.dumps({
"green": False, "stage": "answer",
"note": "It compiles and runs, but the value is wrong.",
"produced": verdict["got"],
})
return json.dumps({k: v for k, v in verdict.items() if k != "expected"})
def mechanically_repaired(src):
"""`zerdali fix` on a submission, in a temporary directory.
Out of process on purpose: this measures the command a person would run,
not a function this file could call differently.
A submission is either one source or a `{module: source}` map, and the map
is the case that matters here -- `use m` resolves against the *directory*,
so the modules have to be written out beside each other before any of them
can be checked, let alone repaired. Passing the map straight to `write()`
is what this used to do, which meant `--fix` had never once run on a
multi-module task.
"""
import subprocess
import tempfile
sources = src if isinstance(src, dict) else {"": src}
# The entry module is the one named "", and on disk it needs some name;
# `use` never refers to it, so any name it cannot collide with will do.
names = {m: (m or "__entry") + ".zd" for m in sources}
with tempfile.TemporaryDirectory() as tmp:
for m, text in sources.items():
with open(os.path.join(tmp, names[m]), "w", encoding="utf-8") as f:
f.write(text)
# Once per file: `fix` repairs the entry it is given, and a module is
# only an entry when it is named as one.
for m in sorted(sources, key=lambda k: (k != "", k)):
subprocess.run(
[sys.executable, "zerdali.py", "fix", os.path.join(tmp, names[m])],
capture_output=True, timeout=120)
out = {m: open(os.path.join(tmp, names[m]), encoding="utf-8").read()
for m in sources}
return out if isinstance(src, dict) else out[""]
def run_task(client, model: str, task: dict, rounds: int, repair: bool = False) -> dict:
cfg = MODELS[model]
# The card is byte-identical on every request of every task, and `tools`
# renders ahead of `system`, so one breakpoint on the system block caches
# both. Nothing volatile is allowed in front of it — the task text goes in
# `messages`, after the breakpoint, precisely so it cannot invalidate this.
tool = SUBMIT_MODULES if "modules" in task else SUBMIT
kwargs = {"model": model, "max_tokens": 16000, "tools": [tool],
"system": [{"type": "text", "text": system_prompt(),
"cache_control": {"type": "ephemeral"}}]}
if cfg["thinking"]:
kwargs["thinking"] = {"type": cfg["thinking"]}
if cfg["effort"]:
kwargs["output_config"] = {"effort": cfg["effort"]}
messages = [{"role": "user", "content": task_prompt(task)}]
tokens_in = tokens_out = cache_write = cache_read = 0
submissions = refusals = 0
last = None
trail = []
started = time.time()
def done(**extra) -> dict:
return {"rounds": submissions, "refusals": refusals, "trail": trail,
"tokens_in": tokens_in, "tokens_out": tokens_out,
"cache_write": cache_write, "cache_read": cache_read,
"seconds": round(time.time() - started, 1), **extra}
for _ in range(rounds * 2): # a turn without a submit still costs a round trip
# A safety classifier can decline a benign request — `capability` reads as
# sandbox-escape prose if you squint. Observed once, not reproducible, and
# scoring it as a language failure would be a lie about the language. Retry
# the turn; only a persistent refusal is recorded as one.
for attempt in range(3):
with client.messages.stream(**kwargs, messages=messages) as stream:
reply = stream.get_final_message()
use = reply.usage
tokens_in += use.input_tokens
cache_write += use.cache_creation_input_tokens or 0
cache_read += use.cache_read_input_tokens or 0
tokens_out += use.output_tokens
if reply.stop_reason != "refusal":
break
refusals += 1
if reply.stop_reason == "refusal":
return done(green=False, stalled="refusal")
calls = [b for b in reply.content if b.type == "tool_use"]
if not calls:
# It stopped without submitting. Say so once and let it continue.
messages.append({"role": "assistant", "content": reply.content})
messages.append({"role": "user", "content":
"Submit with the `submit` tool."})
continue
messages.append({"role": "assistant", "content": reply.content})
results = []
for call in calls:
submissions += 1
try:
src = submitted_source(task, call.input)
except BadSubmission as bad:
# Sayılıyor ve geri bildiriliyor: bir tur harcandı, koşu değil.
last = {"green": False, "stage": "shape", "detail": str(bad)}
trail.append({"round": submissions, "source": "",
"stage": "shape", "rules": []})
results.append({
"type": "tool_result", "tool_use_id": call.id,
"content": f"malformed submission: {bad}",
"is_error": True,
})
continue
if repair:
# Apply the repairs the compiler already computed, before the
# model is asked to reproduce them. A diagnostic carrying one
# fix is an answer, not a suggestion (§72); handing it back for
# the model to retype spends a round on a number the compiler
# printed. Anything with two fixes is a judgement call and is
# left exactly where it was.
src = mechanically_repaired(src)
last = score(task, src)
# The trail is the point of the exercise: a count says a task was
# hard, the trail says which rule made it hard.
trail.append({"round": submissions, "source": src,
"stage": "green" if last["green"] else last["stage"],
"rules": sorted({d.get("rule") or d["code"]
for d in last.get("diagnostics", [])})})
results.append({
"type": "tool_result", "tool_use_id": call.id,
"content": json.dumps({"green": True}) if last["green"]
else verdict_message(last),
"is_error": not last["green"],
})
if last["green"]:
return done(green=True, source=src)
messages.append({"role": "user", "content": results})
if submissions >= rounds:
break
stalled = "no-submission" if last is None else last["stage"]
if last and last.get("diagnostics"):
stalled = f"{last['stage']}:" + ",".join(
sorted({d.get("rule") or d["code"] for d in last["diagnostics"]}))
return done(green=False, stalled=stalled)
def report(model: str, results: dict) -> None:
cfg = MODELS[model]
# With --repeat, a task holds a list of runs. Report the spread rather than
# a single number: on a noisy model the spread *is* the result, and a change
# smaller than it has not been demonstrated.
if any(isinstance(v, list) for v in results.values()):
print(f"\n{model}")
flat = []
for name, runs in results.items():
wins = sum(r["green"] for r in runs)
rs = sorted(r["rounds"] for r in runs)
flat += runs
print(f" {name:<14} green {wins}/{len(runs)} runs "
f"rounds {rs[len(rs) // 2]} (min {rs[0]}, max {rs[-1]})")
cost = None if cfg["in"] is None else sum(
r["tokens_in"] * cfg["in"]
+ r["cache_write"] * cfg["in"] * 1.25
+ r["cache_read"] * cfg["in"] * 0.10
+ r["tokens_out"] * cfg["out"] for r in flat) / 1_000_000
wins = sum(r["green"] for r in flat)
shown = "price unknown" if cost is None else f"${cost:.2f}"
print(f" {wins}/{len(flat)} task-runs green, {shown}")
return
green = [r for r in results.values() if r["green"]]
# Cache writes cost 1.25x and reads 0.1x, so counting them as plain input
# would misprice the run in both directions.
#
# Fiyatı bilinmeyen bir model için `None`: sıfır yazmak onu bedava
# gösterirdi, ve bir ölçümde uydurulmuş bir sayı, yokluğundan kötüdür.
cost = None if cfg["in"] is None else sum(
r["tokens_in"] * cfg["in"]
+ r["cache_write"] * cfg["in"] * 1.25
+ r["cache_read"] * cfg["in"] * 0.10
+ r["tokens_out"] * cfg["out"]
for r in results.values()) / 1_000_000
print(f"\n{model}")
for name, r in results.items():
mark = f"green in {r['rounds']}" if r["green"] else f"STUCK {r['stalled']}"
print(f" {name:<14} {mark:<34} {r['tokens_out']:>6} out {r['seconds']:>5}s")
rounds = sum(r["rounds"] for r in green) / len(green) if green else 0
read = sum(r["cache_read"] for r in results.values())
billed = read + sum(r["tokens_in"] + r["cache_write"]
for r in results.values())
shown = "price unknown" if cost is None else f"${cost:.2f}"
print(f" {len(green)}/{len(results)} green, {rounds:.1f} rounds on average, "
f"{shown}")
if read:
print(f" cache: {100 * read / billed:.0f}% of input served from cache")
else:
print(f" cache: no hits — the prefix is under this model's "
f"{cfg['cache_min']}-token minimum, so the breakpoint is ignored")
# --- self-test -------------------------------------------------------------
#
# Reference solutions. They are here rather than in eval_tasks.py so the task
# file a model might be pointed at never carries the answers — and running them
# proves the task set is solvable at all, which is worth knowing before any
# model failure gets blamed on the model.
SOLUTIONS = {
"pure": "fn double(x: Int) -> Int { x * 2 }",
"effect": ("effect clock { now() -> Int }\n"
"fn stamp() -> Int !{clock.now} { clock.now() }"),
"label": 'fn greet(name: Str@personal) -> Str@personal { "Sayin " + name }',
"declassify": (
"effect db { customer(id: Str@internal) -> Str@personal }\n"
"fn is_known(id: Str@internal) -> Bool !{db.customer} {\n"
" declassify(str_len(db.customer(id)) > 0, {},\n"
' "whether a record exists is not the record")\n}'),
"match": ("data Shape { Circle(r: Int) | Square(side: Int) | Point }\n"
"fn area(s: Shape) -> Int {\n"
" match s { Circle(r) => 3 * r * r, Square(side) => side * side,\n"
" Point => 0 }\n}"),
"recursion": ("data L { Nil | Cons(head: Int, tail: L) }\n"
"fn total(l: L) -> Int "
"{ match l { Nil => 0, Cons(h, t) => h + total(t) } }"),
"diverge": ("fn fact(n: Int) -> Int !{div.loop} "
"{ if n <= 1 { 1 } else { n * fact(n - 1) } }"),
"failure": "fn ratio(a: Int, b: Int) -> Int !{exn.raise} { a / b }",
"examples": (
"fn bonus(salary: Int, years: Int) -> Int\n"
" where bonus(1000, 11) == 200, bonus(1000, 6) == 100, "
"bonus(1000, 2) == 0\n"
"{ if years > 10 { salary * 20 / 100 }\n"
" else { if years > 5 { salary * 10 / 100 } else { 0 } } }"),
"capability": ("effect clock { now() -> Int }\n"
"fn read_twice() -> Int !{clock.now} "
"{ restrict { clock.now } { clock.now() + clock.now() } }"),
"handler": (
"effect emit { item(v: Int) -> Unit }\n"
"data L { Nil | Cons(head: Int, tail: L) }\n"
"fn produce() -> L !{emit.item} "
"{ emit.item(20); emit.item(22); Nil }\n"
"fn collect() -> L "
"{ handle produce() with { emit.item(v) resume k => Cons(v, k(())) } }\n"
"fn total(l: L) -> Int "
"{ match l { Nil => 0, Cons(h, t) => h + total(t) } }"),
"module_label": (
"fn mask(name: Str@personal) -> Str !{exn.raise} {\n"
' declassify(char_at(name, 0), {}, "one initial does not identify")\n}'),
}
LARGE_SOLUTIONS = {
"ledger": {
"book": (
"data Tx { T(who: Str, amount: Int, kind: Str) }\n"
"pub fn net(rows: List Tx, who: Str) -> Int !{exn.raise, div.loop} {\n"
" walk(rows, who, 0)\n"
"}\n"
"fn walk(rows: List Tx, who: Str, i: Int) -> Int "
"!{exn.raise, div.loop} {\n"
" if i >= len(rows) { 0 }\n"
" else { one(nth(rows, i), who) + walk(rows, who, i + 1) }\n"
"}\n"
"fn one(t: Tx, who: Str) -> Int {\n"
" match t {\n"
" T(w, a, k) =>\n"
" if w == who { if k == \"credit\" { a } else { 0 - a } }\n"
" else { 0 }\n"
" }\n"
"}"),
"": "use book",
},
"interp": (
"data E { Num(v: Int) | Var(name: Str) | Add(l: E, r: E)\n"
" | Mul(l: E, r: E) | Let(name: Str, value: E, body: E) }\n"
"data Env { Empty | Bind(name: Str, value: Int, rest: Env) }\n"
"fn lookup(env: Env, name: Str) -> Int !{exn.raise} {\n"
" match env {\n"
" Empty => exn.raise(\"unbound name: \" + name),\n"
" Bind(n, v, rest) => if n == name { v } else { lookup(rest, name) }\n"
" }\n"
"}\n"
"fn eval(e: E, env: Env) -> Int !{exn.raise} {\n"
" match e {\n"
" Num(v) => v,\n"
" Var(n) => lookup(env, n),\n"
" Add(l, r) => eval(l, env) + eval(r, env),\n"
" Mul(l, r) => eval(l, env) * eval(r, env),\n"
" Let(n, value, body) => eval(body, Bind(n, eval(value, env), env))\n"
" }\n"
"}"),
"kanon": (
"data Rec { R(name: Str@personal, city: Str@internal, "
"spend: Int@personal) }\n"
"fn here(r: Rec, city: Str@internal) -> Bool@internal {\n"
" match r { R(n, c, s) => c == city }\n"
"}\n"
"fn tally(rs: List Rec, city: Str@internal, i: Int) -> Int@internal\n"
" !{exn.raise, div.loop} {\n"
" if i >= len(rs) { 0 }\n"
" else {\n"
" (if here(nth(rs, i), city) { 1 } else { 0 })\n"
" + tally(rs, city, i + 1)\n"
" }\n"
"}\n"
"fn spent(rs: List Rec, city: Str@internal, i: Int) -> Int@personal\n"
" !{exn.raise, div.loop} {\n"
" if i >= len(rs) { 0 }\n"
" else {\n"
" (match nth(rs, i) { R(n, c, s) => if c == city { s } else { 0 } })\n"
" + spent(rs, city, i + 1)\n"
" }\n"
"}\n"
"fn city_total(rs: List Rec, city: Str@internal) -> Int\n"
" !{exn.raise, div.loop} {\n"
" declassify(\n"
" if tally(rs, city, 0) >= 3 { spent(rs, city, 0) }\n"
" else { exn.raise(\"fewer than three people: the total is personal\") },\n"
" {},\n"
" \"a total over three or more people is not any one person's spend\")\n"
"}"),
"tokenize": (
"data Tok { Word(text: Str) | Number(text: Str) }\n"
"fn digits(t: Str, i: Int) -> Bool !{exn.raise, div.loop} {\n"
" if i >= str_len(t) { true }\n"
" else if char_at(t, i) >= \"0\" && char_at(t, i) <= \"9\" "
"{ digits(t, i + 1) }\n"
" else { false }\n"
"}\n"
"fn classify(t: Str) -> Tok !{exn.raise, div.loop} {\n"
" if digits(t, 0) { Number(t) } else { Word(t) }\n"
"}\n"
"fn scan(s: Str, i: Int, cur: Str, acc: List Tok) -> List Tok\n"
" !{exn.raise, div.loop} {\n"
" if i >= str_len(s) { push(acc, classify(cur)) }\n"
" else if char_at(s, i) == \" \" "
"{ scan(s, i + 1, \"\", push(acc, classify(cur))) }\n"
" else { scan(s, i + 1, cur + char_at(s, i), acc) }\n"
"}\n"
"fn tokens(s: Str) -> List Tok !{exn.raise, div.loop} "
"{ scan(s, 0, \"\", []) }"),
}
def self_test() -> int:
"""Check the task sets against known-good programs, without touching the API."""
failed = 0
total = 0
for tasks, answers in ((TASKS, SOLUTIONS), (LARGE_TASKS, LARGE_SOLUTIONS)):
for task in tasks:
total += 1
src = answers.get(task["name"])
if src is None:
print(f"MISSING {task['name']}: no reference solution")
failed += 1
continue
verdict = score(task, src)
if verdict["green"]:
print(f"ok {task['name']:<14} solvable")
else:
failed += 1
detail = verdict.get("detail") or verdict.get("diagnostics") or verdict
print(f"FAIL {task['name']:<14} {verdict['stage']}: {detail}")
print(f"\n{total - failed}/{total} tasks have a program that reaches green "
f"— the harness measures the model, not the task set")
return 1 if failed else 0
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--models", default="claude-opus-5")
ap.add_argument("--tasks", default="")
ap.add_argument("--set", default="core", choices=sorted(TASK_SETS),
help="core: the original twelve. large: 50+ line, "
"multi-module tasks where field access and an O(n) "
"push actually come under load.")
ap.add_argument("--rounds", type=int, default=MAX_ROUNDS)
ap.add_argument("--repeat", type=int, default=1,
help="run the set N times. A single run cannot validate a "
"change on a noisy model: haiku-4.5 spans 2/4 to 4/4 "
"green and 2.0 to 4.8 rounds on identical code.")
ap.add_argument("--fix", action="store_true",
help="apply `zerdali fix` to each submission before scoring "
"it -- measures what the mechanical repairs are worth")
ap.add_argument("--label", default="",
help="write eval-<model>-<label>.json, so a run can be "
"kept as a baseline instead of overwriting one")
ap.add_argument("--self-test", action="store_true")
args = ap.parse_args()
if args.self_test:
return self_test()
from eval_vendors import client_for
chosen = TASK_SETS[args.set]
wanted = [t for t in chosen
if not args.tasks or t["name"] in args.tasks.split(",")]
for model in args.models.split(","):
if model not in MODELS:
print(f"unknown model {model!r}; known: {', '.join(MODELS)}",
file=sys.stderr)
return 2
# Satıcı modelin adından. İstemci burada kuruluyor çünkü üçü ayrı
# nesneler, ama `run_task` farkı görmüyor: hepsi aynı yüzeyi taşıyor.
client = client_for(model, MODELS[model].get("effort"))
results = {}
# An hour of measurement should not be thrown away by the API call that
# happens to fail last — an exhausted credit balance did exactly that
# once. Keep what finished, say what stopped it, and write the file.
#
# Writing only at the end still lost everything when the process was
# killed from outside rather than raising: a four-repeat run died with
# nothing on disk. Saving after each task bounds the loss to one task.
stopped = None
def save() -> str:
# A four-task probe must not overwrite a twelve-task record — the
# file is the baseline a later change gets compared against.
partial = "" if len(results) == len(chosen) else ".partial"
label = f"-{args.label}" if args.label else ""
kind = "" if args.set == "core" else f"-{args.set}"
# Kök dizin yüz beş ölçüm dosyasıyla dolmuştu ve dilin kendi
# kaynakları arasında kayboluyorlardı. Kayıt tek klasörde.
os.makedirs("measurements", exist_ok=True)
out = os.path.join("measurements",
f"eval-{model}{kind}{label}{partial}.json")
with open(out, "w", encoding="utf-8") as fh:
json.dump(results, fh, indent=2, sort_keys=True)
return out
for task in wanted:
runs = []
for i in range(args.repeat):
tag = f" ({i + 1}/{args.repeat})" if args.repeat > 1 else ""
print(f" {model} :: {task['name']}{tag} ...", flush=True)
try:
runs.append(run_task(client, model, task, args.rounds,
repair=args.fix))
except Exception as err: # noqa: BLE001
stopped = f"{type(err).__name__}: {err}"
print(f" stopped at {task['name']}: {stopped}",
file=sys.stderr)
break
# After every repeat, not just every task. One task at
# `--repeat 8` is a single task-sized write at the very end,
# which is exactly the window that lost a run twice already.
results[task["name"]] = runs if args.repeat > 1 else runs[0]
save()
if runs:
results[task["name"]] = runs if args.repeat > 1 else runs[0]
save()
if stopped:
break
if results:
report(model, results)
print(f" written to {save()}")
if stopped:
return 1
return 0
if __name__ == "__main__":
os.chdir(os.path.dirname(os.path.abspath(__file__)))
sys.exit(main())