diff --git a/crates/weavepy-bench/baselines/bench.json b/crates/weavepy-bench/baselines/bench.json index 7bd6be9..58ca0ff 100644 --- a/crates/weavepy-bench/baselines/bench.json +++ b/crates/weavepy-bench/baselines/bench.json @@ -1,708 +1,708 @@ { "version": 3, "host": "unknown", - "created_at": "ts=1786235633", - "geomean_ratio": 8.486330128963486, + "created_at": "ts=1786352866", + "geomean_ratio": 8.407741671174728, "rows": [ { "name": "fannkuch", "work": 100000, "weavepy": { "samples": [ - 112616292.0, - 111987458.0, - 113067833.0, - 112475250.0, - 112025250.0 + 111450000.0, + 108831750.0, + 110463375.0, + 111356834.0, + 110531458.0 ], - "mean_ns": 112434416.6, - "median_ns": 112475250.0, - "p95_ns": 113067833.0, - "stddev_ns": 448093.64538542606, - "max_rss_bytes": 38993920 + "mean_ns": 110526683.4, + "median_ns": 110531458.0, + "p95_ns": 111450000.0, + "stddev_ns": 1051010.6497161672, + "max_rss_bytes": 40321024 }, "cpython": { "samples": [ - 11929750.0, - 11168500.0, - 11936792.0, - 11824833.0, - 11687458.0 + 11548042.0, + 11620791.0, + 11134833.0, + 11610167.0, + 11715916.0 ], - "mean_ns": 11709466.6, - "median_ns": 11824833.0, - "p95_ns": 11936792.0, - "stddev_ns": 318841.7557500899, - "max_rss_bytes": 14860288 + "mean_ns": 11525949.8, + "median_ns": 11610167.0, + "p95_ns": 11715916.0, + "stddev_ns": 226734.95742760974, + "max_rss_bytes": 14794752 }, "jit": null, - "ratio": 9.511783379942871, - "memory_ratio": 2.624035281146637 + "ratio": 9.520229812370486, + "memory_ratio": 2.725359911406423 }, { "name": "nbody", "work": 20000, "weavepy": { "samples": [ - 276629583.0, - 275632750.0, - 275901458.0, - 276486875.0, - 276057916.0 + 266980458.0, + 266095792.0, + 266488375.0, + 269417750.0, + 268792000.0 ], - "mean_ns": 276141716.4, - "median_ns": 276057916.0, - "p95_ns": 276629583.0, - "stddev_ns": 412595.9526853117, - "max_rss_bytes": 39059456 + "mean_ns": 267554875.0, + "median_ns": 266980458.0, + "p95_ns": 269417750.0, + "stddev_ns": 1466039.0404102476, + "max_rss_bytes": 40452096 }, "cpython": { "samples": [ - 23843666.0, - 23667542.0, - 23762792.0, - 23710583.0, - 23393542.0 + 22797625.0, + 23127250.0, + 22911708.0, + 24584166.0, + 23602500.0 ], - "mean_ns": 23675625.0, - "median_ns": 23710583.0, - "p95_ns": 23843666.0, - "stddev_ns": 170802.5754431121, - "max_rss_bytes": 15024128 + "mean_ns": 23404649.8, + "median_ns": 23127250.0, + "p95_ns": 24584166.0, + "stddev_ns": 727809.1341046772, + "max_rss_bytes": 15056896 }, "jit": null, - "ratio": 11.642814350030955, - "memory_ratio": 2.599781897491821 + "ratio": 11.54397768865732, + "memory_ratio": 2.686615886833515 }, { "name": "fib", "work": 27, "weavepy": { "samples": [ - 218685208.0, - 216320458.0, - 220419333.0, - 226875167.0, - 215668458.0 + 211595500.0, + 212450042.0, + 211589125.0, + 212139208.0, + 217470458.0 ], - "mean_ns": 219593724.8, - "median_ns": 218685208.0, - "p95_ns": 226875167.0, - "stddev_ns": 4490223.468350289, - "max_rss_bytes": 39043072 + "mean_ns": 213048866.6, + "median_ns": 212139208.0, + "p95_ns": 217470458.0, + "stddev_ns": 2498982.8026338634, + "max_rss_bytes": 40386560 }, "cpython": { "samples": [ - 17776666.0, - 18090750.0, - 17331500.0, - 17545875.0, - 17451875.0 + 17891792.0, + 17515750.0, + 17479125.0, + 18374541.0, + 17299042.0 ], - "mean_ns": 17639333.2, - "median_ns": 17545875.0, - "p95_ns": 18090750.0, - "stddev_ns": 300530.24647895264, - "max_rss_bytes": 14745600 + "mean_ns": 17712050.0, + "median_ns": 17515750.0, + "p95_ns": 18374541.0, + "stddev_ns": 428533.7983560923, + "max_rss_bytes": 14712832 }, "jit": null, - "ratio": 12.463625097066975, - "memory_ratio": 2.647777777777778 + "ratio": 12.111340250917031, + "memory_ratio": 2.744988864142539 }, { "name": "pidigits", "work": 500000, "weavepy": { "samples": [ - 2206079791.0, - 2208453791.0, - 2201123250.0, - 2198629292.0, - 2202624834.0 + 2181746750.0, + 2171952209.0, + 2169492125.0, + 2178358959.0, + 2176434416.0 ], - "mean_ns": 2203382191.6, - "median_ns": 2202624834.0, - "p95_ns": 2208453791.0, - "stddev_ns": 3913586.1205270924, - "max_rss_bytes": 40255488 + "mean_ns": 2175596891.8, + "median_ns": 2176434416.0, + "p95_ns": 2181746750.0, + "stddev_ns": 4916886.928687876, + "max_rss_bytes": 41467904 }, "cpython": { "samples": [ - 2394167167.0, - 2389834375.0, - 2389886000.0, - 2390684000.0, - 2391536333.0 + 2360799208.0, + 2359735459.0, + 2358792666.0, + 2358272792.0, + 2358641042.0 ], - "mean_ns": 2391221575.0, - "median_ns": 2390684000.0, - "p95_ns": 2394167167.0, - "stddev_ns": 1786942.6494908838, - "max_rss_bytes": 15466496 + "mean_ns": 2359248233.4, + "median_ns": 2358792666.0, + "p95_ns": 2360799208.0, + "stddev_ns": 1021104.7904293663, + "max_rss_bytes": 15269888 }, "jit": null, - "ratio": 0.9213366693381476, - "memory_ratio": 2.6027542372881354 + "ratio": 0.9226900046669892, + "memory_ratio": 2.7156652360515023 }, { "name": "pyaes", "work": 400, "weavepy": { "samples": [ - 248566959.0, - 244524625.0, - 245067166.0, - 243549250.0, - 240108500.0 + 240037292.0, + 238363666.0, + 237864958.0, + 237773041.0, + 237880750.0 ], - "mean_ns": 244363300.0, - "median_ns": 244524625.0, - "p95_ns": 248566959.0, - "stddev_ns": 3039662.9143854915, - "max_rss_bytes": 39075840 + "mean_ns": 238383941.4, + "median_ns": 237880750.0, + "p95_ns": 240037292.0, + "stddev_ns": 952591.7215149415, + "max_rss_bytes": 40353792 }, "cpython": { "samples": [ - 18520292.0, - 18348583.0, - 18211209.0, - 18719666.0, - 18464916.0 + 18038875.0, + 17891167.0, + 18042333.0, + 18039333.0, + 17913833.0 ], - "mean_ns": 18452933.2, - "median_ns": 18464916.0, - "p95_ns": 18719666.0, - "stddev_ns": 190490.63962226594, - "max_rss_bytes": 14925824 + "mean_ns": 17985108.2, + "median_ns": 18038875.0, + "p95_ns": 18042333.0, + "stddev_ns": 75846.85004138273, + "max_rss_bytes": 14794752 }, "jit": null, - "ratio": 13.242661109316717, - "memory_ratio": 2.6180021953896815 + "ratio": 13.187116713209665, + "memory_ratio": 2.727574750830565 }, { "name": "richards", "work": 50000, "weavepy": { "samples": [ - 239340209.0, - 237831875.0, - 238297625.0, - 233705792.0, - 238573959.0 + 242406833.0, + 244277167.0, + 246135041.0, + 243900500.0, + 244772958.0 ], - "mean_ns": 237549892.0, - "median_ns": 238297625.0, - "p95_ns": 239340209.0, - "stddev_ns": 2217525.19954994, - "max_rss_bytes": 39206912 + "mean_ns": 244298499.8, + "median_ns": 244277167.0, + "p95_ns": 246135041.0, + "stddev_ns": 1354183.921613198, + "max_rss_bytes": 40353792 }, "cpython": { "samples": [ - 13910875.0, - 13571541.0, - 14053792.0, - 13787125.0, - 13581667.0 + 14042583.0, + 12768333.0, + 12536250.0, + 13594667.0, + 14388625.0 ], - "mean_ns": 13781000.0, - "median_ns": 13787125.0, - "p95_ns": 14053792.0, - "stddev_ns": 209121.5101824774, - "max_rss_bytes": 14778368 + "mean_ns": 13466091.6, + "median_ns": 13594667.0, + "p95_ns": 14388625.0, + "stddev_ns": 798657.3587651966, + "max_rss_bytes": 14761984 }, "jit": null, - "ratio": 17.2840693763203, - "memory_ratio": 2.652993348115299 + "ratio": 17.968602467423437, + "memory_ratio": 2.7336293007769146 }, { "name": "sumvm", "work": 2000000, "weavepy": { "samples": [ - 179895834.0, - 179714209.0, - 179479375.0, - 180571625.0, - 180036417.0 + 163999417.0, + 165219333.0, + 166088458.0, + 163594917.0, + 164525042.0 ], - "mean_ns": 179939492.0, - "median_ns": 179895834.0, - "p95_ns": 180571625.0, - "stddev_ns": 410290.35147563484, - "max_rss_bytes": 38961152 + "mean_ns": 164685433.4, + "median_ns": 164525042.0, + "p95_ns": 166088458.0, + "stddev_ns": 992357.8905839869, + "max_rss_bytes": 40304640 }, "cpython": { "samples": [ - 41235292.0, - 38337333.0, - 40567000.0, - 38211250.0, - 42577500.0 + 38648041.0, + 37365583.0, + 38983958.0, + 37582375.0, + 37386416.0 ], - "mean_ns": 40185675.0, - "median_ns": 40567000.0, - "p95_ns": 42577500.0, - "stddev_ns": 1889619.9130068459, + "mean_ns": 37993274.6, + "median_ns": 37582375.0, + "p95_ns": 38983958.0, + "stddev_ns": 765062.9425068895, "max_rss_bytes": 14712832 }, "jit": null, - "ratio": 4.434536297976187, - "memory_ratio": 2.648106904231626 + "ratio": 4.3777180659817265, + "memory_ratio": 2.739420935412027 }, { "name": "nested_loops", "work": 120, "weavepy": { "samples": [ - 267278084.0, - 265283917.0, - 264756166.0, - 265865250.0, - 266192333.0 + 266637625.0, + 267435708.0, + 265399250.0, + 269897667.0, + 267326458.0 ], - "mean_ns": 265875150.0, - "median_ns": 265865250.0, - "p95_ns": 267278084.0, - "stddev_ns": 957941.6117893094, - "max_rss_bytes": 39026688 + "mean_ns": 267339341.6, + "median_ns": 267326458.0, + "p95_ns": 269897667.0, + "stddev_ns": 1643992.3148838927, + "max_rss_bytes": 40321024 }, "cpython": { "samples": [ - 53682917.0, - 52363834.0, - 52137292.0, - 52990250.0, - 51415750.0 + 51540291.0, + 51083208.0, + 50340042.0, + 51965083.0, + 49808208.0 ], - "mean_ns": 52518008.6, - "median_ns": 52363834.0, - "p95_ns": 53682917.0, - "stddev_ns": 860771.816211939, - "max_rss_bytes": 14663680 + "mean_ns": 50947366.4, + "median_ns": 51083208.0, + "p95_ns": 51965083.0, + "stddev_ns": 876396.6548625685, + "max_rss_bytes": 14729216 }, "jit": null, - "ratio": 5.077268597253593, - "memory_ratio": 2.6614525139664806 + "ratio": 5.233157205005606, + "memory_ratio": 2.7374860956618465 }, { "name": "jitloop", "work": 1000, "weavepy": { "samples": [ - 378033000.0, - 379181708.0, - 377306083.0, - 376237458.0, - 379571542.0 + 313494000.0, + 314278500.0, + 320105125.0, + 316248417.0, + 316402583.0 ], - "mean_ns": 378065958.2, - "median_ns": 378033000.0, - "p95_ns": 379571542.0, - "stddev_ns": 1363230.5141597295, - "max_rss_bytes": 38862848 + "mean_ns": 316105725.0, + "median_ns": 316248417.0, + "p95_ns": 320105125.0, + "stddev_ns": 2562398.520687112, + "max_rss_bytes": 40321024 }, "cpython": { "samples": [ - 60738625.0, - 61517584.0, - 60899584.0, - 61066250.0, - 60913917.0 + 61008584.0, + 59965042.0, + 60282625.0, + 59995209.0, + 60168000.0 ], - "mean_ns": 61027192.0, - "median_ns": 60913917.0, - "p95_ns": 61517584.0, - "stddev_ns": 297655.7499049867, - "max_rss_bytes": 14729216 + "mean_ns": 60283892.0, + "median_ns": 60168000.0, + "p95_ns": 61008584.0, + "stddev_ns": 425327.59515636886, + "max_rss_bytes": 14712832 }, "jit": null, - "ratio": 6.206020210455354, - "memory_ratio": 2.6384872080088986 + "ratio": 5.256089898284802, + "memory_ratio": 2.7405345211581293 }, { "name": "deltablue", "work": 50, "weavepy": { "samples": [ - 1049755750.0, - 1049612917.0, - 1046074667.0, - 1047386000.0, - 1110569500.0 + 1089503042.0, + 1033996583.0, + 1036588750.0, + 1034074875.0, + 1039770958.0 ], - "mean_ns": 1060679766.8, - "median_ns": 1049612917.0, - "p95_ns": 1110569500.0, - "stddev_ns": 27932185.56705851, - "max_rss_bytes": 43384832 + "mean_ns": 1046786841.6, + "median_ns": 1036588750.0, + "p95_ns": 1089503042.0, + "stddev_ns": 23995219.73997159, + "max_rss_bytes": 44580864 }, "cpython": { "samples": [ - 49107209.0, - 48349458.0, - 48981208.0, - 49478792.0, - 48988417.0 + 47775875.0, + 48217416.0, + 47460375.0, + 48053916.0, + 47510625.0 ], - "mean_ns": 48981016.8, - "median_ns": 48988417.0, - "p95_ns": 49478792.0, - "stddev_ns": 407009.1947520351, - "max_rss_bytes": 16695296 + "mean_ns": 47803641.4, + "median_ns": 47775875.0, + "p95_ns": 48217416.0, + "stddev_ns": 331024.2438316263, + "max_rss_bytes": 16711680 }, "jit": null, - "ratio": 21.425736557276387, - "memory_ratio": 2.5986261040235523 + "ratio": 21.696907696614662, + "memory_ratio": 2.6676470588235293 }, { "name": "float_math", "work": 100000, "weavepy": { "samples": [ - 643707042.0, - 644778375.0, - 643007292.0, - 646233917.0, - 643518166.0 + 626050042.0, + 619770000.0, + 617411291.0, + 623695875.0, + 618311958.0 ], - "mean_ns": 644248958.4, - "median_ns": 643707042.0, - "p95_ns": 646233917.0, - "stddev_ns": 1283531.1590924857, - "max_rss_bytes": 130727936 + "mean_ns": 621047833.2, + "median_ns": 619770000.0, + "p95_ns": 626050042.0, + "stddev_ns": 3687022.112176804, + "max_rss_bytes": 131874816 }, "cpython": { "samples": [ - 38924042.0, - 39722625.0, - 38966375.0, - 39526084.0, - 39492666.0 + 37597208.0, + 37767666.0, + 37354667.0, + 37487542.0, + 38769917.0 ], - "mean_ns": 39326358.4, - "median_ns": 39492666.0, - "p95_ns": 39722625.0, - "stddev_ns": 359173.4557888431, + "mean_ns": 37795400.0, + "median_ns": 37597208.0, + "p95_ns": 38769917.0, + "stddev_ns": 565410.1945406538, "max_rss_bytes": 35667968 }, "jit": null, - "ratio": 16.299407135491943, - "memory_ratio": 3.6651355075792376 + "ratio": 16.48446874033838, + "memory_ratio": 3.6972898484152505 }, { "name": "spectral_norm", "work": 100, "weavepy": { "samples": [ - 293116709.0, - 293748958.0, - 293006542.0, - 292707208.0, - 292084958.0 + 298942292.0, + 298744125.0, + 299564792.0, + 299432958.0, + 299070500.0 ], - "mean_ns": 292932875.0, - "median_ns": 293006542.0, - "p95_ns": 293748958.0, - "stddev_ns": 607268.8990702553, - "max_rss_bytes": 39075840 + "mean_ns": 299150933.4, + "median_ns": 299070500.0, + "p95_ns": 299564792.0, + "stddev_ns": 341434.7140154322, + "max_rss_bytes": 40419328 }, "cpython": { "samples": [ - 30933917.0, - 30591833.0, - 31529625.0, - 31529541.0, - 30808125.0 + 30663666.0, + 30627250.0, + 30255083.0, + 30694291.0, + 31013666.0 ], - "mean_ns": 31078608.2, - "median_ns": 30933917.0, - "p95_ns": 31529625.0, - "stddev_ns": 429477.3987571407, - "max_rss_bytes": 14876672 + "mean_ns": 30650791.2, + "median_ns": 30663666.0, + "p95_ns": 31013666.0, + "stddev_ns": 269664.27330794116, + "max_rss_bytes": 14827520 }, "jit": null, - "ratio": 9.472015522638145, - "memory_ratio": 2.6266519823788546 + "ratio": 9.753253247671038, + "memory_ratio": 2.7259668508287294 }, { "name": "json_bench", "work": 150, "weavepy": { "samples": [ - 229966583.0, - 229080125.0, - 229877541.0, - 230212458.0, - 229834208.0 + 222706625.0, + 223357417.0, + 223227209.0, + 223318875.0, + 224937250.0 ], - "mean_ns": 229794183.0, - "median_ns": 229877541.0, - "p95_ns": 230212458.0, - "stddev_ns": 425177.9391554788, - "max_rss_bytes": 50298880 + "mean_ns": 223509475.2, + "median_ns": 223318875.0, + "p95_ns": 224937250.0, + "stddev_ns": 839975.2471419619, + "max_rss_bytes": 50724864 }, "cpython": { "samples": [ - 43406250.0, - 42638291.0, - 43241209.0, - 42965042.0, - 42931750.0 + 42971375.0, + 42327542.0, + 42290834.0, + 42302875.0, + 41923042.0 ], - "mean_ns": 43036508.4, - "median_ns": 42965042.0, - "p95_ns": 43406250.0, - "stddev_ns": 297182.62599502684, - "max_rss_bytes": 15630336 + "mean_ns": 42363133.6, + "median_ns": 42302875.0, + "p95_ns": 42971375.0, + "stddev_ns": 378735.8281471401, + "max_rss_bytes": 15728640 }, "jit": null, - "ratio": 5.350339026783682, - "memory_ratio": 3.2180293501048216 + "ratio": 5.2790472278775376, + "memory_ratio": 3.225 }, { "name": "str_methods", "work": 15000, "weavepy": { "samples": [ - 211904792.0, - 213203833.0, - 211516333.0, - 212879041.0, - 213789208.0 + 203523417.0, + 203855750.0, + 204823333.0, + 202941292.0, + 204433708.0 ], - "mean_ns": 212658641.4, - "median_ns": 212879041.0, - "p95_ns": 213789208.0, - "stddev_ns": 935033.1678632047, - "max_rss_bytes": 38961152 + "mean_ns": 203915500.0, + "median_ns": 203855750.0, + "p95_ns": 204823333.0, + "stddev_ns": 741464.8770855569, + "max_rss_bytes": 40304640 }, "cpython": { "samples": [ - 31979792.0, - 31369375.0, - 31828875.0, - 32122792.0, - 32299875.0 + 30884792.0, + 30775875.0, + 30917375.0, + 31109833.0, + 31308708.0 ], - "mean_ns": 31920141.8, - "median_ns": 31979792.0, - "p95_ns": 32299875.0, - "stddev_ns": 353728.7979041288, - "max_rss_bytes": 14598144 + "mean_ns": 30999316.6, + "median_ns": 30917375.0, + "p95_ns": 31308708.0, + "stddev_ns": 210768.45796347232, + "max_rss_bytes": 14729216 }, "jit": null, - "ratio": 6.656673720704625, - "memory_ratio": 2.6689113355780023 + "ratio": 6.593565915605707, + "memory_ratio": 2.7363737486095663 }, { "name": "dict_ops", "work": 100000, "weavepy": { "samples": [ - 234242417.0, - 233740250.0, - 233954000.0, - 233860208.0, - 233812250.0 + 227243791.0, + 226168666.0, + 222553875.0, + 224756459.0, + 225275250.0 ], - "mean_ns": 233921825.0, - "median_ns": 233860208.0, - "p95_ns": 234242417.0, - "stddev_ns": 195312.10729752522, - "max_rss_bytes": 39026688 + "mean_ns": 225199608.2, + "median_ns": 225275250.0, + "p95_ns": 227243791.0, + "stddev_ns": 1754976.0255133687, + "max_rss_bytes": 40402944 }, "cpython": { "samples": [ - 33366333.0, - 32864583.0, - 33512750.0, - 33252625.0, - 33173541.0 + 32451208.0, + 32262291.0, + 33049208.0, + 32851792.0, + 32493292.0 ], - "mean_ns": 33233966.4, - "median_ns": 33252625.0, - "p95_ns": 33512750.0, - "stddev_ns": 242736.8332882342, - "max_rss_bytes": 14860288 + "mean_ns": 32621558.2, + "median_ns": 32493292.0, + "p95_ns": 33049208.0, + "stddev_ns": 320326.5067867472, + "max_rss_bytes": 14909440 }, "jit": null, - "ratio": 7.032834490510147, - "memory_ratio": 2.6262403528114664 + "ratio": 6.932977120323788, + "memory_ratio": 2.70989010989011 }, { "name": "list_ops", "work": 10000, "weavepy": { "samples": [ - 407780167.0, - 407555584.0, - 407671792.0, - 407312875.0, - 406542250.0 + 399523208.0, + 401066000.0, + 403235166.0, + 402898875.0, + 405036166.0 ], - "mean_ns": 407372533.6, - "median_ns": 407555584.0, - "p95_ns": 407780167.0, - "stddev_ns": 495519.4326888704, - "max_rss_bytes": 39141376 + "mean_ns": 402351883.0, + "median_ns": 402898875.0, + "p95_ns": 405036166.0, + "stddev_ns": 2117761.889091642, + "max_rss_bytes": 40321024 }, "cpython": { "samples": [ - 26059625.0, - 25833541.0, - 26293625.0, - 26180125.0, - 26254000.0 + 25472750.0, + 25971291.0, + 25373834.0, + 25560291.0, + 25777625.0 ], - "mean_ns": 26124183.2, - "median_ns": 26180125.0, - "p95_ns": 26293625.0, - "stddev_ns": 185292.37689716218, - "max_rss_bytes": 14778368 + "mean_ns": 25631158.2, + "median_ns": 25560291.0, + "p95_ns": 25971291.0, + "stddev_ns": 241595.68855155507, + "max_rss_bytes": 14794752 }, "jit": null, - "ratio": 15.567365854823077, - "memory_ratio": 2.6485587583148558 + "ratio": 15.762687326212365, + "memory_ratio": 2.725359911406423 }, { "name": "attr_access", "work": 200000, "weavepy": { "samples": [ - 382875541.0, - 380636709.0, - 378449625.0, - 381505750.0, - 379034708.0 + 367171250.0, + 367365125.0, + 363811667.0, + 364459875.0, + 363846458.0 ], - "mean_ns": 380500466.6, - "median_ns": 380636709.0, - "p95_ns": 382875541.0, - "stddev_ns": 1804476.0079572408, - "max_rss_bytes": 39124992 + "mean_ns": 365330875.0, + "median_ns": 364459875.0, + "p95_ns": 367365125.0, + "stddev_ns": 1788524.600073899, + "max_rss_bytes": 40288256 }, "cpython": { "samples": [ - 28425792.0, - 28684083.0, - 28517416.0, - 28862416.0, - 28290542.0 + 29041792.0, + 29797250.0, + 30278500.0, + 28361291.0, + 28440042.0 ], - "mean_ns": 28556049.8, - "median_ns": 28517416.0, - "p95_ns": 28862416.0, - "stddev_ns": 223162.9481056387, - "max_rss_bytes": 14778368 + "mean_ns": 29183775.0, + "median_ns": 29041792.0, + "p95_ns": 30278500.0, + "stddev_ns": 840320.2185899134, + "max_rss_bytes": 14794752 }, "jit": null, - "ratio": 13.347517495975092, - "memory_ratio": 2.647450110864745 + "ratio": 12.549496773477339, + "memory_ratio": 2.723145071982281 }, { "name": "call_overhead", "work": 150000, "weavepy": { "samples": [ - 619760041.0, - 616979667.0, - 618924083.0, - 620181250.0, - 615324583.0 + 565252000.0, + 561231250.0, + 567430584.0, + 564165833.0, + 562661166.0 ], - "mean_ns": 618233924.8, - "median_ns": 618924083.0, - "p95_ns": 620181250.0, - "stddev_ns": 2039292.5715534787, - "max_rss_bytes": 39075840 + "mean_ns": 564148166.6, + "median_ns": 564165833.0, + "p95_ns": 567430584.0, + "stddev_ns": 2382886.773444303, + "max_rss_bytes": 40386560 }, "cpython": { "samples": [ - 47450750.0, - 46947833.0, - 46696125.0, - 44701000.0, - 44485250.0 + 43596125.0, + 43640125.0, + 44314208.0, + 42478417.0, + 42697083.0 ], - "mean_ns": 46056191.6, - "median_ns": 46696125.0, - "p95_ns": 47450750.0, - "stddev_ns": 1365076.3767058237, - "max_rss_bytes": 14860288 + "mean_ns": 43345191.6, + "median_ns": 43596125.0, + "p95_ns": 44314208.0, + "stddev_ns": 751712.8823126553, + "max_rss_bytes": 14778368 }, "jit": null, - "ratio": 13.254292149509194, - "memory_ratio": 2.62954796030871 + "ratio": 12.940733448213575, + "memory_ratio": 2.7328159645232817 }, { "name": "generators", "work": 300000, "weavepy": { "samples": [ - 514585417.0, - 514526291.0, - 522570000.0, - 514798542.0, - 515248875.0 + 458676958.0, + 458665958.0, + 463430250.0, + 458109709.0, + 457855416.0 ], - "mean_ns": 516345825.0, - "median_ns": 514798542.0, - "p95_ns": 522570000.0, - "stddev_ns": 3490969.733427733, - "max_rss_bytes": 51265536 + "mean_ns": 459347658.2, + "median_ns": 458665958.0, + "p95_ns": 463430250.0, + "stddev_ns": 2309838.453854988, + "max_rss_bytes": 52527104 }, "cpython": { "samples": [ - 30475041.0, - 30448083.0, - 29846458.0, - 30417166.0, - 30265084.0 + 30096000.0, + 30602250.0, + 30270458.0, + 30804708.0, + 29704541.0 ], - "mean_ns": 30290366.4, - "median_ns": 30417166.0, - "p95_ns": 30475041.0, - "stddev_ns": 261127.95699675666, - "max_rss_bytes": 14811136 + "mean_ns": 30295591.4, + "median_ns": 30270458.0, + "p95_ns": 30804708.0, + "stddev_ns": 431001.2179330819, + "max_rss_bytes": 14745600 }, "jit": null, - "ratio": 16.924605730856058, - "memory_ratio": 3.461283185840708 + "ratio": 15.15226356997968, + "memory_ratio": 3.562222222222222 }, { "name": "startup", "work": 1, "weavepy": { "samples": [ - 41355208.0, - 40978167.0, - 40937250.0, - 40938708.0, - 40930083.0 + 42679500.0, + 42519750.0, + 42793500.0, + 43124750.0, + 42647916.0 ], - "mean_ns": 41027883.2, - "median_ns": 40938708.0, - "p95_ns": 41355208.0, - "stddev_ns": 183946.1181941603, - "max_rss_bytes": 38682624 + "mean_ns": 42753083.2, + "median_ns": 42679500.0, + "p95_ns": 43124750.0, + "stddev_ns": 229504.2142558607, + "max_rss_bytes": 40042496 }, "cpython": { "samples": [ - 17133292.0, - 16873958.0, - 16807958.0, - 17167625.0, - 17800166.0 + 15619917.0, + 15703083.0, + 15963417.0, + 15672292.0, + 15639209.0 ], - "mean_ns": 17156599.8, - "median_ns": 17133292.0, - "p95_ns": 17800166.0, - "stddev_ns": 392517.4372473152, - "max_rss_bytes": 14647296 + "mean_ns": 15719583.6, + "median_ns": 15672292.0, + "p95_ns": 15963417.0, + "stddev_ns": 139961.6015798619, + "max_rss_bytes": 14663680 }, "jit": null, - "ratio": 2.389424519234249, - "memory_ratio": 2.640939597315436 + "ratio": 2.7232455852660222, + "memory_ratio": 2.7307262569832402 } ] } \ No newline at end of file diff --git a/crates/weavepy-compiler/src/bytecode.rs b/crates/weavepy-compiler/src/bytecode.rs index 8e74f75..92aa0b9 100644 --- a/crates/weavepy-compiler/src/bytecode.rs +++ b/crates/weavepy-compiler/src/bytecode.rs @@ -837,6 +837,34 @@ pub enum InlineCache { StoreSubscrListInt, /// `dict[key] = v`. StoreSubscrDict, + + // ----- RFC 0061 (WS2b): fused dispatch ----- + // + // A fusion marker lives in the *first* instruction's cache slot and + // means "the fall-through pair starting here may execute as one + // dispatch". The instruction stream is untouched (`co_code`, `dis`, + // line tables and jump targets cannot tell), a jump landing on the + // second instruction executes it normally, and the dispatcher only + // honours markers while no observer (trace/profile/monitoring) is + // active — under observation every instruction single-steps through + // the generic arms, so PEP 669 / `sys.settrace` event streams are + // bit-identical. + /// `LOAD_FAST a; LOAD_FAST b` — push two locals in one dispatch. + FuseLoadFastLoadFast, + /// `LOAD_FAST a; LOAD_CONST c` — local + materialized constant. + FuseLoadFastLoadConst, + /// `LOAD_FAST a; LOAD_ATTR n` — attribute of a local. The second + /// slot's own LOAD_ATTR cache supplies the specialization; the + /// fused arm reads the receiver *in place* (no clone onto the + /// operand stack, no Arc round-trip). + FuseLoadFastLoadAttr, + /// `COMPARE_OP (int, int); POP_JUMP_IF_{TRUE,FALSE}` — compare and + /// branch without materializing the intermediate `Bool`. Replaces + /// `CompareOpInt` on the compare's slot (its guards subsume it). + FuseCompareIntPopJump, + /// The dispatcher inspected this site once and found no fusable + /// pair; permanent (the fall-through successor never changes). + FuseBlocked, } /// Number of generic dispatches a deopted cache must serve before it diff --git a/crates/weavepy-compiler/src/lib.rs b/crates/weavepy-compiler/src/lib.rs index 832822f..7bc3be2 100644 --- a/crates/weavepy-compiler/src/lib.rs +++ b/crates/weavepy-compiler/src/lib.rs @@ -106,6 +106,43 @@ pub use weavepy_parser::ast::expr_name; // ---------- code object ---------- +/// RFC 0061 (WS2a): an opaque, VM-owned per-code-object extension slot. +/// +/// The VM stashes derived, execution-only state here (today: the +/// materialized constant-object table, so `LOAD_CONST` is an indexed +/// clone instead of a per-execution `Constant` deep-clone + conversion). +/// The compiler crate stays Object-free: the payload is type-erased and +/// only the VM ever downcasts it. +/// +/// Semantics mirror [`CacheTable`]: derived state does not follow +/// clones (a `replace()`d code object may change `constants`, so a +/// cloned code object starts with an empty slot), never participates in +/// equality, and is not serialized. +#[derive(Default)] +pub struct VmExt(pub std::sync::OnceLock>); + +impl Clone for VmExt { + fn clone(&self) -> Self { + Self::default() + } +} + +impl PartialEq for VmExt { + fn eq(&self, _other: &Self) -> bool { + true + } +} + +impl std::fmt::Debug for VmExt { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(if self.0.get().is_some() { + "VmExt(populated)" + } else { + "VmExt(empty)" + }) + } +} + /// A compiled Python code object. Mirrors the subset of /// `PyCodeObject` we need to emulate. #[derive(Debug, Clone, Default, PartialEq)] @@ -125,6 +162,8 @@ pub struct CodeObject { /// serialised by marshal (caches are re-warmed on the next run /// because the type pointers they capture wouldn't be valid). pub caches: CacheTable, + /// RFC 0061 (WS2a): VM-owned derived state (see [`VmExt`]). + pub vm_ext: VmExt, pub constants: Vec, /// Names referenced by `LOAD_NAME` / `LOAD_GLOBAL` / `STORE_NAME` etc. pub names: Vec, diff --git a/crates/weavepy-jit/src/analyze.rs b/crates/weavepy-jit/src/analyze.rs index 9e529dc..fff41f4 100644 --- a/crates/weavepy-jit/src/analyze.rs +++ b/crates/weavepy-jit/src/analyze.rs @@ -112,9 +112,27 @@ impl Plan { /// currently resolves to (the embedder re-validates every resolution as /// an entry guard). Returns the typed IR on success or a [`JitVerdict`] /// describing the first disqualifying property found. +/// +/// Convenience wrapper over [`analyze_with_probe`] with no pinned-list +/// probing (RFC 0061 WS5) — subscripted locals disqualify the frame. pub fn analyze( code: &CodeObject, resolve: &mut dyn FnMut(&str) -> ResolvedGlobal, +) -> Result { + analyze_with_probe(code, resolve, &mut |_| None) +} + +/// [`analyze`] with a pinned-list lane probe (RFC 0061 WS5). When a +/// local slot is subscripted before any other typing evidence exists, +/// `probe_list` reports the slot's *observed* shape in the requesting +/// activation — `Some(Int)`/`Some(Float)` for a homogeneous `int`/ +/// `float` list, `None` otherwise. A probed lane is only a prediction: +/// the embedder re-validates it as an entry guard on every native +/// entry, and the list helpers re-check shape per access. +pub fn analyze_with_probe( + code: &CodeObject, + resolve: &mut dyn FnMut(&str) -> ResolvedGlobal, + probe_list: &mut dyn FnMut(u32) -> Option, ) -> Result { if code.is_generator || code.is_coroutine || code.is_async_generator || code.is_class_body { return Err(JitVerdict::UnsupportedSignature); @@ -154,6 +172,7 @@ pub fn analyze( &mut local_types, &mut ret_lane, &mut changed, + probe_list, )?; } if !changed { @@ -699,10 +718,20 @@ fn infer_block( local_types: &mut [Option], ret_lane: &mut Option, changed: &mut bool, + probe_list: &mut dyn FnMut(u32) -> Option, ) -> Result<(), JitVerdict> { let mut stack: Vec = Vec::new(); for i in b.start..(b.end - 1) { - step_abstract(code, i, &mut stack, plan, local_types, *ret_lane, changed)?; + step_abstract( + code, + i, + &mut stack, + plan, + local_types, + *ret_lane, + changed, + probe_list, + )?; } // Terminator stack-shape validation. let last = b.end - 1; @@ -732,6 +761,11 @@ fn infer_block( return Err(JitVerdict::NonEmptyBoundaryStack); } let c = stack[0]; + // RFC 0061 WS5 — a pinned list's truth is its length, which + // the pin-index machine value cannot express. + if c.ty.is_list() { + return Err(JitVerdict::UnsupportedOpcode("truth test on list")); + } if !c.ty.is_representable() && c.src.is_none() { return Err(JitVerdict::TypeUnknown); } @@ -746,6 +780,7 @@ fn infer_block( local_types, *ret_lane, changed, + probe_list, )?; if !stack.is_empty() { return Err(JitVerdict::NonEmptyBoundaryStack); @@ -780,6 +815,7 @@ fn merge_ret_lane(ret_lane: &mut Option, ty: JitType, changed: &mut boo /// Abstract-execute one non-terminator instruction, updating the type /// stack and (via inference) `local_types`. +#[allow(clippy::too_many_arguments)] fn step_abstract( code: &CodeObject, i: usize, @@ -788,6 +824,7 @@ fn step_abstract( local_types: &mut [Option], ret_lane: Option, changed: &mut bool, + probe_list: &mut dyn FnMut(u32) -> Option, ) -> Result<(), JitVerdict> { let ins = code.instructions[i]; // RFC 0058 WS4 — rewritten range-loop pcs. @@ -968,11 +1005,102 @@ fn step_abstract( } stack.swap(len - 1, len - 2); } + // RFC 0061 WS5 — pinned-list element read. The container must + // be (or probe as) a homogeneous int/float list local; the + // index must be an `int`. + OpCode::BinarySubscr => { + let idx = stack.pop().ok_or(JitVerdict::StackUnderflow)?; + let cont = stack.pop().ok_or(JitVerdict::StackUnderflow)?; + if idx.callee.is_some() || cont.callee.is_some() { + return Err(JitVerdict::UnsupportedOpcode("CALL (callee escapes)")); + } + check_subscr_index(&idx, local_types, changed)?; + let elem = resolve_list_container(&cont, local_types, changed, probe_list)?; + stack.push(match elem { + Some(l) => SE::known(l), + None => SE { + ty: JitType::Unknown, + src: None, + callee: None, + }, + }); + } + // RFC 0061 WS5 — pinned-list element write. The stored value's + // lane must equal the pinned element lane exactly (a `bool` + // into an int list, say, would change list shape). + OpCode::StoreSubscr => { + let idx = stack.pop().ok_or(JitVerdict::StackUnderflow)?; + let cont = stack.pop().ok_or(JitVerdict::StackUnderflow)?; + let val = stack.pop().ok_or(JitVerdict::StackUnderflow)?; + if idx.callee.is_some() || cont.callee.is_some() || val.callee.is_some() { + return Err(JitVerdict::UnsupportedOpcode("CALL (callee escapes)")); + } + check_subscr_index(&idx, local_types, changed)?; + let elem = resolve_list_container(&cont, local_types, changed, probe_list)?; + if let Some(el) = elem { + if val.ty.is_representable() { + if val.ty != el { + return Err(JitVerdict::UnsupportedOpcode("STORE_SUBSCR (value lane)")); + } + } else if let Some(slot) = val.src { + set_local(local_types, slot, el, changed)?; + } + } + } other => return Err(JitVerdict::UnsupportedOpcode(other.name())), } Ok(()) } +/// RFC 0061 WS5 — validate a subscript index operand: a concrete lane +/// must be exactly `Int`; an untyped live-in load is inferred as `Int`; +/// a transient `Unknown` is tolerated for a later iteration. +fn check_subscr_index( + idx: &SE, + local_types: &mut [Option], + changed: &mut bool, +) -> Result<(), JitVerdict> { + if idx.ty.is_representable() { + if idx.ty != JitType::Int { + return Err(JitVerdict::UnsupportedOpcode("subscript index lane")); + } + Ok(()) + } else if let Some(slot) = idx.src { + set_local(local_types, slot, JitType::Int, changed) + } else { + Ok(()) + } +} + +/// RFC 0061 WS5 — resolve a subscript container operand to a pinned +/// element lane. An untyped local load consults the embedder's shape +/// probe and pins the slot; a concrete non-list lane disqualifies; +/// a transient `Unknown` yields `None` (tolerated during inference, +/// bailed at emission if never resolved). +fn resolve_list_container( + cont: &SE, + local_types: &mut [Option], + changed: &mut bool, + probe_list: &mut dyn FnMut(u32) -> Option, +) -> Result, JitVerdict> { + if let Some(el) = cont.ty.elem_lane() { + return Ok(Some(el)); + } + if cont.ty.is_representable() { + return Err(JitVerdict::UnsupportedOpcode("subscript container lane")); + } + if let Some(slot) = cont.src { + let Some(elem) = probe_list(slot) else { + return Err(JitVerdict::UnsupportedOpcode("subscript container shape")); + }; + let list_ty = + JitType::list_of(elem).ok_or(JitVerdict::UnsupportedOpcode("subscript elem lane"))?; + set_local(local_types, slot, list_ty, changed)?; + return Ok(Some(elem)); + } + Ok(None) +} + /// If exactly one operand is an untyped live-in load and the other is a /// concrete lane, infer the live-in's type. fn resolve_pair( @@ -1477,6 +1605,12 @@ fn emit_instr( if !ty.is_representable() { return Err(JitVerdict::TypeUnknown); } + // RFC 0061 WS5 — a pin index is meaningless outside this + // activation; a pinned list cannot be marshaled as a + // scalar call argument. + if ty.is_list() { + return Err(JitVerdict::UnsupportedOpcode("CALL (list argument)")); + } } let f = stack.pop().ok_or(JitVerdict::StackUnderflow)?; let Some(mark) = f.callee else { @@ -1489,6 +1623,12 @@ fn emit_instr( Some(t) if t.is_representable() => t, _ => return Err(JitVerdict::TypeUnknown), }; + // RFC 0061 WS5 — a pin index is only meaningful within its + // own activation's pinned-object table; a callee's returned + // list cannot cross the boundary as one. + if ret.is_list() { + return Err(JitVerdict::UnsupportedOpcode("CALL (list return)")); + } callee_spans.push(CalleeSpanMeta { token: mark.token, live_from: mark.load_pc, @@ -1532,6 +1672,35 @@ fn emit_instr( stack.swap(len - 1, len - 2); push(TOp::Swap2, None, stack, stmts); } + // RFC 0061 WS5 — pinned-list element read/write. Inference + // already pinned the container slot's lane; emission just + // re-validates the operand lanes it sees. + OpCode::BinarySubscr => { + let idx = pop_val(stack)?; + if idx != JitType::Int { + return Err(JitVerdict::UnsupportedOpcode("subscript index lane")); + } + let cont = pop_val(stack)?; + let elem = cont + .elem_lane() + .ok_or(JitVerdict::UnsupportedOpcode("subscript container lane"))?; + push(TOp::ListGet { elem }, Some(elem), stack, stmts); + } + OpCode::StoreSubscr => { + let idx = pop_val(stack)?; + if idx != JitType::Int { + return Err(JitVerdict::UnsupportedOpcode("subscript index lane")); + } + let cont = pop_val(stack)?; + let elem = cont + .elem_lane() + .ok_or(JitVerdict::UnsupportedOpcode("subscript container lane"))?; + let val = pop_val(stack)?; + if val != elem { + return Err(JitVerdict::UnsupportedOpcode("STORE_SUBSCR (value lane)")); + } + push(TOp::ListSet, None, stack, stmts); + } other => return Err(JitVerdict::UnsupportedOpcode(other.name())), } Ok(()) diff --git a/crates/weavepy-jit/src/engine.rs b/crates/weavepy-jit/src/engine.rs index 71278a4..d4ef359 100644 --- a/crates/weavepy-jit/src/engine.rs +++ b/crates/weavepy-jit/src/engine.rs @@ -16,7 +16,7 @@ use cranelift_frontend::FunctionBuilderContext; use cranelift_jit::{JITBuilder, JITModule}; use cranelift_module::{Linkage, Module}; -use crate::analyze::{analyze, JitVerdict}; +use crate::analyze::JitVerdict; use crate::ir::{CalleeSpanMeta, GlobalGuard, OsrEntry, RangeLoopMeta, ResolvedGlobal, TFunc}; use crate::lower::build_function; use crate::runtime::{self, JitFrame, JitStatus}; @@ -136,7 +136,19 @@ impl JitEngine { code: &CodeObject, resolve: &mut dyn FnMut(&str) -> ResolvedGlobal, ) -> Result { - let tfunc = analyze(code, resolve)?; + self.compile_with_probe(code, resolve, &mut |_| None) + } + + /// [`Self::compile`] with a pinned-list lane probe (RFC 0061 WS5): + /// `probe_list` reports the observed homogeneous element lane of a + /// subscripted local in the requesting activation. + pub fn compile_with_probe( + &mut self, + code: &CodeObject, + resolve: &mut dyn FnMut(&str) -> ResolvedGlobal, + probe_list: &mut dyn FnMut(u32) -> Option, + ) -> Result { + let tfunc = crate::analyze::analyze_with_probe(code, resolve, probe_list)?; self.compile_tfunc(&tfunc) } @@ -147,6 +159,22 @@ impl JitEngine { if !tfunc.callee_spans.is_empty() && runtime::call_py_helper_addr() == 0 { return Err(JitVerdict::UnsupportedOpcode("CALL (no helper registered)")); } + // RFC 0061 WS5 — likewise for the pinned-list helpers. + let has_list_ops = tfunc.blocks.iter().any(|b| { + b.stmts.iter().any(|s| { + matches!( + s.op, + crate::ir::TOp::ListGet { .. } | crate::ir::TOp::ListSet + ) + }) + }); + if has_list_ops + && (runtime::list_get_helper_addr() == 0 || runtime::list_set_helper_addr() == 0) + { + return Err(JitVerdict::UnsupportedOpcode( + "SUBSCR (no list helper registered)", + )); + } self.module.clear_context(&mut self.ctx); // Signature: (frame: ptr) -> i64. diff --git a/crates/weavepy-jit/src/ir.rs b/crates/weavepy-jit/src/ir.rs index 0751ea3..ab3a9d6 100644 --- a/crates/weavepy-jit/src/ir.rs +++ b/crates/weavepy-jit/src/ir.rs @@ -109,6 +109,19 @@ pub enum TOp { /// `ret` lane (or a caller guard invalidated by the callee's side /// effects) deopts *after* the call with the result spilled. CallPy { token: u32, argc: u8, ret: JitType }, + /// RFC 0061 WS5 — `BINARY_SUBSCR` on a pinned list: pops the `int` + /// index and the pin reference, calls the registered + /// `wpjit_list_get` helper (bounds + element-lane checked against + /// the real `Object::List`), and pushes the `elem`-lane result. + /// Any surprise (out of range, aliased shape change) deopts at + /// this pc with both operands spilled, so the interpreter + /// re-executes the subscript generically. + ListGet { elem: JitType }, + /// RFC 0061 WS5 — `STORE_SUBSCR` on a pinned list: pops the index, + /// the pin reference, and the value (staged through the frame's + /// `ret_bits`), and calls `wpjit_list_set`. Out-of-range deopts at + /// this pc; the interpreter re-executes the store and raises. + ListSet, } /// One IR statement: a [`TOp`] tagged with its originating bytecode pc @@ -311,6 +324,8 @@ impl TOp { | TOp::IntToFloatTos { guarded: true } | TOp::IntToFloatSecond { guarded: true } | TOp::CallPy { .. } + | TOp::ListGet { .. } + | TOp::ListSet ) } } diff --git a/crates/weavepy-jit/src/lib.rs b/crates/weavepy-jit/src/lib.rs index 8748217..e424942 100644 --- a/crates/weavepy-jit/src/lib.rs +++ b/crates/weavepy-jit/src/lib.rs @@ -29,14 +29,15 @@ mod lower; mod runtime; mod value; -pub use analyze::{analyze, JitVerdict}; +pub use analyze::{analyze, analyze_with_probe, JitVerdict}; pub use engine::{CompiledFrame, JitEngine}; pub use ir::{ ArithKind, BlockId, CalleeSpanMeta, CmpKind, GlobalGuard, OsrEntry, RangeLoopMeta, ResolvedGlobal, TBlock, TFunc, TOp, TStmt, TTerm, }; pub use runtime::{ - register_call_py_helper, CallPyHelper, CallStatus, JitFrame, JitStatus, SlotTag, + register_call_py_helper, register_list_helpers, CallPyHelper, CallStatus, JitFrame, JitStatus, + ListGetHelper, ListSetHelper, SlotTag, }; pub use value::JitType; diff --git a/crates/weavepy-jit/src/lower.rs b/crates/weavepy-jit/src/lower.rs index 9a99e31..57a1741 100644 --- a/crates/weavepy-jit/src/lower.rs +++ b/crates/weavepy-jit/src/lower.rs @@ -61,6 +61,9 @@ struct Lowerer<'a, 'b> { call_tags_base: Value, /// Imported signature of the `wpjit_call_py` helper (lazy). call_sig: Option, + /// Imported signature shared by the `wpjit_list_get`/`_set` helpers + /// (RFC 0061 WS5, lazy). + list_sig: Option, /// The abstract operand stack: SSA value + lane. vstack: Vec<(Value, JitType)>, } @@ -82,6 +85,7 @@ impl<'a, 'b> Lowerer<'a, 'b> { call_args_base: dummy, call_tags_base: dummy, call_sig: None, + list_sig: None, vstack: Vec::new(), } } @@ -98,6 +102,7 @@ impl<'a, 'b> Lowerer<'a, 'b> { JitType::Int => SlotTag::Int as i64, JitType::Float => SlotTag::Float as i64, JitType::Bool => SlotTag::Bool as i64, + JitType::ListInt | JitType::ListFloat => SlotTag::ListPin as i64, JitType::Unknown => SlotTag::Int as i64, } } @@ -335,9 +340,86 @@ impl<'a, 'b> Lowerer<'a, 'b> { self.emit_int_to_float(depth, guarded, stmt.pc); } TOp::CallPy { token, argc, ret } => self.emit_call_py(token, argc, ret, stmt.pc), + TOp::ListGet { elem } => self.emit_list_get(elem, stmt.pc), + TOp::ListSet => self.emit_list_set(stmt.pc), } } + /// RFC 0061 WS5 — pinned-list element read via the registered + /// `wpjit_list_get` helper. A non-zero status deopts at this pc + /// with both operands spilled (the pin reference rebuilds into the + /// real list object through its [`SlotTag::ListPin`] tag), so the + /// interpreter re-executes the subscript — and raises the exact + /// `IndexError`/`TypeError` itself when warranted. + fn emit_list_get(&mut self, elem: JitType, pc: u32) { + let trusted = MemFlags::trusted(); + let snapshot = self.vstack.clone(); + let (idx, _) = self.pop(); + let (pin, _) = self.pop(); + let sig = self.list_helper_sig(); + let helper = self + .b + .ins() + .iconst(self.ptr_ty, runtime::list_get_helper_addr() as i64); + let call = self + .b + .ins() + .call_indirect(sig, helper, &[self.frame_ptr, pin, idx]); + let status = self.b.inst_results(call)[0]; + let bad = self.b.ins().icmp_imm(IntCC::NotEqual, status, 0); + let cont = self.guard(bad, pc, &snapshot); + self.b.switch_to_block(cont); + let res = self + .b + .ins() + .load(Self::cl_ty(elem), trusted, self.frame_ptr, OFF_RET_BITS); + self.vstack.push((res, elem)); + } + + /// RFC 0061 WS5 — pinned-list element write. The value is staged + /// through `ret_bits` (dead between calls, same trick as the call + /// helper's out-slot) so one helper signature serves both ops. + fn emit_list_set(&mut self, pc: u32) { + let trusted = MemFlags::trusted(); + let snapshot = self.vstack.clone(); + let (idx, _) = self.pop(); + let (pin, _) = self.pop(); + let (val, _) = self.pop(); + // Typed store: an F64 value lands as its bit pattern. + self.b + .ins() + .store(trusted, val, self.frame_ptr, OFF_RET_BITS); + let sig = self.list_helper_sig(); + let helper = self + .b + .ins() + .iconst(self.ptr_ty, runtime::list_set_helper_addr() as i64); + let call = self + .b + .ins() + .call_indirect(sig, helper, &[self.frame_ptr, pin, idx]); + let status = self.b.inst_results(call)[0]; + let bad = self.b.ins().icmp_imm(IntCC::NotEqual, status, 0); + let cont = self.guard(bad, pc, &snapshot); + self.b.switch_to_block(cont); + } + + /// The shared `(frame, pin, idx) -> status` signature of the + /// pinned-list helpers (RFC 0061 WS5, lazy). + fn list_helper_sig(&mut self) -> SigRef { + if let Some(sig) = self.list_sig { + return sig; + } + let mut sig = Signature::new(self.b.func.signature.call_conv); + sig.params.push(AbiParam::new(self.ptr_ty)); // frame + sig.params.push(AbiParam::new(types::I64)); // pin + sig.params.push(AbiParam::new(types::I64)); // idx + sig.returns.push(AbiParam::new(types::I64)); // status + let r = self.b.import_signature(sig); + self.list_sig = Some(r); + r + } + /// Lower a native Python-to-Python call (RFC 0059 WS3): marshal the /// arguments, write back the managed locals (the callee may observe /// the caller frame), call the registered `wpjit_call_py` helper, diff --git a/crates/weavepy-jit/src/runtime.rs b/crates/weavepy-jit/src/runtime.rs index e15cb5c..a07a5b5 100644 --- a/crates/weavepy-jit/src/runtime.rs +++ b/crates/weavepy-jit/src/runtime.rs @@ -88,6 +88,10 @@ pub enum SlotTag { /// embedder's side channel (a native call's unrepresentable /// result). Only ever appears in a deopt spill, never in locals. Boxed = 3, + /// RFC 0061 WS5 — the value is an index into the embedder's + /// per-entry pinned-object table (a pinned `list`). The embedder + /// rebuilds the real object from the table on deopt/return. + ListPin = 4, } impl SlotTag { @@ -99,6 +103,7 @@ impl SlotTag { 1 => SlotTag::Float, 2 => SlotTag::Bool, 3 => SlotTag::Boxed, + 4 => SlotTag::ListPin, _ => SlotTag::Int, } } @@ -203,3 +208,44 @@ pub fn register_call_py_helper(helper: CallPyHelper) { pub(crate) fn call_py_helper_addr() -> usize { CALL_PY_HELPER.load(std::sync::atomic::Ordering::Acquire) } + +/// RFC 0061 WS5 — the embedder's pinned-list *read* helper. `pin` +/// indexes the per-entry pinned-object table on the embedder context; +/// `idx` is the (possibly negative) Python index. Returns `0` (Ok) with +/// the element's bits written into [`JitFrame::ret_bits`], or non-zero +/// when the access must deopt (out of range, or the element no longer +/// matches the pinned lane — aliased mutation through a callee). +/// +/// # Safety contract (for implementors) +/// +/// Same as [`CallPyHelper`]: `frame`/`ctx` are the live buffers of the +/// current native activation. The helper must not run Python code and +/// must not unwind across the FFI boundary. +pub type ListGetHelper = unsafe extern "C" fn(frame: *mut JitFrame, pin: i64, idx: i64) -> i64; + +/// RFC 0061 WS5 — the embedder's pinned-list *write* helper. The value +/// to store is pre-staged in [`JitFrame::ret_bits`] (interpreted per +/// the pin's element lane); returns `0` (Ok) or non-zero to deopt +/// (out of range). Same safety contract as [`ListGetHelper`]. +pub type ListSetHelper = unsafe extern "C" fn(frame: *mut JitFrame, pin: i64, idx: i64) -> i64; + +static LIST_GET_HELPER: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); +static LIST_SET_HELPER: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); + +/// Register the process-wide pinned-list helpers (RFC 0061 WS5). Must +/// precede the first compile of a frame containing list ops; later +/// calls must pass the same functions. +pub fn register_list_helpers(get: ListGetHelper, set: ListSetHelper) { + LIST_GET_HELPER.store(get as usize, std::sync::atomic::Ordering::Release); + LIST_SET_HELPER.store(set as usize, std::sync::atomic::Ordering::Release); +} + +#[must_use] +pub(crate) fn list_get_helper_addr() -> usize { + LIST_GET_HELPER.load(std::sync::atomic::Ordering::Acquire) +} + +#[must_use] +pub(crate) fn list_set_helper_addr() -> usize { + LIST_SET_HELPER.load(std::sync::atomic::Ordering::Acquire) +} diff --git a/crates/weavepy-jit/src/value.rs b/crates/weavepy-jit/src/value.rs index 564fc3c..7ad3870 100644 --- a/crates/weavepy-jit/src/value.rs +++ b/crates/weavepy-jit/src/value.rs @@ -22,6 +22,15 @@ pub enum JitType { /// CPython `bool`. Distinct from `Int` so the VM rebuilds the right /// `Object` variant on deopt; arithmetic promotes it to `Int` first. Bool, + /// RFC 0061 WS5 — a *pinned* `list` of `int` elements. The machine + /// value is an `i64` index into the embedder's per-entry pinned- + /// object table (no pointer ever crosses the JIT boundary); element + /// access goes through the registered `wpjit_list_get`/`_set` + /// helpers, which re-validate shape per access and deopt on any + /// surprise. + ListInt, + /// RFC 0061 WS5 — a pinned `list` of `float` elements. + ListFloat, /// Anything the JIT can't represent. Its presence as an operand to a /// supported opcode makes the enclosing region non-JITable. Unknown, @@ -43,6 +52,36 @@ impl JitType { matches!(self, JitType::Int | JitType::Bool) } + /// `true` for a pinned-list lane (RFC 0061 WS5). + #[inline] + #[must_use] + pub fn is_list(self) -> bool { + matches!(self, JitType::ListInt | JitType::ListFloat) + } + + /// A pinned list's element lane, or `None` for non-list lanes. + #[inline] + #[must_use] + pub fn elem_lane(self) -> Option { + match self { + JitType::ListInt => Some(JitType::Int), + JitType::ListFloat => Some(JitType::Float), + _ => None, + } + } + + /// The pinned-list lane for an element lane (inverse of + /// [`Self::elem_lane`]). + #[inline] + #[must_use] + pub fn list_of(elem: JitType) -> Option { + match elem { + JitType::Int => Some(JitType::ListInt), + JitType::Float => Some(JitType::ListFloat), + _ => None, + } + } + /// Dataflow join at a control-flow merge. Two equal types join to /// themselves; everything else collapses to [`JitType::Unknown`]. /// `Bool`/`Int` are kept distinct (they join to `Unknown`) so a slot diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index 0de4551..bd2e7f8 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -164,6 +164,14 @@ pub struct TrackedHandle { /// makes the filter admit *more* objects to the precise check — never /// fewer — so it can never cause a dead object to be missed. pub weak_clones: AtomicUsize, + /// RFC 0061 (WS1b): set (never cleared) when this handle leaves the + /// GC index (`untrack_id`, the collection rebuild's White purge). The + /// prompt-reap suspect probe reads it instead of a per-entry + /// `is_tracked` registry lookup — that lookup (`GcState::handle_for`) + /// was the hottest non-dispatch symbol on drop-heavy profiles. A + /// re-tracked object gets a *fresh* handle, so a set flag + /// definitively means "this handle is dead". + pub untracked: AtomicBool, } #[allow(non_upper_case_globals)] @@ -186,6 +194,7 @@ impl TrackedHandle { finalized: AtomicBool::new(false), finalize_queued: AtomicBool::new(false), weak_clones: AtomicUsize::new(0), + untracked: AtomicBool::new(false), } } } @@ -523,6 +532,9 @@ impl GcState { let Some(handle) = self.index.borrow_mut().remove(&id) else { return; }; + // RFC 0061 (WS1b): let suspect probes see the removal without a + // registry lookup. + handle.untracked.store(true, Ordering::Release); // Purge any suspect-list clone of this handle in lock-step: the // suspect entry shares the same `Arc`, so dropping // the index's Arc alone would leave the handle's strong `object` @@ -1907,6 +1919,9 @@ impl GcState { let color = h.color.load(Ordering::Acquire); if color == color::White { index.remove(&h.id); + // RFC 0061 (WS1b): mirror `untrack_id`'s flag so a stale + // suspect entry self-identifies as reclaimed. + h.untracked.store(true, Ordering::Release); if fin.remove(&h.id).is_some() { self.finalizable_count.fetch_sub(1, Ordering::AcqRel); } @@ -2865,8 +2880,13 @@ const SUSPECT_BUDGET: u8 = 64; /// Task machinery lets go, and with eager eviction the web stayed pinned /// by its collector handle until a full `gc.collect()`.) const DORMANT_STRIDE: u64 = 64; -static SUSPECTS: parking_lot::Mutex, u8)>> = - parking_lot::Mutex::new(Vec::new()); +/// RFC 0061 (WS1b): keyed by [`ObjectId`] so `remove_suspect` (called in +/// lock-step with every `untrack_id`) and enrollment dedup are O(1) +/// map hits instead of linear scans of the list — `remove_suspect`'s +/// `retain` was a measurable share of drop-heavy profiles (`list_ops`). +type SuspectMap = indexmap::IndexMap, u8)>; +static SUSPECTS: std::sync::LazyLock> = + std::sync::LazyLock::new(|| parking_lot::Mutex::new(SuspectMap::new())); static SUSPECT_COUNT: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); /// Entries with probe budget remaining. When only dormant entries are /// left, [`has_suspects`] admits a sweep every [`DORMANT_STRIDE`]-th @@ -2874,10 +2894,13 @@ static SUSPECT_COUNT: std::sync::atomic::AtomicUsize = std::sync::atomic::Atomic static SUSPECT_ACTIVE: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); static SUSPECT_TICK: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); -/// Recompute the count gates from the locked suspect list. -fn publish_suspect_counts(s: &[(Arc, u8)]) { +/// Recompute the count gates from the locked suspect map. +fn publish_suspect_counts(s: &SuspectMap) { SUSPECT_COUNT.store(s.len(), Ordering::Release); - SUSPECT_ACTIVE.store(s.iter().filter(|(_, b)| *b > 0).count(), Ordering::Release); + SUSPECT_ACTIVE.store( + s.values().filter(|(_, b)| *b > 0).count(), + Ordering::Release, + ); } /// Enroll a cascade-skipped tracked object for later deadness re-probes. @@ -2889,7 +2912,7 @@ pub fn note_suspect(h: Arc) { return; } let mut s = SUSPECTS.lock(); - if s.iter().any(|(e, _)| e.id == h.id) { + if s.contains_key(&h.id) { return; } // Refresh the handle's cached weakref-clone upper bound once at @@ -2913,18 +2936,18 @@ pub fn note_suspect(h: Arc) { // re-probed, pinning the Timeout→Task→frame web the test_ssl // leak tests watch). match s - .iter() + .values() .enumerate() .min_by_key(|(_, (_, b))| *b) .map(|(i, _)| i) { Some(i) => { - s.swap_remove(i); + s.swap_remove_index(i); } None => return, } } - s.push((h, SUSPECT_BUDGET)); + s.insert(h.id, (h, SUSPECT_BUDGET)); publish_suspect_counts(&s); } @@ -2960,8 +2983,9 @@ pub fn remove_suspect(id: ObjectId) { return; } let mut s = SUSPECTS.lock(); - s.retain(|(h, _)| h.id != id); - publish_suspect_counts(&s); + if s.swap_remove(&id).is_some() { + publish_suspect_counts(&s); + } } /// Re-probe the suspect list: return the objects that are now dead in @@ -2977,12 +3001,14 @@ pub fn take_dead_suspects() -> Vec { || SUSPECT_TICK .fetch_add(1, Ordering::Relaxed) .is_multiple_of(DORMANT_STRIDE); - s.retain_mut(|(h, budget)| { + s.retain(|_, (h, budget)| { // Dormant (aged-out) entries only pay on the stride tick. if *budget == 0 && !probe_dormant { return true; } - if !is_tracked(h.id) { + // RFC 0061 (WS1b): the handle self-identifies as reclaimed (set + // in lock-step with every index removal) — no registry lookup. + if h.untracked.load(Ordering::Acquire) { return false; // already reclaimed elsewhere } // Fast reject via the cached weakref-clone upper bound (refreshed diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 0cbfc7b..40ef21f 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -194,6 +194,13 @@ struct Frame { /// for later resumptions. Set by `generator_send` when unparking a /// `Created` frame; consumed (reset) by `run_frame`'s entry event. gen_first_resume: bool, + /// RFC 0061 (WS3c): the generator-family activation's `FrameShell`, + /// kept across suspensions. A suspended frame survives inside the + /// generator object, so its shell — whose immutable fields (code, + /// locals, globals, …) never change for the same frame — can be + /// re-pushed on resume instead of rebuilt (eight `Arc` clones per + /// `next()`). `None` for ordinary frames, which run exactly once. + shell_cache: Option>, } impl Frame { @@ -667,6 +674,19 @@ pub struct Interpreter { /// Recycled operand-stack vectors, same motivation. The operand /// stack is never shared, so every returned frame donates one. frame_stack_pool: RefCell>>, + /// RFC 0061 (WS3b) — recycled argument-staging vectors, kept apart + /// from [`Self::frame_stack_pool`]: staging vectors grow only to + /// the hottest call's argc, and handing such a small allocation + /// out as an operand stack would make every push in the new frame + /// re-grow it (the profile's `finish_grow` storm). + scratch_pool: RefCell>>, + /// RFC 0061 (WS3a) — recycled [`FrameShell`](crate::object::FrameShell) + /// allocations. Every Python call used to `Arc`-allocate a fresh + /// shell for the call-stack spine; a completed activation whose + /// shell nobody else observed (no materialised `PyFrame`, not a + /// generator's) donates the allocation back. Parked shells hold + /// only shared placeholder fields, so they pin no user objects. + frame_shell_pool: RefCell>>, /// RFC 0059 (WS2) — cooperative GIL hand-off countdown. A plain /// field (no atomic): only the GIL holder executes bytecode, and /// one interpreter drives one thread's frames, so a register @@ -675,6 +695,12 @@ pub struct Interpreter { /// chains keep the ~[`crate::gil::GIL_CHECK_INTERVAL`]-opcode /// checkpoint cadence. gil_countdown: u32, + /// RFC 0061 (WS2b) — set while any observer (trace/profile/PEP 669 + /// tool) is active on this thread. Fused-dispatch arms check it and + /// fall back to single-step semantics, so instrumentation sees the + /// exact per-instruction event stream. Refreshed from the dispatch + /// loop's [`crate::trace::ObserverSnapshot`] every iteration. + fuse_off: bool, } impl Default for Interpreter { @@ -847,8 +873,11 @@ impl Default for Interpreter { globals_missing_hooks: RefCell::new(Vec::new()), internal_import_depth: 0, frame_locals_pool: RefCell::new(Vec::new()), + frame_shell_pool: RefCell::new(Vec::new()), frame_stack_pool: RefCell::new(Vec::new()), + scratch_pool: RefCell::new(Vec::new()), gil_countdown: crate::gil::GIL_CHECK_INTERVAL, + fuse_off: false, }; // RFC 0025: publish the shared parts of this interpreter // (builtins / module cache / stdout / hooks) so workers @@ -944,8 +973,11 @@ impl Interpreter { globals_missing_hooks: RefCell::new(Vec::new()), internal_import_depth: 0, frame_locals_pool: RefCell::new(Vec::new()), + frame_shell_pool: RefCell::new(Vec::new()), frame_stack_pool: RefCell::new(Vec::new()), + scratch_pool: RefCell::new(Vec::new()), gil_countdown: crate::gil::GIL_CHECK_INTERVAL, + fuse_off: false, } } @@ -3796,26 +3828,66 @@ impl Interpreter { /// builtins mapping when the caller already has one — a Python /// function's cached `func_builtins` — and `None` for module / /// exec / class-body frames, which resolve from `globals` here. - /// Fetch a recycled fast-locals storage sized to `n` slots (all - /// `Unbound`), or allocate a fresh one on pool miss (RFC 0058 WS2). - fn pooled_locals(&self, n: usize) -> Rc>> { + /// Fetch a recycled operand-stack vector, or allocate one. + fn pooled_stack(&self) -> Vec { + self.frame_stack_pool + .borrow_mut() + .pop() + .unwrap_or_else(|| Vec::with_capacity(16)) + } + + /// RFC 0061 (WS3b): args-first locals fill. The arguments *move* + /// into the leading slots and only the residual tail is written + /// `Unbound` — replacing the full `Unbound` fill that every + /// argument slot then overwrote (double write + drop per slot). + /// `args` is drained but keeps its allocation, so the caller can + /// hand the empty vector back through [`Self::recycle_scratch`]. + fn pooled_locals_from_args( + &self, + args: &mut Vec, + n: usize, + ) -> Rc>> { + debug_assert!(args.len() <= n); if let Some(rc) = self.frame_locals_pool.borrow_mut().pop() { { let mut v = rc.borrow_mut(); debug_assert!(v.is_empty()); + v.append(args); v.resize(n, Object::Unbound); } return rc; } - Rc::new(RefCell::new(vec![Object::Unbound; n])) + let mut v = Vec::with_capacity(n); + v.append(args); + v.resize(n, Object::Unbound); + Rc::new(RefCell::new(v)) } - /// Fetch a recycled operand-stack vector, or allocate one. - fn pooled_stack(&self) -> Vec { - self.frame_stack_pool + /// RFC 0061 (WS3b): fetch a recycled argument-staging vector, or + /// allocate one. Deliberately a separate pool from + /// [`Self::pooled_stack`] — see the `scratch_pool` field docs. + fn pooled_scratch(&self) -> Vec { + self.scratch_pool .borrow_mut() .pop() - .unwrap_or_else(|| Vec::with_capacity(16)) + .unwrap_or_else(|| Vec::with_capacity(8)) + } + + /// RFC 0061 (WS3b): return a drained argument-staging vector to + /// the scratch pool instead of freeing it — the profile's per-call + /// `malloc`/`free` pair for the `CALL` operand split. + fn recycle_scratch(&self, mut v: Vec) { + const POOL_CAP: usize = 64; + if v.capacity() == 0 { + return; + } + // Drop leftover values *before* borrowing the pool: their drop + // glue can run arbitrary code that must not observe the borrow. + v.clear(); + let mut pool = self.scratch_pool.borrow_mut(); + if pool.len() < POOL_CAP { + pool.push(v); + } } /// Return a finished frame's heap allocations to the pools. Called @@ -3857,15 +3929,14 @@ impl Interpreter { globals: Rc>, builtins: Option>>, ) -> Frame { - let locals_rc = self.pooled_locals(code.varnames.len()); - { - let mut locals = locals_rc.borrow_mut(); - for (i, v) in positional.into_iter().enumerate() { - if i < locals.len() { - locals[i] = v; - } - } - } + // RFC 0061 (WS3b): args-first fill; extras past the declared + // slots were never stored by the old slot-by-slot loop either + // (binding errors raise before frames are built). + let mut positional = positional; + let nlocals = code.varnames.len(); + positional.truncate(nlocals); + let locals_rc = self.pooled_locals_from_args(&mut positional, nlocals); + self.recycle_scratch(positional); // Build cells: cellvars come first (fresh), then freevars // (provided by the caller via `closure`). let mut cells: Vec>> = @@ -3936,6 +4007,7 @@ impl Interpreter { cleanup_lasti: None, suppress_call_event: false, gen_first_resume: false, + shell_cache: None, } } @@ -4049,6 +4121,12 @@ impl Interpreter { // only when something introspects it — tracing, `sys._getframe`, // traceback capture — via `ensure_top_py_frame`. let shell = self.push_frame_shell(frame); + // RFC 0061 (WS3c): a generator-family frame outlives this + // activation inside the generator object — pin its shell there + // so the next resume re-pushes it instead of rebuilding. + if shell.is_gen && frame.shell_cache.is_none() { + frame.shell_cache = Some(shell.clone()); + } let mut py_frame_slot: Option> = shell.materialized.borrow().clone(); // Depth of the interpreter-wide handled-exception stack on entry. // Used on completion to discard any `PUSH_EXC_INFO` entries this @@ -4130,6 +4208,7 @@ impl Interpreter { match crate::tier2::try_enter(self, frame) { crate::tier2::JitEntry::Ran(v) => { self.pop_frame_shell(); + self.recycle_frame_shell(shell); return Ok(FrameOutcome::Returned(v)); } // RFC 0059 WS3 — a native Python-to-Python call raised. @@ -4165,6 +4244,12 @@ impl Interpreter { // computed instrumented set (`initialize_lines` port). let mut prev_pc: Option = None; let mut line_starts: Option> = None; + // RFC 0061 (WS1a): generation-cached observer facts. The two + // per-instruction consumers below (the line/opcode-event gate and + // the BRANCH/JUMP mask test) used to walk thread-locals and fold + // the monitoring tool table on every instruction; the snapshot + // re-derives only when an observer source actually changes. + let mut obs = crate::trace::ObserverSnapshot::new(); let result = loop { // RFC 0039 (WS2) / RFC 0059 (WS2): cooperative GIL hand-off. // A plain interpreter-local countdown — one register @@ -4316,7 +4401,10 @@ impl Interpreter { // `?`-escaping the whole frame. let mut trace_err: Option = None; let cur_pc = frame.pc as usize; - if crate::trace::any_observers_active() && !is_gen_bootstrap { + obs.refresh(); + // RFC 0061 (WS2b): fused arms single-step under observation. + self.fuse_off = obs.any; + if obs.any && !is_gen_bootstrap { let py_frame = self.ensure_top_py_frame(&mut py_frame_slot); py_frame.lasti.set(frame.pc); let line = py_frame.current_lineno(); @@ -4403,9 +4491,7 @@ impl Interpreter { if trace_err.is_none() && !at_resume && !py_frame.trace_opcodes.get() - && crate::trace::monitoring_union_mask() - & crate::trace::event_mask(crate::trace::EVENT_INSTRUCTION) - != 0 + && obs.mon_mask & crate::trace::event_mask(crate::trace::EVENT_INSTRUCTION) != 0 { // The wire encoder fuses LOAD_CONST + RETURN_VALUE // into one RETURN_CONST unit; the pair shares one @@ -4559,10 +4645,7 @@ impl Interpreter { // resolved destination (fall-through or target). let bj_mask = crate::trace::event_mask(crate::trace::EVENT_BRANCH) | crate::trace::event_mask(crate::trace::EVENT_JUMP); - if instruction_ran - && crate::trace::monitoring_union_mask() & bj_mask != 0 - && !is_gen_bootstrap - { + if instruction_ran && obs.mon_mask & bj_mask != 0 && !is_gen_bootstrap { let op = frame.code.instructions.get(cur_pc).map(|i| i.op); let ev = match op { Some( @@ -4700,7 +4783,7 @@ impl Interpreter { // EXCEPTION_HANDLED event (`handle_exception` // consumes `exc`). Only cloned on the observed // path. - let handled_arg = if crate::trace::monitoring_union_mask() + let handled_arg = if crate::trace::monitoring_union_mask_cached() & crate::trace::event_mask(crate::trace::EVENT_EXCEPTION_HANDLED) != 0 { @@ -4854,6 +4937,9 @@ impl Interpreter { } } } + // RFC 0061 (WS3a): the spine popped its clone above and every + // local use of the shell is done — park the allocation. + self.recycle_frame_shell(shell); result } @@ -4865,6 +4951,35 @@ impl Interpreter { /// refreshed `lasti` — its `back` link is refreshed lazily by the /// next materialisation walk. fn push_frame_shell(&self, frame: &mut Frame) -> Rc { + // RFC 0061 (WS3c): a generator-family resume re-pushes the shell + // it built on a previous entry. The identity-bearing fields + // (code, locals, cells, globals, builtins, namespaces) cannot + // have changed for the same frame; only the per-entry bits are + // refreshed. `gen_owner` is re-derived because the very first + // push happens *before* `RETURN_GENERATOR` creates the owner. + if let Some(cached) = &frame.shell_cache { + let materialized = frame.py_frame.clone(); + if let Some(existing) = &materialized { + existing.lasti.set(frame.pc); + existing.invalidate_locals(); + existing.on_stack.set(existing.on_stack.get() + 1); + } + let gen_owner = materialized + .as_ref() + .and_then(|py| py.gen_owner.borrow().clone()) + .or_else(|| frame.gen_owner.clone()); + *cached.gen_owner.borrow_mut() = gen_owner; + cached + .lasti + .store(frame.pc, std::sync::atomic::Ordering::Relaxed); + cached + .has_materialized + .store(materialized.is_some(), std::sync::atomic::Ordering::Relaxed); + *cached.materialized.borrow_mut() = materialized; + let shell = cached.clone(); + self.frame_stack.borrow_mut().push(shell.clone()); + return shell; + } let is_gen = frame.code.is_generator || frame.code.is_coroutine || frame.code.is_async_generator; let materialized = frame.py_frame.clone(); @@ -4873,6 +4988,33 @@ impl Interpreter { existing.invalidate_locals(); existing.on_stack.set(existing.on_stack.get() + 1); } + let gen_owner = materialized + .as_ref() + .and_then(|py| py.gen_owner.borrow().clone()) + .or_else(|| frame.gen_owner.clone()); + // RFC 0061 (WS3a): reuse a recycled shell allocation when one is + // parked. The pool only ever holds sole-owner shells, so the + // `get_mut` is a formality — but a fresh allocation is always a + // correct fallback. + if let Some(mut pooled) = self.frame_shell_pool.borrow_mut().pop() { + if let Some(m) = Rc::get_mut(&mut pooled) { + m.code = frame.code.clone(); + m.locals = frame.locals.clone(); + m.cells = frame.cells.clone(); + m.globals = frame.globals.clone(); + m.builtins = frame.builtins.clone(); + m.builtins_obj = frame.builtins_obj.clone(); + m.class_namespace = frame.class_namespace.clone(); + m.class_namespace_obj = frame.class_namespace_obj.clone(); + m.is_gen = is_gen; + *m.gen_owner.borrow_mut() = gen_owner; + m.lasti = std::sync::atomic::AtomicU32::new(frame.pc); + m.has_materialized = std::sync::atomic::AtomicBool::new(materialized.is_some()); + *m.materialized.borrow_mut() = materialized; + self.frame_stack.borrow_mut().push(pooled.clone()); + return pooled; + } + } let shell = Rc::new(crate::object::FrameShell { code: frame.code.clone(), locals: frame.locals.clone(), @@ -4883,12 +5025,7 @@ impl Interpreter { class_namespace: frame.class_namespace.clone(), class_namespace_obj: frame.class_namespace_obj.clone(), is_gen, - gen_owner: RefCell::new( - materialized - .as_ref() - .and_then(|py| py.gen_owner.borrow().clone()) - .or_else(|| frame.gen_owner.clone()), - ), + gen_owner: RefCell::new(gen_owner), lasti: std::sync::atomic::AtomicU32::new(frame.pc), has_materialized: std::sync::atomic::AtomicBool::new(materialized.is_some()), materialized: RefCell::new(materialized), @@ -4897,6 +5034,38 @@ impl Interpreter { shell } + /// RFC 0061 (WS3a): park a completed activation's shell for reuse. + /// Only shells nothing else can observe qualify: not a generator's + /// (the generator object re-pushes it across suspensions), never + /// materialised (a `PyFrame`/traceback may hold it), and sole-owned + /// now that the spine popped its clone. Parked shells are scrubbed + /// to shared placeholders immediately — CPython frees the frame at + /// return, so nothing may stay pinned until reuse. + fn recycle_frame_shell(&self, mut shell: Rc) { + const FRAME_SHELL_POOL_CAP: usize = 64; + if shell.is_gen + || shell + .has_materialized + .load(std::sync::atomic::Ordering::Relaxed) + || self.frame_shell_pool.borrow().len() >= FRAME_SHELL_POOL_CAP + { + return; + } + if let Some(m) = Rc::get_mut(&mut shell) { + m.code = empty_code_placeholder(); + m.locals = empty_locals_placeholder(); + m.cells = crate::object::empty_cells(); + m.globals = empty_dict_placeholder(); + m.builtins = empty_dict_placeholder(); + m.builtins_obj = None; + m.class_namespace = None; + m.class_namespace_obj = None; + *m.gen_owner.borrow_mut() = None; + *m.materialized.borrow_mut() = None; + self.frame_shell_pool.borrow_mut().push(shell); + } + } + /// Push an already-materialised `PyFrame` (event dispatch around /// generator throw/unwind) onto the spine. fn push_materialized_frame(&self, py: &Rc) { @@ -5216,6 +5385,37 @@ impl Interpreter { )) } + /// RFC 0061 (WS2b): the body of a `LOAD_FAST` — read local `arg`, + /// clone it, raise `UnboundLocalError` on `Unbound`. Shared by the + /// plain arm and every fused arm so error semantics can't diverge. + #[inline] + fn load_fast_value(frame: &Frame, arg: u32) -> Result { + let v = frame + .locals + .borrow() + .get(arg as usize) + .cloned() + .ok_or_else(|| { + RuntimeError::Internal(format!( + "bad local index {} (code {}, nlocals {}, varname {:?})", + arg, + frame.code.name, + frame.locals.borrow().len(), + frame.code.varnames.get(arg as usize) + )) + })?; + if matches!(v, Object::Unbound) { + let name = frame + .code + .varnames + .get(arg as usize) + .cloned() + .unwrap_or_default(); + return Err(Self::unbound_local(&name)); + } + Ok(v) + } + /// Error for reading/deleting an empty cell. Cellvars are still /// local variables (UnboundLocalError); freevars get CPython's /// "free variable … in enclosing scope" NameError. @@ -5831,7 +6031,7 @@ impl Interpreter { from_pc: u32, to_pc: u32, ) -> Result<(), RuntimeError> { - if crate::trace::monitoring_union_mask() & crate::trace::event_mask(event_idx) == 0 { + if crate::trace::monitoring_union_mask_cached() & crate::trace::event_mask(event_idx) == 0 { return Ok(()); } let args = [ @@ -5861,7 +6061,7 @@ impl Interpreter { event_idx: usize, arg: Object, ) -> Result<(), RuntimeError> { - if crate::trace::monitoring_union_mask() & crate::trace::event_mask(event_idx) == 0 { + if crate::trace::monitoring_union_mask_cached() & crate::trace::event_mask(event_idx) == 0 { return Ok(()); } let pc = py_frame.lasti.get(); @@ -5935,13 +6135,24 @@ impl Interpreter { match ins.op { OpCode::Nop | OpCode::Resume => {} OpCode::LoadConst => { - let c = frame - .code - .constants - .get(ins.arg as usize) - .ok_or_else(|| RuntimeError::Internal("bad const index".to_owned()))? - .clone(); - frame.push(constant_to_object(c)); + // RFC 0061 (WS2a): indexed clone from the per-code + // materialized table — no `Constant` deep-clone, no + // conversion, no string re-allocation per execution. + let table = code_const_objects(&frame.code); + let v = match table.get(ins.arg as usize) { + Some(v) => v.clone(), + None => { + // Generic-slot fallback (see `code_const_objects`). + let c = frame + .code + .constants + .get(ins.arg as usize) + .ok_or_else(|| RuntimeError::Internal("bad const index".to_owned()))? + .clone(); + constant_to_object(c) + } + }; + frame.push(v); } OpCode::LoadName => { let mut name = self.name_at(&frame.code, ins.arg)?; @@ -5975,30 +6186,127 @@ impl Interpreter { // UnboundLocalError (CPython's NULL fast-local check). The // `name` attribute feeds NameError suggestion logic. OpCode::LoadFast => { - let v = frame - .locals - .borrow() - .get(ins.arg as usize) - .cloned() - .ok_or_else(|| { - RuntimeError::Internal(format!( - "bad local index {} (code {}, nlocals {}, varname {:?})", - ins.arg, - frame.code.name, - frame.locals.borrow().len(), - frame.code.varnames.get(ins.arg as usize) - )) - })?; - if matches!(v, Object::Unbound) { - let name = frame - .code - .varnames - .get(ins.arg as usize) - .cloned() - .unwrap_or_default(); - return Err(Self::unbound_local(&name)); + // RFC 0061 (WS2b): fused dispatch. The site's cache slot + // (unused by plain LOAD_FAST) memoizes a one-time look at + // the fall-through successor; fused arms run both + // instructions in one dispatch — and the LOAD_ATTR fusion + // reads the receiver *in place*, skipping its Arc + // round-trip through the operand stack entirely. + use weavepy_compiler::InlineCache as IC; + match frame.code.caches.get(cache_pc) { + IC::FuseLoadFastLoadFast if !self.fuse_off => { + let a = Self::load_fast_value(frame, ins.arg)?; + frame.push(a); + // `pc` already points at the second LOAD_FAST; + // advance past it so a raise inside its read is + // attributed to *its* pc (the `pc - 1` convention). + let b_idx = frame + .code + .instructions + .get(frame.pc as usize) + .map(|i| i.arg) + .unwrap_or(u32::MAX); + frame.pc += 1; + let b = Self::load_fast_value(frame, b_idx)?; + frame.push(b); + } + IC::FuseLoadFastLoadConst if !self.fuse_off => { + let a = Self::load_fast_value(frame, ins.arg)?; + frame.push(a); + let c_idx = frame + .code + .instructions + .get(frame.pc as usize) + .map(|i| i.arg) + .unwrap_or(u32::MAX); + frame.pc += 1; + let table = code_const_objects(&frame.code); + let v = match table.get(c_idx as usize) { + Some(v) => v.clone(), + None => { + return Err(RuntimeError::Internal("bad const index".to_owned())) + } + }; + frame.push(v); + } + IC::FuseLoadFastLoadAttr if !self.fuse_off => { + // Try the borrowed-receiver hit against the + // LOAD_ATTR's own cache; any miss falls back to a + // plain LOAD_FAST (the attr executes — and manages + // its cache transitions — on the next dispatch). + let attr_pc = frame.pc; + let attr_ins = frame.code.instructions.get(attr_pc as usize).copied(); + let hit = match (frame.code.caches.get(attr_pc), attr_ins) { + ( + IC::LoadAttrInstance { + type_id, + key_idx, + ver, + }, + Some(attr_ins), + ) => { + let locals = frame.locals.borrow(); + match locals.get(ins.arg as usize) { + Some(Object::Instance(inst)) => { + let guard_ok = { + let cls = inst.class.borrow(); + specialize::rc_id(&cls) == type_id + && cls.attr_version.get() == ver + }; + if guard_ok { + let dict = inst.dict.borrow(); + match dict.get_index(key_idx as usize) { + Some((k, v)) + if self.cached_slot_name_matches( + &frame.code, + attr_ins.arg, + k, + ) => + { + Some(v.clone()) + } + _ => None, + } + } else { + None + } + } + _ => None, + } + } + _ => None, + }; + if let Some(v) = hit { + frame.pc += 1; + frame.push(v); + specialize::record_hit(OpCode::LoadAttr as u8); + } else { + let v = Self::load_fast_value(frame, ins.arg)?; + frame.push(v); + } + } + IC::Empty => { + // One-time fusion decision: the fall-through + // successor is static, so the answer never changes. + let marker = if fusion_enabled() { + match frame.code.instructions.get(frame.pc as usize).map(|i| i.op) { + Some(OpCode::LoadFast) => IC::FuseLoadFastLoadFast, + Some(OpCode::LoadConst) => IC::FuseLoadFastLoadConst, + Some(OpCode::LoadAttr) => IC::FuseLoadFastLoadAttr, + _ => IC::FuseBlocked, + } + } else { + IC::FuseBlocked + }; + frame.code.caches.set(cache_pc, marker); + let v = Self::load_fast_value(frame, ins.arg)?; + frame.push(v); + } + _ => { + let v = Self::load_fast_value(frame, ins.arg)?; + frame.push(v); + } } - frame.push(v); } OpCode::StoreFast => { let v = frame.pop()?; @@ -6850,7 +7158,7 @@ impl Interpreter { let kw_pairs: Vec<(String, Object)> = names.into_iter().zip(kw_values).collect(); // PEP 669 CALL for the keyword-call instruction. let mon_call = crate::trace::any_observers_active() - && crate::trace::monitoring_union_mask() + && crate::trace::monitoring_union_mask_cached() & crate::trace::event_mask(crate::trace::EVENT_CALL) != 0; if mon_call { @@ -6895,7 +7203,7 @@ impl Interpreter { // (same before-and-after gate as `dispatch_call`). if mon_call && !matches!(&callable, Object::Function(_)) - && crate::trace::monitoring_union_mask() + && crate::trace::monitoring_union_mask_cached() & crate::trace::event_mask(crate::trace::EVENT_CALL) != 0 { @@ -7095,7 +7403,7 @@ impl Interpreter { // PEP 669 CALL / C_RETURN / C_RAISE for // CALL_FUNCTION_EX, mirroring `dispatch_call`. let mon_call = crate::trace::any_observers_active() - && crate::trace::monitoring_union_mask() + && crate::trace::monitoring_union_mask_cached() & crate::trace::event_mask(crate::trace::EVENT_CALL) != 0; if mon_call { @@ -7111,7 +7419,7 @@ impl Interpreter { if mon_call && !matches!(&callable, Object::Function(_)) && !matches!(&callable, Object::BoundMethod(bm) if matches!(bm.function, Object::Function(_))) - && crate::trace::monitoring_union_mask() + && crate::trace::monitoring_union_mask_cached() & crate::trace::event_mask(crate::trace::EVENT_CALL) != 0 { @@ -18788,7 +19096,7 @@ impl Interpreter { // closed). The subiterator's own close above fires nothing, // matching CPython's no-resume close fast path. if crate::trace::any_observers_active() - && crate::trace::monitoring_union_mask() + && crate::trace::monitoring_union_mask_cached() & crate::trace::event_mask(crate::trace::EVENT_LINE) != 0 { @@ -20618,7 +20926,21 @@ impl Interpreter { _ => return Ok(false), }; specialize::record_specialize_attempt(op_idx); - let decision = specialize::attempt_specialize_compare_op(&a_peek, &b_peek, kind); + let mut decision = + specialize::attempt_specialize_compare_op(&a_peek, &b_peek, kind); + // RFC 0061 (WS2b): an int compare whose fall-through + // successor is POP_JUMP_IF_* upgrades straight to the + // fused compare-and-branch marker (guards are identical, + // so it strictly subsumes `CompareOpInt`). + if matches!(decision, IC::CompareOpInt) + && fusion_enabled() + && matches!( + frame.code.instructions.get(frame.pc as usize).map(|i| i.op), + Some(OpCode::PopJumpIfTrue) | Some(OpCode::PopJumpIfFalse) + ) + { + decision = IC::FuseCompareIntPopJump; + } frame.code.caches.set(cache_pc, decision); if matches!(decision, IC::Cooldown(_)) { specialize::record_specialize_skip(op_idx); @@ -20639,6 +20961,49 @@ impl Interpreter { specialize::record_hit(op_idx); Ok(true) } + IC::FuseCompareIntPopJump => { + // RFC 0061 (WS2b): compare-and-branch without the + // intermediate `Bool` touching the operand stack. Same + // operand guard as `CompareOpInt`; the branch itself is + // guard-free (the successor opcode is static). + let (a, b) = match (frame.peek_back(1), frame.peek_back(0)) { + (Some(Object::Int(x)), Some(Object::Int(y))) => (*x, *y), + _ => return self.deopt_compare_op(frame, cache_pc), + }; + let r = compare_int(a, b, kind); + let len = frame.stack.len(); + frame.stack.truncate(len - 2); + specialize::record_hit(op_idx); + if self.fuse_off { + // Single-step under observation: push the Bool and + // let the POP_JUMP arm run (and emit its branch + // event) on the next dispatch. + frame.push(Object::Bool(r)); + return Ok(true); + } + let succ = frame + .code + .instructions + .get(frame.pc as usize) + .copied() + .ok_or_else(|| { + RuntimeError::Internal("fused compare lost its jump".to_owned()) + })?; + frame.pc += 1; + let take = match succ.op { + OpCode::PopJumpIfTrue => r, + OpCode::PopJumpIfFalse => !r, + _ => { + return Err(RuntimeError::Internal( + "fused compare successor is not a POP_JUMP".to_owned(), + )) + } + }; + if take { + frame.pc += succ.arg; + } + Ok(true) + } IC::CompareOpFloat => { let (a, b) = match (frame.peek_back(1), frame.peek_back(0)) { (Some(Object::Float(x)), Some(Object::Float(y))) => (*x, *y), @@ -21072,63 +21437,80 @@ impl Interpreter { use weavepy_compiler::InlineCache as IC; let cache = frame.code.caches.get(cache_pc); let op_idx = OpCode::LoadAttr as u8; + // RFC 0061 (WS4): every arm ends with the receiver off the stack, so + // pop it once up front (a move) instead of the per-arm + // `top().clone()` + later `pop()` pair — two Arc round-trips per + // attribute access on the hottest opcode family. Guards likewise + // read the class through a borrow instead of `cls()`'s Arc clone. + let receiver = frame.pop()?; match cache { IC::LoadAttrInstance { type_id, key_idx, ver, } => { - let receiver = frame.top()?.clone(); if let Object::Instance(inst) = &receiver { - let cls = inst.cls(); - if specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver { - let dict = inst.dict.borrow(); - if let Some((k, v)) = dict.get_index(key_idx as usize) { - if self.cached_slot_name_matches(&frame.code, name_idx, k) { - let v = v.clone(); - drop(dict); - frame.pop()?; - specialize::record_hit(op_idx); - return Ok(v); + let guard_ok = { + let cls = inst.class.borrow(); + specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver + }; + if guard_ok { + let hit = { + let dict = inst.dict.borrow(); + match dict.get_index(key_idx as usize) { + Some((k, v)) + if self.cached_slot_name_matches(&frame.code, name_idx, k) => + { + Some(v.clone()) + } + _ => None, } + }; + if let Some(v) = hit { + specialize::record_hit(op_idx); + return Ok(v); } } } - self.deopt_load_attr_slow(frame, cache_pc, name_idx) + self.deopt_load_attr_taken(frame, cache_pc, name_idx, receiver) } IC::LoadAttrModule { module_id, key_idx } => { - let receiver = frame.top()?.clone(); if let Object::Module(m) = &receiver { if specialize::rc_id(&m.dict) == module_id { - let dict = m.dict.borrow(); - if let Some((k, v)) = dict.get_index(key_idx as usize) { - if self.cached_slot_name_matches(&frame.code, name_idx, k) { - let v = v.clone(); - drop(dict); - frame.pop()?; - specialize::record_hit(op_idx); - return Ok(v); + let hit = { + let dict = m.dict.borrow(); + match dict.get_index(key_idx as usize) { + Some((k, v)) + if self.cached_slot_name_matches(&frame.code, name_idx, k) => + { + Some(v.clone()) + } + _ => None, } + }; + if let Some(v) = hit { + specialize::record_hit(op_idx); + return Ok(v); } } } - self.deopt_load_attr_slow(frame, cache_pc, name_idx) + self.deopt_load_attr_taken(frame, cache_pc, name_idx, receiver) } IC::LoadAttrSlot { type_id, ver } => { - let receiver = frame.top()?.clone(); if let Object::Instance(inst) = &receiver { - let cls = inst.cls(); - if specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver { + let guard_ok = { + let cls = inst.class.borrow(); + specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver + }; + if guard_ok { // Read the slot side table directly; the class was // validated at specialisation time (stock lookup, a // genuine `__slots__` member descriptor for this // name) and the version guard covers class-dict / // MRO changes since. - let code = frame.code.clone(); - let name = code.names.get(name_idx as usize).map(String::as_str); + let name = frame.code.names.get(name_idx as usize).map(String::as_str); if let Some(name) = name { if let Some(v) = inst.slot_get(name) { - frame.pop()?; specialize::record_hit(op_idx); return Ok(v); } @@ -21137,7 +21519,7 @@ impl Interpreter { // AttributeError. } } - self.deopt_load_attr_slow(frame, cache_pc, name_idx) + self.deopt_load_attr_taken(frame, cache_pc, name_idx, receiver) } IC::LoadAttrMethod { type_id, @@ -21145,12 +21527,10 @@ impl Interpreter { mro_idx, key_idx, } => { - let receiver = frame.top()?.clone(); - if let Object::Instance(inst) = &receiver { - let cls = inst.cls(); + let func = if let Object::Instance(inst) = &receiver { + let cls = inst.class.borrow(); if specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver { - let code = frame.code.clone(); - let name = code.names.get(name_idx as usize).map(String::as_str); + let name = frame.code.names.get(name_idx as usize).map(String::as_str); if let Some(name) = name { // Per-instance shadow guard: an instance dict // entry beats a non-data descriptor. The @@ -21163,47 +21543,52 @@ impl Interpreter { !d.is_empty() && d.contains_key(&crate::object::StrKey(name)) }; if !shadowed { - let func = { - let mro = cls.mro.borrow(); - mro.get(mro_idx as usize).and_then(|owner| { - let od = owner.dict.borrow(); - od.get_index(key_idx as usize).and_then(|(k, v)| { - // The slot must still hold this - // very name and a plain function - // (version guard makes churn - // rare, not impossible-to-miss). - match (&k.0, v) { - (Object::Str(s), Object::Function(_)) - if &**s == name => - { - Some(v.clone()) - } - _ => None, + let mro = cls.mro.borrow(); + mro.get(mro_idx as usize).and_then(|owner| { + let od = owner.dict.borrow(); + od.get_index(key_idx as usize).and_then(|(k, v)| { + // The slot must still hold this + // very name and a plain function + // (version guard makes churn + // rare, not impossible-to-miss). + match (&k.0, v) { + (Object::Str(s), Object::Function(_)) + if &**s == name => + { + Some(v.clone()) } - }) + _ => None, + } }) - }; - if let Some(func) = func { - frame.pop()?; - specialize::record_hit(op_idx); - return Ok(Object::BoundMethod(Rc::new(BoundMethod::new( - receiver, func, - )))); - } + }) + } else { + None } + } else { + None } + } else { + None } + } else { + None + }; + if let Some(func) = func { + specialize::record_hit(op_idx); + // The receiver moves into the bound method — no clone. + return Ok(Object::BoundMethod(Rc::new(BoundMethod::new( + receiver, func, + )))); } - self.deopt_load_attr_slow(frame, cache_pc, name_idx) + self.deopt_load_attr_taken(frame, cache_pc, name_idx, receiver) } IC::LoadAttrType { type_id, key_idx, ver, } => { - let receiver = frame.top()?.clone(); - if let Object::Instance(inst) = &receiver { - let cls = inst.cls(); + let hit = if let Object::Instance(inst) = &receiver { + let cls = inst.class.borrow(); if specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver { // Per-instance shadow guard: an instance dict entry // beats a plain class attribute. The specialised @@ -21224,49 +21609,52 @@ impl Interpreter { } }; if shadowed { - return self.deopt_load_attr_slow(frame, cache_pc, name_idx); - } - let dict = cls.dict.borrow(); - if let Some((k, v)) = dict.get_index(key_idx as usize) { - if self.cached_slot_name_matches(&frame.code, name_idx, k) { - let v = v.clone(); - drop(dict); - frame.pop()?; - specialize::record_hit(op_idx); - // For function descriptors found on the - // type we'd normally bind to the - // instance — bail to the slow path - // when the value is callable, so the - // generic descriptor protocol runs. - // (Bound-method specialization is RFC - // 0022 territory.) `Instance` values are - // deopted too: one could be a non-data - // descriptor (`cached_property`) whose - // `__get__` must run. - if matches!( - v, - Object::Function(_) - | Object::Builtin(_) - | Object::Property(_) - | Object::ClassMethod(_) - | Object::StaticMethod(_) - | Object::SlotDescriptor(_) - | Object::Instance(_) - ) { - // Push receiver back and deopt. - frame.push(receiver); - return self.deopt_load_attr_slow(frame, cache_pc, name_idx); + None + } else { + let dict = cls.dict.borrow(); + match dict.get_index(key_idx as usize) { + Some((k, v)) + if self.cached_slot_name_matches(&frame.code, name_idx, k) => + { + Some(v.clone()) } - return Ok(v); + _ => None, } } + } else { + None + } + } else { + None + }; + if let Some(v) = hit { + // For function descriptors found on the type we'd + // normally bind to the instance — bail to the slow + // path when the value is callable, so the generic + // descriptor protocol runs. (Bound-method + // specialization is RFC 0022 territory.) `Instance` + // values are deopted too: one could be a non-data + // descriptor (`cached_property`) whose `__get__` + // must run. + if !matches!( + v, + Object::Function(_) + | Object::Builtin(_) + | Object::Property(_) + | Object::ClassMethod(_) + | Object::StaticMethod(_) + | Object::SlotDescriptor(_) + | Object::Instance(_) + ) { + specialize::record_hit(op_idx); + return Ok(v); } } - self.deopt_load_attr_slow(frame, cache_pc, name_idx) + self.deopt_load_attr_taken(frame, cache_pc, name_idx, receiver) } IC::Empty => { let name = self.name_at(&frame.code, name_idx)?; - let obj = frame.pop()?; + let obj = receiver; let result = self.load_attr(&obj, &name); // Specialize *after* the generic lookup. The decision's // dict-index probe hashes `name` into the instance dict and @@ -21295,34 +21683,34 @@ impl Interpreter { IC::Empty }; frame.code.caches.set(cache_pc, next); - let obj = frame.pop()?; let name = self.name_at(&frame.code, name_idx)?; - self.load_attr(&obj, &name) + self.load_attr(&receiver, &name) } _ => { - let obj = frame.pop()?; let name = self.name_at(&frame.code, name_idx)?; - self.load_attr(&obj, &name) + self.load_attr(&receiver, &name) } } } - /// Deopt a `LOAD_ATTR` cache and run the generic handler. + /// Deopt a `LOAD_ATTR` cache and run the generic handler. The + /// receiver has already been popped by `specialized_load_attr`'s + /// shared prologue (RFC 0061 WS4). #[inline] - fn deopt_load_attr_slow( + fn deopt_load_attr_taken( &mut self, frame: &mut Frame, cache_pc: u32, name_idx: u32, + receiver: Object, ) -> Result { specialize::record_miss(OpCode::LoadAttr as u8); frame .code .caches .set(cache_pc, weavepy_compiler::InlineCache::Cooldown(COOLDOWN)); - let obj = frame.pop()?; let name = self.name_at(&frame.code, name_idx)?; - self.load_attr(&obj, &name) + self.load_attr(&receiver, &name) } /// Specialized `STORE_ATTR`. Stack discipline matches the @@ -21338,16 +21726,22 @@ impl Interpreter { use weavepy_compiler::InlineCache as IC; let cache = frame.code.caches.get(cache_pc); let op_idx = OpCode::StoreAttr as u8; + // RFC 0061 (WS4): pop the receiver once (a move) instead of the + // per-arm `top().clone()` + later `pop()` pair; guards read the + // class through a borrow instead of `cls()`'s Arc clone. + let receiver = frame.pop()?; match cache { IC::StoreAttrInstance { type_id, key_idx, ver, } => { - let receiver = frame.top()?.clone(); if let Object::Instance(inst) = &receiver { - let cls = inst.cls(); - if specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver { + let guard_ok = { + let cls = inst.class.borrow(); + specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver + }; + if guard_ok { // Validate the cached index still holds *this* name: // a `del` on an earlier attribute shift-renumbers // every later slot (same guard as LOAD_ATTR). @@ -21358,7 +21752,6 @@ impl Interpreter { }) }; if name_ok { - frame.pop()?; let val = frame.pop()?; // Mirror the slow path: a bound method stored // on the instance must join the cycle collector @@ -21387,13 +21780,15 @@ impl Interpreter { } } } - self.deopt_store_attr_slow(frame, cache_pc, name_idx) + self.deopt_store_attr_taken(frame, cache_pc, name_idx, receiver) } IC::StoreAttrNewKey { type_id, ver } => { - let receiver = frame.top()?.clone(); if let Object::Instance(inst) = &receiver { - let cls = inst.cls(); - if specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver { + let guard_ok = { + let cls = inst.class.borrow(); + specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver + }; + if guard_ok { // Validated at specialisation time (and guarded by // `ver` since): stock `__setattr__`, no data // descriptor for this name anywhere on the MRO, @@ -21403,7 +21798,6 @@ impl Interpreter { // by a single hash probe. let code = frame.code.clone(); if let Some(name) = code.names.get(name_idx as usize) { - frame.pop()?; let val = frame.pop()?; // Mirror `generic_setattr_instance`: a bound // method escaping into an instance attribute @@ -21435,13 +21829,15 @@ impl Interpreter { } } } - self.deopt_store_attr_slow(frame, cache_pc, name_idx) + self.deopt_store_attr_taken(frame, cache_pc, name_idx, receiver) } IC::StoreAttrSlot { type_id, ver } => { - let receiver = frame.top()?.clone(); if let Object::Instance(inst) = &receiver { - let cls = inst.cls(); - if specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver { + let guard_ok = { + let cls = inst.class.borrow(); + specialize::rc_id(&cls) == type_id && cls.attr_version.get() == ver + }; + if guard_ok { // The class was validated at specialisation time: a // genuine `__slots__` member descriptor for this // name on a stock-lookup class. The version guard @@ -21452,7 +21848,6 @@ impl Interpreter { // prompt-reaps). let code = frame.code.clone(); if let Some(name) = code.names.get(name_idx as usize) { - frame.pop()?; let val = frame.pop()?; inst.slot_set(name, val); specialize::record_hit(op_idx); @@ -21460,10 +21855,9 @@ impl Interpreter { } } } - self.deopt_store_attr_slow(frame, cache_pc, name_idx) + self.deopt_store_attr_taken(frame, cache_pc, name_idx, receiver) } IC::Empty => { - let receiver = frame.top()?.clone(); let name = self.name_at(&frame.code, name_idx)?; specialize::record_specialize_attempt(op_idx); let decision = specialize::attempt_specialize_store_attr(&receiver, &name); @@ -21473,9 +21867,8 @@ impl Interpreter { } else { specialize::record_specialize_success(op_idx); } - let obj = frame.pop()?; let val = frame.pop()?; - self.store_attr(&obj, &name, val) + self.store_attr(&receiver, &name, val) } IC::Cooldown(n) => { let next = if n > 0 { @@ -21484,37 +21877,36 @@ impl Interpreter { IC::Empty }; frame.code.caches.set(cache_pc, next); - let obj = frame.pop()?; let val = frame.pop()?; let name = self.name_at(&frame.code, name_idx)?; - self.store_attr(&obj, &name, val) + self.store_attr(&receiver, &name, val) } _ => { - let obj = frame.pop()?; let val = frame.pop()?; let name = self.name_at(&frame.code, name_idx)?; - self.store_attr(&obj, &name, val) + self.store_attr(&receiver, &name, val) } } } - /// Deopt a `STORE_ATTR` cache. + /// Deopt a `STORE_ATTR` cache. The receiver has already been popped + /// by `specialized_store_attr`'s shared prologue (RFC 0061 WS4). #[inline] - fn deopt_store_attr_slow( + fn deopt_store_attr_taken( &mut self, frame: &mut Frame, cache_pc: u32, name_idx: u32, + receiver: Object, ) -> Result<(), RuntimeError> { specialize::record_miss(OpCode::StoreAttr as u8); frame .code .caches .set(cache_pc, weavepy_compiler::InlineCache::Cooldown(COOLDOWN)); - let obj = frame.pop()?; let val = frame.pop()?; let name = self.name_at(&frame.code, name_idx)?; - self.store_attr(&obj, &name, val) + self.store_attr(&receiver, &name, val) } /// Specialized `FOR_ITER`. Returns `Ok(true)` when the fast @@ -29576,14 +29968,18 @@ impl Interpreter { use weavepy_compiler::InlineCache as IC; let op_idx = OpCode::Call as u8; let split_at = frame.stack.len().saturating_sub(argc); - let mut args: Vec = frame.stack.split_off(split_at); + // RFC 0061 (WS3b): stage the operands through a pooled vector + // (`drain` moves them without the fresh `split_off` allocation); + // the shared tail hands the emptied vector back to the pool. + let mut args: Vec = self.pooled_scratch(); + args.extend(frame.stack.drain(split_at..)); let callable = frame.pop()?; // PEP 669 CALL fires for every `CALL`-family instruction in // monitored code, before the target runs — with the operands // as seen at the call site (a zero-arg `super()` reports // MISSING, before the `__class__`/`self` injection below). let mon_call = crate::trace::any_observers_active() - && crate::trace::monitoring_union_mask() + && crate::trace::monitoring_union_mask_cached() & crate::trace::event_mask(crate::trace::EVENT_CALL) != 0; if mon_call { @@ -29661,7 +30057,7 @@ impl Interpreter { }; let code = frame.code.clone(); let result = self.call_from_bytecode(&callable, &mut args, &frame.globals); - let still_on = crate::trace::monitoring_union_mask() + let still_on = crate::trace::monitoring_union_mask_cached() & crate::trace::event_mask(crate::trace::EVENT_CALL) != 0; match result { @@ -29777,9 +30173,13 @@ impl Interpreter { _ => {} } let f = f.clone(); - let mut combined: Vec = Vec::with_capacity(args.len() + 1); + // RFC 0061 (WS3b): pooled receiver+args staging; the + // drained `args` goes straight back to the pool and + // `combined` is recycled by `make_frame`'s fill. + let mut combined: Vec = self.pooled_scratch(); combined.push(bm.receiver.clone()); combined.append(&mut args); + self.recycle_scratch(args); drop(callable); let r = self.call_python_owned(&f, combined, Vec::new())?; frame.push(r); @@ -29998,6 +30398,7 @@ impl Interpreter { // the call has returned and the operands are about to be dropped. self.reap_call_receiver(callable); self.reap_call_args(&mut args); + self.recycle_scratch(args); Ok(()) } @@ -30096,16 +30497,13 @@ impl Interpreter { fn run_py_exact_nofree( &mut self, f: &Rc, - args: Vec, + mut args: Vec, ) -> Result { let code = f.code(); - let locals_rc = self.pooled_locals(code.varnames.len()); - { - let mut locals = locals_rc.borrow_mut(); - for (slot, v) in args.into_iter().enumerate() { - locals[slot] = v; - } - } + // RFC 0061 (WS3b): move the arguments straight into the leading + // locals slots and recycle the drained staging vector. + let locals_rc = self.pooled_locals_from_args(&mut args, code.varnames.len()); + self.recycle_scratch(args); let mut frame = Frame { code, locals: locals_rc, @@ -30125,6 +30523,7 @@ impl Interpreter { cleanup_lasti: None, suppress_call_event: false, gen_first_resume: false, + shell_cache: None, }; self.run_frame(&mut frame) } @@ -39182,6 +39581,67 @@ pub fn constant_to_object_public(c: Constant) -> Object { constant_to_object(c) } +/// RFC 0061 (WS2b): `WEAVEPY_NO_FUSE=1` disables fused-dispatch marker +/// installation — a bisection escape hatch; already-installed markers +/// are inert because sites are only marked when this returns true. +fn fusion_enabled() -> bool { + static ON: std::sync::OnceLock = std::sync::OnceLock::new(); + *ON.get_or_init(|| std::env::var_os("WEAVEPY_NO_FUSE").is_none()) +} + +/// RFC 0061 (WS3a): shared placeholder fields for parked +/// [`FrameShell`](crate::object::FrameShell)s. A pooled shell is +/// unreachable by construction (the pool owns its only `Rc`), so the +/// placeholders are never read or mutated through it — they exist only +/// so parking pins no user objects and allocates nothing. +fn empty_code_placeholder() -> Rc { + static P: std::sync::OnceLock> = std::sync::OnceLock::new(); + P.get_or_init(|| Rc::new(CodeObject::default())).clone() +} + +fn empty_locals_placeholder() -> Rc>> { + static P: std::sync::OnceLock>>> = std::sync::OnceLock::new(); + P.get_or_init(|| Rc::new(RefCell::new(Vec::new()))).clone() +} + +fn empty_dict_placeholder() -> Rc> { + static P: std::sync::OnceLock>> = + std::sync::OnceLock::new(); + P.get_or_init(|| Rc::new(RefCell::new(crate::object::DictData::default()))) + .clone() +} + +/// RFC 0061 (WS2a): the per-code-object materialized constant table, +/// living in the code object's [`weavepy_compiler::VmExt`] slot. Built +/// once on first access; `LOAD_CONST` becomes an indexed `Object` clone +/// instead of a per-execution `Constant` deep-clone (string constants +/// allocated twice per execution before this). +struct CodeConstObjects { + objects: Vec, +} + +/// The materialized constants for `code`, built on first use. The +/// returned reference is tied to the code object (the `OnceLock` pins +/// the `Arc` for the code object's lifetime). +#[inline] +fn code_const_objects(code: &CodeObject) -> &[Object] { + let arc = code.vm_ext.0.get_or_init(|| { + std::sync::Arc::new(CodeConstObjects { + objects: code + .constants + .iter() + .map(|c| constant_to_object(c.clone())) + .collect(), + }) + }); + match arc.downcast_ref::() { + Some(t) => &t.objects, + // Another subsystem claimed the slot first (none exists today, + // but the slot is deliberately generic): fall back per-call. + None => &[], + } +} + fn constant_to_object(c: Constant) -> Object { match c { Constant::None => Object::None, @@ -41619,6 +42079,83 @@ mod tests { assert_eq!(out, run(src)); } + // RFC 0061 WS5 — pinned-list subscript lanes. + + #[cfg(feature = "jit")] + #[test] + fn jit_list_int_sum_compiles_clean() { + // `xs[i]` on a homogeneous int list: the shape probe pins the + // lane, reads go through `wpjit_list_get`, and a clean kernel + // never deopts. + let src = "def ssum(xs, n):\n t = 0\n i = 0\n\ + \x20 while i < n:\n t = t + xs[i]\n i = i + 1\n\ + \x20 return t\n\ + xs = [x * 2 for x in range(30)]\n\ + r = 0\nk = 0\n\ + while k < 60:\n r = ssum(xs, 30)\n k = k + 1\n\ + print(r)\n"; + let (out, compiled, deopts) = run_jit(src); + assert!(compiled >= 1, "JIT never compiled the int-list kernel"); + assert_eq!(deopts, 0, "clean int-list kernel should not deopt"); + assert_eq!(out, "870\n"); + assert_eq!(out, run(src)); + } + + #[cfg(feature = "jit")] + #[test] + fn jit_list_float_lane_and_store() { + // Float lane, plus `xs[i] = v` through `wpjit_list_set`: the + // list's contents after native execution must match the + // interpreter's exactly. + let src = "def scale(xs, n):\n i = 0\n\ + \x20 while i < n:\n xs[i] = xs[i] * 1.5\n i = i + 1\n\ + \x20 return 0\n\ + xs = [float(x) for x in range(20)]\n\ + k = 0\n\ + while k < 60:\n scale(xs, 20)\n k = k + 1\n\ + print(xs[1])\nprint(xs[19])\n"; + let (out, compiled, _deopts) = run_jit(src); + assert!(compiled >= 1, "JIT never compiled the float-list kernel"); + assert_eq!(out, run(src), "float store diverged from interpreter"); + } + + #[cfg(feature = "jit")] + #[test] + fn jit_list_negative_index_and_bounds_deopt() { + // Negative indices resolve natively; an out-of-range access + // deopts and the interpreter raises the real IndexError. + let src = "def pick(xs, i):\n return xs[i]\n\ + xs = [x for x in range(10)]\n\ + r = 0\nk = 0\n\ + while k < 60:\n r = r + pick(xs, 0 - 1)\n k = k + 1\n\ + print(r)\n\ + try:\n pick(xs, 99)\nexcept IndexError:\n print('bounds ok')\n"; + let (out, compiled, deopts) = run_jit(src); + assert!(compiled >= 1, "JIT never compiled the pick kernel"); + assert!(deopts >= 1, "out-of-range subscript must deopt"); + assert_eq!(out, "540\nbounds ok\n"); + assert_eq!(out, run(src)); + } + + #[cfg(feature = "jit")] + #[test] + fn jit_list_heterogeneous_element_deopts() { + // The list goes heterogeneous *after* the compile: the per- + // access lane check in `wpjit_list_get` catches the str element + // and the interpreter finishes the concatenation-free path. + let src = "def pick(xs, i):\n return xs[i]\n\ + xs = [x for x in range(10)]\n\ + k = 0\n\ + while k < 60:\n pick(xs, 3)\n k = k + 1\n\ + xs[3] = 'boom'\n\ + print(pick(xs, 3))\n"; + let (out, compiled, deopts) = run_jit(src); + assert!(compiled >= 1, "JIT never compiled the pick kernel"); + assert!(deopts >= 1, "heterogeneous element must deopt"); + assert_eq!(out, "boom\n"); + assert_eq!(out, run(src)); + } + #[test] fn list_comprehension() { let src = "xs = [x * x for x in range(4)]\nprint(xs)\n"; diff --git a/crates/weavepy-vm/src/stdlib/marshal_mod.rs b/crates/weavepy-vm/src/stdlib/marshal_mod.rs index 371c818..29efa59 100644 --- a/crates/weavepy-vm/src/stdlib/marshal_mod.rs +++ b/crates/weavepy-vm/src/stdlib/marshal_mod.rs @@ -1126,6 +1126,7 @@ impl<'a> MarshalReader<'a> { qualname: co_qualname, filename: string_of(&filename, "co_filename")?, caches: CacheTable::with_len(decoded.instructions.len()), + vm_ext: weavepy_compiler::VmExt::default(), instructions: decoded.instructions, constants: tuple_to_constants(&consts)?, names: tuple_of_strings(&names, "co_names")?, diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index a7bea47..f684143 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -120,6 +120,8 @@ impl JitState { // containing calls. Registered unconditionally (it only stores a // fn pointer) so late enabling, e.g. via the test hook, works. weavepy_jit::register_call_py_helper(wpjit_call_py); + // RFC 0061 WS5 — same for the pinned-list access helpers. + weavepy_jit::register_list_helpers(wpjit_list_get, wpjit_list_set); JitState { enabled, threshold, @@ -135,13 +137,16 @@ impl JitState { /// frame's namespaces (used both to classify globals for analysis /// and to snapshot the guard expectations); `ret_lane_of` reports a /// candidate Python callee's stable scalar return lane (RFC 0059 - /// WS3). Returns the compiled frame + guard snapshot + callee table - /// when one is available. + /// WS3); `probe_list` reports a subscripted local's observed element + /// lane in the requesting activation (RFC 0061 WS5). Returns the + /// compiled frame + guard snapshot + callee table when one is + /// available. fn get_compiled( &mut self, code: &Rc, resolve_obj: &mut dyn FnMut(&str) -> Option, ret_lane_of: &mut dyn FnMut(&Rc, &Rc) -> Option, + probe_list: &mut dyn FnMut(u32) -> Option, ) -> Option { let key = Rc::as_ptr(code).cast::(); { @@ -212,7 +217,7 @@ impl JitState { } classify_global(obj.as_ref()) }; - let (tier, out) = match engine.compile(code, &mut classify) { + let (tier, out) = match engine.compile_with_probe(code, &mut classify, probe_list) { Ok(cf) => { self.stats.frames_compiled += 1; // Snapshot the exact objects the guards must keep @@ -351,24 +356,45 @@ fn callee_ret_lane( lane } +/// One activation's pinned lists (RFC 0061 WS5): slot bits tagged +/// [`SlotTag::ListPin`] index this table, which pairs the list storage +/// with the element lane the compile assumed. +type PinTable = Vec<(Rc>>, JitType)>; + /// Reconstruct an [`Object`] from a `(bits, tag)` slot. `Boxed` never /// appears in locals or ordinary spills (the parked result travels -/// through [`CallCtx::parked`]); map it defensively to `None`. +/// through [`CallCtx::parked`]); map it defensively to `None`, likewise +/// a `ListPin` reaching a context without pin-table access. fn unpack(bits: u64, tag: u32) -> Object { match SlotTag::from_raw(tag) { SlotTag::Int => Object::Int(bits as i64), SlotTag::Float => Object::Float(f64::from_bits(bits)), SlotTag::Bool => Object::Bool(bits != 0), - SlotTag::Boxed => Object::None, + SlotTag::Boxed | SlotTag::ListPin => Object::None, + } +} + +/// As [`unpack`] with the activation's pin table at hand, so a +/// [`SlotTag::ListPin`] slot rebuilds into its real list object +/// (RFC 0061 WS5). +fn unpack_pins(bits: u64, tag: u32, pins: &PinTable) -> Object { + match SlotTag::from_raw(tag) { + SlotTag::ListPin => pins + .get(bits as usize) + .map_or(Object::None, |(l, _)| Object::List(l.clone())), + _ => unpack(bits, tag), } } /// Reconstruct an [`Object`] from a slot whose lane is statically known. -fn unpack_ty(bits: u64, ty: JitType) -> Object { +fn unpack_ty(bits: u64, ty: JitType, pins: &PinTable) -> Object { match ty { JitType::Int => Object::Int(bits as i64), JitType::Float => Object::Float(f64::from_bits(bits)), JitType::Bool => Object::Bool(bits != 0), + JitType::ListInt | JitType::ListFloat => pins + .get(bits as usize) + .map_or(Object::None, |(l, _)| Object::List(l.clone())), JitType::Unknown => Object::None, } } @@ -384,6 +410,50 @@ fn pack(obj: &Object, ty: JitType) -> Option { } } +/// Entry-guard check for one managed local (RFC 0061 WS5): a scalar +/// lane must pack; a pinned-list lane must hold a `list` whose *first* +/// element matches the compiled element lane (an O(1) proxy for the +/// probe's full scan — the access helpers re-validate per element, so +/// a heterogeneous tail costs a deopt, never correctness). +fn entry_local_ok(obj: &Object, ty: JitType) -> bool { + let Some(elem) = ty.elem_lane() else { + return pack(obj, ty).is_some(); + }; + let Object::List(l) = obj else { + return false; + }; + matches!( + (l.borrow().first(), elem), + (None, _) | (Some(Object::Int(_)), JitType::Int) | (Some(Object::Float(_)), JitType::Float) + ) +} + +/// The compile-time shape probe (RFC 0061 WS5): report the element lane +/// of local `slot` when it currently holds a homogeneous non-empty +/// `int` or `float` list, `None` otherwise (an empty list has no +/// evidence to predict a lane from). +fn probe_list_lane(frame: &super::Frame, slot: u32) -> Option { + let locals = frame.locals.borrow(); + let Some(Object::List(l)) = locals.get(slot as usize) else { + return None; + }; + let items = l.borrow(); + let mut lane: Option = None; + for it in items.iter() { + let t = match it { + Object::Int(_) => JitType::Int, + Object::Float(_) => JitType::Float, + _ => return None, + }; + match lane { + None => lane = Some(t), + Some(cur) if cur == t => {} + Some(_) => return None, + } + } + lane +} + /// Bump the back-edge hot counter for a code object. Returns `true` /// when the caller should attempt an OSR entry (RFC 0059 WS3b); always /// `false` when the JIT is disabled. @@ -463,6 +533,9 @@ struct CallCtx { parked: Option, /// A raised callee's exception, parked for the `Raised` exit. raised: Option, + /// RFC 0061 WS5 — this activation's pinned lists, indexed by the + /// pin bits native code carries in `SlotTag::ListPin` slots. + pins: PinTable, } /// `true` while every burned-in resolution still holds: each guarded @@ -553,7 +626,9 @@ unsafe extern "C" fn wpjit_call_py( SlotTag::Int => JitType::Int, SlotTag::Float => JitType::Float, SlotTag::Bool => JitType::Bool, - SlotTag::Boxed => JitType::Unknown, + // A list-lane call result is rejected at emission; + // `Unknown` never packs, forcing the boxed path. + SlotTag::Boxed | SlotTag::ListPin => JitType::Unknown, }; if let Some(bits) = pack(&v, expect) { jf.ret_bits = bits; @@ -567,6 +642,83 @@ unsafe extern "C" fn wpjit_call_py( } } +/// The `wpjit_list_get` helper (RFC 0061 WS5): read one element of a +/// pinned list. Returns `0` with the element's bits in +/// [`JitFrame::ret_bits`], or non-zero to deopt — out of range, or the +/// element no longer matches the pinned lane (aliased mutation through +/// a callee). Never runs Python code and never drops a heap object. +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]. +unsafe extern "C" fn wpjit_list_get(frame: *mut JitFrame, pin: i64, idx: i64) -> i64 { + // SAFETY: see wpjit_call_py — same live-buffer contract. + let jf = unsafe { &mut *frame }; + #[allow(clippy::cast_ptr_alignment)] + let ctx = unsafe { &mut *jf.ctx.cast::() }; + let Some((list, elem)) = ctx.pins.get(pin as usize) else { + return 1; + }; + let items = list.borrow(); + let len = items.len() as i64; + let i = if idx < 0 { idx + len } else { idx }; + if i < 0 || i >= len { + return 1; + } + match (&items[i as usize], elem) { + (Object::Int(v), JitType::Int) => { + jf.ret_bits = *v as u64; + 0 + } + (Object::Float(f), JitType::Float) => { + jf.ret_bits = f.to_bits(); + 0 + } + _ => 1, + } +} + +/// The `wpjit_list_set` helper (RFC 0061 WS5): write one element of a +/// pinned list. The value's bits are pre-staged in +/// [`JitFrame::ret_bits`], interpreted per the pin's element lane. +/// Deopts (non-zero) when out of range or when the displaced element +/// is a heap object — replacing it here would drop it inside the +/// helper, and the drop-site machinery (prompt reap, parked +/// finalizers) belongs to the interpreter's store path. +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]. +unsafe extern "C" fn wpjit_list_set(frame: *mut JitFrame, pin: i64, idx: i64) -> i64 { + // SAFETY: see wpjit_call_py — same live-buffer contract. + let jf = unsafe { &mut *frame }; + #[allow(clippy::cast_ptr_alignment)] + let ctx = unsafe { &mut *jf.ctx.cast::() }; + let Some((list, elem)) = ctx.pins.get(pin as usize) else { + return 1; + }; + let v = match elem { + JitType::Int => Object::Int(jf.ret_bits as i64), + JitType::Float => Object::Float(f64::from_bits(jf.ret_bits)), + _ => return 1, + }; + let mut items = list.borrow_mut(); + let len = items.len() as i64; + let i = if idx < 0 { idx + len } else { idx }; + if i < 0 || i >= len { + return 1; + } + let dst = &mut items[i as usize]; + if !matches!( + dst, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) { + return 1; + } + *dst = v; + 0 +} + /// Offer a fresh frame (pc 0, empty stack) to the JIT. See [`JitEntry`]. pub(crate) fn try_enter(interp: &mut super::Interpreter, frame: &mut super::Frame) -> JitEntry { // Phase 1: counter + compilation, holding the state borrow briefly. @@ -580,7 +732,8 @@ pub(crate) fn try_enter(interp: &mut super::Interpreter, frame: &mut super::Fram let frame_ref: &super::Frame = frame; let mut resolve = |name: &str| resolve_plain_global(interp_ref, frame_ref, name); let mut ret_of = |f: &Rc, c: &Rc| callee_ret_lane(interp_ref, f, c); - st.get_compiled(&frame.code, &mut resolve, &mut ret_of) + let mut probe = |slot: u32| probe_list_lane(frame_ref, slot); + st.get_compiled(&frame.code, &mut resolve, &mut ret_of, &mut probe) }); let Some(entry) = entry else { return JitEntry::Skip; @@ -608,8 +761,7 @@ pub(crate) fn try_enter(interp: &mut super::Interpreter, frame: &mut super::Fram }; let ok = locals .get(slot as usize) - .and_then(|o| pack(o, ty)) - .is_some(); + .is_some_and(|o| entry_local_ok(o, ty)); if !ok { JIT.with(|cell| cell.borrow_mut().stats.entry_guard_failures += 1); return JitEntry::Skip; @@ -636,7 +788,8 @@ pub(crate) fn try_enter_osr(interp: &mut super::Interpreter, frame: &mut super:: let frame_ref: &super::Frame = frame; let mut resolve = |name: &str| resolve_plain_global(interp_ref, frame_ref, name); let mut ret_of = |f: &Rc, c: &Rc| callee_ret_lane(interp_ref, f, c); - st.get_compiled(&frame.code, &mut resolve, &mut ret_of) + let mut probe = |slot: u32| probe_list_lane(frame_ref, slot); + st.get_compiled(&frame.code, &mut resolve, &mut ret_of, &mut probe) }); let Some(entry) = entry else { return JitEntry::Skip; @@ -667,7 +820,7 @@ pub(crate) fn try_enter_osr(interp: &mut super::Interpreter, frame: &mut super:: let n_real = frame.code.varnames.len(); for slot in 0..n_real { if let Some(ty) = cf.local_types.get(slot).copied().flatten() { - if locals.get(slot).and_then(|o| pack(o, ty)).is_none() { + if !locals.get(slot).is_some_and(|o| entry_local_ok(o, ty)) { drop(locals); return fail(&frame.code); } @@ -726,10 +879,21 @@ fn enter_compiled( let cf = &entry.cf; let n = cf.n_locals as usize; let mut locals_buf = vec![0u64; n]; + // RFC 0061 WS5 — pin every list-lane local: the slot carries an + // index into `pins`, and the table (not the slot) keeps the list + // alive and reachable for the access helpers and the deopt rebuild. + let mut pins: PinTable = Vec::new(); { let locals = frame.locals.borrow(); for (slot, dst) in locals_buf.iter_mut().enumerate() { if let Some(ty) = cf.local_types[slot] { + if let Some(elem) = ty.elem_lane() { + if let Some(Object::List(l)) = locals.get(slot) { + *dst = pins.len() as u64; + pins.push((l.clone(), elem)); + } + continue; + } *dst = locals.get(slot).and_then(|o| pack(o, ty)).unwrap_or(0); } } @@ -751,6 +915,7 @@ fn enter_compiled( builtins: frame.builtins.clone(), parked: None, raised: None, + pins, }; let mut jf = JitFrame { locals: locals_buf.as_mut_ptr(), @@ -785,7 +950,7 @@ fn enter_compiled( }); match status { - JitStatus::Returned => JitEntry::Ran(unpack(jf.ret_bits, jf.ret_tag)), + JitStatus::Returned => JitEntry::Ran(unpack_pins(jf.ret_bits, jf.ret_tag, &ctx.pins)), JitStatus::Deopt | JitStatus::Raised => { // Write back managed locals (synthetic range slots have no // interpreter home — they feed the iterator rebuild below), @@ -796,12 +961,12 @@ fn enter_compiled( for (slot, &bits) in locals_buf.iter().enumerate() { if let Some(ty) = cf.local_types[slot] { if let Some(dst) = locals.get_mut(slot) { - *dst = unpack_ty(bits, ty); + *dst = unpack_ty(bits, ty, &ctx.pins); } } } } - rebuild_stack(frame, entry, &locals_buf, &spill, &tags, &jf); + rebuild_stack(frame, entry, &locals_buf, &spill, &tags, &jf, &ctx.pins); if matches!(status, JitStatus::Raised) { // As though the CALL instruction just executed and // raised: pc points past it (`handle_exception` uses @@ -828,6 +993,7 @@ fn enter_compiled( /// live range iterators of enclosing rewritten loops (bottom), then the /// spilled temporaries with any *erased* callee objects re-inserted at /// their recorded interpreter-stack depths (RFC 0059 WS3). +#[allow(clippy::too_many_arguments)] fn rebuild_stack( frame: &mut super::Frame, entry: &CompiledEntry, @@ -835,6 +1001,7 @@ fn rebuild_stack( spill: &[u64], tags: &[u32], jf: &JitFrame, + pins: &PinTable, ) { let cf = &entry.cf; // RFC 0058 WS4 — iterators from the synthetic slots, outermost first. @@ -869,7 +1036,7 @@ fn rebuild_stack( .push(entry.callees[open[next].token as usize].0.clone()); next += 1; } - frame.stack.push(unpack(spill[i], tags[i])); + frame.stack.push(unpack_pins(spill[i], tags[i], pins)); } while next < open.len() { frame diff --git a/crates/weavepy-vm/src/trace.rs b/crates/weavepy-vm/src/trace.rs index 5b44d87..0cef4f9 100644 --- a/crates/weavepy-vm/src/trace.rs +++ b/crates/weavepy-vm/src/trace.rs @@ -53,6 +53,67 @@ fn observer_transition(was: bool, is: bool) { } _ => {} } + bump_observer_gen(); +} + +/// RFC 0061 (WS1a): monotonically bumped whenever any observer source — +/// per-thread trace/profile hooks, the all-threads fallbacks, or a +/// monitoring tool's event mask — changes. Hot paths cache derived +/// observer state ([`ObserverSnapshot`], [`monitoring_union_mask_cached`]) +/// keyed by this generation, so the per-instruction cost is one relaxed +/// load + compare instead of TLS walks and mask folds. +static OBSERVER_GEN: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); + +#[inline] +pub fn observer_gen() -> u64 { + OBSERVER_GEN.load(Ordering::Relaxed) +} + +#[inline] +fn bump_observer_gen() { + OBSERVER_GEN.fetch_add(1, Ordering::Release); +} + +/// A generation-stamped cache of the two observer facts the eval loop +/// needs per instruction: "is any observer active on this thread?" and +/// "which PEP 669 events does any tool want?". Kept as a *local* in the +/// dispatch loop (not on the `Interpreter`) because both derive from +/// thread-local hook tables and an interpreter can be driven from +/// different OS threads across C-API entries; a stack local can never +/// outlive its thread. +#[derive(Debug)] +pub struct ObserverSnapshot { + gen: u64, + pub any: bool, + pub mon_mask: u32, +} + +impl ObserverSnapshot { + /// A stale snapshot; the first [`Self::refresh`] derives. + pub fn new() -> Self { + Self { + gen: 0, + any: false, + mon_mask: 0, + } + } + + /// One relaxed load + compare on the no-change path. + #[inline(always)] + pub fn refresh(&mut self) { + let g = observer_gen(); + if g != self.gen { + self.gen = g; + self.any = any_observers_active(); + self.mon_mask = monitoring_union_mask(); + } + } +} + +impl Default for ObserverSnapshot { + fn default() -> Self { + Self::new() + } } thread_local! { @@ -534,6 +595,28 @@ pub fn monitoring_union_mask() -> u32 { MONITORING_TOOLS.with(|cell| cell.borrow().union_mask()) } +/// RFC 0061 (WS1a): generation-cached [`monitoring_union_mask`] for hot +/// opcode arms that gate per-instruction events (BRANCH/JUMP, +/// INSTRUCTION, the call-boundary PY_* probes). One thread-local read + +/// one relaxed load on the no-change path; the borrow-and-fold only runs +/// when an observer source actually changed. +#[inline] +pub fn monitoring_union_mask_cached() -> u32 { + thread_local! { + static CACHED: std::cell::Cell<(u64, u32)> = const { std::cell::Cell::new((0, 0)) }; + } + CACHED.with(|c| { + let (gen, mask) = c.get(); + let g = observer_gen(); + if g == gen { + return mask; + } + let fresh = monitoring_union_mask(); + c.set((g, fresh)); + fresh + }) +} + // The `sys.monitoring.DISABLE` / `MISSING` sentinels. Identity // objects: the dispatcher compares callback return values against // `DISABLE` by pointer, so the module namespace and the dispatcher diff --git a/docs/rfcs/0061-performance-wave-3-dormant-gates-fused-dispatch.md b/docs/rfcs/0061-performance-wave-3-dormant-gates-fused-dispatch.md new file mode 100644 index 0000000..8443dc5 --- /dev/null +++ b/docs/rfcs/0061-performance-wave-3-dormant-gates-fused-dispatch.md @@ -0,0 +1,558 @@ +# RFC 0061: Performance wave 3 — dormant-gate burn-down, fused dispatch, and allocation-free calls + +- **Status**: Accepted +- **Authors**: WeavePy authors +- **Created**: 2026-08-09 +- **Tracking issue**: TBD +- **Builds on**: RFC 0059 (the unified eval breaker and precision-gate + philosophy this wave extends to the observability and GC layers), + RFC 0058 (the measured bench lane, frame pools, and IC families), + RFC 0032/0059 (tier-2 Cranelift JIT + OSR, extended here with its + first container lane), RFC 0021 (the `InlineCache` side-table model + that WS2's fusion slots and WS4's pointer guards live in), + RFC 0049/0057/0060 (the conformance baseline and protocol that act + as this wave's no-regression guard), RFC 0055/0056 (the ecosystem + lane, same role). + +## Summary + +RFC 0059 cut the geomean from 9.92× to 8.51× of CPython by deleting +per-instruction dead weight and teaching the JIT calls + OSR. Its +Results section left the residual at "dispatch and `Object::clone` +traffic", and its future-work list sketched this wave. Before writing +a line of design we re-profiled the post-0059 binary (macOS `sample`, +1ms cadence, release + debuginfo) on the three fixtures that anchor +the slow tail — `attr_access` (13.35×), `call_overhead` (13.25×), and +`list_ops` (15.57×). The top-of-stack sample shares are unambiguous, +and three of the five biggest line items are *dormant machinery*, +exactly the class of tax RFC 0059 WS1 burned once already — one layer +down: + +1. **The PEP 669 union mask is recomputed through TLS per gate.** + `trace::monitoring_union_mask()` does a `thread_local!` access plus + a `GilCell` borrow plus a fold over the tool table — and the eval + loop consults it at several per-instruction and per-call gates. + With **zero monitoring tools registered** it is 6.5–7.9% of + on-CPU samples on all three fixtures (165/2530 `attr_access`, + 294 `call_overhead`, 363 `list_ops` top-of-stack samples). +2. **The prompt-reap suspect sweep scans instead of probing.** + `gc_trace::take_dead_suspects()` — the between-bytecodes + refcount-death probe behind CPython-faithful `__del__` timing — + locks the global `SUSPECTS` vector and walks **every** entry + (`is_tracked` → `GcState::handle_for` registry lookup + + `strong_count_for`) each time the `maybe_dead` gate fires, and + `remove_suspect` is a linear `retain` over the same vector on + every untrack. A method call that drops one bound temporary pays a + full sweep. Together `take_dead_suspects` + `handle_for` + + `remove_suspect` are **10.8% of `attr_access`, ~13% of + `list_ops`** (598 `handle_for` top-of-stack samples there — the + single hottest non-dispatch symbol). +3. **The call path allocates.** In `call_overhead`, the allocator + family (`xzm_malloc_tiny`/`xzm_free`/`malloc_zone_malloc`/`free`) + plus `memset`/`memmove` is **~12–14% of samples**: a fresh + `Arc` per call, locals-vector zero-fill, and argument + staging buffers — all on top of the RFC 0058 locals/stack pools + that were supposed to make calls allocation-free. +4. **IC guards re-derive types through the global registry.** + The `LOAD_ATTR`/`STORE_ATTR` fast paths guard by `type_id`, which + costs a `GilCell>>` borrow of the process-wide + type registry per hit (105–109 samples on every profiled fixture), + plus `PyInstance::cls` lookups. The guard should be one pointer + compare. +5. **Dispatch itself.** The loop prologue + (`run_until_yield_or_return_impl`, 17–18% self time) plus `step` + decode/match (15–16%) still dominate; `Object::clone` + + `drop_glue` (Arc refcount traffic feeding `LOAD_FAST`'s + clone-to-stack discipline) add ~10%; `_tlv_get_addr` — the TLS + walk feeding `MAYBE_DEAD`, the monitoring table, and the stats + gates — adds ~5%; and `constant_to_object` re-materializes + `LOAD_CONST` operands from compiler constants on **every + execution** (1.5% on `attr_access`, more on string-heavy code). + +None of this is architecture; all of it is rent paid per instruction +or per call for machinery that is off, or work redone that could be +cached. This wave burns the five line items in profile order, and +extends the tier-2 JIT with its first container lane (list subscript +load/store) so the `pyaes`/`list_ops`-shaped kernels stop being +permanently interpreter-bound under `WEAVEPY_JIT=1`. + +## Motivation + +Unchanged from RFC 0058/0059, sharpened by the conformance endgame: +RFC 0060 moved the `Lib/test` baseline to 515/548 with the ecosystem +lane at 29/29 — the compatibility bar for "drop-in" is now met for +every real package in the matrix, so speed is the loudest remaining +gap between the README's promise ("dramatically improving execution +speed") and the measured 8.49× geomean. The two prior perf waves +retired the structural excuses (an unusable harness, eager frames, +missing IC families, a JIT that couldn't run real functions). What's +left, per the fresh profiles, is precisely enumerable dead weight — +and this wave's philosophy is inherited verbatim from RFC 0059: + +1. **Precision before machinery.** The monitoring mask, the suspect + sweep, and the registry-borrow guards are not architectural + problems; they are imprecise gates and uncached derivations. + Making them precise is compatibility-neutral by construction (the + slow paths are unchanged, only entered less / fed cheaper). +2. **Fusion over layout.** The twice-deferred "contiguous + frame/data-stack layout" stays deferred, now with a reason + measured rather than assumed: the locals indirection + (`Rc>>`) is load-bearing for PEP 667 live + `f_locals` views and generator frame ownership, and the profile + shows the tax is not the allocation layout (pools already recycle + it) but the per-access *discipline* — borrows, clones, bounds. A + superinstruction layer that fuses the dominant instruction pairs + (and skips the receiver clone entirely for `LOAD_FAST`-fed + attribute/subscript ops) attacks the measured cost at a fraction + of the rewrite risk. +3. **The JIT grows one lane at a time.** RFC 0059 taught it calls and + OSR; the enumerated next multiplier is list element access. One + narrow, helper-backed, guard-checked container lane — not a + general object model in Cranelift. +4. **Measured, gated, honest.** Every claim lands in + `baselines/bench.json` under the RFC 0058 methodology, the CI gate + ratchets, and the full regrtest + ecosystem sweeps must hold at + baseline (`unexpected 0`, 29/29). + +## CPython reference + +- CPython 3.13's `sys.monitoring` (PEP 669) keeps per-tool event + masks in `_PyInterpreterState.monitoring_matrix` and consults a + **pre-folded per-code-object `_co_instrumentation` version** — the + union is never recomputed on the fast path; instrumentation is + compiled *into* the bytecode when tools attach + (`Python/instrumentation.c`). +- CPython frees objects the instant their refcount hits zero + (`Py_DECREF` → `_Py_Dealloc`), i.e. death detection is **event- + driven at the drop site**, never a scan. WeavePy's suspect sweep + emulates the observable ordering under `Arc` semantics; this wave + moves it to drop-site probing, matching CPython's cost model as + well as its semantics (the acceptance harness for timing stays + `test_io.test_error_through_destructor`, `test_eval_breaker.py`, + and the finalizer regrtests). +- CPython 3.13 calls are allocation-free in steady state: frames live + on a contiguous per-thread data stack (`_PyThreadState.datastack`), + and `_PyInterpreterFrame` is not a `PyObject` until someone asks + (`PyFrameObject` materializes lazily — the model RFC 0058 adopted; + this wave finishes the job for the *shell* allocation). +- CPython 3.11+ superinstructions (`LOAD_FAST__LOAD_FAST`, …) and the + PEP 659 adaptive interpreter fuse/specialize in place, gated off + when instrumentation is active — the same discipline WS2 adopts + (fusion is invisible under `sys.settrace`, which re-routes to the + unfused generic path). +- CPython type checks on IC fast paths are pointer compares + (`Py_TYPE(obj) == cached_type` + `tp_version_tag`), never registry + lookups — the WS4 model. + +## Detailed design + +### WS1 — Dormant-gate burn-down, round 2 + +**WS1a: a folded monitoring mask.** A process-wide +`static MONITORING_UNION: AtomicU32`, maintained at the *mutation* +sites (`sys.monitoring.register_callback` / `set_events` / +`set_local_events` / tool free, plus `settrace`/`setprofile` bridge +attach/detach). Every hot-path `monitoring_union_mask()` caller reads +one relaxed load; the TLS + `GilCell` + fold path survives only inside +the mutators. Since monitoring mutation already takes the tool-table +borrow, publication is a store at the end of the existing critical +sections. Cross-thread visibility piggybacks on the GIL hand-off +(SeqCst not required; the GIL is a full barrier). + +**WS1b: drop-site probing for prompt reap.** The suspect structure +becomes an `IndexMap, budget)>` (O(1) +`remove_suspect`, no more linear `retain`), and the per-instruction +sweep stops scanning: + +- Drop sites that displace a suspect-eligible object (the existing + `may_anchor_finalizable` predicate from RFC 0059 WS1b) additionally + record the *candidate's `ObjectId`* in a small interpreter-local + ring (`RecentDrops`, fixed 16 entries, overflow degrades to the + full-sweep flag — never lost precision, only lost cheapness). +- The eval-loop gate, when it fires, probes **only the ringed ids** + against the suspect map (O(ring) instead of O(suspects)), running + the exact per-entry logic that exists today (`is_tracked` → + strong-count fast reject → weakref-clone bound → reap). The + dormant-stride full re-probe survives unchanged as the periodic + backstop for entries whose death no drop site witnessed (weakref + clears, cross-thread drops), so no reap becomes *later* than today + — the stride tick is the same; the per-instruction scan between + ticks becomes a targeted probe. +- `GcState::handle_for` gets out of the probe entirely: the suspect + entry already owns the `Arc`; `is_tracked` becomes a + generation check on the handle itself rather than a registry + lookup. + +**WS1c: hot flags move onto the `Interpreter`.** The `MAYBE_DEAD` +thread-local `Cell`, the stats gate, and the observer/ +monitoring caches become plain fields on the `Interpreter` (one +pointer already in a register), refreshed where they can change +(observer registration bumps the existing generation counter; the +breaker word covers cross-thread setters). `_tlv_get_addr` leaves the +per-instruction path; the TLS cells stay as the cross-crate fallback +for paths that lack an interpreter (`weavepy-capi` release hooks). +`specialize::record_hit`/`record_miss` get `#[inline(always)]` +single-load fast gates (the profile shows a real call frame today). + +### WS2 — Dispatch and operand de-taxing + +**WS2a: a VM-side constant-object table.** Each `CodeObject` gains a +lazily-built, VM-owned `Vec` (one slot per `constants` entry, +built on first execution of the code object, living in the existing +VM side-structure keyed by code identity so `weavepy-compiler` stays +Object-free). `LOAD_CONST` becomes an indexed clone; string interning +and bigint materialization happen once per code object instead of per +execution. `constant_to_object` survives for the cold paths +(`co_consts` introspection, marshal). + +**WS2b: fused dispatch via the IC side table.** Superinstruction +fusion happens at *cache-population time*, never in the instruction +stream: `instructions`, `co_code`, `dis`, line tables, and jump +targets are untouched (the RFC 0033 introspection surface cannot tell +this wave happened). New `InlineCache` variants memoize a decoded +pair at the first instruction's `cache_pc`; the second slot's cache +stays whatever it was (a jump landing on the second instruction +executes it normally — fusion only short-circuits the *fall-through* +path). Fusion arms are gated on the interpreter's cached +observers-off flag (WS1c), so `sys.settrace`/`sys.monitoring` +attachment sees pure single-step execution, and every fused arm +re-checks its operand guards with the same deopt-to-generic + +`Cooldown` protocol every existing IC uses. The pairs, chosen from +static frequency in the bench corpus + the profiled fixtures: + +| fused cache | shape | why | +|---|---|---| +| `FuseLoadFastLoadFast` | two locals pushed | the most common pair in every corpus | +| `FuseLoadFastLoadConst` | local + constant | binop feeders | +| `FuseLoadFastLoadAttr{Instance,Slot,Method}` | attribute of a local | **skips the receiver clone/drop pair entirely** — the local is read in place, only the result is pushed (`self.x` costs one Arc bump, not three) | +| `FuseLoadFastBinarySubscr{ListInt,…}` | `local[stack_top]` | same borrow trick for the container | +| `FuseCompareIntPopJump{True,False}` | int compare + branch | loop conditions; no `Bool` push/pop/re-dispatch | +| `FuseBinaryOpIntStoreFast` / `FuseBinaryOpFloatStoreFast` | arith → local | `total += …` tails; result lands in the local without a stack round-trip | + +Fusion candidacy is decided when the *first* op's cache specializes +(the generic arm already proved the shape once) and requires: same +source line (line-event exactness under later trace attach), the pair +not spanning an exception-table boundary, and the second pc not being +a jump target for the *fused semantics* arms (`FuseCompareIntPopJump`, +`FuseBinaryOpStoreFast`) where skipping the intermediate stack state +must be unobservable. A `WEAVEPY_VM_STATS` counter family +(`fuse_hit`/`fuse_miss`/`fuse_blocked`) makes the fusion rate +auditable. + +**WS2c: prologue and fetch hygiene.** The instruction fetch drops the +per-step bounds check behind a compiler-guaranteed invariant (jump +targets are validated at code-object construction; a +`debug_assert!` keeps the check in debug builds). The prompt- +finalization and observer gates read the WS1c interpreter fields +(plain loads, no TLS). The `gil_countdown` and breaker check merge +into one decrement-and-test. Expected shape: ≤ 4 branches of prologue +for the nothing-pending case. + +### WS3 — Allocation-free calls + +**WS3a: `FrameShell` recycling.** A bounded per-interpreter freelist +(64 entries, mirroring the RFC 0058 locals/stack pools) recycles the +`Arc` allocation: on frame teardown, a shell that is +sole-owned (`Arc::strong_count == 1` — nobody materialized a +`PyFrame`, no traceback holds it) is reset via `Arc::get_mut` +(fields overwritten in place, `lasti`/flags re-armed) and pushed; +`push_frame_shell` pops before allocating. Shells that escaped stay +heap-owned exactly as today. + +**WS3b: argument staging without malloc.** Profile-driven burn of the +remaining per-call allocations in the `CallPyExact*` / +`CallBoundMethodExact` / `CallPyDefaults` fast paths: argument +vectors staged through a reusable interpreter scratch buffer instead +of fresh `Vec`s, defaults tails cloned slot-wise into pooled locals +(no intermediate collect), and the locals fill writing arguments +first + `Unbound` only for the residual slots (no full zero-fill + +overwrite). The acceptance probe is the allocator share of the +`call_overhead` profile dropping under 5%. + +**WS3c: generator resume de-taxing.** The `generators` fixture pays +call-shaped cost per `next()`: shell re-push, recursion-guard +arithmetic, and prologue re-entry. The suspended generator keeps its +shell `Arc` alive across yields (it already owns the frame); resume +re-pushes the *same* shell without re-deriving flags, and the +resume path skips the fresh-call bookkeeping that cannot apply to a +frame that already ran (annotations setup, defaults binding). + +### WS4 — Pointer-identity IC guards + +`LoadAttr{Instance,Slot,Method}` / `StoreAttr{Instance,Slot,NewKey}` +cache entries replace their `type_id`-keyed guard (which costs a +global type-registry `GilCell` borrow per hit) with: + +- the cached type's address (`usize` from `Arc::as_ptr`) — compared + against the instance's type pointer, one load + compare; +- the existing `attr_version` u32, unchanged (mutation invalidation + keeps its current semantics). + +To make the instance side a field read, `PyInstance` carries its +`Arc` directly (today's `cls()` resolves through the +registry — 29 top-of-stack samples on `attr_access`). The registry +stays authoritative for identity/GC bookkeeping; the instance's Arc +is a cache of the same value the registry holds, kept coherent +because a live instance pins its class (CPython semantics: an +instance's `__class__` assignment goes through the guarded setter, +which updates the carried Arc and bumps `attr_version`). The type +registry borrow leaves the attribute hot path entirely; `FOR_ITER`, +`BINARY_SUBSCR`, and `CALL` IC guards get the same treatment where +they currently key through the registry. + +### WS5 — Tier-2 JIT: the list lane + +One container lane, helper-backed, in the RFC 0059 WS3 mold (the JIT +still never sees the object model): + +- **Analysis**: `BinarySubscr`/`StoreSubscr` where the container is a + *pinned local* — a local slot whose type the fixpoint proves is + `list` throughout the region (assigned before the loop from a + shape the analyzer trusts: a parameter guarded at entry, or a + `BuildList` result) — and the index lane is `Int`. Element lanes: + `Int` or `Float`, established by an entry guard (homogeneity check) + and *maintained* by the lane itself (a store of the wrong lane is + impossible by construction; stores from unknown lanes bail at + analysis). +- **ABI**: a second registered helper family: + `wpjit_list_get(ctx, slot, idx, out_bits) -> ListStatus` and + `wpjit_list_set(ctx, slot, idx, bits, tag) -> ListStatus`. The VM + packs pinned-list locals as opaque entries in a per-entry + `PinnedObj` table on the `CallCtx` (the JIT sees only the slot + index — no pointers cross the boundary), the helper does the + bounds-checked, homogeneity-checked access against the real + `Object::List`, and `OutOfRange`/`WrongShape`/`Resized` statuses + deopt through the existing side-exit spill with a new + `PinnedListMeta` (mirroring `RangeLoopMeta`) that reinstates the + list object on the interpreter stack/locals at the deopt pc. +- **Guards**: list identity is pinned at entry/OSR pack time; there + is no aliasing hazard *within* the region because the only writes + the analyzer admits to that slot are the lane's own stores (a + `StoreFast` to the pinned slot bails analysis), and calls out + (`CallPy`) conservatively unpin — a region containing both a + pinned list op and an interpreted-fallback call keeps the current + `Unrepresentable`-style bail. +- **Scope honesty**: `append`/`len` and attribute lanes stay out of + this wave; the lane exists to prove the pinned-object ABI and to + move the `pyaes`-shaped kernels. The JIT remains off by default + behind the `jit` feature + `WEAVEPY_JIT=1`. + +## Compatibility + +- WS1 changes *when* dormant machinery is probed and *how* death + candidates are found, never what reaping does: the per-entry logic + is byte-identical, the dormant stride is unchanged, and the + finalizer-timing regrtests (`test_finalizers.py`, + `test_eval_breaker.py`, `test_gc_basic.py`, plus vendored + `test_io`'s destructor-error case) are the acceptance harness. +- WS2's fusion is invisible to introspection by construction (the + instruction stream, `co_code` re-encoding, `dis`, and line tables + are untouched) and to tracing by gating (observers-active routes + through the unfused generic arms; attach mid-loop deopts fused + slots via the existing cache-invalidation path). `test_sys_settrace` + and `test_monitoring` must not move from their measured rows. +- WS3's recycling preserves identity semantics: a shell or frame that + anyone can still observe (`PyFrame` materialized, traceback, + `gi_frame`) is never recycled — same sole-owner rule the RFC 0058 + pools established. +- WS4 changes the *representation* of the guard, not its strictness: + pointer + version subsumes id + version (an id can be reused only + after the type dies, which the carried Arc prevents while any + instance lives; `__class__` assignment and `type.__setattr__` + invalidation keep their existing version-bump discipline). +- WS5 extends the analyzer's accepted subset; everything else stays + `NotJitable`. Deopt fidelity gets dedicated regrtests (resize + mid-loop via an admitted store, out-of-range index, heterogeneous + seed data, tracing attach while native frames are live). +- `bench.json` stays v3; no schema change this wave. + +## Testing + +1. `cargo test --workspace` plus new unit tests: monitoring-mask + publication points, suspect-map probe/stride equivalence (a + property test driving random drop orders against the old scan as + oracle), shell-recycle sole-owner discipline, fusion guard + matrices, pointer-guard invalidation on `__class__` assignment + and type mutation, pinned-list deopt state reconstruction. +2. New regrtests under `tests/regrtest/`: `test_fused_dispatch.py` + (fusion + trace-attach mid-loop + dis/co_code invariance), + `test_prompt_reap_probe.py` (finalizer timing under drop-site + probing: `__del__` ordering, weakref callbacks, resurrection), + `test_jit_list_lane.py` (auto-skips off-`jit` builds). +3. The RFC 0049-protocol verification sweep: full + `regrtest --all-cpython --mode subprocess` (must hold + `unexpected 0` against the RFC 0060 baseline), ecosystem lane + 29/29 offline, `cargo fmt` / `clippy -D warnings`. +4. `cargo xbench run --update-baseline` + `gate` on the final binary; + the `--jit` column re-measured. + +## Acceptance criteria + +1. **Interpreted geomean ≤ 6.5×** CPython on the 20-fixture suite + (from 8.49×; ≥ 1.3× wall-clock at the geomean), no fixture + regressing beyond the gate's 10%. Stretch (non-blocking): ≤ 5.5×. +2. **The three profiled fixtures each improve ≥ 25%**: + `attr_access` ≤ 10.0×, `call_overhead` ≤ 9.9×, `list_ops` ≤ 11.7×. +3. **The dormant taxes are gone from the profile**: re-sampled + `attr_access` shows `monitoring_union_mask` + suspect-sweep + + type-registry-borrow symbols at ≤ 1.5% combined (from ~22%). +4. **`call_overhead`'s allocator share ≤ 5%** of samples (from + ~12–14%). +5. **The list lane is demonstrably live** under `WEAVEPY_JIT=1`: + `WEAVEPY_VM_STATS` counters show pinned-list compilation + native + hits on a list-kernel microbench, with `test_jit_list_lane.py` + covering the deopt matrix; the `--jit` column is reported for the + suite. +6. **Startup ratio ≤ 2.42×** (no regression), RSS column reported. +7. **Zero conformance cost**: full-sweep `unexpected 0`, ecosystem + 29/29, fmt/clippy/tests green. + +## Implementation notes and results + +### What shipped (deltas from the design sketch) + +- **WS1–WS4** landed as designed (observer snapshot + cached + monitoring mask, `untracked`-flag suspects over an `IndexMap`, + the per-code-object constant table behind `VmExt`, four fused + arms gated by `fuse_off`, FrameShell/locals/arg-staging pools, + pop-once receivers with `class`-field pointer guards). +- **WS3b amendment — separate scratch pool.** Validation profiling + caught the argument-staging vectors being recycled into the + *operand-stack* pool: staging vectors grow only to the hottest + call's argc, and handing such a small allocation out as an operand + stack made every push in the new frame re-grow it (a + `finish_grow` storm worth ~4% of `attr_access`). Staging now has + its own `scratch_pool`; `frame_stack_pool` is exclusively fed by + retired operand stacks again. +- **WS5** shipped with a simpler ABI than sketched: the helpers are + `wpjit_list_get/set(frame, pin, idx) -> status` with the value + staged through the dead-between-calls `ret_bits` slot, and the pin + table (`CallCtx::pins`) pairs each list with its element lane. The + element lane comes from an embedder *probe* over the entering + activation's local (homogeneous non-empty `int`/`float` list); + the entry guard is O(1) (is-a-list + first-element lane) because + the helpers re-validate shape per access and deopt on any surprise + — aliased mutation through a callee included. A pinned list may be + *returned* (the pin rebuilds through the table) but never passed + as a call argument, never truth-tested, and a lane-changing store + disqualifies at analysis. Deopt spills carry `SlotTag::ListPin` + and rebuild the real object from the table. + +### Landscape shift: the RFC 0060 endgame commit + +Between this RFC's baseline (`1623904`, the wave-2 tip that recorded +geomean 8.49×) and this wave's merge base, the conformance-endgame +commit (`b0ce10f`) landed and regressed the interpreter hot path +**25–37%** across the suite (pervasive `GcState::handle_for` lookups, +`monitoring_union_mask` recomputation, per-drop weakref-registry +counting, and finalizable-index sweeps). Measured on the bench +fixtures (self-timed region, same machine, back-to-back binaries): + +| fixture | wave-2 `1623904` | endgame `b0ce10f` | this wave | +|---|---|---|---| +| richards | 229ms | 311ms | 246ms | +| attr_access | 373ms | 492ms | 365ms | +| generators | 496ms | 670ms | 455ms | +| fib | 211ms | 294ms | 212ms | +| deltablue | 1015ms | 1277ms | 1035ms | +| call_overhead | 599ms | 747ms | 566ms | +| list_ops | 396ms | 548ms | 401ms | + +Against its **actual merge base** this wave delivers 15–30% per +fixture. Against the *recorded* wave-2 baseline the suite geomean +moves 8.49× → **8.41×** (new `bench.json` baseline): the wave +effectively paid down the endgame commit's regression first and +banked the remainder. The acceptance targets in this document +(geomean ≤ 6.5×, per-fixture ≤ 25%) were set against a merge base +that no longer exists; the dormant-tax and allocator-share criteria +(3, 4) hold on the re-sampled profiles, and criterion 5 holds (the +list lane compiles, runs natively, and deopts correctly under the +new `jit_list_*` unit tests). + +### Immediate follow-ups (endgame-commit hot-path burn-down) + +Re-profiling after this wave shows the remaining top taxes are the +endgame commit's own machinery, in priority order: + +1. per-drop `weakref_registry` `count`/`strong_clone_count` calls + (thread-local + `GilCell` borrow even on the bloom-filtered + miss path) — ~2%; +2. `GcState::handle_for` on paths that usually miss — ~1.5%; +3. the finalizable-index sweep cadence when *any* generator is + alive (`has_any_finalizable` is population-, not shape-gated) — + ~1%; +4. suspects `IndexMap` traffic at frame exits — ~1.5%. + +## Drawbacks + +- Fusion adds a second dispatch dimension (cache-variant × opcode) to + an interpreter that is already the project's most complex function; + the mitigation is that every fused arm is a composition of two + existing, tested arms plus a guard, the stats counters make the + layer observable, and `WEAVEPY_NO_FUSE=1` (debug escape hatch) + disables population for bisection. +- Drop-site probing rests on the ring capturing the common death + sites; a workload whose deaths are all cross-thread or + weakref-driven degrades to today's stride-gated scan (never worse, + but the win is workload-dependent). +- Carrying `Arc` on every instance costs one word per + instance and a refcount edge that the cycle GC must know about + (types already participate in tracing; the edge is added to the + trace function). +- The pinned-list ABI is the JIT's first object-adjacent surface; + scoping it to slot indices + helpers keeps `weavepy-jit` free of + the object model, but the deopt metadata grows another variant to + maintain. + +## Alternatives + +- **Contiguous frame + data-stack rewrite** (the CPython layout): + rejected again this wave, now on profile evidence — the allocation + layout is already pooled and the measured tax is access discipline + and dormant gates; the rewrite risks the PEP 667/generator + ownership model for a win the profile does not currently promise. + Revisit when the cheaper levers are spent. +- **Compile-time superinstructions** (fused opcodes in the + instruction stream): rejected — it would perturb `co_code` + re-encoding, `dis`, line tables, and the RFC 0033 introspection + contract, for the same runtime effect the IC-slot memoization + achieves invisibly. +- **A general tagged-pointer / NaN-boxed object model**: the deep fix + for clone traffic, and a full-runtime rewrite; the fusion layer + buys the dominant share (loads feeding one consumer) at ~1% of the + risk. Reconsider alongside free-threading, which changes the + refcount story anyway. + +## Prior art + +- CPython 3.11–3.13: superinstructions, PEP 659 inline caches with + pointer+version guards, instrumentation-compiled monitoring + (PEP 669), lazy frame objects, per-thread data stack. +- PyPy: guard-based list strategies (homogeneous int/float lists) — + the WS5 lane is the bounded, helper-backed cousin. +- V8/JSC: polymorphic inline caches keyed by map/shape pointer — + WS4's pointer-identity guard is the monomorphic form. + +## Unresolved questions + +- Should fused arms cover `LOAD_DEREF`-fed pairs (closure-heavy code + like `deltablue`)? Deferred until the stats counters say the + `LOAD_FAST` family is saturated. +- Whether the suspect ring should be per-thread rather than + per-interpreter once free-threading lands (same deferral as every + RFC since 0058). +- Whether `PinnedObj` slots should generalize to tuples/strings in a + follow-up lane, or whether attribute lanes (via WS4's pointer + guards burned into JIT entry guards) are the better next JIT + increment — decide on post-wave profiles. + +## Future work + +- Attribute lanes in tier-2 (the WS4 pointer guards are the + prerequisite this wave lands). +- `list.append` / `len` in the pinned-list lane; tuple/str lanes. +- The tagged-value model, revisited with free-threading. +- Windows/Linux measured bench baselines (the harness runs there; + the checked-in ratios are macOS-arm64). +- Memory-ratio gating (two waves of `max_rss_bytes` history now + exist — flip the reported column to a gated one next wave).